Skip to content

Commit b1c97ad

Browse files
authored
Define the ETC algorithm (#15)
2 parents dee19a3 + 070e2a3 commit b1c97ad

9 files changed

Lines changed: 330 additions & 87 deletions

File tree

‎LeanBandits.lean‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,5 @@
1+
import LeanBandits.AlgorithmBuilding
12
import LeanBandits.Bandit
3+
import LeanBandits.ETC
24
import LeanBandits.Regret
35
import LeanBandits.UCB

‎LeanBandits/AlgorithmBuilding.lean‎

Lines changed: 132 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,132 @@
1+
/-
2+
Copyright (c) 2025 Rémy Degenne. All rights reserved.
3+
Released under Apache 2.0 license as described in the file LICENSE.
4+
Authors: Rémy Degenne
5+
-/
6+
import LeanBandits.Bandit
7+
8+
/-! # Tools to build bandit algorithms
9+
10+
-/
11+
12+
open MeasureTheory ProbabilityTheory Finset
13+
open scoped ENNReal NNReal
14+
15+
section MeasurableArgmax -- copied from PR #27579 (and changed from argmin to argmax)
16+
17+
lemma measurable_encode {α : Type*} {_ : MeasurableSpace α} [Encodable α]
18+
[MeasurableSingletonClass α] :
19+
Measurable (Encodable.encode (α := α)) := by
20+
refine measurable_to_nat fun a ↦ ?_
21+
have : Encodable.encode ⁻¹' {Encodable.encode a} = {a} := by ext; simp
22+
rw [this]
23+
exact measurableSet_singleton _
24+
25+
lemma measurableEmbedding_encode (α : Type*) {_ : MeasurableSpace α} [Encodable α]
26+
[MeasurableSingletonClass α] :
27+
MeasurableEmbedding (Encodable.encode (α := α)) where
28+
injective := Encodable.encode_injective
29+
measurable := measurable_encode
30+
measurableSet_image' _ _ := .of_discrete
31+
32+
section Finite
33+
34+
variable {𝓧 𝓨 α : Type*} {m𝓧 : MeasurableSpace 𝓧} {m𝓨 : MeasurableSpace 𝓨}
35+
{mα : MeasurableSpace α} [TopologicalSpace α] [LinearOrder α]
36+
[OpensMeasurableSpace α] [OrderClosedTopology α] [SecondCountableTopology α]
37+
38+
lemma measurableSet_isMax [Countable 𝓨]
39+
{f : 𝓧 → 𝓨 → α} (hf : ∀ y, Measurable (fun x ↦ f x y)) (y : 𝓨) :
40+
MeasurableSet {x | ∀ z, f x z ≤ f x y} := by
41+
rw [show {x | ∀ y', f x y' ≤ f x y} = ⋂ y', {x | f x y' ≤ f x y} by ext; simp]
42+
exact MeasurableSet.iInter fun z ↦ measurableSet_le (by fun_prop) (by fun_prop)
43+
44+
lemma exists_isMaxOn' {α : Type*} [LinearOrder α]
45+
[Nonempty 𝓨] [Finite 𝓨] [Encodable 𝓨] (f : 𝓧 → 𝓨 → α) (x : 𝓧) :
46+
∃ n : ℕ, ∃ y, n = Encodable.encode y ∧ ∀ z, f x z ≤ f x y := by
47+
obtain ⟨y, h⟩ := Finite.exists_max (f x)
48+
exact ⟨Encodable.encode y, y, rfl, h⟩
49+
50+
/-- A measurable argmax function. -/
51+
noncomputable
52+
def measurableArgmax [Nonempty 𝓨] [Finite 𝓨] [Encodable 𝓨] [MeasurableSingletonClass 𝓨]
53+
(f : 𝓧 → 𝓨 → α)
54+
[∀ x, DecidablePred fun n ↦ ∃ y, n = Encodable.encode y ∧ ∀ (z : 𝓨), f x z ≤ f x y]
55+
(x : 𝓧) :
56+
𝓨 :=
57+
(measurableEmbedding_encode 𝓨).invFun (Nat.find (exists_isMaxOn' f x))
58+
59+
lemma measurable_measurableArgmax [Nonempty 𝓨] [Finite 𝓨] [Encodable 𝓨] [MeasurableSingletonClass 𝓨]
60+
{f : 𝓧 → 𝓨 → α}
61+
[∀ x, DecidablePred fun n ↦ ∃ y, n = Encodable.encode y ∧ ∀ (z : 𝓨), f x z ≤ f x y]
62+
(hf : ∀ y, Measurable (fun x ↦ f x y)) :
63+
Measurable (measurableArgmax f) := by
64+
refine (MeasurableEmbedding.measurable_invFun (measurableEmbedding_encode 𝓨)).comp ?_
65+
refine measurable_find _ fun n ↦ ?_
66+
have : {x | ∃ y, n = Encodable.encode y ∧ ∀ (z : 𝓨), f x z ≤ f x y}
67+
= ⋃ y, ({x | n = Encodable.encode y} ∩ {x | ∀ z, f x z ≤ f x y}) := by ext; simp
68+
rw [this]
69+
refine MeasurableSet.iUnion fun y ↦ (MeasurableSet.inter (by simp) ?_)
70+
exact measurableSet_isMax (by fun_prop) y
71+
72+
lemma isMaxOn_measurableArgmax {α : Type*} [LinearOrder α]
73+
[Nonempty 𝓨] [Finite 𝓨] [Encodable 𝓨] [MeasurableSingletonClass 𝓨]
74+
(f : 𝓧 → 𝓨 → α)
75+
[∀ x, DecidablePred fun n ↦ ∃ y, n = Encodable.encode y ∧ ∀ (z : 𝓨), f x z ≤ f x y]
76+
(x : 𝓧) (z : 𝓨) :
77+
f x z ≤ f x (measurableArgmax f x) := by
78+
obtain ⟨y, h_eq, h_le⟩ := Nat.find_spec (exists_isMaxOn' f x)
79+
refine le_trans (h_le z) (le_of_eq ?_)
80+
rw [measurableArgmax, h_eq,
81+
MeasurableEmbedding.leftInverse_invFun (measurableEmbedding_encode 𝓨) y]
82+
83+
end Finite
84+
end MeasurableArgmax
85+
86+
namespace Bandits
87+
88+
variable {α : Type*} [DecidableEq α] [MeasurableSpace α]
89+
90+
/-- Number of pulls of arm `a` up to (and including) time `n`. -/
91+
noncomputable
92+
def pullCount' (n : ℕ) (h : Iic n → α × ℝ) (a : α) := #{s | (h s).1 = a}
93+
94+
/-- Sum of rewards of arm `a` up to (and including) time `n`. -/
95+
noncomputable
96+
def sumRewards' (n : ℕ) (h : Iic n → α × ℝ) (a : α) :=
97+
∑ s, if (h s).1 = a then (h s).2 else 0
98+
99+
/-- Empirical mean of arm `a` at time `n`. -/
100+
noncomputable
101+
def empMean' (n : ℕ) (h : Iic n → α × ℝ) (a : α) :=
102+
(sumRewards' n h a) / (pullCount' n h a)
103+
104+
omit [MeasurableSpace α] in
105+
lemma pullCount'_eq_sum (n : ℕ) (h : Iic n → α × ℝ) (a : α) :
106+
pullCount' n h a = ∑ s : Iic n, if (h s).1 = a then 1 else 0 := by simp [pullCount']
107+
108+
@[fun_prop]
109+
lemma measurable_pullCount' [MeasurableSingletonClass α] (n : ℕ) (a : α) :
110+
Measurable (fun h ↦ pullCount' n h a) := by
111+
simp_rw [pullCount'_eq_sum]
112+
have h_meas s : Measurable (fun (h : Iic n → α × ℝ) ↦ if (h s).1 = a then 1 else 0) := by
113+
refine Measurable.ite ?_ (by fun_prop) (by fun_prop)
114+
exact (measurableSet_singleton _).preimage (by fun_prop)
115+
fun_prop
116+
117+
@[fun_prop]
118+
lemma measurable_sumRewards' [MeasurableSingletonClass α] (n : ℕ) (a : α) :
119+
Measurable (fun h ↦ sumRewards' n h a) := by
120+
simp_rw [sumRewards']
121+
have h_meas s : Measurable (fun (h : Iic n → α × ℝ) ↦ if (h s).1 = a then (h s).2 else 0) := by
122+
refine Measurable.ite ?_ (by fun_prop) (by fun_prop)
123+
exact (measurableSet_singleton _).preimage (by fun_prop)
124+
fun_prop
125+
126+
@[fun_prop]
127+
lemma measurable_empMean' [MeasurableSingletonClass α] (n : ℕ) (a : α) :
128+
Measurable (fun h ↦ empMean' n h a) := by
129+
unfold empMean'
130+
fun_prop
131+
132+
end Bandits

‎LeanBandits/Bandit.lean‎

Lines changed: 67 additions & 60 deletions
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
/-
22
Copyright (c) 2025 Rémy Degenne. All rights reserved.
33
Released under Apache 2.0 license as described in the file LICENSE.
4-
Authors: Rémy Degenne
4+
Authors: Rémy Degenne, Paulo Rauber
55
-/
66
import Mathlib
77

@@ -22,115 +22,122 @@ def MeasurableEquiv.piIicZero (α : Type*) [MeasurableSpace α] :
2222

2323
namespace Bandits
2424

25-
variable {α : Type*} {mα : MeasurableSpace α}
25+
variable {α R : Type*} {mα : MeasurableSpace α} {mR : MeasurableSpace R}
2626

2727
section MeasureSpace
2828

29-
/-- A bandit interaction between an agent described by a policy and an environment given by
30-
reward distributions. -/
31-
structure Bandit (α : Type*) [MeasurableSpace α] where
32-
/-- Conditional distribution of the rewards given the arm pulled. -/
33-
ν : Kernel α ℝ
34-
hν : IsMarkovKernel ν
29+
/-- A stochastic, sequential algorithm. -/
30+
structure Algorithm (α R : Type*) [MeasurableSpace α] [MeasurableSpace R] where
3531
/-- Policy or sampling rule: distribution of the next pull. -/
36-
policy : (n : ℕ) → Kernel (Iic n → α × ℝ) α
37-
h_policy n : IsMarkovKernel (policy n)
32+
policy : (n : ℕ) → Kernel (Iic n → α × R) α
33+
[h_policy : ∀ n, IsMarkovKernel (policy n)]
3834
/-- Distribution of the first pull. -/
3935
p0 : Measure α
40-
hp0 : IsProbabilityMeasure p0
36+
[hp0 : IsProbabilityMeasure p0]
4137

42-
instance (b : Bandit α) : IsMarkovKernel b.ν := b.hν
43-
instance (b : Bandit α) (n : ℕ) : IsMarkovKernel (b.policy n) := b.h_policy n
44-
instance (b : Bandit α) : IsProbabilityMeasure b.p0 := b.hp0
38+
instance (alg : Algorithm α R) (n : ℕ) : IsMarkovKernel (alg.policy n) := alg.h_policy n
39+
instance (alg : Algorithm α R) : IsProbabilityMeasure alg.p0 := alg.hp0
4540

4641
namespace Bandit
4742

4843
/-- Kernel describing the distribution of the next arm-reward pair given the history up to `n`. -/
4944
noncomputable
50-
def stepKernel (b : Bandit α) (n : ℕ) : Kernel (Iic n → α × ℝ) (α × ℝ) :=
51-
(b.policy n) ⊗ₖ b.ν.prodMkLeft (Iic n → α × ℝ)
45+
def stepKernel (alg : Algorithm α R) (ν : Kernel α R) (n : ℕ) : Kernel (Iic n → α × R) (α × R) :=
46+
(alg.policy n) ⊗ₖ ν.prodMkLeft (Iic n → α × R)
5247

53-
instance (b : Bandit α) (n : ℕ) : IsMarkovKernel (b.stepKernel n) := by
48+
instance (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
49+
IsMarkovKernel (stepKernel alg ν n) := by
5450
rw [stepKernel]
5551
infer_instance
5652

5753
@[simp]
58-
lemma fst_stepKernel (b : Bandit α) (n : ℕ) : (b.stepKernel n).fst = b.policy n := by
54+
lemma fst_stepKernel (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
55+
(stepKernel alg ν n).fst = alg.policy n := by
5956
rw [stepKernel, Kernel.fst_compProd]
6057

6158
@[simp]
62-
lemma snd_stepKernel (b : Bandit α) (n : ℕ) : (b.stepKernel n).snd = b.ν ∘ₖ b.policy n := by
59+
lemma snd_stepKernel (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
60+
(stepKernel alg ν n).snd = ν ∘ₖ alg.policy n := by
6361
rw [stepKernel, Kernel.snd_compProd_prodMkLeft]
6462

6563
/-- Kernel sending a partial trajectory of the bandit interaction `Iic n → α × ℝ` to a measure
6664
on `ℕ → α × ℝ`, supported on full trajectories that start with the partial one. -/
67-
noncomputable def traj (b : Bandit α) (n : ℕ) : Kernel (Iic n → α × ℝ) (ℕ → α × ℝ) :=
68-
ProbabilityTheory.Kernel.traj (X := fun _ ↦ α × ℝ) b.stepKernel n
69-
70-
instance (b : Bandit α) (n : ℕ) : IsMarkovKernel (b.traj n) := by
71-
rw [traj]
72-
infer_instance
65+
noncomputable def traj (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
66+
Kernel (Iic n → α × R) (ℕ → α × R) :=
67+
ProbabilityTheory.Kernel.traj (X := fun _ ↦ α × R) (stepKernel alg ν) n
68+
deriving IsMarkovKernel
7369

7470
/-- Measure on the sequence of arms pulled and rewards observed generated by the bandit. -/
7571
noncomputable
76-
def trajMeasure (b : Bandit α) : Measure (ℕ → α × ℝ) :=
77-
(b.traj 0) ∘ₘ ((b.p0 ⊗ₘ b.ν).map (MeasurableEquiv.piIicZero _).symm)
72+
def trajMeasure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] : Measure (ℕ → α × R) :=
73+
(traj alg ν 0) ∘ₘ ((alg.p0 ⊗ₘ ν).map (MeasurableEquiv.piIicZero _).symm)
7874

7975
/-- Measure of an infinite stream of rewards from each arm. -/
8076
noncomputable
81-
def streamMeasure (b : Bandit α) : Measure (ℕ → α → ℝ) :=
82-
Measure.infinitePi fun _ ↦ Measure.infinitePi b.ν
77+
def streamMeasure (ν : Kernel α R) [IsMarkovKernel ν] : Measure (ℕ → α → R) :=
78+
Measure.infinitePi fun _ ↦ Measure.infinitePi ν
79+
deriving IsProbabilityMeasure
80+
81+
instance (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] :
82+
IsProbabilityMeasure (trajMeasure alg ν) := by
83+
rw [trajMeasure]
84+
have : IsProbabilityMeasure ((alg.p0 ⊗ₘ ν).map (MeasurableEquiv.piIicZero _).symm) :=
85+
isProbabilityMeasure_map <| by fun_prop
86+
infer_instance
8387

8488
/-- Joint distribution of the sequence of arm pulled and rewards, and a stream of independent
8589
rewards from all arms. -/
8690
noncomputable
87-
def measure (b : Bandit α) : Measure ((ℕ → α × ℝ) × (ℕ → α → ℝ)) :=
88-
(b.trajMeasure).prod (b.streamMeasure)
89-
90-
instance (b : Bandit α) : IsProbabilityMeasure b.trajMeasure := by
91-
rw [Bandit.trajMeasure]
92-
have : IsProbabilityMeasure ((b.p0 ⊗ₘ b.ν).map (MeasurableEquiv.piIicZero _).symm) :=
93-
isProbabilityMeasure_map <| by fun_prop
94-
infer_instance
95-
96-
instance (b : Bandit α) : IsProbabilityMeasure b.streamMeasure := by
97-
rw [streamMeasure]
98-
infer_instance
99-
100-
instance (b : Bandit α) : IsProbabilityMeasure b.measure := by
101-
rw [measure]
102-
infer_instance
91+
def measure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] :
92+
Measure ((ℕ → α × R) × (ℕ → α → R)) :=
93+
(trajMeasure alg ν).prod (streamMeasure ν)
94+
deriving IsProbabilityMeasure
10395

10496
end Bandit
10597

10698
/-- `arm n` is the arm pulled at time `n`. This is a random variable on the measurable space
10799
`ℕ → α × ℝ`. -/
108-
def arm (n : ℕ) (h : ℕ → α × ℝ) : α := (h n).1
100+
def arm (n : ℕ) (h : ℕ → α × R) : α := (h n).1
109101

110102
/-- `reward n` is the reward at time `n`. This is a random variable on the measurable space
111-
`ℕ → α × ℝ`. -/
112-
def reward (n : ℕ) (h : ℕ → α × ℝ) : ℝ := (h n).2
103+
`ℕ → α × R`. -/
104+
def reward (n : ℕ) (h : ℕ → α × R) : R := (h n).2
113105

114106
/-- `hist n` is the history up to time `n`. This is a random variable on the measurable space
115-
`ℕ → α × ℝ`. -/
116-
def hist (n : ℕ) (h : ℕ → α × ℝ) : Iic n → α × ℝ := fun i ↦ h i
107+
`ℕ → α × R`. -/
108+
def hist (n : ℕ) (h : ℕ → α × R) : Iic n → α × R := fun i ↦ h i
109+
110+
@[fun_prop]
111+
lemma measurable_arm (n : ℕ) : Measurable (arm n (α := α) (R := R)) := by unfold arm; fun_prop
112+
113+
@[fun_prop]
114+
lemma measurable_reward (n : ℕ) : Measurable (reward n (α := α) (R := R)) := by
115+
unfold reward; fun_prop
116+
117+
@[fun_prop]
118+
lemma measurable_hist (n : ℕ) : Measurable (hist n (α := α) (R := R)) := by unfold hist; fun_prop
117119

118120
/-- Filtration of the bandit process. -/
119121
def ℱ (α : Type*) [MeasurableSpace α] :
120-
Filtration ℕ (inferInstance : MeasurableSpace (ℕ → α × ℝ)) :=
121-
MeasureTheory.Filtration.piLE (X := fun _ ↦ α × ℝ)
122-
123-
lemma condDistrib_arm_reward [StandardBorelSpace α] [Nonempty α] (b : Bandit α) (n : ℕ) :
124-
condDistrib (fun h ↦ (arm n h, reward n h)) (hist n) b.trajMeasure = b.stepKernel n := by
122+
Filtration ℕ (inferInstance : MeasurableSpace (ℕ → α × R)) :=
123+
MeasureTheory.Filtration.piLE (X := fun _ ↦ α × R)
124+
125+
lemma condDistrib_arm_reward [StandardBorelSpace α] [Nonempty α]
126+
[StandardBorelSpace R] [Nonempty R] (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν]
127+
(n : ℕ) :
128+
condDistrib (fun h ↦ (arm n h, reward n h)) (hist n) (Bandit.trajMeasure alg ν)
129+
= Bandit.stepKernel alg ν n := by
125130
sorry
126131

127-
lemma condDistrib_reward (b : Bandit α) (n : ℕ) :
128-
condDistrib (reward n) (arm n) b.trajMeasure = b.ν := by
132+
lemma condDistrib_reward [StandardBorelSpace R] [Nonempty R] (alg : Algorithm α R)
133+
(ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
134+
condDistrib (reward n) (arm n) (Bandit.trajMeasure alg ν) = ν := by
129135
sorry
130136

131-
lemma condDistrib_arm [StandardBorelSpace α] [Nonempty α] (b : Bandit α) (n : ℕ) :
132-
condDistrib (arm n) (hist n) b.trajMeasure = b.policy n := by
133-
rw [← b.fst_stepKernel, ← condDistrib_arm_reward]
137+
lemma condDistrib_arm [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R]
138+
(alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
139+
condDistrib (arm n) (hist n) (Bandit.trajMeasure alg ν) = alg.policy n := by
140+
rw [← Bandit.fst_stepKernel alg ν n, ← condDistrib_arm_reward alg ν n]
134141
sorry
135142

136143
end MeasureSpace

‎LeanBandits/ETC.lean‎

Lines changed: 61 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,61 @@
1+
/-
2+
Copyright (c) 2025 Rémy Degenne. All rights reserved.
3+
Released under Apache 2.0 license as described in the file LICENSE.
4+
Authors: Rémy Degenne
5+
-/
6+
import Mathlib.Probability.Moments.SubGaussian
7+
import LeanBandits.AlgorithmBuilding
8+
9+
/-! # The Explore-Then-Commit Algorithm
10+
11+
-/
12+
13+
open MeasureTheory ProbabilityTheory Finset
14+
open scoped ENNReal NNReal
15+
16+
namespace Bandits
17+
18+
variable {K : ℕ}
19+
20+
/-- Arm pulled by the ETC algorithm at time `n + 1`. -/
21+
noncomputable
22+
def etcNextArm (hK : 0 < K) (m n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K :=
23+
have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK
24+
if hn : n < K * m - 1 then
25+
⟨(n + 1) % K, Nat.mod_lt _ hK⟩ -- for `n = 0` we have pulled arm 0 already, and we pull arm 1
26+
else
27+
if hn_eq : n = K * m - 1 then measurableArgmax (empMean' n) h
28+
else (h ⟨n - 1, by simp⟩).1
29+
30+
@[fun_prop]
31+
lemma measurable_etcNextArm (hK : 0 < K) (m n : ℕ) : Measurable (etcNextArm hK m n) := by
32+
have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK
33+
unfold etcNextArm
34+
simp only [dite_eq_ite]
35+
refine Measurable.ite (by simp) (by fun_prop) ?_
36+
refine Measurable.ite (by simp) ?_ (by fun_prop)
37+
exact measurable_measurableArgmax fun a ↦ by fun_prop
38+
39+
/-- The Explore-Then-Commit algorithm. -/
40+
noncomputable
41+
def etcAlgorithm (hK : 0 < K) (m : ℕ) : Algorithm (Fin K) ℝ where
42+
policy n := Kernel.deterministic (etcNextArm hK m n) (by fun_prop)
43+
p0 := Measure.dirac ⟨0, hK⟩
44+
45+
lemma ETC.arm_zero (hK : 0 < K) (m : ℕ) (ν : Kernel (Fin K) ℝ) [IsMarkovKernel ν] :
46+
arm 0 =ᵐ[Bandit.trajMeasure (etcAlgorithm hK m) ν] fun h ↦ ⟨0, hK⟩ := by
47+
suffices h : (Bandit.trajMeasure (etcAlgorithm hK m) ν).map (arm 0) = (etcAlgorithm hK m).p0 by
48+
have h_eq : ∀ᵐ x ∂((Bandit.trajMeasure (etcAlgorithm hK m) ν).map (arm 0)), x = ⟨0, hK⟩ := by
49+
rw [h]
50+
simp [etcAlgorithm]
51+
exact ae_of_ae_map (by fun_prop) h_eq
52+
-- extract lemma
53+
sorry
54+
55+
lemma ETC.arm_ae_eq_etcNextArm (hK : 0 < K) (m : ℕ) (ν : Kernel (Fin K) ℝ) [IsMarkovKernel ν]
56+
(n : ℕ) :
57+
arm (n + 1) =ᵐ[(Bandit.trajMeasure (etcAlgorithm hK m) ν)]
58+
fun h ↦ etcNextArm hK m n (fun i ↦ h i) := by
59+
sorry
60+
61+
end Bandits

0 commit comments

Comments
 (0)