Skip to content

Commit 4f349ad

Browse files
authored
Delete unnecessary definitions (#59)
2 parents bb9410b + 6fb6110 commit 4f349ad

7 files changed

Lines changed: 65 additions & 222 deletions

File tree

‎LeanBandits/Bandit/Bandit.lean‎

Lines changed: 18 additions & 72 deletions
Original file line numberDiff line numberDiff line change
@@ -8,7 +8,6 @@ import LeanBandits.ForMathlib.IndepFun
88
import LeanBandits.ForMathlib.IndepInfinitePi
99
import LeanBandits.ForMathlib.KernelRepresentation
1010
import LeanBandits.ForMathlib.StandardBorel
11-
import LeanBandits.SequentialLearning.Deterministic
1211
import LeanBandits.SequentialLearning.FiniteActions
1312
import LeanBandits.SequentialLearning.StationaryEnv
1413

@@ -26,32 +25,6 @@ variable {α R : Type*} {mα : MeasurableSpace α} {mR : MeasurableSpace R}
2625

2726
section MeasureSpace
2827

29-
namespace Bandit
30-
31-
/-- Kernel describing the distribution of the next action-reward pair given the history up to
32-
time `n`. -/
33-
noncomputable
34-
def stepKernel (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
35-
Kernel (Iic n → α × R) (α × R) :=
36-
Learning.stepKernel alg (stationaryEnv ν) n
37-
deriving IsMarkovKernel
38-
39-
@[simp]
40-
lemma fst_stepKernel (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
41-
(stepKernel alg ν n).fst = alg.policy n := by
42-
rw [stepKernel, Learning.fst_stepKernel]
43-
44-
@[simp]
45-
lemma snd_stepKernel (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
46-
(stepKernel alg ν n).snd = ν ∘ₖ alg.policy n := by
47-
rw [stepKernel, Learning.stepKernel, stationaryEnv_feedback, Kernel.snd_compProd_prodMkLeft]
48-
49-
/-- Measure on the sequence of actions pulled and rewards observed generated by the bandit. -/
50-
noncomputable
51-
def trajMeasure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] : Measure (ℕ → α × R) :=
52-
Learning.trajMeasure alg (stationaryEnv ν)
53-
deriving IsProbabilityMeasure
54-
5528
/-- Measure of an infinite stream of rewards from each action. -/
5629
noncomputable
5730
def streamMeasure (ν : Kernel α R) : Measure (ℕ → α → R) :=
@@ -61,26 +34,6 @@ instance (ν : Kernel α R) [IsMarkovKernel ν] : IsProbabilityMeasure (streamMe
6134
unfold streamMeasure
6235
infer_instance
6336

64-
/-- Joint distribution of the sequence of action pulled and rewards, and a stream of independent
65-
rewards from all actions. -/
66-
noncomputable
67-
def measure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] :
68-
Measure ((ℕ → α × R) × (ℕ → α → R)) :=
69-
(trajMeasure alg ν).prod (streamMeasure ν)
70-
deriving IsProbabilityMeasure
71-
72-
@[simp]
73-
lemma fst_measure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] :
74-
(measure alg ν).fst = trajMeasure alg ν := by
75-
rw [measure, Measure.fst_prod]
76-
77-
@[simp]
78-
lemma snd_measure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν] :
79-
(measure alg ν).snd = streamMeasure ν := by
80-
rw [measure, Measure.snd_prod]
81-
82-
end Bandit
83-
8437
section StreamMeasure
8538

8639
lemma _root_.hasLaw_eval_infinitePi {ι : Type*} {X : ι → Type*} {mX : ∀ i, MeasurableSpace (X i)}
@@ -90,15 +43,15 @@ lemma _root_.hasLaw_eval_infinitePi {ι : Type*} {X : ι → Type*} {mX : ∀ i,
9043
map_eq := by exact (measurePreserving_eval_infinitePi μ i).map_eq
9144

9245
lemma hasLaw_eval_streamMeasure (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
93-
HasLaw (fun h : ℕ → α → R ↦ h n) (Measure.infinitePi ν) (Bandit.streamMeasure ν) :=
46+
HasLaw (fun h : ℕ → α → R ↦ h n) (Measure.infinitePi ν) (streamMeasure ν) :=
9447
hasLaw_eval_infinitePi (fun _ ↦ Measure.infinitePi ν) n
9548

9649
lemma hasLaw_eval_eval_streamMeasure (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) (a : α) :
97-
HasLaw (fun h : ℕ → α → R ↦ h n a) (ν a) (Bandit.streamMeasure ν) :=
50+
HasLaw (fun h : ℕ → α → R ↦ h n a) (ν a) (streamMeasure ν) :=
9851
(hasLaw_eval_infinitePi ν a).comp (hasLaw_eval_streamMeasure ν n)
9952

10053
lemma identDistrib_eval_eval_id_streamMeasure (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) (a : α) :
101-
IdentDistrib (fun h : ℕ → α → R ↦ h n a) id (Bandit.streamMeasure ν) (ν a) where
54+
IdentDistrib (fun h : ℕ → α → R ↦ h n a) id (streamMeasure ν) (ν a) where
10255
aemeasurable_fst := Measurable.aemeasurable (by fun_prop)
10356
aemeasurable_snd := Measurable.aemeasurable (by fun_prop)
10457
map_eq := by
@@ -118,47 +71,40 @@ lemma Integrable.congr_identDistrib {Ω Ω' : Type*}
11871

11972
lemma integrable_eval_streamMeasure (ν : Kernel α ℝ) [IsMarkovKernel ν] (n : ℕ) (a : α)
12073
(h_int : Integrable id (ν a)) :
121-
Integrable (fun h : ℕ → α → ℝ ↦ h n a) (Bandit.streamMeasure ν) :=
74+
Integrable (fun h : ℕ → α → ℝ ↦ h n a) (streamMeasure ν) :=
12275
Integrable.congr_identDistrib h_int (identDistrib_eval_eval_id_streamMeasure ν n a).symm
12376

12477
lemma integral_eval_streamMeasure (ν : Kernel α ℝ) [IsMarkovKernel ν] (n : ℕ) (a : α) :
125-
∫ h, h n a ∂(Bandit.streamMeasure ν) = (ν a)[id] := by
126-
calc ∫ h, h n a ∂(Bandit.streamMeasure ν)
127-
_ = ∫ x, x ∂((Bandit.streamMeasure ν).map (fun h ↦ h n a)) := by
78+
∫ h, h n a ∂(streamMeasure ν) = (ν a)[id] := by
79+
calc ∫ h, h n a ∂(streamMeasure ν)
80+
_ = ∫ x, x ∂((streamMeasure ν).map (fun h ↦ h n a)) := by
12881
rw [integral_map (Measurable.aemeasurable (by fun_prop)) (by fun_prop)]
12982
_ = (ν a)[id] := by simp [(hasLaw_eval_eval_streamMeasure ν n a).map_eq]
13083

13184
lemma iIndepFun_eval_streamMeasure' (ν : Kernel α R) [IsMarkovKernel ν] :
132-
iIndepFun (fun n ω ↦ ω n) (Bandit.streamMeasure ν) :=
85+
iIndepFun (fun n ω ↦ ω n) (streamMeasure ν) :=
13386
iIndepFun_infinitePi (P := fun (_ : ℕ) ↦ Measure.infinitePi ν) (Ω := fun _ ↦ α → R)
13487
(X := fun i u ↦ u) (fun i ↦ by fun_prop)
13588

13689
lemma iIndepFun_eval_streamMeasure'' (ν : Kernel α R) [IsMarkovKernel ν] (a : α) :
137-
iIndepFun (fun n ω ↦ ω n a) (Bandit.streamMeasure ν) :=
90+
iIndepFun (fun n ω ↦ ω n a) (streamMeasure ν) :=
13891
(iIndepFun_eval_streamMeasure' ν).comp (g := fun i ω ↦ ω a) (by fun_prop)
13992

14093
lemma iIndepFun_eval_streamMeasure (ν : Kernel α R) [IsMarkovKernel ν] :
141-
iIndepFun (fun (p : ℕ × α) ω ↦ ω p.1 p.2) (Bandit.streamMeasure ν) :=
94+
iIndepFun (fun (p : ℕ × α) ω ↦ ω p.1 p.2) (streamMeasure ν) :=
14295
iIndepFun_uncurry_infinitePi' (X := fun _ _ ↦ id) (fun _ ↦ ν) (by fun_prop)
14396

14497
lemma indepFun_eval_streamMeasure (ν : Kernel α R) [IsMarkovKernel ν] {n m : ℕ} {a b : α}
14598
(h : n ≠ m ∨ a ≠ b) :
146-
IndepFun (fun ω ↦ ω n a) (fun ω ↦ ω m b) (Bandit.streamMeasure ν) := by
99+
IndepFun (fun ω ↦ ω n a) (fun ω ↦ ω m b) (streamMeasure ν) := by
147100
change IndepFun (fun ω ↦ ω (n, a).1 (n, a).2) (fun ω ↦ ω (m, b).1 (m, b).2)
148-
(Bandit.streamMeasure ν)
101+
(streamMeasure ν)
149102
exact (iIndepFun_eval_streamMeasure ν).indepFun (by grind)
150103

151104
lemma indepFun_eval_streamMeasure' (ν : Kernel α R) [IsMarkovKernel ν] {a b : α} (h : a ≠ b) :
152-
IndepFun (fun ω n ↦ ω n a) (fun ω n ↦ ω n b) (Bandit.streamMeasure ν) :=
105+
IndepFun (fun ω n ↦ ω n a) (fun ω n ↦ ω n b) (streamMeasure ν) :=
153106
indepFun_proj_infinitePi_infinitePi h
154107

155-
lemma indepFun_eval_snd_measure (alg : Algorithm α R) (ν : Kernel α R) [IsMarkovKernel ν]
156-
{a b : α} (h : a ≠ b) :
157-
IndepFun (fun ω n ↦ ω.2 n a) (fun ω n ↦ ω.2 n b) (Bandit.measure alg ν) := by
158-
refine indepFun_snd_prod ?_ ?_ (indepFun_eval_streamMeasure' ν h) (Bandit.trajMeasure alg ν)
159-
· exact Measurable.aemeasurable (by fun_prop)
160-
· exact Measurable.aemeasurable (by fun_prop)
161-
162108
end StreamMeasure
163109

164110
namespace ArrayModel
@@ -181,7 +127,7 @@ instance {α R : Type*} [Countable α] [MeasurableSpace R] [StandardBorelSpace R
181127
/-- Probability measure for the array model of stochastic bandits. -/
182128
noncomputable
183129
def arrayMeasure (ν : Kernel α R) : Measure (probSpace α R) :=
184-
(Measure.infinitePi fun _ ↦ volume).prod (Bandit.streamMeasure ν)
130+
(Measure.infinitePi fun _ ↦ volume).prod (streamMeasure ν)
185131

186132
instance (ν : Kernel α R) [IsMarkovKernel ν] : IsProbabilityMeasure (arrayMeasure ν) :=
187133
Measure.prod.instIsProbabilityMeasure _ _
@@ -606,7 +552,7 @@ lemma map_snd_apply_arrayMeasure {ν : Kernel α R} [IsMarkovKernel ν] (n : ℕ
606552
rw [Measure.snd, Measure.map_map (by fun_prop) (by fun_prop)]
607553
rfl
608554
_ = ν a := by
609-
rw [arrayMeasure, Measure.snd_prod, Bandit.streamMeasure]
555+
rw [arrayMeasure, Measure.snd_prod, streamMeasure]
610556
have : (fun ω ↦ ω n a) = (fun h : α → R ↦ h a) ∘ (fun ω : ℕ → α → R ↦ ω n) := rfl
611557
rw [this, ← Measure.map_map (by fun_prop) (by fun_prop), Measure.infinitePi_map_eval,
612558
Measure.infinitePi_map_eval]
@@ -629,7 +575,7 @@ omit [DecidableEq α] [Nonempty α] [StandardBorelSpace α] in
629575
lemma indepFun_fst_add_one_aux (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) :
630576
(fun ω ↦ ω.1 (n + 1)) ⟂ᵢ[arrayMeasure ν] (fun ω ↦ (fun (i : Iic n) ↦ ω.1 i, ω.2)) := by
631577
let μ₁ : Measure (ℕ → I) := Measure.infinitePi fun _ ↦ volume
632-
let μ₂ : Measure (ℕ → α → R) := Bandit.streamMeasure ν
578+
let μ₂ : Measure (ℕ → α → R) := streamMeasure ν
633579
-- Coordinates of μ₁ are independent
634580
have h_indep : iIndepFun (fun i (ω : ℕ → I) ↦ ω i) μ₁ :=
635581
iIndepFun_infinitePi (fun _ ↦ measurable_id)
@@ -1016,10 +962,10 @@ lemma hasCondDistrib_action' (alg : Algorithm α R) (ν : Kernel α R) [IsMarkov
1016962
rw [h_indep']
1017963
congr
1018964
simp only [arrayMeasure]
1019-
calc ((Measure.infinitePi fun x ↦ ℙ).prod (Bandit.streamMeasure ν)).map (fun ω ↦ ω.1 (n + 1))
965+
calc ((Measure.infinitePi fun x ↦ ℙ).prod (streamMeasure ν)).map (fun ω ↦ ω.1 (n + 1))
1020966
_ = (Measure.infinitePi fun x ↦ ℙ).map (Function.eval (n + 1)) := by
1021967
nth_rw 2 [← Measure.fst_prod (μ := Measure.infinitePi fun x ↦ ℙ)
1022-
(ν := Bandit.streamMeasure ν)]
968+
(ν := streamMeasure ν)]
1023969
rw [Measure.fst, Measure.map_map (by fun_prop) (by fun_prop)]
1024970
rfl
1025971
_ = ℙ := by rw [Measure.infinitePi_map_eval]

‎LeanBandits/Bandit/RewardByCountMeasure.lean‎

Lines changed: 5 additions & 97 deletions
Original file line numberDiff line numberDiff line change
@@ -20,7 +20,7 @@ variable {α Ω : Type*} {mα : MeasurableSpace α} {mΩ : MeasurableSpace Ω} [
2020
{alg : Algorithm α ℝ} {ν : Kernel α ℝ} [IsMarkovKernel ν]
2121
{h_inter : IsAlgEnvSeq A R alg (stationaryEnv ν) P}
2222

23-
local notation "𝔓'" => P.prod (Bandit.streamMeasure ν)
23+
local notation "𝔓'" => P.prod (streamMeasure ν)
2424

2525
omit [DecidableEq α] [StandardBorelSpace α] [Nonempty α] in
2626
lemma hasLaw_Z (a : α) (m : ℕ) :
@@ -30,10 +30,10 @@ lemma hasLaw_Z (a : α) (m : ℕ) :
3030
_ = ((𝔓').snd).map (fun ω ↦ ω m a) := by
3131
rw [Measure.snd, Measure.map_map (by fun_prop) (by fun_prop)]
3232
rfl
33-
_ = (Bandit.streamMeasure ν).map (fun ω ↦ ω m a) := by simp
33+
_ = (streamMeasure ν).map (fun ω ↦ ω m a) := by simp
3434
_ = ((Measure.infinitePi fun _ ↦ Measure.infinitePi ν).map (fun ω ↦ ω m)).map
3535
(fun ω ↦ ω a) := by
36-
rw [Bandit.streamMeasure, Measure.map_map (by fun_prop) (by fun_prop)]
36+
rw [streamMeasure, Measure.map_map (by fun_prop) (by fun_prop)]
3737
rfl
3838
_ = ν a := by simp_rw [(measurePreserving_eval_infinitePi _ _).map_eq]
3939

@@ -44,9 +44,6 @@ notation "𝓛[" Y " | " X " in " s "; " μ "]" => Measure.map Y (μ[|X ⁻¹' s
4444
/-- Law of `Y` conditioned on the event that `X` equals `x`. -/
4545
notation "𝓛[" Y " | " X " ← " x "; " μ "]" => Measure.map Y (μ[|X ⁻¹' {x}])
4646

47-
local notation "𝔓t" => Bandit.trajMeasure alg ν
48-
local notation "𝔓" => Bandit.measure alg ν
49-
5047
omit [DecidableEq α] in
5148
lemma condDistrib_reward'' [Countable α]
5249
(h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (n : ℕ) :
@@ -107,7 +104,7 @@ lemma condIndepFun_reward_stepsUntil_action [StandardBorelSpace Ω] [Countable
107104
(fun ω ↦ R n ω.1) ({ω | stepsUntil A a m ω.1 = ↑n}.indicator (fun _ ↦ 1)) 𝔓' := by
108105
have hA := h.measurable_A
109106
have hR := h.measurable_R
110-
exact condIndepFun_fst_prod (ν := Bandit.streamMeasure ν)
107+
exact condIndepFun_fst_prod (ν := streamMeasure ν)
111108
(measurable_indicator_stepsUntil_eq hA hR a m n) (by fun_prop) (by fun_prop)
112109
(condIndepFun_reward_stepsUntil_action' h a m n)
113110

@@ -229,97 +226,8 @@ lemma identDistrib_rewardByCount_id [StandardBorelSpace Ω] [Countable α]
229226

230227
lemma identDistrib_rewardByCount_eval [StandardBorelSpace Ω] [Countable α]
231228
(h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (a : α) (n m : ℕ) (hn : n ≠ 0) :
232-
IdentDistrib (rewardByCount A R a n) (fun ω ↦ ω m a) 𝔓' (Bandit.streamMeasure ν) :=
229+
IdentDistrib (rewardByCount A R a n) (fun ω ↦ ω m a) 𝔓' (streamMeasure ν) :=
233230
(identDistrib_rewardByCount_id h a n hn).trans
234231
(identDistrib_eval_eval_id_streamMeasure ν m a).symm
235232

236-
-- lemma indepFun_rewardByCount_Iic [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
237-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (a : α)
238-
-- (n : ℕ) :
239-
-- (rewardByCount A R a (n + 1)) ⟂ᵢ[𝔓'] fun ω (i : Iic n) ↦ rewardByCount A R a i ω := by
240-
-- sorry
241-
242-
-- lemma iIndepFun_rewardByCount' [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
243-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (a : α) :
244-
-- iIndepFun (rewardByCount A R a) 𝔓' := by
245-
-- have hA := h.measurable_A
246-
-- have hR := h.measurable_R
247-
-- rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)]
248-
-- exact indepFun_rewardByCount_Iic h a
249-
250-
-- lemma iIndepFun_rewardByCount [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
251-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) :
252-
-- iIndepFun (fun (p : α × ℕ) ↦ rewardByCount A R p.1 (p.2 + 1)) 𝔓' := by
253-
-- sorry
254-
255-
-- lemma identDistrib_rewardByCount_stream_all [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
256-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) :
257-
-- IdentDistrib (fun ω (p : α × ℕ) ↦ rewardByCount A R p.1 (p.2 + 1) ω)
258-
-- (fun ω p ↦ ω p.2 p.1) 𝔓' (Bandit.streamMeasure ν) := by
259-
-- refine IdentDistrib.pi (fun p ↦ ?_) ?_ ?_
260-
-- · refine identDistrib_rewardByCount_eval h p.1 (p.2 + 1) p.2 (by simp) (ν := ν)
261-
-- · exact iIndepFun_rewardByCount h
262-
-- · sorry
263-
264-
-- lemma identDistrib_rewardByCount_stream' [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
265-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (a : α) :
266-
-- IdentDistrib (fun ω n ↦ rewardByCount A R a (n + 1) ω) (fun ω n ↦ ω n a)
267-
-- 𝔓' (Bandit.streamMeasure ν) := by
268-
-- refine IdentDistrib.pi (fun n ↦ ?_) ?_ ?_
269-
-- · refine identDistrib_rewardByCount_eval h a (n + 1) n (by simp) (ν := ν)
270-
-- · have h_indep := iIndepFun_rewardByCount' h a
271-
-- exact iIndepFun.precomp (g := fun n ↦ n + 1) (fun i j hij ↦ by grind) h_indep
272-
-- · exact iIndepFun_eval_streamMeasure'' ν a
273-
274-
omit [DecidableEq α] [StandardBorelSpace α] [Nonempty α] in
275-
lemma identDistrib_eval_streamMeasure_measure (a : α) :
276-
IdentDistrib (fun ω n ↦ ω n a) (fun ω n ↦ ω.2 n a)
277-
(Bandit.streamMeasure ν) 𝔓 := by
278-
refine IdentDistrib.pi (fun n ↦ ?_) ?_ ?_
279-
· rw [← Bandit.snd_measure alg ν, Measure.snd,
280-
identDistrib_map_left_iff (by fun_prop) (by fun_prop)
281-
(Measurable.aemeasurable <| by fun_prop)]
282-
exact IdentDistrib.refl (by fun_prop)
283-
· exact iIndepFun_eval_streamMeasure'' ν a
284-
· change iIndepFun (fun n ↦ ((fun ω ↦ ω n a) ∘ Prod.snd)) 𝔓
285-
rw [← iIndepFun_map_iff (by fun_prop) (fun _ ↦ Measurable.aemeasurable (by fun_prop))]
286-
rw [← Measure.snd, Bandit.snd_measure]
287-
exact iIndepFun_eval_streamMeasure'' ν a
288-
289-
-- lemma identDistrib_rewardByCount_stream [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
290-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (a : α) :
291-
-- IdentDistrib (fun ω n ↦ rewardByCount A R a (n + 1) ω) (fun ω n ↦ ω.2 n a) 𝔓' 𝔓 :=
292-
-- (identDistrib_rewardByCount_stream' h a).trans (identDistrib_eval_streamMeasure_measure a)
293-
294-
-- lemma indepFun_rewardByCount_of_ne [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
295-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) {a b : α} (hab : a ≠ b) :
296-
-- IndepFun (fun ω s ↦ rewardByCount A R a s ω) (fun ω s ↦ rewardByCount A R b s ω) 𝔓' := by
297-
-- sorry
298-
299-
-- lemma identDistrib_sum_Icc_rewardByCount [StandardBorelSpace Ω] [Nonempty Ω] [Countable α]
300-
-- (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (m : ℕ) (a : α) :
301-
-- IdentDistrib (fun ω ↦ ∑ s ∈ Icc 1 m, rewardByCount A R a s ω)
302-
-- (fun ω ↦ ∑ s ∈ range m, ω.2 s a) 𝔓' 𝔓 := by
303-
-- have h1 (a : α) :
304-
-- IdentDistrib (fun ω s ↦ rewardByCount A R a (s + 1) ω) (fun ω s ↦ ω.2 s a) 𝔓' 𝔓 :=
305-
-- identDistrib_rewardByCount_stream h a
306-
-- have h_eq (ω : Ω × (ℕ → α → ℝ)) : ∑ s ∈ Icc 1 m, rewardByCount A R a s ω
307-
-- = ∑ s ∈ range m, rewardByCount A R a (s + 1) ω := by
308-
-- let e : Icc 1 m ≃ range m :=
309-
-- { toFun x := ⟨x - 1, by have h := x.2; simp only [mem_Icc] at h; simp; grind⟩
310-
-- invFun x := ⟨x + 1, by
311-
-- have h := x.2
312-
-- simp only [mem_Icc, le_add_iff_nonneg_left, zero_le, true_and, ge_iff_le]
313-
-- simp only [mem_range] at h
314-
-- grind⟩
315-
-- left_inv x := by have h := x.2; simp only [mem_Icc] at h; grind
316-
-- right_inv x := by have h := x.2; grind }
317-
-- rw [← sum_coe_sort (Icc 1 m), ← sum_coe_sort (range m), sum_equiv e]
318-
-- · simp
319-
-- · simp only [univ_eq_attach, mem_attach, forall_const, Subtype.forall, mem_Icc,
320-
-- forall_and_index]
321-
-- grind
322-
-- simp_rw [h_eq]
323-
-- exact IdentDistrib.comp (h1 a) (u := fun p ↦ ∑ s ∈ range m, p s) (by fun_prop)
324-
325233
end Bandits

0 commit comments

Comments
 (0)