Skip to content

Commit a3747a9

Browse files
committed
Refactor BayesStationaryEnv
1 parent d3e9b99 commit a3747a9

1 file changed

Lines changed: 25 additions & 41 deletions

File tree

‎LeanBandits/SequentialLearning/BayesStationaryEnv.lean‎

Lines changed: 25 additions & 41 deletions
Original file line numberDiff line numberDiff line change
@@ -104,7 +104,7 @@ lemma hasLaw_IT_action_zero (h : IsBayesAlgEnvSeq Q κ alg E A R' P) :
104104
∀ᵐ e ∂Q, HasLaw (IT.action 0) alg.p0 (condDistrib (trajectory A R') E P e) := by
105105
rw [← h.hasLaw_env.map_eq]
106106
filter_upwards [condDistrib_comp E
107-
(measurable_trajectory h.measurable_A h.measurable_R).aemeasurable
107+
((measurable_trajectory h.measurable_A h.measurable_R).aemeasurable)
108108
(IT.measurable_action (α := α) (R := R) 0),
109109
h.hasCondDistrib_action_zero.condDistrib_eq] with e hc hcd
110110
exact ⟨(IT.measurable_action 0).aemeasurable, by
@@ -131,38 +131,25 @@ lemma hasCondDistrib_IT_action (h : IsBayesAlgEnvSeq Q κ alg E A R' P) (n : ℕ
131131
rwa [Kernel.sectR_prodMkLeft] at he
132132

133133
lemma hasCondDistrib_IT_reward [IsFiniteKernel κ] (h : IsBayesAlgEnvSeq Q κ alg E A R' P) (n : ℕ) :
134-
∀ᵐ e ∂Q, HasCondDistrib (IT.reward (n + 1)) (fun x ↦ (IT.hist n x, IT.action (n + 1) x))
134+
∀ᵐ e ∂Q, HasCondDistrib (IT.reward (n + 1)) (fun τ ↦ (IT.hist n τ, IT.action (n + 1) τ))
135135
((κ.sectR e).prodMkLeft _) (condDistrib (trajectory A R') E P e) := by
136136
rw [← h.hasLaw_env.map_eq]
137-
have hmt := measurable_trajectory h.measurable_A h.measurable_R
138-
have hm := IsAlgEnvSeq.measurable_hist h.measurable_A h.measurable_R n
139137
have h_reorder : HasCondDistrib (R' (n + 1))
140-
(fun ω ↦ ((IsAlgEnvSeq.hist A R' n ω, A (n + 1) ω), E ω))
141-
((κ.prodMkLeft _).comap (fun ((h, a), e) ↦ (h, (e, a))) (by fun_prop)) P := by
142-
convert (h.hasCondDistrib_reward n).comp_right
143-
((MeasurableEquiv.prodCongr (.refl _) .prodComm).trans MeasurableEquiv.prodAssoc.symm) using 2
144-
filter_upwards [h_reorder.ae_hasCondDistrib_sectL
145-
((IT.measurable_hist n).prodMk (IT.measurable_action (n + 1)))
146-
(IT.measurable_reward (n + 1))
147-
hmt.aemeasurable h.measurable_E.aemeasurable] with e he
148-
have hk : ((κ.prodMkLeft _).comap (fun ((h, a), e) ↦ (h, (e, a))) (by fun_prop)).sectL e =
149-
(κ.sectR e).prodMkLeft (↥(Iic n) → α × R) :=
150-
Kernel.ext fun ⟨_, a⟩ ↦ by
151-
simp [Kernel.sectL_apply, Kernel.comap_apply, Kernel.prodMkLeft_apply]
152-
rw [hk] at he; exact he
138+
(fun ω ↦ (E ω, IsAlgEnvSeq.hist A R' n ω, A (n + 1) ω))
139+
(κ.comap (fun (e, _, a) ↦ (e, a)) (by fun_prop)) P :=
140+
(h.hasCondDistrib_reward n).comp_right (MeasurableEquiv.prodAssoc.symm.trans
141+
((MeasurableEquiv.prodCongr .prodComm (.refl _)).trans .prodAssoc))
142+
exact h_reorder.ae_hasCondDistrib_sectR ((IT.measurable_hist n).prodMk
143+
(IT.measurable_action (n + 1))) (IT.measurable_reward (n + 1))
144+
(measurable_trajectory h.measurable_A h.measurable_R).aemeasurable h.measurable_E.aemeasurable
153145

154146
lemma ae_IsAlgEnvSeq [IsMarkovKernel κ] (h : IsBayesAlgEnvSeq Q κ alg E A R' P) :
155147
∀ᵐ e ∂Q, IsAlgEnvSeq IT.action IT.reward alg (stationaryEnv (κ.sectR e))
156148
(condDistrib (trajectory A R') E P e) := by
157149
filter_upwards [hasLaw_IT_action_zero h, hasCondDistrib_IT_reward_zero h,
158150
ae_all_iff.2 (hasCondDistrib_IT_action h), ae_all_iff.2 (hasCondDistrib_IT_reward h)]
159-
with _ h_a0 h_r0 h_a h_r
160-
exact {
161-
hasLaw_action_zero := h_a0
162-
hasCondDistrib_reward_zero := h_r0
163-
hasCondDistrib_action := h_a
164-
hasCondDistrib_reward := h_r
165-
}
151+
with _ ha0 hr0 hA hR
152+
exact ⟨IT.measurable_action, IT.measurable_reward, ha0, hr0, hA, hR⟩
166153

167154
end CondDistribIsAlgEnvSeq
168155

@@ -194,28 +181,26 @@ lemma IsAlgEnvSeq.isBayesAlgEnvSeq
194181
apply HasCondDistrib.hasLaw_of_const
195182
simpa [bayesStationaryEnv] using h.hasCondDistrib_reward_zero.fst
196183
hasCondDistrib_action_zero := by
197-
have hfst : HasCondDistrib (fun ω ↦ (R' 0 ω).1) (A 0) (Kernel.const α Q) P := by
184+
have hfst : HasCondDistrib (fun ω ↦ (R' 0 ω).1) (A 0) (Kernel.const _ Q) P := by
198185
simpa [bayesStationaryEnv] using h.hasCondDistrib_reward_zero.fst
199186
simpa [h.hasLaw_action_zero.map_eq, Algorithm.prod_left] using hfst.swap_const
200-
hasCondDistrib_reward_zero := by
201-
have h0 := h.hasCondDistrib_reward_zero
202-
simp only [bayesStationaryEnv] at h0
203-
convert h0.of_compProd.comp_right (MeasurableEquiv.prodComm : α × 𝓔 ≃ᵐ 𝓔 × α) using 2
187+
hasCondDistrib_reward_zero :=
188+
h.hasCondDistrib_reward_zero.of_compProd.comp_right MeasurableEquiv.prodComm
204189
hasCondDistrib_action n := by
205190
let f : (Iic n → α × 𝓔 × R) → 𝓔 × (Iic n → α × R) :=
206191
fun h ↦ ((h ⟨0, by simp⟩).2.1, fun i ↦ ((h i).1, (h i).2.2))
207-
suffices h' : HasCondDistrib (A (n + 1)) (IsAlgEnvSeq.hist A R' n)
208-
(((alg.policy n).comap Prod.snd (by fun_prop)).comap f (by fun_prop)) P from
209-
h'.comp_left (f := f)
210-
exact h.hasCondDistrib_action n
192+
have hc : HasCondDistrib (A (n + 1)) (IsAlgEnvSeq.hist A R' n)
193+
(((alg.policy n).comap Prod.snd (by fun_prop)).comap f (by fun_prop)) P :=
194+
h.hasCondDistrib_action n
195+
exact hc.comp_left (f := f)
211196
hasCondDistrib_reward n := by
212197
let f : (Iic n → α × 𝓔 × R) × α → (Iic n → α × R) × 𝓔 × α :=
213198
fun p ↦ ((fun i ↦ ((p.1 i).1, (p.1 i).2.2)), (p.1 ⟨0, by simp⟩).2.1, p.2)
214-
have hf : Measurable f := by fun_prop
215-
suffices h' : HasCondDistrib (fun ω ↦ (R' (n + 1) ω).2)
199+
have hc : HasCondDistrib (fun ω ↦ (R' (n + 1) ω).2)
216200
(fun ω ↦ (IsAlgEnvSeq.hist A R' n ω, A (n + 1) ω))
217-
((Kernel.prodMkLeft (↥(Iic n) → α × R) κ).comap f hf) P from h'.comp_left hf
218-
simpa [bayesStationaryEnv, Kernel.snd_prod] using (h.hasCondDistrib_reward n).snd
201+
((Kernel.prodMkLeft ((Iic n) → α × R) κ).comap f (by fun_prop)) P := by
202+
simpa [bayesStationaryEnv, Kernel.snd_prod] using (h.hasCondDistrib_reward n).snd
203+
exact hc.comp_left (by fun_prop)
219204

220205
end IsAlgEnvSeq
221206

@@ -234,13 +219,12 @@ lemma isBayesAlgEnvSeq_bayesTrajMeasure
234219
(Q : Measure 𝓔) [IsProbabilityMeasure Q] (κ : Kernel (𝓔 × α) R) [IsMarkovKernel κ]
235220
(alg : Algorithm α R) :
236221
IsBayesAlgEnvSeq Q κ alg (fun ω ↦ (ω 0).2.1) action (fun n ω ↦ (ω n).2.2)
237-
(bayesTrajMeasure Q κ alg) :=
238-
(isAlgEnvSeq_trajMeasure _ _).isBayesAlgEnvSeq
222+
(bayesTrajMeasure Q κ alg) := (isAlgEnvSeq_trajMeasure _ _).isBayesAlgEnvSeq
239223

240224
noncomputable
241225
def bayesTrajMeasurePosterior [StandardBorelSpace 𝓔] [Nonempty 𝓔]
242-
(Q : Measure 𝓔) [IsProbabilityMeasure Q] (κ : Kernel (𝓔 × α) ℝ) [IsMarkovKernel κ]
243-
(alg : Algorithm α ℝ) (n : ℕ) : Kernel (Iic n → α × ℝ) 𝓔 :=
226+
(Q : Measure 𝓔) (κ : Kernel (𝓔 × α) R) [IsMarkovKernel κ]
227+
(alg : Algorithm α R) (n : ℕ) : Kernel (Iic n → α × R) 𝓔 :=
244228
condDistrib (fun ω ↦ (ω 0).2.1) (IsAlgEnvSeq.hist action (fun n ω ↦ (ω n).2.2) n)
245229
(bayesTrajMeasure Q κ alg)
246230
deriving IsMarkovKernel

0 commit comments

Comments
 (0)