@@ -104,7 +104,7 @@ lemma hasLaw_IT_action_zero (h : IsBayesAlgEnvSeq Q κ alg E A R' P) :
104104 ∀ᵐ e ∂Q, HasLaw (IT.action 0 ) alg.p0 (condDistrib (trajectory A R') E P e) := by
105105 rw [← h.hasLaw_env.map_eq]
106106 filter_upwards [condDistrib_comp E
107- (measurable_trajectory h.measurable_A h.measurable_R).aemeasurable
107+ (( measurable_trajectory h.measurable_A h.measurable_R).aemeasurable)
108108 (IT.measurable_action (α := α) (R := R) 0 ),
109109 h.hasCondDistrib_action_zero.condDistrib_eq] with e hc hcd
110110 exact ⟨(IT.measurable_action 0 ).aemeasurable, by
@@ -131,38 +131,25 @@ lemma hasCondDistrib_IT_action (h : IsBayesAlgEnvSeq Q κ alg E A R' P) (n : ℕ
131131 rwa [Kernel.sectR_prodMkLeft] at he
132132
133133lemma hasCondDistrib_IT_reward [IsFiniteKernel κ] (h : IsBayesAlgEnvSeq Q κ alg E A R' P) (n : ℕ) :
134- ∀ᵐ e ∂Q, HasCondDistrib (IT.reward (n + 1 )) (fun x ↦ (IT.hist n x , IT.action (n + 1 ) x ))
134+ ∀ᵐ e ∂Q, HasCondDistrib (IT.reward (n + 1 )) (fun τ ↦ (IT.hist n τ , IT.action (n + 1 ) τ ))
135135 ((κ.sectR e).prodMkLeft _) (condDistrib (trajectory A R') E P e) := by
136136 rw [← h.hasLaw_env.map_eq]
137- have hmt := measurable_trajectory h.measurable_A h.measurable_R
138- have hm := IsAlgEnvSeq.measurable_hist h.measurable_A h.measurable_R n
139137 have h_reorder : HasCondDistrib (R' (n + 1 ))
140- (fun ω ↦ ((IsAlgEnvSeq.hist A R' n ω, A (n + 1 ) ω), E ω))
141- ((κ.prodMkLeft _).comap (fun ((h, a), e) ↦ (h, (e, a))) (by fun_prop)) P := by
142- convert (h.hasCondDistrib_reward n).comp_right
143- ((MeasurableEquiv.prodCongr (.refl _) .prodComm).trans MeasurableEquiv.prodAssoc.symm) using 2
144- filter_upwards [h_reorder.ae_hasCondDistrib_sectL
145- ((IT.measurable_hist n).prodMk (IT.measurable_action (n + 1 )))
146- (IT.measurable_reward (n + 1 ))
147- hmt.aemeasurable h.measurable_E.aemeasurable] with e he
148- have hk : ((κ.prodMkLeft _).comap (fun ((h, a), e) ↦ (h, (e, a))) (by fun_prop)).sectL e =
149- (κ.sectR e).prodMkLeft (↥(Iic n) → α × R) :=
150- Kernel.ext fun ⟨_, a⟩ ↦ by
151- simp [Kernel.sectL_apply, Kernel.comap_apply, Kernel.prodMkLeft_apply]
152- rw [hk] at he; exact he
138+ (fun ω ↦ (E ω, IsAlgEnvSeq.hist A R' n ω, A (n + 1 ) ω))
139+ (κ.comap (fun (e, _, a) ↦ (e, a)) (by fun_prop)) P :=
140+ (h.hasCondDistrib_reward n).comp_right (MeasurableEquiv.prodAssoc.symm.trans
141+ ((MeasurableEquiv.prodCongr .prodComm (.refl _)).trans .prodAssoc))
142+ exact h_reorder.ae_hasCondDistrib_sectR ((IT.measurable_hist n).prodMk
143+ (IT.measurable_action (n + 1 ))) (IT.measurable_reward (n + 1 ))
144+ (measurable_trajectory h.measurable_A h.measurable_R).aemeasurable h.measurable_E.aemeasurable
153145
154146lemma ae_IsAlgEnvSeq [IsMarkovKernel κ] (h : IsBayesAlgEnvSeq Q κ alg E A R' P) :
155147 ∀ᵐ e ∂Q, IsAlgEnvSeq IT.action IT.reward alg (stationaryEnv (κ.sectR e))
156148 (condDistrib (trajectory A R') E P e) := by
157149 filter_upwards [hasLaw_IT_action_zero h, hasCondDistrib_IT_reward_zero h,
158150 ae_all_iff.2 (hasCondDistrib_IT_action h), ae_all_iff.2 (hasCondDistrib_IT_reward h)]
159- with _ h_a0 h_r0 h_a h_r
160- exact {
161- hasLaw_action_zero := h_a0
162- hasCondDistrib_reward_zero := h_r0
163- hasCondDistrib_action := h_a
164- hasCondDistrib_reward := h_r
165- }
151+ with _ ha0 hr0 hA hR
152+ exact ⟨IT.measurable_action, IT.measurable_reward, ha0, hr0, hA, hR⟩
166153
167154end CondDistribIsAlgEnvSeq
168155
@@ -194,28 +181,26 @@ lemma IsAlgEnvSeq.isBayesAlgEnvSeq
194181 apply HasCondDistrib.hasLaw_of_const
195182 simpa [bayesStationaryEnv] using h.hasCondDistrib_reward_zero.fst
196183 hasCondDistrib_action_zero := by
197- have hfst : HasCondDistrib (fun ω ↦ (R' 0 ω).1 ) (A 0 ) (Kernel.const α Q) P := by
184+ have hfst : HasCondDistrib (fun ω ↦ (R' 0 ω).1 ) (A 0 ) (Kernel.const _ Q) P := by
198185 simpa [bayesStationaryEnv] using h.hasCondDistrib_reward_zero.fst
199186 simpa [h.hasLaw_action_zero.map_eq, Algorithm.prod_left] using hfst.swap_const
200- hasCondDistrib_reward_zero := by
201- have h0 := h.hasCondDistrib_reward_zero
202- simp only [bayesStationaryEnv] at h0
203- convert h0.of_compProd.comp_right (MeasurableEquiv.prodComm : α × 𝓔 ≃ᵐ 𝓔 × α) using 2
187+ hasCondDistrib_reward_zero :=
188+ h.hasCondDistrib_reward_zero.of_compProd.comp_right MeasurableEquiv.prodComm
204189 hasCondDistrib_action n := by
205190 let f : (Iic n → α × 𝓔 × R) → 𝓔 × (Iic n → α × R) :=
206191 fun h ↦ ((h ⟨0 , by simp⟩).2 .1 , fun i ↦ ((h i).1 , (h i).2 .2 ))
207- suffices h' : HasCondDistrib (A (n + 1 )) (IsAlgEnvSeq.hist A R' n)
208- (((alg.policy n).comap Prod.snd (by fun_prop)).comap f (by fun_prop)) P from
209- h'.comp_left (f := f)
210- exact h.hasCondDistrib_action n
192+ have hc : HasCondDistrib (A (n + 1 )) (IsAlgEnvSeq.hist A R' n)
193+ (((alg.policy n).comap Prod.snd (by fun_prop)).comap f (by fun_prop)) P :=
194+ h.hasCondDistrib_action n
195+ exact hc.comp_left (f := f)
211196 hasCondDistrib_reward n := by
212197 let f : (Iic n → α × 𝓔 × R) × α → (Iic n → α × R) × 𝓔 × α :=
213198 fun p ↦ ((fun i ↦ ((p.1 i).1 , (p.1 i).2 .2 )), (p.1 ⟨0 , by simp⟩).2 .1 , p.2 )
214- have hf : Measurable f := by fun_prop
215- suffices h' : HasCondDistrib (fun ω ↦ (R' (n + 1 ) ω).2 )
199+ have hc : HasCondDistrib (fun ω ↦ (R' (n + 1 ) ω).2 )
216200 (fun ω ↦ (IsAlgEnvSeq.hist A R' n ω, A (n + 1 ) ω))
217- ((Kernel.prodMkLeft (↥(Iic n) → α × R) κ).comap f hf) P from h'.comp_left hf
218- simpa [bayesStationaryEnv, Kernel.snd_prod] using (h.hasCondDistrib_reward n).snd
201+ ((Kernel.prodMkLeft ((Iic n) → α × R) κ).comap f (by fun_prop)) P := by
202+ simpa [bayesStationaryEnv, Kernel.snd_prod] using (h.hasCondDistrib_reward n).snd
203+ exact hc.comp_left (by fun_prop)
219204
220205end IsAlgEnvSeq
221206
@@ -234,13 +219,12 @@ lemma isBayesAlgEnvSeq_bayesTrajMeasure
234219 (Q : Measure 𝓔) [IsProbabilityMeasure Q] (κ : Kernel (𝓔 × α) R) [IsMarkovKernel κ]
235220 (alg : Algorithm α R) :
236221 IsBayesAlgEnvSeq Q κ alg (fun ω ↦ (ω 0 ).2 .1 ) action (fun n ω ↦ (ω n).2 .2 )
237- (bayesTrajMeasure Q κ alg) :=
238- (isAlgEnvSeq_trajMeasure _ _).isBayesAlgEnvSeq
222+ (bayesTrajMeasure Q κ alg) := (isAlgEnvSeq_trajMeasure _ _).isBayesAlgEnvSeq
239223
240224noncomputable
241225def bayesTrajMeasurePosterior [StandardBorelSpace 𝓔] [Nonempty 𝓔]
242- (Q : Measure 𝓔) [IsProbabilityMeasure Q] (κ : Kernel (𝓔 × α) ℝ ) [IsMarkovKernel κ]
243- (alg : Algorithm α ℝ ) (n : ℕ) : Kernel (Iic n → α × ℝ ) 𝓔 :=
226+ (Q : Measure 𝓔) (κ : Kernel (𝓔 × α) R ) [IsMarkovKernel κ]
227+ (alg : Algorithm α R ) (n : ℕ) : Kernel (Iic n → α × R ) 𝓔 :=
244228 condDistrib (fun ω ↦ (ω 0 ).2 .1 ) (IsAlgEnvSeq.hist action (fun n ω ↦ (ω n).2 .2 ) n)
245229 (bayesTrajMeasure Q κ alg)
246230deriving IsMarkovKernel
0 commit comments