@@ -208,6 +208,43 @@ lemma action_zero_detAlgorithm [MeasurableSingletonClass α] : action 0 =ᵐ[
208208 simp [detAlgorithm]
209209 exact ae_of_ae_map (by fun_prop) h_eq
210210
211+ lemma action_eq_eval_comp_hist (n : ℕ) :
212+ action (α := α) (R := R) n = (fun x ↦ (x ⟨n, by simp⟩).1 ) ∘ (hist n) := rfl
213+
214+ lemma reward_eq_eval_comp_hist (n : ℕ) :
215+ reward (α := α) (R := R) n = (fun x ↦ (x ⟨n, by simp⟩).2 ) ∘ (hist n) := rfl
216+
217+ lemma measurable_hist_filtration (n : ℕ) : Measurable[Learning.filtration α R n] (hist n) := by
218+ simp [Learning.filtration, Filtration.piLE_eq_comap_frestrictLe, ← hist_eq_frestrictLe,
219+ measurable_iff_comap_le]
220+
221+ -- todo: due to the type of `Adapted` and the fact that `Iic n → α × R` depends on `n`, we cannot
222+ -- state that `hist` is adapted.
223+
224+ lemma measurable_action_filtration (n : ℕ) : Measurable[Learning.filtration α R n] (action n) := by
225+ simp only [Learning.filtration, Filtration.piLE_eq_comap_frestrictLe, ← hist_eq_frestrictLe]
226+ rw [action_eq_eval_comp_hist, measurable_iff_comap_le, ← MeasurableSpace.comap_comp]
227+ refine MeasurableSpace.comap_mono ?_
228+ rw [← measurable_iff_comap_le]
229+ fun_prop
230+
231+ lemma adapted_action [TopologicalSpace α] [TopologicalSpace.PseudoMetrizableSpace α]
232+ [SecondCountableTopology α] [OpensMeasurableSpace α] :
233+ Adapted (Learning.filtration α R) action :=
234+ fun n ↦ (measurable_action_filtration n).stronglyMeasurable
235+
236+ lemma measurable_reward_filtration (n : ℕ) : Measurable[Learning.filtration α R n] (reward n) := by
237+ simp only [Learning.filtration, Filtration.piLE_eq_comap_frestrictLe, ← hist_eq_frestrictLe]
238+ rw [reward_eq_eval_comp_hist, measurable_iff_comap_le, ← MeasurableSpace.comap_comp]
239+ refine MeasurableSpace.comap_mono ?_
240+ rw [← measurable_iff_comap_le]
241+ fun_prop
242+
243+ lemma adapted_reward [TopologicalSpace R] [TopologicalSpace.PseudoMetrizableSpace R]
244+ [SecondCountableTopology R] [OpensMeasurableSpace R] :
245+ Adapted (Learning.filtration α R) reward :=
246+ fun n ↦ (measurable_reward_filtration n).stronglyMeasurable
247+
211248lemma action_detAlgorithm_ae_eq
212249 [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R]
213250 (n : ℕ) :
0 commit comments