77
88public import LeanMachineLearning.ForMathlib.MeasureTheory.Measurable
99public import LeanMachineLearning.ForMathlib.Probability.Kernel.IonescuTulcea.Traj
10+ public import LeanMachineLearning.ForMathlib.Probability.Kernel.MeasurableSpace
1011
1112/-!
1213# Algorithms and environments
@@ -110,6 +111,15 @@ structure Algorithm (π π π¨ : Type*) [MeasurableSpace π] [MeasurableS
110111instance (alg : Algorithm π π π¨) (n : β) : IsMarkovKernel (alg.policy n) :=
111112 alg.isMarkovKernel_policy n
112113
114+ instance : MeasurableSpace (Algorithm π π π¨) :=
115+ MeasurableSpace.comap (fun alg β¦ alg.policy) inferInstance
116+
117+ lemma measurable_algorithm_iff (f : Ξ© β Algorithm π π π¨) :
118+ Measurable f β β n, Measurable fun x β¦ (f x).policy n := by
119+ unfold instMeasurableSpaceAlgorithm
120+ rw [measurable_comap_iff, measurable_pi_iff]
121+ simp
122+
113123/-- A stochastic environment.
114124At each round, an observation is drawn prior to the algorithm taking an action. Then the environment
115125provides feedback based on the observation and the action. -/
@@ -129,6 +139,16 @@ instance (env : Environment π π π¨) (n : β) : IsMarkovKernel (env.obs
129139instance (env : Environment π π π¨) (n : β) : IsMarkovKernel (env.feedback n) :=
130140 env.isMarkovKernel_feedback n
131141
142+ instance : MeasurableSpace (Environment π π π¨) :=
143+ MeasurableSpace.comap (fun env β¦ (env.obs, env.feedback)) inferInstance
144+
145+ lemma measurable_environment_iff (f : Ξ© β Environment π π π¨) :
146+ Measurable f β
147+ β n, Measurable (fun x β¦ (f x).obs n) β§ Measurable (fun x β¦ (f x).feedback n) := by
148+ simp_rw [measurable_comap_iff, measurable_fun_prod, forall_and, measurable_pi_iff,
149+ measurable_kernel_iff]
150+ rfl
151+
132152/-- Distribution of the first observation: the observation kernel at time `0` applied to the empty
133153history. -/
134154def Environment.obs0 (env : Environment π π π¨) : Measure π :=
0 commit comments