@@ -97,14 +97,13 @@ variable {A : ℕ → Ω → 𝓐} {Y : ℕ → Ω → 𝓨} {alg : Algorithm
9797 {P : Measure Ω} [IsFiniteMeasure P] {N : ℕ}
9898
9999/-- Step of the algorithm-environment sequence: the action-feedback pair at time `n`. -/
100- def IsAlgEnvSeq. step (A : ℕ → Ω → 𝓐) (Y : ℕ → Ω → 𝓨) (n : ℕ) (ω : Ω) : 𝓐 × 𝓨 :=
100+ def step (A : ℕ → Ω → 𝓐) (Y : ℕ → Ω → 𝓨) (n : ℕ) (ω : Ω) : 𝓐 × 𝓨 :=
101101 (A n ω, Y n ω)
102102
103103@[fun_prop]
104- lemma IsAlgEnvSeq.measurable_step (n : ℕ) (hA : Measurable (A n))
105- (hY : Measurable (Y n)) :
106- Measurable (IsAlgEnvSeq.step A Y n) := by
107- unfold IsAlgEnvSeq.step
104+ lemma measurable_step (n : ℕ) (hA : Measurable (A n)) (hY : Measurable (Y n)) :
105+ Measurable (step A Y n) := by
106+ unfold step
108107 fun_prop
109108
110109/-- A random variable that gives the sequence of action-feedback pairs. -/
@@ -117,24 +116,24 @@ lemma measurable_trajectory {A : ℕ → Ω → 𝓐} {Y : ℕ → Ω → 𝓨}
117116 fun_prop
118117
119118/-- History of the algorithm-environment sequence up to time `n`. -/
120- def IsAlgEnvSeq.hist (A : ℕ → Ω → 𝓐) (Y : ℕ → Ω → 𝓨) (n : ℕ) (ω : Ω) : Iic n → 𝓐 × 𝓨 :=
119+ def history (A : ℕ → Ω → 𝓐) (Y : ℕ → Ω → 𝓨) (n : ℕ) (ω : Ω) : Iic n → 𝓐 × 𝓨 :=
121120 fun i ↦ (A i ω, Y i ω)
122121
123122@[fun_prop]
124- lemma IsAlgEnvSeq.measurable_hist (hA : ∀ n, Measurable (A n))
123+ lemma measurable_history (hA : ∀ n, Measurable (A n))
125124 (hY : ∀ n, Measurable (Y n)) (n : ℕ) :
126- Measurable (IsAlgEnvSeq.hist A Y n) := by
127- unfold IsAlgEnvSeq.hist
125+ Measurable (history A Y n) := by
126+ unfold history
128127 fun_prop
129128
130- lemma IsAlgEnvSeq.eval_comp_hist (n : ℕ) :
131- (fun x ↦ x ⟨n, by simp⟩) ∘ (hist A Y n) = step A Y n := rfl
129+ lemma eval_comp_history (n : ℕ) :
130+ (fun x ↦ x ⟨n, by simp⟩) ∘ (history A Y n) = step A Y n := rfl
132131
133- lemma IsAlgEnvSeq.fst_eval_comp_hist (n : ℕ) :
134- (fun x ↦ (x ⟨n, by simp⟩).1 ) ∘ (hist A Y n) = A n := rfl
132+ lemma fst_eval_comp_history (n : ℕ) :
133+ (fun x ↦ (x ⟨n, by simp⟩).1 ) ∘ (history A Y n) = A n := rfl
135134
136- lemma IsAlgEnvSeq.snd_eval_comp_hist (n : ℕ) :
137- (fun x ↦ (x ⟨n, by simp⟩).2 ) ∘ (hist A Y n) = Y n := rfl
135+ lemma snd_eval_comp_history (n : ℕ) :
136+ (fun x ↦ (x ⟨n, by simp⟩).2 ) ∘ (history A Y n) = Y n := rfl
138137
139138section IsAlgEnvSeq
140139
@@ -155,11 +154,11 @@ structure IsAlgEnvSeq
155154 hasCondDistrib_feedback_zero : HasCondDistrib (Y 0 ) (A 0 ) env.ν0 P
156155 /-- The next action has the correct conditional distribution given the history. -/
157156 hasCondDistrib_action n :
158- HasCondDistrib (A (n + 1 )) (IsAlgEnvSeq.hist A Y n) (alg.policy n) P
157+ HasCondDistrib (A (n + 1 )) (history A Y n) (alg.policy n) P
159158 /-- The next feedback has the correct conditional distribution given the history and
160159 next action. -/
161160 hasCondDistrib_feedback n :
162- HasCondDistrib (Y (n + 1 )) (fun ω ↦ (IsAlgEnvSeq.hist A Y n ω, A (n + 1 ) ω))
161+ HasCondDistrib (Y (n + 1 )) (fun ω ↦ (history A Y n ω, A (n + 1 ) ω))
163162 (env.feedback n) P
164163
165164/-- An algorithm-environment sequence: a sequence of actions and feedbacks generated
@@ -177,11 +176,11 @@ structure IsAlgEnvSeqUntil
177176 hasCondDistrib_feedback_zero : HasCondDistrib (Y 0 ) (A 0 ) env.ν0 P
178177 /-- The next action has the correct conditional distribution given the history. -/
179178 hasCondDistrib_action n (hn : n < N) :
180- HasCondDistrib (A (n + 1 )) (IsAlgEnvSeq.hist A Y n) (alg.policy n) P
179+ HasCondDistrib (A (n + 1 )) (history A Y n) (alg.policy n) P
181180 /-- The next feedback has the correct conditional distribution given the history and
182181 next action. -/
183182 hasCondDistrib_feedback n (hn : n < N) :
184- HasCondDistrib (Y (n + 1 )) (fun ω ↦ (IsAlgEnvSeq.hist A Y n ω, A (n + 1 ) ω))
183+ HasCondDistrib (Y (n + 1 )) (fun ω ↦ (history A Y n ω, A (n + 1 ) ω))
185184 (env.feedback n) P
186185
187186lemma IsAlgEnvSeqUntil.mono (h : IsAlgEnvSeqUntil A Y alg env P N) {N' : ℕ} (hN : N' ≤ N) :
@@ -202,47 +201,61 @@ lemma IsAlgEnvSeq.isAlgEnvSeqUntil (h : IsAlgEnvSeq A Y alg env P) (N : ℕ) :
202201 hasCondDistrib_action n _ := h.hasCondDistrib_action n
203202 hasCondDistrib_feedback n _ := h.hasCondDistrib_feedback n
204203
204+ @[fun_prop]
205+ lemma IsAlgEnvSeq.measurable_step (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) :
206+ Measurable (step A Y n) := by
207+ have hA := h.measurable_action
208+ have hY := h.measurable_feedback
209+ fun_prop
210+
211+ @[fun_prop]
212+ lemma IsAlgEnvSeq.measurable_history (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) :
213+ Measurable (history A Y n) := by
214+ have hA := h.measurable_action
215+ have hY := h.measurable_feedback
216+ fun_prop
217+
205218lemma IsAlgEnvSeq.hasLaw_step_zero (h : IsAlgEnvSeq A Y alg env P) :
206219 HasLaw (step A Y 0 ) (alg.p0 ⊗ₘ env.ν0 ) P :=
207220 HasLaw.prod_of_hasCondDistrib h.hasLaw_action_zero h.hasCondDistrib_feedback_zero
208221
209222lemma IsAlgEnvSeqUntil.hasLaw_step_zero (h : IsAlgEnvSeqUntil A Y alg env P N) :
210- HasLaw (IsAlgEnvSeq. step A Y 0 ) (alg.p0 ⊗ₘ env.ν0 ) P :=
223+ HasLaw (step A Y 0 ) (alg.p0 ⊗ₘ env.ν0 ) P :=
211224 HasLaw.prod_of_hasCondDistrib h.hasLaw_action_zero h.hasCondDistrib_feedback_zero
212225
213226lemma IsAlgEnvSeq.hasCondDistrib_step (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) :
214- HasCondDistrib (step A Y (n + 1 )) (hist A Y n) (stepKernel alg env n) P :=
227+ HasCondDistrib (step A Y (n + 1 )) (history A Y n) (stepKernel alg env n) P :=
215228 HasCondDistrib.prod (h.hasCondDistrib_action n) (h.hasCondDistrib_feedback n)
216229
217230lemma IsAlgEnvSeqUntil.hasCondDistrib_step (h : IsAlgEnvSeqUntil A Y alg env P N)
218231 (n : ℕ) (hn : n < N) :
219- HasCondDistrib (IsAlgEnvSeq. step A Y (n + 1 )) (IsAlgEnvSeq.hist A Y n)
232+ HasCondDistrib (step A Y (n + 1 )) (history A Y n)
220233 (stepKernel alg env n) P :=
221234 HasCondDistrib.prod (h.hasCondDistrib_action n hn) (h.hasCondDistrib_feedback n hn)
222235
223- lemma IsAlgEnvSeq.hasLaw_hist_zero (h : IsAlgEnvSeq A Y alg env P) : HasLaw (hist A Y 0 )
236+ lemma IsAlgEnvSeq.hasLaw_history_zero (h : IsAlgEnvSeq A Y alg env P) : HasLaw (history A Y 0 )
224237 ((P.map (step A Y 0 )).map (MeasurableEquiv.piUnique (fun _ : Iic 0 ↦ 𝓐 × 𝓨)).symm) P where
225- aemeasurable := (measurable_hist h.measurable_action h.measurable_feedback 0 ).aemeasurable
238+ aemeasurable := (h.measurable_history 0 ).aemeasurable
226239 map_eq := by
227240 have he : (MeasurableEquiv.piUnique (fun _ : Iic 0 ↦ 𝓐 × 𝓨)).symm ∘ step A Y 0 =
228- hist A Y 0 := by
241+ history A Y 0 := by
229242 funext _ ⟨0 , _⟩
230243 rfl
231244 rw [← he]
232245 have hA := h.measurable_action
233246 have hY := h.measurable_feedback
234247 exact (Measure.map_map (by fun_prop) (by fun_prop)).symm
235248
236- lemma IsAlgEnvSeq.hasLaw_hist_succ (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) :
237- HasLaw (hist A Y (n + 1 ))
238- ((P.map (hist A Y n) ⊗ₘ condDistrib (step A Y (n + 1 )) (hist A Y n) P).map
249+ lemma IsAlgEnvSeq.hasLaw_history_succ (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) :
250+ HasLaw (history A Y (n + 1 ))
251+ ((P.map (history A Y n) ⊗ₘ condDistrib (step A Y (n + 1 )) (history A Y n) P).map
239252 (MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm) P where
240- aemeasurable := (measurable_hist h.measurable_action h.measurable_feedback (n + 1 )).aemeasurable
253+ aemeasurable := (h.measurable_history (n + 1 )).aemeasurable
241254 map_eq := by
242255 have he : (MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm ∘
243- (fun ω ↦ (hist A Y n ω, step A Y (n + 1 ) ω)) = hist A Y (n + 1 ) := by
256+ (fun ω ↦ (history A Y n ω, step A Y (n + 1 ) ω)) = history A Y (n + 1 ) := by
244257 funext ω
245- exact (MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm_apply_apply (hist A Y (n + 1 ) ω)
258+ exact (MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm_apply_apply (history A Y (n + 1 ) ω)
246259 have hA := h.measurable_action
247260 have hY := h.measurable_feedback
248261 rw [← he, ← Measure.map_map (by fun_prop) (by fun_prop)]
@@ -254,49 +267,49 @@ end IsAlgEnvSeq
254267/-- Filtration generated by the history up to time `n`. -/
255268def IsAlgEnvSeq.filtration (hA : ∀ n, Measurable (A n)) (hY : ∀ n, Measurable (Y n)) :
256269 Filtration ℕ mΩ where
257- seq i := MeasurableSpace.comap (hist A Y i) inferInstance
270+ seq i := MeasurableSpace.comap (history A Y i) inferInstance
258271 mono' i j hij := by
259272 simp only
260273 rw [← measurable_iff_comap_le]
261- have : hist A Y i = (fun h k ↦ h ⟨k.1 , by grind⟩) ∘ hist A Y j := rfl
274+ have : history A Y i = (fun h k ↦ h ⟨k.1 , by grind⟩) ∘ history A Y j := rfl
262275 rw [this]
263276 exact measurable_comp_comap _ (by fun_prop)
264277 le' i := by
265278 rw [← measurable_iff_comap_le]
266- exact measurable_hist hA hY i
279+ exact Learning.measurable_history hA hY i
267280
268- lemma IsAlgEnvSeq.adapted_hist
281+ lemma IsAlgEnvSeq.adapted_history
269282 (hA : ∀ n, Measurable (A n)) (hY : ∀ n, Measurable (Y n)) :
270- Adapted (filtration hA hY) (IsAlgEnvSeq.hist A Y) :=
283+ Adapted (filtration hA hY) (history A Y) :=
271284 fun _ ↦ measurable_iff_comap_le.mpr le_rfl
272285
273286lemma IsAlgEnvSeq.adapted_step
274287 (hA : ∀ n, Measurable (A n)) (hY : ∀ n, Measurable (Y n)) :
275288 Adapted (filtration hA hY) (step A Y) := by
276289 intro n
277- have : step A Y n = (fun h ↦ (h ⟨n, by simp⟩)) ∘ (hist A Y n) := by
290+ have : step A Y n = (fun h ↦ (h ⟨n, by simp⟩)) ∘ (history A Y n) := by
278291 ext ω : 1
279- simp [hist , step]
292+ simp [history , step]
280293 rw [this]
281294 exact measurable_comp_comap _ (by fun_prop)
282295
283296lemma IsAlgEnvSeq.adapted_action
284297 (hA : ∀ n, Measurable (A n)) (hY : ∀ n, Measurable (Y n)) :
285298 Adapted (filtration hA hY) A := by
286299 intro n
287- have : A n = (fun h ↦ (h ⟨n, by simp⟩).1 ) ∘ (hist A Y n) := by
300+ have : A n = (fun h ↦ (h ⟨n, by simp⟩).1 ) ∘ (history A Y n) := by
288301 ext ω : 1
289- simp [IsAlgEnvSeq.hist ]
302+ simp [history ]
290303 rw [this]
291304 exact measurable_comp_comap _ (by fun_prop)
292305
293306lemma IsAlgEnvSeq.adapted_feedback
294307 (hA : ∀ n, Measurable (A n)) (hY : ∀ n, Measurable (Y n)) :
295308 Adapted (filtration hA hY) Y := by
296309 intro n
297- have : Y n = (fun h ↦ (h ⟨n, by simp⟩).2 ) ∘ (hist A Y n) := by
310+ have : Y n = (fun h ↦ (h ⟨n, by simp⟩).2 ) ∘ (history A Y n) := by
298311 ext ω : 1
299- simp [IsAlgEnvSeq.hist ]
312+ simp [history ]
300313 rw [this]
301314 exact measurable_comp_comap _ (by fun_prop)
302315
@@ -351,7 +364,7 @@ lemma IsAlgEnvSeq.filtrationAction_zero_eq_comap
351364lemma IsAlgEnvSeq.filtrationAction_eq_comap
352365 {hA : ∀ n, Measurable (A n)} {hY : ∀ n, Measurable (Y n)} (n : ℕ) (hn : n ≠ 0 ) :
353366 filtrationAction hA hY n =
354- MeasurableSpace.comap (fun ω ↦ (hist A Y (n - 1 ) ω, A n ω)) inferInstance := by
367+ MeasurableSpace.comap (fun ω ↦ (history A Y (n - 1 ) ω, A n ω)) inferInstance := by
355368 simp only [filtrationAction, filtration, ← MeasurableSpace.comap_prodMk, hn, ↓reduceIte]
356369 rfl
357370
0 commit comments