Skip to content

Commit 86f4300

Browse files
committed
Merge branch 'main' into issue-112
2 parents fbbfe2a + cce2458 commit 86f4300

5 files changed

Lines changed: 341 additions & 1 deletion

File tree

‎LeanMachineLearning.lean‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -18,10 +18,13 @@ public import LeanMachineLearning.Probability.Independence.IndepInfinitePi
1818
public import LeanMachineLearning.Probability.Integrable
1919
public import LeanMachineLearning.Probability.Kernel.Basic
2020
public import LeanMachineLearning.Probability.Kernel.Composition.MapComap
21+
public import LeanMachineLearning.Probability.Kernel.Composition.MeasureCompProd
2122
public import LeanMachineLearning.Probability.Kernel.IonescuTulcea.Traj
2223
public import LeanMachineLearning.Probability.Kernel.KernelSub
2324
public import LeanMachineLearning.Probability.Moments.SubGaussian
25+
public import LeanMachineLearning.Probability.WithDensity
2426
public import LeanMachineLearning.SequentialLearning.Algorithm
27+
public import LeanMachineLearning.SequentialLearning.AlgorithmDensity
2528
public import LeanMachineLearning.SequentialLearning.Algorithms.RandomSampling
2629
public import LeanMachineLearning.SequentialLearning.Algorithms.RoundRobin
2730
public import LeanMachineLearning.SequentialLearning.Deterministic
Lines changed: 31 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,31 @@
1+
/-
2+
Copyright (c) 2026 Paulo Rauber. All rights reserved.
3+
Released under Apache 2.0 license as described in the file LICENSE.
4+
Authors: Paulo Rauber
5+
-/
6+
module
7+
8+
public import Mathlib.Probability.Kernel.Composition.MeasureCompProd
9+
10+
@[expose] public section
11+
12+
open ProbabilityTheory
13+
14+
namespace MeasureTheory.Measure
15+
16+
variable {α β : Type*} {mα : MeasurableSpace α} {mβ : MeasurableSpace β} {κ η : Kernel α β}
17+
18+
section AbsolutelyContinuous
19+
20+
lemma AbsolutelyContinuous.compProd_left_apply {γ : Type*} {mγ : MeasurableSpace γ}
21+
[IsSFiniteKernel η] {a : α} (hac : κ a ≪ η a) (ξ : Kernel (α × β) γ) :
22+
(κ ⊗ₖ ξ) a ≪ (η ⊗ₖ ξ) a := by
23+
by_cases hκ : IsSFiniteKernel κ
24+
· by_cases hξ : IsSFiniteKernel ξ
25+
· simp_rw [Kernel.compProd_apply_eq_compProd_sectR, hac.compProd_left _]
26+
· simp [Kernel.compProd_of_not_isSFiniteKernel_right _ _ hξ]
27+
· simp [Kernel.compProd_of_not_isSFiniteKernel_left _ _ hκ]
28+
29+
end AbsolutelyContinuous
30+
31+
end MeasureTheory.Measure
Lines changed: 125 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,125 @@
1+
/-
2+
Copyright (c) 2026 Paulo Rauber. All rights reserved.
3+
Released under Apache 2.0 license as described in the file LICENSE.
4+
Authors: Paulo Rauber
5+
-/
6+
module
7+
8+
public import Mathlib.Probability.Kernel.CompProdEqIff
9+
public import Mathlib.Probability.Kernel.Composition.MeasureComp
10+
11+
@[expose] public section
12+
13+
open MeasureTheory ProbabilityTheory
14+
15+
open scoped ENNReal
16+
17+
variable {α β γ : Type*} {mα : MeasurableSpace α} {mβ : MeasurableSpace β} {mγ : MeasurableSpace γ}
18+
variable {μ : Measure α}
19+
20+
namespace MeasureTheory
21+
22+
lemma map_withDensity_comp {g : α → γ} {f : γ → ℝ≥0∞} (hg : Measurable g) (hf : Measurable f) :
23+
(μ.withDensity (f ∘ g)).map g = (μ.map g).withDensity f := by
24+
ext s hs
25+
rw [Measure.map_apply hg hs, withDensity_apply _ (hg hs), withDensity_apply _ hs,
26+
setLIntegral_map hs hf hg]
27+
rfl
28+
29+
lemma map_equiv_withDensity {e : α ≃ᵐ β} {f : α → ℝ≥0∞} (hf : Measurable f) :
30+
(μ.withDensity f).map e = (μ.map e).withDensity (f ∘ e.symm) := by
31+
simp_rw [← map_withDensity_comp e.measurable (hf.comp e.symm.measurable),
32+
Function.comp_assoc, MeasurableEquiv.symm_comp_self]
33+
rfl
34+
35+
lemma map_swap_withDensity_comp_snd {μ : Measure (α × β)} {f : β → ℝ≥0∞} (hf : Measurable f) :
36+
(μ.withDensity (fun ab ↦ f ab.2)).map Prod.swap =
37+
(μ.map Prod.swap).withDensity (fun ba ↦ f ba.1) := by
38+
rw [← map_withDensity_comp measurable_swap (by fun_prop)]
39+
rfl
40+
41+
end MeasureTheory
42+
43+
namespace MeasureTheory.Measure
44+
45+
lemma compProd_withDensity_left [SFinite μ] {κ : Kernel α β} [IsSFiniteKernel κ] {f : α → ℝ≥0∞}
46+
(hf : Measurable f) : (μ.withDensity f) ⊗ₘ κ = (μ ⊗ₘ κ).withDensity (fun ab ↦ f ab.1) := by
47+
refine ext_of_lintegral _ fun g hg ↦ ?_
48+
calc ∫⁻ ab, g ab ∂((μ.withDensity f) ⊗ₘ κ)
49+
= ∫⁻ a, ∫⁻ b, g (a, b) ∂κ a ∂(μ.withDensity f) :=
50+
lintegral_compProd hg
51+
_ = ∫⁻ a, f a * ∫⁻ b, g (a, b) ∂κ a ∂μ :=
52+
lintegral_withDensity_eq_lintegral_mul _ hf hg.lintegral_kernel_prod_right'
53+
_ = ∫⁻ a, ∫⁻ b, f a * g (a, b) ∂κ a ∂μ :=
54+
lintegral_congr fun a ↦ (lintegral_const_mul _ (by fun_prop)).symm
55+
_ = ∫⁻ ab, (fun ab ↦ f ab.1) ab * g ab ∂(μ ⊗ₘ κ) :=
56+
(lintegral_compProd ((hf.comp measurable_fst).mul hg)).symm
57+
_ = ∫⁻ ab, g ab ∂((μ ⊗ₘ κ).withDensity (fun ab ↦ f ab.1)) :=
58+
(lintegral_withDensity_eq_lintegral_mul _ (hf.comp measurable_fst) hg).symm
59+
60+
lemma compProd_withDensity_withDensity [SFinite μ] {κ : Kernel α β} [IsSFiniteKernel κ]
61+
{f : α → ℝ≥0∞} {g : α → β → ℝ≥0∞} (hf : Measurable f) (hg : Measurable (Function.uncurry g))
62+
[IsSFiniteKernel (κ.withDensity g)] :
63+
(μ.withDensity f) ⊗ₘ (κ.withDensity g) =
64+
(μ ⊗ₘ κ).withDensity (fun ac ↦ f ac.1 * g ac.1 ac.2) := by
65+
rw [compProd_withDensity hg, compProd_withDensity_left hf]
66+
exact (withDensity_mul _ (hf.comp measurable_fst) hg).symm
67+
68+
lemma compProd_eq_compProd_withDensity_comp_snd [SFinite μ] {κ η : Kernel α β} [IsSFiniteKernel κ]
69+
[IsSFiniteKernel η] {f : β → ℝ≥0∞} (hf : Measurable f)
70+
(h : κ =ᵐ[μ] η.withDensity (fun _ b ↦ f b)) :
71+
μ ⊗ₘ κ = (μ ⊗ₘ η).withDensity (fun ab ↦ f ab.2) := by
72+
/- A proof based on `compProd_congr` requires `IsSFiniteKernel (η.withDensity fun _ b ↦ f b)`. -/
73+
refine ext_of_lintegral _ fun g hg ↦ ?_
74+
calc ∫⁻ ab, g ab ∂(μ ⊗ₘ κ)
75+
= ∫⁻ a, ∫⁻ b, g (a, b) ∂κ a ∂μ :=
76+
lintegral_compProd hg
77+
_ = ∫⁻ a, ∫⁻ b, g (a, b) ∂((η a).withDensity f) ∂μ := by
78+
apply lintegral_congr_ae
79+
filter_upwards [h] with a ha
80+
rw [ha, Kernel.withDensity_apply _ (by fun_prop)]
81+
_ = ∫⁻ a, ∫⁻ b, f b * g (a, b) ∂η a ∂μ := by
82+
congr with a
83+
exact lintegral_withDensity_eq_lintegral_mul _ hf (by fun_prop)
84+
_ = ∫⁻ ab, f ab.2 * g ab ∂(μ ⊗ₘ η) :=
85+
(lintegral_compProd ((hf.comp measurable_snd).mul hg)).symm
86+
_ = ∫⁻ ab, g ab ∂((μ ⊗ₘ η).withDensity (fun ab ↦ f ab.2)) :=
87+
(lintegral_withDensity_eq_lintegral_mul _ (hf.comp measurable_snd) hg).symm
88+
89+
end MeasureTheory.Measure
90+
91+
namespace ProbabilityTheory.Kernel
92+
93+
lemma comp_withDensity_eq_withDensity_comp {κ : Kernel α β} [IsSFiniteKernel κ] {f : β → ℝ≥0∞}
94+
(hf : Measurable f) : (κ.withDensity (fun _ b ↦ f b)) ∘ₘ μ = (κ ∘ₘ μ).withDensity f := by
95+
refine Measure.ext_of_lintegral _ fun g hg ↦ ?_
96+
calc ∫⁻ b, g b ∂((κ.withDensity (fun _ b ↦ f b)) ∘ₘ μ)
97+
= ∫⁻ a, ∫⁻ b, g b ∂(κ.withDensity (fun _ b ↦ f b)) a ∂μ :=
98+
Measure.lintegral_bind (measurable _).aemeasurable hg.aemeasurable
99+
_ = ∫⁻ a, ∫⁻ b, f b * g b ∂κ a ∂μ := by
100+
congr with a
101+
exact lintegral_withDensity _ (by fun_prop) _ hg
102+
_ = ∫⁻ b, f b * g b ∂(κ ∘ₘ μ) :=
103+
(Measure.lintegral_bind (measurable _).aemeasurable (hf.mul hg).aemeasurable).symm
104+
_ = ∫⁻ b, g b ∂((κ ∘ₘ μ).withDensity f) :=
105+
(lintegral_withDensity_eq_lintegral_mul _ hf hg).symm
106+
107+
lemma compProd_withDensity_left {κ : Kernel α β} {η : Kernel (α × β) γ} {f : α → β → ℝ≥0∞}
108+
[IsSFiniteKernel κ] [IsSFiniteKernel η] [IsSFiniteKernel (κ.withDensity f)]
109+
(hf : Measurable (Function.uncurry f)) :
110+
(κ.withDensity f) ⊗ₖ η = (κ ⊗ₖ η).withDensity (fun a bc ↦ f a bc.1) := by
111+
ext a : 1
112+
calc ((κ.withDensity f) ⊗ₖ η) a
113+
= (κ a).withDensity (f a) ⊗ₘ η.sectR a := by
114+
rw [compProd_apply_eq_compProd_sectR, Kernel.withDensity_apply _ hf]
115+
_ = ((κ a) ⊗ₘ (η.sectR a)).withDensity (fun bc ↦ f a bc.1) :=
116+
Measure.compProd_withDensity_left (by fun_prop)
117+
_ = ((κ ⊗ₖ η).withDensity (fun a bc ↦ f a bc.1)) a := by
118+
rw [← compProd_apply_eq_compProd_sectR, Kernel.withDensity_apply _ (by fun_prop)]
119+
120+
lemma withDensity_rnDeriv_eq' {κ η : Kernel α β} [MeasurableSpace.CountableOrCountablyGenerated α β]
121+
[IsFiniteKernel κ] [IsFiniteKernel η] (h : ∀ a, κ a ≪ η a) :
122+
η.withDensity (κ.rnDeriv η) = κ :=
123+
Kernel.ext fun a ↦ withDensity_rnDeriv_eq (h a)
124+
125+
end ProbabilityTheory.Kernel

‎LeanMachineLearning/SequentialLearning/Algorithm.lean‎

Lines changed: 39 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -19,11 +19,13 @@ an algorithm interacting with an environment.
1919
2020
* `Algorithm 𝓐 𝓨`: a stochastic, sequential algorithm.
2121
* `Environment 𝓐 𝓨`: a stochastic environment.
22-
* `IsAlgEnvSeq A 𝓨' alg env P`: an algorithm-environment sequence. That is, a sequence of
22+
* `IsAlgEnvSeq A 𝓨 alg env P`: an algorithm-environment sequence. That is, a sequence of
2323
actions `A` and feedback `Y` that have the correct conditional distributions to be generated by
2424
an algorithm `alg` interacting with an environment `env`, defined on a probability space `(Ω, P)`.
2525
* `IsAlgEnvSeqUntil A Y alg env P N`: `A` and `Y` form an algorithm-environment sequence until
2626
time `N`.
27+
* `prod_left alg`: an `Algorithm 𝓐 (𝓧 × 𝓨)` obtained from an algorithm `alg : Algorithm 𝓐 𝓨` by
28+
ignoring the `𝓧` component of each observation.
2729
2830
## Notes
2931
@@ -55,6 +57,13 @@ structure Algorithm (𝓐 𝓨 : Type*) [MeasurableSpace 𝓐] [MeasurableSpace
5557
instance (alg : Algorithm 𝓐 𝓨) (n : ℕ) : IsMarkovKernel (alg.policy n) := alg.h_policy n
5658
instance (alg : Algorithm 𝓐 𝓨) : IsProbabilityMeasure alg.p0 := alg.hp0
5759

60+
/-- An algorithm with observations in `𝓧 × 𝓨` obtained from an algorithm with observations in `𝓨`
61+
by ignoring the `𝓧` component of each observation. -/
62+
def Algorithm.prod_left (𝓧 : Type*) [MeasurableSpace 𝓧] (alg : Algorithm 𝓐 𝓨) :
63+
Algorithm 𝓐 (𝓧 × 𝓨) where
64+
policy n := (alg.policy n).comap (fun h i ↦ ((h i).1, (h i).2.2)) (by fun_prop)
65+
p0 := alg.p0
66+
5867
/-- A stochastic environment. -/
5968
-- ANCHOR: Environment
6069
structure Environment (𝓐 𝓨 : Type*) [MeasurableSpace 𝓐] [MeasurableSpace 𝓨] where
@@ -190,6 +199,35 @@ lemma IsAlgEnvSeqUntil.hasCondDistrib_step (h : IsAlgEnvSeqUntil A Y alg env P N
190199
(stepKernel alg env n) P :=
191200
HasCondDistrib.prod (h.hasCondDistrib_action n hn) (h.hasCondDistrib_feedback n hn)
192201

202+
lemma IsAlgEnvSeq.hasLaw_hist_zero (h : IsAlgEnvSeq A Y alg env P) : HasLaw (hist A Y 0)
203+
((P.map (step A Y 0)).map (MeasurableEquiv.piUnique (fun _ : Iic 0 ↦ 𝓐 × 𝓨)).symm) P where
204+
aemeasurable := (measurable_hist h.measurable_action h.measurable_feedback 0).aemeasurable
205+
map_eq := by
206+
have he : (MeasurableEquiv.piUnique (fun _ : Iic 0 ↦ 𝓐 × 𝓨)).symm ∘ step A Y 0 =
207+
hist A Y 0 := by
208+
funext _ ⟨0, _⟩
209+
rfl
210+
rw [← he]
211+
have hA := h.measurable_action
212+
have hY := h.measurable_feedback
213+
exact (Measure.map_map (by fun_prop) (by fun_prop)).symm
214+
215+
lemma IsAlgEnvSeq.hasLaw_hist_succ (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) :
216+
HasLaw (hist A Y (n + 1))
217+
((P.map (hist A Y n) ⊗ₘ condDistrib (step A Y (n + 1)) (hist A Y n) P).map
218+
(MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm) P where
219+
aemeasurable := (measurable_hist h.measurable_action h.measurable_feedback (n + 1)).aemeasurable
220+
map_eq := by
221+
have he : (MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm ∘
222+
(fun ω ↦ (hist A Y n ω, step A Y (n + 1) ω)) = hist A Y (n + 1) := by
223+
funext ω
224+
exact (MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n).symm_apply_apply (hist A Y (n + 1) ω)
225+
have hA := h.measurable_action
226+
have hY := h.measurable_feedback
227+
rw [← he, ← Measure.map_map (by fun_prop) (by fun_prop)]
228+
congr
229+
exact (compProd_map_condDistrib (by fun_prop)).symm
230+
193231
end IsAlgEnvSeq
194232

195233
/-- Filtration generated by the history up to time `n`. -/
Lines changed: 143 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,143 @@
1+
/-
2+
Copyright (c) 2026 Paulo Rauber. All rights reserved.
3+
Released under Apache 2.0 license as described in the file LICENSE.
4+
Authors: Paulo Rauber
5+
-/
6+
module
7+
8+
public import LeanMachineLearning.Probability.Kernel.Composition.MeasureCompProd
9+
public import LeanMachineLearning.Probability.WithDensity
10+
public import LeanMachineLearning.SequentialLearning.Algorithm
11+
12+
/-!
13+
# Algorithm density
14+
15+
We define a density function that allows obtaining the law of the history under one algorithm from
16+
the law of the history under another algorithm when they are interacting with the same
17+
environment. This also requires one algorithm to be absolutely continuous with respect to another, a
18+
concept that we also introduce here.
19+
20+
## Main definitions
21+
22+
* `AbsolutelyContinuous alg alg₀`: `alg` is absolutely continuous with respect to `alg₀` (also
23+
denoted `alg ≪ₐ alg₀`) when, in every situation, a set of actions with probability zero under
24+
`alg₀` also has probability zero under `alg`. Intuitively, `alg` never acts in a way that `alg₀`
25+
would never act.
26+
* `density alg alg₀ n`: a density function that allows obtaining the law of the history at time `n`
27+
under `alg` from the law of the history at time `n` under `alg₀` when they are interacting with
28+
the same environment and `alg ≪ₐ alg₀`.
29+
30+
## Main results
31+
32+
* `absolutelyContinuous_map_hist`: the law of the history at time `n` under `alg` is absolutely
33+
continuous with respect to the law of the history at time `n` under `alg₀` when they
34+
are interacting with the same environment and `alg ≪ₐ alg₀`.
35+
* `hasLaw_hist_withDensity`: the law of the history at time `n` under `alg` is the law of the
36+
history at time `n` under `alg₀` with density `alg.density alg₀ n` when they are interacting
37+
with the same environment and `alg ≪ₐ alg₀`.
38+
39+
-/
40+
41+
@[expose] public section
42+
43+
open MeasureTheory ProbabilityTheory Finset
44+
45+
open scoped ENNReal
46+
47+
namespace Learning
48+
49+
variable {𝓐 𝓨 : Type*} [MeasurableSpace 𝓐] [MeasurableSpace 𝓨]
50+
51+
namespace Algorithm
52+
53+
/-- For every time and history, the distribution over actions according to `alg` is absolutely
54+
continuous with respect to the distribution over actions according to `alg₀`. -/
55+
structure AbsolutelyContinuous (alg alg₀ : Algorithm 𝓐 𝓨) : Prop where
56+
p0 : alg.p0 ≪ alg₀.p0
57+
policy n h : alg.policy n h ≪ alg₀.policy n h
58+
59+
@[inherit_doc AbsolutelyContinuous]
60+
scoped notation:50 alg " ≪ₐ " alg₀ => AbsolutelyContinuous alg alg₀
61+
62+
/-- If the algorithm `alg` is absolutely continuous with respect to the algorithm `alg₀` and they
63+
are both interacting with the same environment, then the law of the history at time `n` under `alg`
64+
is the law of the history at time `n` under `alg₀` with density `alg.density alg₀ n`. -/
65+
noncomputable
66+
def density [MeasurableSpace.CountablyGenerated 𝓐] (alg alg₀ : Algorithm 𝓐 𝓨) :
67+
(n : ℕ) → (Iic n → 𝓐 × 𝓨) → ℝ≥0∞
68+
| 0, h => (alg.p0.rnDeriv alg₀.p0 (h ⟨0, by simp⟩).1)
69+
| n + 1, h =>
70+
let p := MeasurableEquiv.IicSuccProd (fun _ ↦ 𝓐 × 𝓨) n h
71+
alg.density alg₀ n p.1 * (alg.policy n).rnDeriv (alg₀.policy n) p.1 p.2.1
72+
73+
@[fun_prop]
74+
lemma measurable_density [MeasurableSpace.CountablyGenerated 𝓐] (alg alg₀ : Algorithm 𝓐 𝓨) (n : ℕ) :
75+
Measurable (alg.density alg₀ n) := by
76+
induction n with
77+
| zero => simp_rw [density]; fun_prop
78+
| succ n ih => simp_rw [density]; fun_prop
79+
80+
end Algorithm
81+
82+
namespace IsAlgEnvSeq
83+
84+
variable {Ω : Type*} [MeasurableSpace Ω]
85+
variable [StandardBorelSpace 𝓐] [Nonempty 𝓐] [StandardBorelSpace 𝓨] [Nonempty 𝓨]
86+
variable {alg : Algorithm 𝓐 𝓨} {env : Environment 𝓐 𝓨}
87+
variable {A : ℕ → Ω → 𝓐} {Y : ℕ → Ω → 𝓨}
88+
variable {P : Measure Ω} [IsFiniteMeasure P]
89+
90+
variable {Ω₀ : Type*} [MeasurableSpace Ω₀]
91+
variable {alg₀ : Algorithm 𝓐 𝓨}
92+
variable {A₀ : ℕ → Ω₀ → 𝓐} {Y₀ : ℕ → Ω₀ → 𝓨}
93+
variable {P₀ : Measure Ω₀} [IsProbabilityMeasure P₀]
94+
95+
open scoped Algorithm
96+
97+
lemma absolutelyContinuous_map_hist (h : IsAlgEnvSeq A Y alg env P)
98+
(h₀ : IsAlgEnvSeq A₀ Y₀ alg₀ env P₀) (hc : alg ≪ₐ alg₀) (n : ℕ) :
99+
P.map (IsAlgEnvSeq.hist A Y n) ≪ P₀.map (IsAlgEnvSeq.hist A₀ Y₀ n) := by
100+
induction n with
101+
| zero =>
102+
rw [h.hasLaw_hist_zero.map_eq, h₀.hasLaw_hist_zero.map_eq]
103+
apply Measure.AbsolutelyContinuous.map _ (by fun_prop)
104+
rw [h.hasLaw_step_zero.map_eq, h₀.hasLaw_step_zero.map_eq]
105+
exact Measure.AbsolutelyContinuous.compProd_left hc.p0 _
106+
| succ n ih =>
107+
rw [(h.hasLaw_hist_succ n).map_eq, (h₀.hasLaw_hist_succ n).map_eq]
108+
apply Measure.AbsolutelyContinuous.map _ (by fun_prop)
109+
rw [Measure.compProd_congr (h.hasCondDistrib_step n).condDistrib_eq,
110+
Measure.compProd_congr (h₀.hasCondDistrib_step n).condDistrib_eq]
111+
apply Measure.AbsolutelyContinuous.compProd ih
112+
filter_upwards with h' using Measure.AbsolutelyContinuous.compProd_left_apply (hc.policy n h') _
113+
114+
lemma hasLaw_hist_withDensity (h : IsAlgEnvSeq A Y alg env P) (h₀ : IsAlgEnvSeq A₀ Y₀ alg₀ env P₀)
115+
(hc : alg ≪ₐ alg₀) (n : ℕ) : HasLaw (IsAlgEnvSeq.hist A Y n)
116+
((P₀.map (IsAlgEnvSeq.hist A₀ Y₀ n)).withDensity (alg.density alg₀ n)) P where
117+
aemeasurable :=
118+
(IsAlgEnvSeq.measurable_hist h.measurable_action h.measurable_feedback n).aemeasurable
119+
map_eq := by
120+
induction n with
121+
| zero =>
122+
rw [h.hasLaw_hist_zero.map_eq, h₀.hasLaw_hist_zero.map_eq, h.hasLaw_step_zero.map_eq,
123+
h₀.hasLaw_step_zero.map_eq]
124+
rw [← Measure.withDensity_rnDeriv_eq _ _ hc.p0,
125+
Measure.compProd_withDensity_left (by fun_prop)]
126+
exact map_equiv_withDensity (by fun_prop)
127+
| succ n ih =>
128+
let ρ h' (ar : 𝓐 × 𝓨) := Kernel.rnDeriv (alg.policy n) (alg₀.policy n) h' ar.1
129+
have hs : stepKernel alg env n = (stepKernel alg₀ env n).withDensity ρ := by
130+
rw [stepKernel, ← Kernel.withDensity_rnDeriv_eq' (hc.policy n)]
131+
exact Kernel.compProd_withDensity_left (Kernel.measurable_rnDeriv _ _)
132+
have : IsMarkovKernel ((stepKernel alg₀ env n).withDensity ρ) := by
133+
rw [← hs]
134+
infer_instance
135+
rw [(h.hasLaw_hist_succ n).map_eq, (h₀.hasLaw_hist_succ n).map_eq,
136+
Measure.compProd_congr (h.hasCondDistrib_step n).condDistrib_eq,
137+
Measure.compProd_congr (h₀.hasCondDistrib_step n).condDistrib_eq, ih, hs,
138+
Measure.compProd_withDensity_withDensity (by fun_prop) (by fun_prop)]
139+
exact map_equiv_withDensity (by fun_prop)
140+
141+
end IsAlgEnvSeq
142+
143+
end Learning

0 commit comments

Comments
 (0)