From 9c99704ef7e8858fcd9bad2f56969d02013d5470 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sun, 26 Apr 2026 16:11:46 +0200 Subject: [PATCH 01/43] alg/env classes and sgd --- LeanMachineLearning.lean | 2 + .../Algorithms/GradientDescent.lean | 171 +++++++++++ .../SequentialLearning/Deterministic.lean | 277 ++++++++++++++---- .../SequentialLearning/EvaluationEnv.lean | 148 ++++++++++ .../SequentialLearning/StationaryEnv.lean | 190 +++++++++--- 5 files changed, 706 insertions(+), 82 deletions(-) create mode 100644 LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean create mode 100644 LeanMachineLearning/SequentialLearning/EvaluationEnv.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 64e000e2..c3dff20a 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -9,6 +9,7 @@ public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.Optimization.Algorithms.GradientDescent public import LeanMachineLearning.Probability.HasCondDistrib public import LeanMachineLearning.Probability.Independence.CondDistrib public import LeanMachineLearning.Probability.Independence.CondIndepFun @@ -22,6 +23,7 @@ public import LeanMachineLearning.SequentialLearning.Algorithm public import LeanMachineLearning.SequentialLearning.Algorithms.RandomSampling public import LeanMachineLearning.SequentialLearning.Algorithms.RoundRobin public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.SequentialLearning.EvaluationEnv public import LeanMachineLearning.SequentialLearning.FiniteActions public import LeanMachineLearning.SequentialLearning.IonescuTulceaSpace public import LeanMachineLearning.SequentialLearning.StationaryEnv diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean new file mode 100644 index 00000000..5e59e3b6 --- /dev/null +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -0,0 +1,171 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.SequentialLearning.EvaluationEnv +public import Mathlib + +/-! +# Online gradient descent + +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset +open scoped Gradient ENNReal NNReal RealInnerProductSpace + +namespace Learning + +variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [CompleteSpace E] [BorelSpace E] + [MeasurableSub₂ E] [SecondCountableTopology E] + {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} {x x₀ : E} + {g : ℕ → E → E} {hg : ∀ n, Measurable (g n)} + {env : Environment E E} + {X G : ℕ → Ω → E} {γ : ℕ → ℝ} + +-- todo: write a process version? with `X : ℕ → Ω → E`, as a `ℕ → Ω → F` +def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := + ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) + +noncomputable def linearizedLoss (f : ℕ → E → ℝ) (x : ℕ → E) : ℕ → E → ℝ := + fun n y ↦ ⟪y, ∇ (f n) (x n)⟫ + +/-- Online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`, +without projection. + +It is an algorithm that chooses actions in `E` and gets feedback in `E` (gradient of the function at +the queried point). -/ +noncomputable +def gradientDescent (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := + detAlgorithm (fun n hist ↦ (hist ⟨n, by grind⟩).1 - γ n • (hist ⟨n, by grind⟩).2) (by fun_prop) x₀ + +lemma action_gradientDescent_ae_eq (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) env P) (n : ℕ) : + X (n + 1) =ᵐ[P] X n - γ n • G n := h_seq.action_detAlgorithm_ae_eq n + +lemma action_gradientDescent_ae_all_eq (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) env P) : + ∀ᵐ ω ∂P, X 0 ω = x₀ ∧ ∀ n, X (n + 1) ω = X n ω - γ n • G n ω := + h_seq.action_detAlgorithm_ae_all_eq + +lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) env P) (n : ℕ) : + X n =ᵐ[P] fun ω ↦ x₀ - ∑ i ∈ Finset.range n, γ i • G i ω := by + filter_upwards [h_seq.action_detAlgorithm_ae_all_eq] with ω ⟨hω0, hω⟩ + induction n with + | zero => simpa + | succ n ih => rw [hω n, Finset.sum_range_succ, ← sub_sub]; congr + +section Convex + +lemma todo' {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x y : E) : + f x - f y ≤ ⟪x - y, ∇ f x⟫ := by + sorry + +lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by + calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := sorry + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by gcongr; exact todo' hf (x _) y + +lemma todo'' (x y g : E) (η : ℝ) : + ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by + sorry + +lemma todo (x y g : ℕ → E) (η : ℕ → ℝ) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * η i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - η i • g i) - y i‖ ^ 2) + + (η i / 2) * ‖g i‖ ^ 2) := by + gcongr with i hi + rw [todo'' (x i) (y i) (g i) (η i)] + +lemma todo''' (x g : ℕ → E) (y : E) + (η : ℝ) (hη : 0 ≤ η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + grw [todo x (fun _ ↦ y) g (fun _ ↦ η) n] + rw [sum_add_distrib, ← mul_sum, ← mul_sum] + gcongr + refine le_of_eq ?_ + simp_rw [← hx] + sorry + +lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 ≤ η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + grw [todo''' x g y η hη hx n] + gcongr + exact sub_le_self _ (sq_nonneg _) + +end Convex + +section Stochastic + +variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] + +-- use `obliviousEnv gradKernel` as the environment for stochastic gradient descent +example (gradKernel : ℕ → Kernel E E) [∀ n, IsMarkovKernel (gradKernel n)] : + Environment E E := obliviousEnv gradKernel + +-- use the deterministic equality wrt any sequence +lemma todo1 {η : ℝ} (hη : 0 ≤ η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (y : E) (n : ℕ) : + ∀ᵐ ω ∂P, ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫ ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 := by + filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω + refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ + congr + exact hω.1 + +lemma sfdsf {η : ℝ} (hη : 0 ≤ η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (y : E) (n : ℕ) : + P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by + sorry + +lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) + {η : ℝ} (hη : 0 ≤ η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (y : E) (n : ℕ) : + P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by + rw [sfdsf hη h h_unbiased y n] + gcongr + · refine Integrable.sub ?_ (integrable_const _) + sorry + · simp only [inner_sub_left] + refine Integrable.sub ?_ ?_ + · sorry + · sorry + · exact fun ω ↦ todo' (hf n) (X n ω) y + +lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) + {η : ℝ} (hη : 0 ≤ η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (y : E) (n : ℕ) : + P[fun ω ↦ ∑ i ∈ Finset.range n, f n (X n ω) - f n y] ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + sorry + +lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + {η : ℝ} (hη : 0 ≤ η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) + (y : E) (n : ℕ) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X n ω) - f y] ≤ + (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + + (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + sorry + +end Stochastic + +end Learning diff --git a/LeanMachineLearning/SequentialLearning/Deterministic.lean b/LeanMachineLearning/SequentialLearning/Deterministic.lean index 3f84fda8..5c3b2e5e 100644 --- a/LeanMachineLearning/SequentialLearning/Deterministic.lean +++ b/LeanMachineLearning/SequentialLearning/Deterministic.lean @@ -34,23 +34,219 @@ open MeasureTheory ProbabilityTheory Filter Real Finset open scoped ENNReal NNReal +/-- Two deterministic kernels are equal if and only if their underlying functions are equal. -/ +@[simp] +lemma ProbabilityTheory.Kernel.deterministic_eq_deterministic_iff + {α β : Type*} {mα : MeasurableSpace α} {mβ : MeasurableSpace β} + [MeasurableSpace.SeparatesPoints β] + {f g : α → β} {hf : Measurable f} {hg : Measurable g} : + Kernel.deterministic f hf = Kernel.deterministic g hg ↔ f = g := by + simp [Kernel.ext_iff, Kernel.deterministic_apply, dirac_eq_dirac_iff, funext_iff] + namespace Learning variable {α R : Type*} {mα : MeasurableSpace α} {mR : MeasurableSpace R} +/-- An algorithm is deterministic if its initial action and subsequent actions are determined by +measurable functions (and not possibly random kernels). -/ +class IsDeterministicAlg (alg : Algorithm α R) : Prop where + exists_action0 : ∃ action0, alg.p0 = Measure.dirac action0 + exists_nextAction n : ∃ (nextAction : (Iic n → α × R) → α) (h_meas : Measurable nextAction), + alg.policy n = Kernel.deterministic nextAction h_meas + +/-- The initial action of a deterministic algorithm. -/ +noncomputable +def actionZero (alg : Algorithm α R) [h_det : IsDeterministicAlg alg] : α := + h_det.exists_action0.choose + +/-- The next action of a deterministic algorithm after step `n`. -/ +noncomputable +def nextAction (alg : Algorithm α R) [h_det : IsDeterministicAlg alg] (n : ℕ) : + (Iic n → α × R) → α := + (h_det.exists_nextAction n).choose + +@[fun_prop] +lemma measurable_nextAction (alg : Algorithm α R) [IsDeterministicAlg alg] (n : ℕ) : + Measurable (nextAction alg n) := + (IsDeterministicAlg.exists_nextAction n).choose_spec.choose + +lemma p0_eq_dirac (alg : Algorithm α R) [h_det : IsDeterministicAlg alg] : + alg.p0 = Measure.dirac (actionZero alg) := + h_det.exists_action0.choose_spec + +lemma policy_eq_deterministic (alg : Algorithm α R) [h_det : IsDeterministicAlg alg] (n : ℕ) : + alg.policy n = Kernel.deterministic (nextAction alg n) (measurable_nextAction alg n) := + (IsDeterministicAlg.exists_nextAction n).choose_spec.choose_spec + +namespace IsDeterministicAlg + +variable {Ω : Type*} {mΩ : MeasurableSpace Ω} + [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] + {alg : Algorithm α R} {env : Environment α R} {P : Measure Ω} [IsFiniteMeasure P] + {A : ℕ → Ω → α} {R' : ℕ → Ω → R} {n N : ℕ} + +lemma hasLaw_action_zero_of_IsAlgEnvSeqUntil [h_det : IsDeterministicAlg alg] + (h : IsAlgEnvSeqUntil A R' alg env P N) : + HasLaw (A 0) (Measure.dirac (actionZero alg)) P where + aemeasurable := have hA := h.measurable_A; by fun_prop + map_eq := (h.hasLaw_action_zero).map_eq.trans (p0_eq_dirac alg) + +lemma action_zero_of_IsAlgEnvSeqUntil [h_det : IsDeterministicAlg alg] + (h : IsAlgEnvSeqUntil A R' alg env P N) : + A 0 =ᵐ[P] fun _ ↦ actionZero alg := by + have h_eq : ∀ᵐ x ∂(P.map (A 0)), x = actionZero alg := by + simp [(hasLaw_action_zero_of_IsAlgEnvSeqUntil h).map_eq] + have hA := h.measurable_A + exact ae_of_ae_map (by fun_prop) h_eq + +lemma action_ae_eq_of_IsAlgEnvSeqUntil [h_det : IsDeterministicAlg alg] + (h : IsAlgEnvSeqUntil A R' alg env P N) (hn : n < N) : + A (n + 1) =ᵐ[P] fun ω ↦ nextAction alg n (IsAlgEnvSeq.hist A R' n ω) := by + have hA := h.measurable_A + have hR' := h.measurable_R + have h_eq := (h.hasCondDistrib_action n hn).condDistrib_eq + rw [policy_eq_deterministic alg n] at h_eq + refine ae_eq_of_condDistrib_eq_deterministic (by fun_prop : Measurable (nextAction alg n)) + (by fun_prop) (by fun_prop) h_eq + +lemma hasLaw_action_zero [h_det : IsDeterministicAlg alg] (h : IsAlgEnvSeq A R' alg env P) : + HasLaw (A 0) (Measure.dirac (actionZero alg)) P where + aemeasurable := have hA := h.measurable_A; by fun_prop + map_eq := (h.hasLaw_action_zero).map_eq.trans (p0_eq_dirac alg) + +lemma action_zero_ae_eq [h_det : IsDeterministicAlg alg] (h : IsAlgEnvSeq A R' alg env P) : + A 0 =ᵐ[P] fun _ ↦ actionZero alg := + action_zero_of_IsAlgEnvSeqUntil (h.isAlgEnvSeqUntil 0) + +lemma action_ae_eq [h_det : IsDeterministicAlg alg] (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : + A (n + 1) =ᵐ[P] fun ω ↦ nextAction alg n (IsAlgEnvSeq.hist A R' n ω) := + action_ae_eq_of_IsAlgEnvSeqUntil (h.isAlgEnvSeqUntil (n + 1)) (by simp) + +lemma action_ae_all_eq [h_det : IsDeterministicAlg alg] (h : IsAlgEnvSeq A R' alg env P) : + ∀ᵐ ω ∂P, A 0 ω = actionZero alg ∧ + ∀ n, A (n + 1) ω = nextAction alg n (IsAlgEnvSeq.hist A R' n ω) := by + rw [eventually_and, ae_all_iff] + exact ⟨action_zero_ae_eq h, action_ae_eq h⟩ + +end IsDeterministicAlg + +/-- An environment is deterministic if its initial feedbacks are determined by +measurable functions (and not possibly random kernels). -/ +class IsDeterministicEnv (env : Environment α R) : Prop where + exists_f0 : ∃ (f0 : α → R) (hf0 : Measurable f0), env.ν0 = Kernel.deterministic f0 hf0 + exists_f : ∀ n, ∃ (f : ((Iic n → α × R) × α) → R) (hf : Measurable f), + env.feedback n = Kernel.deterministic f hf + +/-- The initial feedback function of a deterministic environment. -/ +noncomputable +def feedbackFunZero (env : Environment α R) [h_det : IsDeterministicEnv env] : α → R := + h_det.exists_f0.choose + +@[fun_prop] +lemma measurable_feedbackFunZero (env : Environment α R) [IsDeterministicEnv env] : + Measurable (feedbackFunZero env) := + (IsDeterministicEnv.exists_f0).choose_spec.choose + +lemma ν0_eq_deterministic (env : Environment α R) [IsDeterministicEnv env] : + env.ν0 = Kernel.deterministic (feedbackFunZero env) (measurable_feedbackFunZero env) := + (IsDeterministicEnv.exists_f0).choose_spec.choose_spec + +/-- The feedback function of a deterministic environment at step `n`. -/ +noncomputable +def feedbackFun (env : Environment α R) [h_det : IsDeterministicEnv env] (n : ℕ) : + ((Iic n → α × R) × α) → R := + (h_det.exists_f n).choose + +@[fun_prop] +lemma measurable_feedbackFun (env : Environment α R) [IsDeterministicEnv env] (n : ℕ) : + Measurable (feedbackFun env n) := + (IsDeterministicEnv.exists_f n).choose_spec.choose + +lemma feedback_eq_deterministic (env : Environment α R) [IsDeterministicEnv env] (n : ℕ) : + env.feedback n = Kernel.deterministic (feedbackFun env n) (measurable_feedbackFun env n) := + (IsDeterministicEnv.exists_f n).choose_spec.choose_spec + +namespace IsDeterministicEnv + +variable {Ω : Type*} {mΩ : MeasurableSpace Ω} + [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] + {alg : Algorithm α R} {env : Environment α R} {P : Measure Ω} [IsFiniteMeasure P] + {A : ℕ → Ω → α} {R' : ℕ → Ω → R} + {f : (n : ℕ) → ((Iic n → α × R) × α) → R} {hf : ∀ n, Measurable (f n)} + {f0 : α → R} {hf0 : Measurable f0} + +lemma hasCondDistrib_reward_zero [h_det : IsDeterministicEnv env] + (h : IsAlgEnvSeq A R' alg env P) : + HasCondDistrib (R' 0) (A 0) + (Kernel.deterministic (feedbackFunZero env) (measurable_feedbackFunZero env)) P := by + rw [← ν0_eq_deterministic] + exact h.hasCondDistrib_reward_zero + +lemma hasCondDistrib_reward [h_det : IsDeterministicEnv env] + (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : + HasCondDistrib (R' (n + 1)) (fun ω ↦ (IsAlgEnvSeq.hist A R' n ω, A (n + 1) ω)) + (Kernel.deterministic (feedbackFun env n) (measurable_feedbackFun env n)) P := by + rw [← feedback_eq_deterministic] + exact h.hasCondDistrib_reward n + +end IsDeterministicEnv + +variable {nextA : (n : ℕ) → (Iic n → α × R) → α} {h_next : ∀ n, Measurable (nextA n)} + {action0 : α} {env : Environment α R} + {f0 : α → R} {hf0 : Measurable f0} + {f : (n : ℕ) → ((Iic n → α × R) × α) → R} {hf : ∀ n, Measurable (f n)} + /-- A deterministic algorithm, which chooses the action given by the function `nextAction`. -/ @[simps] noncomputable -- ANCHOR: detAlgorithm -def detAlgorithm (nextAction : (n : ℕ) → (Iic n → α × R) → α) - (h_next : ∀ n, Measurable (nextAction n)) (action0 : α) : +def detAlgorithm (nextA : (n : ℕ) → (Iic n → α × R) → α) + (h_next : ∀ n, Measurable (nextA n)) (action0 : α) : Algorithm α R where - policy n := Kernel.deterministic (nextAction n) (h_next n) + policy n := Kernel.deterministic (nextA n) (h_next n) p0 := Measure.dirac action0 -- ANCHOR_END: detAlgorithm -variable {nextAction : (n : ℕ) → (Iic n → α × R) → α} {h_next : ∀ n, Measurable (nextAction n)} - {action0 : α} {env : Environment α R} +instance : IsDeterministicAlg (detAlgorithm nextA h_next action0) where + exists_action0 := ⟨action0, rfl⟩ + exists_nextAction n := ⟨nextA n, h_next n, rfl⟩ + +@[simp] +lemma actionZero_detAlgorithm [MeasurableSpace.SeparatesPoints α] : + actionZero (detAlgorithm nextA h_next action0) = action0 := by + have h_eq := p0_eq_dirac (detAlgorithm nextA h_next action0) + simp only [detAlgorithm] at h_eq + rw [dirac_eq_dirac_iff] at h_eq + exact h_eq.symm + +@[simp] +lemma nextAction_detAlgorithm [MeasurableSpace.SeparatesPoints α] (n : ℕ) : + nextAction (detAlgorithm nextA h_next action0) n = nextA n := by + have h_eq := policy_eq_deterministic (detAlgorithm nextA h_next action0) n + simpa [detAlgorithm] using h_eq.symm + +/-- A deterministic environment, where the feedback is given by evaluating +fixed measurable functions. -/ +noncomputable def detEnvironment + (f0 : α → R) (hf0 : Measurable f0) + (f : (n : ℕ) → ((Iic n → α × R) × α) → R) (hf : ∀ n, Measurable (f n)) : + Environment α R where + feedback n := (Kernel.deterministic (f n) (hf n)) + ν0 := Kernel.deterministic f0 hf0 + +instance : IsDeterministicEnv (detEnvironment f0 hf0 f hf) where + exists_f0 := ⟨f0, hf0, rfl⟩ + exists_f n := ⟨f n, hf n, rfl⟩ + +@[simp] +lemma feedbackFunZero_detEnvironment [MeasurableSpace.SeparatesPoints R] : + feedbackFunZero (detEnvironment f0 hf0 f hf) = f0 := by + simpa [detEnvironment] using (ν0_eq_deterministic (detEnvironment f0 hf0 f hf)).symm + +@[simp] +lemma feedbackFun_detEnvironment [MeasurableSpace.SeparatesPoints R] (n : ℕ) : + feedbackFun (detEnvironment f0 hf0 f hf) n = f n := by + simpa [detEnvironment] using (feedback_eq_deterministic (detEnvironment f0 hf0 f hf) n).symm namespace IsAlgEnvSeq @@ -59,34 +255,25 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {ν : Kernel α R} [IsMarkovKernel ν] {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} -lemma HasLaw_action_zero_detAlgorithm - (h : IsAlgEnvSeq A R' (detAlgorithm nextAction h_next action0) env P) : - HasLaw (A 0) (Measure.dirac action0) P where - aemeasurable := have hA := h.measurable_A; by fun_prop - map_eq := (hasLaw_action_zero h).map_eq +lemma hasLaw_action_zero_detAlgorithm + (h : IsAlgEnvSeq A R' (detAlgorithm nextA h_next action0) env P) : + HasLaw (A 0) (Measure.dirac action0) P := by + simpa using IsDeterministicAlg.hasLaw_action_zero h lemma action_zero_detAlgorithm - (h : IsAlgEnvSeq A R' (detAlgorithm nextAction h_next action0) env P) : - A 0 =ᵐ[P] fun _ ↦ action0 := by - have h_eq : ∀ᵐ x ∂(P.map (A 0)), x = action0 := by - rw [(hasLaw_action_zero h).map_eq] - simp [detAlgorithm] - have hA := h.measurable_A - exact ae_of_ae_map (by fun_prop) h_eq + (h : IsAlgEnvSeq A R' (detAlgorithm nextA h_next action0) env P) : + A 0 =ᵐ[P] fun _ ↦ action0 := + (IsDeterministicAlg.action_zero_ae_eq h).trans (by simp) lemma action_detAlgorithm_ae_eq - (h : IsAlgEnvSeq A R' (detAlgorithm nextAction h_next action0) env P) (n : ℕ) : - A (n + 1) =ᵐ[P] fun ω ↦ nextAction n (hist A R' n ω) := by - have hA := h.measurable_A - have hR' := h.measurable_R - exact ae_eq_of_condDistrib_eq_deterministic (by fun_prop) (by fun_prop) (by fun_prop) - (h.hasCondDistrib_action n).condDistrib_eq + (h : IsAlgEnvSeq A R' (detAlgorithm nextA h_next action0) env P) (n : ℕ) : + A (n + 1) =ᵐ[P] fun ω ↦ nextA n (hist A R' n ω) := + (IsDeterministicAlg.action_ae_eq h n).trans (by simp) lemma action_detAlgorithm_ae_all_eq - (h : IsAlgEnvSeq A R' (detAlgorithm nextAction h_next action0) env P) : - ∀ᵐ ω ∂P, A 0 ω = action0 ∧ ∀ n, A (n + 1) ω = nextAction n (hist A R' n ω) := by - rw [eventually_and, ae_all_iff] - exact ⟨action_zero_detAlgorithm h, action_detAlgorithm_ae_eq h⟩ + (h : IsAlgEnvSeq A R' (detAlgorithm nextA h_next action0) env P) : + ∀ᵐ ω ∂P, A 0 ω = action0 ∧ ∀ n, A (n + 1) ω = nextA n (hist A R' n ω) := by + filter_upwards [IsDeterministicAlg.action_ae_all_eq h] with ω hω using by simp [hω] end IsAlgEnvSeq @@ -97,34 +284,26 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {ν : Kernel α R} [IsMarkovKernel ν] {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} {N n : ℕ} -lemma HasLaw_action_zero_detAlgorithm - (h : IsAlgEnvSeqUntil A R' (detAlgorithm nextAction h_next action0) env P N) : - HasLaw (A 0) (Measure.dirac action0) P where - aemeasurable := have hA := h.measurable_A; by fun_prop - map_eq := (hasLaw_action_zero h).map_eq +lemma hasLaw_action_zero_detAlgorithm + (h : IsAlgEnvSeqUntil A R' (detAlgorithm nextA h_next action0) env P N) : + HasLaw (A 0) (Measure.dirac action0) P := by + simpa using IsDeterministicAlg.hasLaw_action_zero_of_IsAlgEnvSeqUntil h lemma action_zero_detAlgorithm - (h : IsAlgEnvSeqUntil A R' (detAlgorithm nextAction h_next action0) env P N) : - A 0 =ᵐ[P] fun _ ↦ action0 := by - have h_eq : ∀ᵐ x ∂(P.map (A 0)), x = action0 := by - rw [(hasLaw_action_zero h).map_eq] - simp [detAlgorithm] - have hA := h.measurable_A - exact ae_of_ae_map (by fun_prop) h_eq + (h : IsAlgEnvSeqUntil A R' (detAlgorithm nextA h_next action0) env P N) : + A 0 =ᵐ[P] fun _ ↦ action0 := + (IsDeterministicAlg.action_zero_of_IsAlgEnvSeqUntil h).trans (by simp) lemma action_detAlgorithm_ae_eq - (h : IsAlgEnvSeqUntil A R' (detAlgorithm nextAction h_next action0) env P N) (hn : n < N) : - A (n + 1) =ᵐ[P] fun ω ↦ nextAction n (IsAlgEnvSeq.hist A R' n ω) := by - have hA := h.measurable_A - have hR' := h.measurable_R - exact ae_eq_of_condDistrib_eq_deterministic (by fun_prop) (by fun_prop) (by fun_prop) - (h.hasCondDistrib_action n hn).condDistrib_eq + (h : IsAlgEnvSeqUntil A R' (detAlgorithm nextA h_next action0) env P N) (hn : n < N) : + A (n + 1) =ᵐ[P] fun ω ↦ nextA n (IsAlgEnvSeq.hist A R' n ω) := + (IsDeterministicAlg.action_ae_eq_of_IsAlgEnvSeqUntil h hn).trans (by simp) end IsAlgEnvSeqUntil namespace IT -local notation "𝔓" => trajMeasure (detAlgorithm nextAction h_next action0) env +local notation "𝔓" => trajMeasure (detAlgorithm nextA h_next action0) env lemma HasLaw_action_zero_detAlgorithm : HasLaw (IT.action 0) (Measure.dirac action0) 𝔓 where map_eq := (IT.hasLaw_action_zero _ _).map_eq @@ -137,13 +316,13 @@ lemma action_zero_detAlgorithm [MeasurableSingletonClass α] : exact ae_of_ae_map (by fun_prop) h_eq lemma action_detAlgorithm_ae_eq [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] - [Nonempty R] (n : ℕ) : IT.action (n + 1) =ᵐ[𝔓] fun h ↦ nextAction n (IT.hist n h) := + [Nonempty R] (n : ℕ) : IT.action (n + 1) =ᵐ[𝔓] fun h ↦ nextA n (IT.hist n h) := ae_eq_of_condDistrib_eq_deterministic (by fun_prop) (by fun_prop) (by fun_prop) - (IT.condDistrib_action (detAlgorithm nextAction h_next action0) env n) + (IT.condDistrib_action (detAlgorithm nextA h_next action0) env n) lemma action_detAlgorithm_ae_all_eq [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] : - ∀ᵐ h ∂𝔓, IT.action 0 h = action0 ∧ ∀ n, IT.action (n + 1) h = nextAction n (IT.hist n h) := by + ∀ᵐ h ∂𝔓, IT.action 0 h = action0 ∧ ∀ n, IT.action (n + 1) h = nextA n (IT.hist n h) := by rw [eventually_and, ae_all_iff] exact ⟨action_zero_detAlgorithm, action_detAlgorithm_ae_eq⟩ diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean new file mode 100644 index 00000000..8b889c7e --- /dev/null +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -0,0 +1,148 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ +module + +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.SequentialLearning.StationaryEnv +public import LeanMachineLearning.Probability.Independence.CondDistrib + +/-! +# Function evaluation environments + +A stationary environment where the reward is given by evaluating a fixed measurable function `f` at +the chosen action. + +## Main definitions + +* `evalEnv hf`: A stationary environment where the reward is given by a deterministic kernel that + evaluates a fixed measurable function at the chosen action. + +## Main statements + +* `reward_ae_eq_evals_actions`: For almost all `ω`, the reward at time `n` is equal to `f` + evaluated at the action taken at time `n`. +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory + +@[simp] +lemma ProbabilityTheory.Kernel.prodMkLeft_deterministic {α β γ : Type*} + {mα : MeasurableSpace α} {mβ : MeasurableSpace β} {mγ : MeasurableSpace γ} + {f : α → β} (hf : Measurable f) : + (Kernel.deterministic f hf).prodMkLeft γ = + Kernel.deterministic (fun p ↦ f p.2) (by fun_prop) := by + ext + simp [Kernel.deterministic_apply] + +namespace Learning + +variable {α R : Type*} {mα : MeasurableSpace α} {mR : MeasurableSpace R} + {g : ℕ → α → R} {hg : ∀ n, Measurable (g n)} + {f : α → R} {hf : Measurable f} + +/-- The evaluation environment where the reward is given by evaluating a fixed measurable function +`f` at the chosen action. -/ +noncomputable def onlineEvalEnv (g : ℕ → α → R) (hg : ∀ n, Measurable (g n)) := + obliviousEnv (fun n ↦ Kernel.deterministic (g n) (hg n)) + +instance : IsObliviousEnv (onlineEvalEnv g hg) := + ⟨⟨fun n ↦ Kernel.deterministic (g n) (hg n), fun _ ↦ inferInstance, rfl, fun _ ↦ rfl⟩⟩ + +instance : IsDeterministicEnv (onlineEvalEnv g hg) where + exists_f0 := ⟨g 0, hg 0, rfl⟩ + exists_f n := ⟨fun p ↦ g (n + 1) p.2, by fun_prop, rfl⟩ + +@[simp] +lemma feedbackCondAction_onlineEvalEnv (n : ℕ) : + feedbackCondAction (onlineEvalEnv g hg) n = Kernel.deterministic (g n) (hg n) := by + simp [onlineEvalEnv] + +@[simp] +lemma feedbackFunZero_onlineEvalEnv [MeasurableSpace.SeparatesPoints R] : + feedbackFunZero (onlineEvalEnv g hg) = g 0 := by + have h_eq := ν0_eq_deterministic (onlineEvalEnv g hg) + simpa only [onlineEvalEnv, ν0_obliviousEnv, Kernel.prodMkLeft_deterministic, + Kernel.deterministic_eq_deterministic_iff] using h_eq.symm + +@[simp] +lemma feedbackFun_onlineEvalEnv [MeasurableSpace.SeparatesPoints R] (n : ℕ) : + feedbackFun (onlineEvalEnv g hg) n = fun p ↦ g (n + 1) p.2 := by + have h_eq := feedback_eq_deterministic (onlineEvalEnv g hg) n + simpa only [onlineEvalEnv, feedback_obliviousEnv, Kernel.prodMkLeft_deterministic, + Kernel.deterministic_eq_deterministic_iff] using h_eq.symm + +namespace OnlineEvalEnv + +variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] + {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} + {g : ℕ → α → R} {hg : ∀ n, Measurable (g n)} + {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} + +lemma hascondDistrib_reward (h : IsAlgEnvSeq A R' alg (onlineEvalEnv g hg) P) (n : ℕ) : + HasCondDistrib (R' n) (A n) (Kernel.deterministic (g n) (hg n)) P := by + simpa using IsObliviousEnv.hasCondDistrib_reward h n + +lemma reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (onlineEvalEnv g hg) P) (n : ℕ) : + R' n =ᵐ[P] g n ∘ A n := + ae_eq_of_condDistrib_eq_deterministic (hg n) (h.measurable_A n).aemeasurable + (h.measurable_R n).aemeasurable (hascondDistrib_reward h n).condDistrib_eq + +lemma forall_reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (onlineEvalEnv g hg) P) : + ∀ᵐ ω ∂P, ∀ n, R' n ω = g n (A n ω) := by + rw [ae_all_iff] + intro n + exact reward_ae_eq_eval_action h n + +end OnlineEvalEnv + +/-- The evaluation environment where the reward is given by evaluating a fixed measurable function +`f` at the chosen action. -/ +noncomputable def evalEnv (f : α → R) (hf : Measurable f) := onlineEvalEnv (fun _ ↦ f) (fun _ ↦ hf) + +instance : IsObliviousEnv (evalEnv f hf) := by unfold evalEnv; infer_instance + +instance : IsDeterministicEnv (evalEnv f hf) := by unfold evalEnv; infer_instance + +@[simp] +lemma feedbackCondAction_evalEnv (n : ℕ) : + feedbackCondAction (evalEnv f hf) n = Kernel.deterministic f hf := by simp [evalEnv] + +@[simp] +lemma feedbackFunZero_evalEnv [MeasurableSpace.SeparatesPoints R] : + feedbackFunZero (evalEnv f hf) = f := by simp [evalEnv] + +@[simp] +lemma feedbackFun_evalEnv [MeasurableSpace.SeparatesPoints R] (n : ℕ) : + feedbackFun (evalEnv f hf) n = fun p ↦ f p.2 := by simp [evalEnv] + +namespace EvalEnv + +variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] + {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {f : α → R} {hf : Measurable f} + {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} + +lemma hascondDistrib_reward (h : IsAlgEnvSeq A R' alg (evalEnv f hf) P) (n : ℕ) : + HasCondDistrib (R' n) (A n) (Kernel.deterministic f hf) P := by + simpa using IsObliviousEnv.hasCondDistrib_reward h n + +lemma reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv f hf) P) (n : ℕ) : + R' n =ᵐ[P] f ∘ A n := OnlineEvalEnv.reward_ae_eq_eval_action h n + +lemma forall_reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv f hf) P) : + ∀ᵐ ω ∂P, ∀ n, R' n ω = f (A n ω) := OnlineEvalEnv.forall_reward_ae_eq_eval_action h + +open Finset in +lemma reward_ae_eq_eval_action_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (evalEnv f hf) P) {n : ℕ} + (g : (Iic n → R) → β) : + ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by + filter_upwards [forall_reward_ae_eq_eval_action h] with ω hω + simp_rw [hω] + +end EvalEnv + +end Learning diff --git a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean index 88e841cb..5555ec0a 100644 --- a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean +++ b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean @@ -27,44 +27,64 @@ open MeasureTheory ProbabilityTheory Filter Real Finset open scoped ENNReal NNReal +@[simp] +lemma ProbabilityTheory.Kernel.prodMkLeft_eq_prodMkLeft {α β γ : Type*} [h_nonempty : Nonempty γ] + {mα : MeasurableSpace α} {mβ : MeasurableSpace β} {mγ : MeasurableSpace γ} + (κ ν : Kernel α β) : + κ.prodMkLeft γ = ν.prodMkLeft γ ↔ κ = ν := by + simp only [Kernel.ext_iff, Kernel.prodMkLeft_apply, Prod.forall] + exact ⟨fun h b ↦ h h_nonempty.some b, fun h _ b ↦ h b⟩ + namespace Learning variable {α R : Type*} {mα : MeasurableSpace α} {mR : MeasurableSpace R} -/-- A stationary environment, in which the distribution of the next reward depends only on the last -action. -/ -@[simps] --- ANCHOR: stationaryEnv -def stationaryEnv (ν : Kernel α R) [IsMarkovKernel ν] : Environment α R where - feedback _ := ν.prodMkLeft _ - ν0 := ν --- ANCHOR_END: stationaryEnv +/-- An environment is oblivious if the distribution of the next feedback depends only on +the last action and not on the past history. -/ +class IsObliviousEnv (env : Environment α R) : Prop where + exists_eq_prodMkLeft : ∃ ν : ℕ → Kernel α R, (∀ n, IsMarkovKernel (ν n)) ∧ + (env.ν0 = ν 0) ∧ (∀ n, env.feedback n = (ν (n + 1)).prodMkLeft _) + +/-- The kernel representing the conditional distribution of the feedback given the action +at time `n` in an oblivious environment. -/ +noncomputable +def feedbackCondAction (env : Environment α R) [h_obl : IsObliviousEnv env] (n : ℕ) : Kernel α R := + h_obl.exists_eq_prodMkLeft.choose n + +instance (env : Environment α R) [IsObliviousEnv env] (n : ℕ) : + IsMarkovKernel (feedbackCondAction env n) := + IsObliviousEnv.exists_eq_prodMkLeft.choose_spec.1 n + +lemma ν0_eq_feedbackCondAction (env : Environment α R) [IsObliviousEnv env] : + env.ν0 = feedbackCondAction env 0 := + IsObliviousEnv.exists_eq_prodMkLeft.choose_spec.2.1 + +lemma feedback_eq_feedbackCondAction (env : Environment α R) [IsObliviousEnv env] (n : ℕ) : + env.feedback n = (feedbackCondAction env (n + 1)).prodMkLeft _ := + IsObliviousEnv.exists_eq_prodMkLeft.choose_spec.2.2 n + +namespace IsObliviousEnv variable {Ω : Type*} {mΩ : MeasurableSpace Ω} [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] - {alg : Algorithm α R} {ν : Kernel α R} [IsMarkovKernel ν] - {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} + {alg : Algorithm α R} {env : Environment α R} {P : Measure Ω} [IsFiniteMeasure P] + {A : ℕ → Ω → α} {R' : ℕ → Ω → R} {n N : ℕ} + {ν : ℕ → Kernel α R} [∀ n, IsMarkovKernel (ν n)] -namespace IsAlgEnvSeq - -/-- The conditional distribution of the reward at time `n` given the action at time `n` is `ν`. -/ -lemma condDistrib_reward_stationaryEnv - (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : - condDistrib (R' n) (A n) P =ᵐ[P.map (A n)] ν := by +lemma hasCondDistrib_reward [IsObliviousEnv env] (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : + HasCondDistrib (R' n) (A n) (feedbackCondAction env n) P := by have hA := h.measurable_A have hR' := h.measurable_R cases n with - | zero => - rw [condDistrib_ae_eq_iff_measure_eq_compProd _ (by fun_prop)] - change P.map (step A R' 0) = P.map (A 0) ⊗ₘ ν - rw [(hasLaw_action_zero h).map_eq, (hasLaw_step_zero h).map_eq, stationaryEnv_ν0] + | zero => rw [← ν0_eq_feedbackCondAction]; exact h.hasCondDistrib_reward_zero | succ n => + refine ⟨by fun_prop, by fun_prop, ?_⟩ have h_eq := (h.hasCondDistrib_reward n).condDistrib_eq rw [condDistrib_ae_eq_iff_measure_eq_compProd _ (by fun_prop)] at h_eq ⊢ have : P.map (A (n + 1)) = - (P.map (fun x ↦ (hist A R' n x, A (n + 1) x))).snd := by + (P.map (fun x ↦ (IsAlgEnvSeq.hist A R' n x, A (n + 1) x))).snd := by rw [Measure.snd_map_prodMk (by fun_prop)] - simp only [stationaryEnv_feedback] at h_eq + simp only [feedback_eq_feedbackCondAction] at h_eq rw [this, ← Measure.snd_prodAssoc_compProd_prodMkLeft, ← h_eq, Measure.snd_map_prodMk (by fun_prop), Measure.map_map (by fun_prop) (by fun_prop)] congr @@ -72,29 +92,133 @@ lemma condDistrib_reward_stationaryEnv /-- The reward at time `n + 1` is conditionally independent of the history up to time `n` given the action at time `n + 1`. -/ lemma condIndepFun_reward_hist_action [StandardBorelSpace Ω] - (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : - R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A _ ; P] hist A R' n := by + [IsObliviousEnv env] (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : + R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A _ ; P] IsAlgEnvSeq.hist A R' n := by have hA := h.measurable_A have hR' := h.measurable_R - exact condIndepFun_of_exists_condDistrib_prod_ae_eq_prodMkLeft - (by fun_prop) (by fun_prop) (by fun_prop) (h.hasCondDistrib_reward n).condDistrib_eq + refine condIndepFun_of_exists_condDistrib_prod_ae_eq_prodMkLeft + (η := feedbackCondAction env (n + 1)) + (by fun_prop) (by fun_prop) (by fun_prop) ?_ + refine HasCondDistrib.condDistrib_eq ?_ + rw [← feedback_eq_feedbackCondAction] + exact h.hasCondDistrib_reward n lemma condIndepFun_reward_hist_action_action [StandardBorelSpace Ω] - (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : + [IsObliviousEnv env] (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A (n + 1); P] - (fun ω ↦ (hist A R' n ω, A (n + 1) ω)) := by - have h_indep : R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A (n + 1); P] hist A R' n := by - convert h.condIndepFun_reward_hist_action n + (fun ω ↦ (IsAlgEnvSeq.hist A R' n ω, A (n + 1) ω)) := by + have h_indep : R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A (n + 1); P] IsAlgEnvSeq.hist A R' n := + condIndepFun_reward_hist_action h n have hA := h.measurable_A have hR' := h.measurable_R exact h_indep.prod_right (by fun_prop) (by fun_prop) (by fun_prop) lemma condIndepFun_reward_hist_action_action' [StandardBorelSpace Ω] - (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) (hn : n ≠ 0) : - R' n ⟂ᵢ[A n, h.measurable_A n; P] (fun ω ↦ (hist A R' (n - 1) ω, A n ω)) := by - have := h.condIndepFun_reward_hist_action_action (n - 1) + [IsObliviousEnv env] (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) (hn : n ≠ 0) : + R' n ⟂ᵢ[A n, h.measurable_A n; P] (fun ω ↦ (IsAlgEnvSeq.hist A R' (n - 1) ω, A n ω)) := by + have := condIndepFun_reward_hist_action_action h (n - 1) grind +end IsObliviousEnv + +/-- An oblivious environment, in which the distribution of the next reward depends only on the last +action, but in a possibly time-dependent manner. -/ +@[simps] +def obliviousEnv (ν : ℕ → Kernel α R) [∀ n, IsMarkovKernel (ν n)] : Environment α R where + feedback n := (ν (n + 1)).prodMkLeft _ + ν0 := ν 0 + +@[simp] +lemma feedback_obliviousEnv (ν : ℕ → Kernel α R) [∀ n, IsMarkovKernel (ν n)] (n : ℕ) : + (obliviousEnv ν).feedback n = (ν (n + 1)).prodMkLeft _ := by simp [obliviousEnv] + +@[simp] +lemma ν0_obliviousEnv (ν : ℕ → Kernel α R) [∀ n, IsMarkovKernel (ν n)] : + (obliviousEnv ν).ν0 = ν 0 := by simp [obliviousEnv] + +instance (ν : ℕ → Kernel α R) [∀ n, IsMarkovKernel (ν n)] : + IsObliviousEnv (obliviousEnv ν) where + exists_eq_prodMkLeft := ⟨fun n ↦ ν n, inferInstance,rfl, fun _ ↦ rfl⟩ + +@[simp] +lemma feedbackCondAction_obliviousEnv (ν : ℕ → Kernel α R) [hν : ∀ n, IsMarkovKernel (ν n)] + (n : ℕ) : + feedbackCondAction (obliviousEnv ν) n = ν n := by + rcases isEmpty_or_nonempty α with hα | hα + · ext a : 1 + exact hα.elim a + rcases isEmpty_or_nonempty R with hR | hR + · refine absurd (hν 0) ?_ + simp only [Subsingleton.eq_zero ν, Pi.zero_apply] + exact Kernel.not_isMarkovKernel_zero + have : Nonempty (Iic n → α × R) := ⟨fun _ ↦ (hα.some, hR.some)⟩ + have h_eq_zero := ν0_eq_feedbackCondAction (obliviousEnv ν) + have h_eq := feedback_eq_feedbackCondAction (obliviousEnv ν) (n - 1) + cases n with + | zero => exact h_eq_zero.symm + | succ n => + simp only [Nat.add_one_sub_one, obliviousEnv_feedback, add_tsub_cancel_right] at h_eq + rw [← Kernel.prodMkLeft_eq_prodMkLeft (γ := Iic n → α × R)] + exact h_eq.symm + +/-- A stationary environment, in which the distribution of the next reward depends only on the last +action. -/ +-- ANCHOR: stationaryEnv +def stationaryEnv (ν : Kernel α R) [IsMarkovKernel ν] : Environment α R := obliviousEnv (fun _ ↦ ν) +-- ANCHOR_END: stationaryEnv + +@[simp] +lemma feedback_stationaryEnv (ν : Kernel α R) [IsMarkovKernel ν] (n : ℕ) : + (stationaryEnv ν).feedback n = ν.prodMkLeft _ := by simp [stationaryEnv] + +@[simp] +lemma ν0_stationaryEnv (ν : Kernel α R) [IsMarkovKernel ν] : (stationaryEnv ν).ν0 = ν := by + simp [stationaryEnv] + +instance (ν : Kernel α R) [IsMarkovKernel ν] : IsObliviousEnv (stationaryEnv ν) where + exists_eq_prodMkLeft := ⟨fun _ ↦ ν, inferInstance, rfl, fun _ ↦ rfl⟩ + +@[simp] +lemma feedbackCondAction_stationaryEnv (ν : Kernel α R) [hν : IsMarkovKernel ν] (n : ℕ) : + feedbackCondAction (stationaryEnv ν) n = ν := feedbackCondAction_obliviousEnv _ _ + +variable {Ω : Type*} {mΩ : MeasurableSpace Ω} + [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] + {alg : Algorithm α R} {ν : Kernel α R} [IsMarkovKernel ν] + {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} + +namespace IsAlgEnvSeq + +/-- The conditional distribution of the reward at time `n` given the action at time `n` is `ν`. -/ +lemma hasCondDistrib_reward_stationaryEnv + (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : + HasCondDistrib (R' n) (A n) ν P := by + simpa using IsObliviousEnv.hasCondDistrib_reward h n + +/-- The conditional distribution of the reward at time `n` given the action at time `n` is `ν`. -/ +lemma condDistrib_reward_stationaryEnv + (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : + condDistrib (R' n) (A n) P =ᵐ[P.map (A n)] ν := + (hasCondDistrib_reward_stationaryEnv h n).condDistrib_eq + +/-- The reward at time `n + 1` is conditionally independent of the history up to time `n` +given the action at time `n + 1`. -/ +lemma condIndepFun_reward_hist_action [StandardBorelSpace Ω] + (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : + R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A _ ; P] hist A R' n := + IsObliviousEnv.condIndepFun_reward_hist_action h n + +lemma condIndepFun_reward_hist_action_action [StandardBorelSpace Ω] + (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : + R' (n + 1) ⟂ᵢ[A (n + 1), h.measurable_A (n + 1); P] + (fun ω ↦ (hist A R' n ω, A (n + 1) ω)) := + IsObliviousEnv.condIndepFun_reward_hist_action_action h n + +lemma condIndepFun_reward_hist_action_action' [StandardBorelSpace Ω] + (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) (hn : n ≠ 0) : + R' n ⟂ᵢ[A n, h.measurable_A n; P] (fun ω ↦ (hist A R' (n - 1) ω, A n ω)) := + IsObliviousEnv.condIndepFun_reward_hist_action_action' h n hn + end IsAlgEnvSeq namespace IT From 46b6fe4920bae5450bf7bee789ff0ac6e4a1fcb7 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Fri, 1 May 2026 21:36:34 +0200 Subject: [PATCH 02/43] progress --- .../Algorithms/GradientDescent.lean | 68 ++++++++++++------- 1 file changed, 43 insertions(+), 25 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 5e59e3b6..4f6c2d1a 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -28,7 +28,7 @@ variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} {x x₀ : E} {g : ℕ → E → E} {hg : ∀ n, Measurable (g n)} {env : Environment E E} - {X G : ℕ → Ω → E} {γ : ℕ → ℝ} + {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} -- todo: write a process version? with `X : ℕ → Ω → E`, as a `ℕ → Ω → F` def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := @@ -62,33 +62,44 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) en section Convex -lemma todo' {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x y : E) : +lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x y : E) : f x - f y ≤ ⟪x - y, ∇ f x⟫ := by + simp only [tsub_le_iff_right] + rw [add_comm] sorry -lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) : +lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := sorry - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by gcongr; exact todo' hf (x _) y - -lemma todo'' (x y g : E) (η : ℝ) : + _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by + simp_rw [smul_sum] + grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] + _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by + simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] + rw [← sum_mul] + simp + field + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by + gcongr + exact hf.sub_le_inner_gradient (x _) y + +lemma todo'' (x y g : E) (hη : 0 < η) : ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by sorry -lemma todo (x y g : ℕ → E) (η : ℕ → ℝ) (n : ℕ) : +lemma todo (x y g : ℕ → E) (hη : ∀ n, 0 < γ n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ ∑ i ∈ Finset.range n, - ((2 * η i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - η i • g i) - y i‖ ^ 2) + - (η i / 2) * ‖g i‖ ^ 2) := by + ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + + (γ i / 2) * ‖g i‖ ^ 2) := by gcongr with i hi - rw [todo'' (x i) (y i) (g i) (η i)] + rw [todo'' (x i) (y i) (g i) (hη i)] lemma todo''' (x g : ℕ → E) (y : E) - (η : ℝ) (hη : 0 ≤ η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo x (fun _ ↦ y) g (fun _ ↦ η) n] + grw [todo x (fun _ ↦ y) g (fun _ ↦ hη) n] rw [sum_add_distrib, ← mul_sum, ← mul_sum] gcongr refine le_of_eq ?_ @@ -96,10 +107,10 @@ lemma todo''' (x g : ℕ → E) (y : E) sorry lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) - (hη : 0 ≤ η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo''' x g y η hη hx n] + grw [todo''' x g y hη hx n] gcongr exact sub_le_self _ (sq_nonneg _) @@ -114,7 +125,7 @@ example (gradKernel : ℕ → Kernel E E) [∀ n, IsMarkovKernel (gradKernel n)] Environment E E := obliviousEnv gradKernel -- use the deterministic equality wrt any sequence -lemma todo1 {η : ℝ} (hη : 0 ≤ η) +lemma todo1 (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : ∀ᵐ ω ∂P, ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫ ≤ @@ -124,15 +135,24 @@ lemma todo1 {η : ℝ} (hη : 0 ≤ η) congr exact hω.1 -lemma sfdsf {η : ℝ} (hη : 0 ≤ η) +lemma sfdsf (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (y : E) (n : ℕ) : P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by - sorry + let M n := MeasurableSpace.comap (IsAlgEnvSeq.hist X G n) inferInstance + calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] + _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | M n] ω] := by + sorry + _ = P[fun ω ↦ ⟪X n ω - y, P[G n | M n] ω⟫] := by + sorry + _ = P[fun ω ↦ ⟪X n ω - y, P[G n | MeasurableSpace.comap (X n) inferInstance] ω⟫] := by + sorry + _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by + sorry + _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] -lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) - {η : ℝ} (hη : 0 ≤ η) +lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (y : E) (n : ℕ) : @@ -145,10 +165,9 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) refine Integrable.sub ?_ ?_ · sorry · sorry - · exact fun ω ↦ todo' (hf n) (X n ω) y + · exact fun ω ↦ (hf n).sub_le_inner_gradient (X n ω) y -lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) - {η : ℝ} (hη : 0 ≤ η) +lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (y : E) (n : ℕ) : @@ -156,8 +175,7 @@ lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by sorry -lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) - {η : ℝ} (hη : 0 ≤ η) +lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) (y : E) (n : ℕ) : From 4fab1efeaea62ff9e61865adaa9fed21f1a4575f Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 07:04:13 +0200 Subject: [PATCH 03/43] work --- .../Algorithms/GradientDescent.lean | 57 ++++++++++++++++--- 1 file changed, 48 insertions(+), 9 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 4f6c2d1a..48adfe63 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -62,13 +62,44 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) en section Convex -lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x y : E) : +omit [SecondCountableTopology E] in +lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : f x - f y ≤ ⟪x - y, ∇ f x⟫ := by + -- TODO: clean this proof (it's from Aristotle) simp only [tsub_le_iff_right] rw [add_comm] - sorry - -lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + have h_convex : ∀ t ∈ Set.Ioo (0 : ℝ) 1, + f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by + intro t ht + have h1 : x + t • (y - x) = (1 - t) • x + t • y := by + rw [smul_sub, sub_smul, one_smul]; abel + rw [h1] + have h2 := hf.2 (Set.mem_univ x) (Set.mem_univ y) + (show 0 ≤ 1 - t by linarith [ht.2]) (show 0 ≤ t by linarith [ht.1]) + (by linarith [ht.1, ht.2]) + simp only [smul_eq_mul] at h2; linarith + have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by + simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] + have h_path_deriv : HasDerivAt (fun t : ℝ => f (x + t • (y - x))) + ((fderiv ℝ f x) (y - x)) 0 := by + have h1 : HasDerivAt (fun t : ℝ => x + t • (y - x)) (y - x) 0 := by + simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) + have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by + simp only [zero_smul, add_zero]; exact hfx.hasFDerivAt + exact h2.comp_hasDerivAt _ h1 + have h_tendsto := h_path_deriv.tendsto_slope_zero_right + have h_combined : (fderiv ℝ f x) (y - x) ≤ f y - f x := by + refine le_of_tendsto h_tendsto (Filter.eventually_of_mem + (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht => ?_) + simp only [zero_add, zero_smul, add_zero, smul_eq_mul, inv_mul_le_iff₀ ht.1] + linarith [h_convex t ht] + linarith [show ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ from by + rw [show x - y = -(y - x) from by abel, inner_neg_left]] + +omit [SecondCountableTopology E] in +lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by @@ -81,12 +112,17 @@ lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) field _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by gcongr - exact hf.sub_le_inner_gradient (x _) y + exact hf.sub_le_inner_gradient hdf.differentiableAt y +omit [CompleteSpace E] [SecondCountableTopology E] in lemma todo'' (x y g : E) (hη : 0 < η) : ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by - sorry + have hsub : (x - η • g) - y = (x - y) - η • g := by abel + rw [hsub, norm_sub_sq_real (x - y) (η • g)] + simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] + field +omit [CompleteSpace E] [SecondCountableTopology E] in lemma todo (x y g : ℕ → E) (hη : ∀ n, 0 < γ n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ ∑ i ∈ Finset.range n, @@ -95,6 +131,7 @@ lemma todo (x y g : ℕ → E) (hη : ∀ n, 0 < γ n) (n : ℕ) : gcongr with i hi rw [todo'' (x i) (y i) (g i) (hη i)] +omit [CompleteSpace E] [SecondCountableTopology E] in lemma todo''' (x g : ℕ → E) (y : E) (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ @@ -104,8 +141,9 @@ lemma todo''' (x g : ℕ → E) (y : E) gcongr refine le_of_eq ?_ simp_rw [← hx] - sorry + exact Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n +omit [CompleteSpace E] [SecondCountableTopology E] in lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ @@ -152,7 +190,8 @@ lemma sfdsf (hη : 0 < η) sorry _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] -lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) +lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (y : E) (n : ℕ) : @@ -165,7 +204,7 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) refine Integrable.sub ?_ ?_ · sorry · sorry - · exact fun ω ↦ (hf n).sub_le_inner_gradient (X n ω) y + · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) From 5afb754d30c055ec6c11a42f2569ec3e7819878d Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 07:29:43 +0200 Subject: [PATCH 04/43] extract convexity lemmas --- .../Algorithms/GradientDescent.lean | 78 ++++++++++++------- 1 file changed, 50 insertions(+), 28 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 48adfe63..42701b00 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -62,40 +62,61 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) en section Convex +omit [CompleteSpace E] [SecondCountableTopology E] in +lemma _root_.ConvexOn.fderiv_sub_le_sub {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + fderiv ℝ f x (y - x) ≤ f y - f x := by + have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : + f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by + have h1 : x + t • (y - x) = (1 - t) • x + t • y := by module + have h2 : f ((1 - t) • x + t • y) ≤ (1 - t) • f x + t • f y := + hf.2 (Set.mem_univ x) (Set.mem_univ y) (by grind) (by grind) (by simp) + simp only [smul_eq_mul] at h2 + grind + have h_path_deriv : HasDerivAt (fun t : ℝ ↦ f (x + t • (y - x))) + (fderiv ℝ f x (y - x)) 0 := by + have h1 : HasDerivAt (fun t : ℝ ↦ x + t • (y - x)) (y - x) 0 := by + simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) + have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by + simpa using hfx.hasFDerivAt + exact h2.comp_hasDerivAt _ h1 + refine le_of_tendsto h_path_deriv.tendsto_slope_zero_right (Filter.eventually_of_mem + (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht ↦ ?_) + simp [inv_mul_le_iff₀ ht.1] + grind + +omit [CompleteSpace E] [SecondCountableTopology E] in +lemma _root_.ConvexOn.add_fderiv_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x + fderiv ℝ f x (y - x) ≤ f y := by + suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind + exact hf.fderiv_sub_le_sub hfx y + +omit [SecondCountableTopology E] in +lemma _root_.ConvexOn.add_inner_gradient_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x + ⟪y - x, ∇ f x⟫ ≤ f y := by + have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by + simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] + rw [← hfderiv] + exact hf.add_fderiv_le hfx y + +omit [SecondCountableTopology E] in +lemma _root_.ConvexOn.le_add_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x ≤ f y + ⟪x - y, ∇ f x⟫ := by + have h_add_le := hf.add_inner_gradient_le hfx y + have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by + rw [show x - y = -(y - x) from by abel, inner_neg_left] + grind + omit [SecondCountableTopology E] in lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x - f y ≤ ⟪x - y, ∇ f x⟫ := by - -- TODO: clean this proof (it's from Aristotle) simp only [tsub_le_iff_right] rw [add_comm] - have h_convex : ∀ t ∈ Set.Ioo (0 : ℝ) 1, - f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by - intro t ht - have h1 : x + t • (y - x) = (1 - t) • x + t • y := by - rw [smul_sub, sub_smul, one_smul]; abel - rw [h1] - have h2 := hf.2 (Set.mem_univ x) (Set.mem_univ y) - (show 0 ≤ 1 - t by linarith [ht.2]) (show 0 ≤ t by linarith [ht.1]) - (by linarith [ht.1, ht.2]) - simp only [smul_eq_mul] at h2; linarith - have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by - simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] - have h_path_deriv : HasDerivAt (fun t : ℝ => f (x + t • (y - x))) - ((fderiv ℝ f x) (y - x)) 0 := by - have h1 : HasDerivAt (fun t : ℝ => x + t • (y - x)) (y - x) 0 := by - simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) - have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by - simp only [zero_smul, add_zero]; exact hfx.hasFDerivAt - exact h2.comp_hasDerivAt _ h1 - have h_tendsto := h_path_deriv.tendsto_slope_zero_right - have h_combined : (fderiv ℝ f x) (y - x) ≤ f y - f x := by - refine le_of_tendsto h_tendsto (Filter.eventually_of_mem - (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht => ?_) - simp only [zero_add, zero_smul, add_zero, smul_eq_mul, inv_mul_le_iff₀ ht.1] - linarith [h_convex t ht] - linarith [show ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ from by - rw [show x - y = -(y - x) from by abel, inner_neg_left]] + exact hf.le_add_inner_gradient hfx y omit [SecondCountableTopology E] in lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) @@ -187,6 +208,7 @@ lemma sfdsf (hη : 0 < η) _ = P[fun ω ↦ ⟪X n ω - y, P[G n | MeasurableSpace.comap (X n) inferInstance] ω⟫] := by sorry _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by + refine integral_congr_ae ?_ sorry _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] From 5163b782a8eddd08f78725a130296f40ffcf1270 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 07:55:57 +0200 Subject: [PATCH 05/43] add integrable inner --- .../Algorithms/GradientDescent.lean | 37 +++++++++++++++---- 1 file changed, 30 insertions(+), 7 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 42701b00..e79da34d 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -107,7 +107,7 @@ lemma _root_.ConvexOn.le_add_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ . f x ≤ f y + ⟪x - y, ∇ f x⟫ := by have h_add_le := hf.add_inner_gradient_le hfx y have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by - rw [show x - y = -(y - x) from by abel, inner_neg_left] + rw [show x - y = -(y - x) by abel, inner_neg_left] grind omit [SecondCountableTopology E] in @@ -179,10 +179,6 @@ section Stochastic variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] --- use `obliviousEnv gradKernel` as the environment for stochastic gradient descent -example (gradKernel : ℕ → Kernel E E) [∀ n, IsMarkovKernel (gradKernel n)] : - Environment E E := obliviousEnv gradKernel - -- use the deterministic equality wrt any sequence lemma todo1 (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) @@ -212,6 +208,34 @@ lemma sfdsf (hη : 0 < η) sorry _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] +omit [IsProbabilityMeasure P] [InnerProductSpace ℝ E] [CompleteSpace E] + [SecondCountableTopology E] in +theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_two_norm_lt_top {f : Ω → E} + (hf : MemLp f 2 P) : + eLpNorm (fun x ↦ ‖f x‖ ^ (2 : ℝ)) 1 P < ∞ := by + simpa [eLpNorm_one_eq_lintegral_enorm] using + (hf.integrable_enorm_rpow (by simp) (by simp)).hasFiniteIntegral + +omit [IsProbabilityMeasure P] [CompleteSpace E] [SecondCountableTopology E] in +lemma _root_.MeasureTheory.MemLp.integrable_inner {f g : Ω → E} + (hf : MemLp f 2 P) (hg : MemLp g 2 P) : + Integrable (fun ω ↦ ⟪f ω, g ω⟫) P := by + rw [← memLp_one_iff_integrable] + constructor + · exact hf.aestronglyMeasurable.inner hg.aestronglyMeasurable + have h x : ‖⟪f x, g x⟫‖ ≤ ‖‖f x‖ ^ (2 : ℝ) + ‖g x‖ ^ (2 : ℝ)‖ := by + norm_cast + calc ‖⟪f x, g x⟫‖ ≤ ‖f x‖ * ‖g x‖ := norm_inner_le_norm _ _ + _ ≤ 2 * ‖f x‖ * ‖g x‖ := by + gcongr + exact le_mul_of_one_le_left (norm_nonneg _) one_le_two + _ ≤ ‖‖f x‖ ^ 2 + ‖g x‖ ^ 2‖ := (two_mul_le_add_sq _ _).trans (le_abs_self _) + refine (eLpNorm_mono h).trans_lt ((eLpNorm_add_le ?_ ?_ le_rfl).trans_lt ?_) + · exact (hf.norm.aemeasurable.pow_const _).aestronglyMeasurable + · exact (hg.norm.aemeasurable.pow_const _).aestronglyMeasurable + rw [ENNReal.add_lt_top] + exact ⟨hf.eLpNorm_rpow_two_norm_lt_top, hg.eLpNorm_rpow_two_norm_lt_top⟩ + lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) @@ -222,8 +246,7 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) gcongr · refine Integrable.sub ?_ (integrable_const _) sorry - · simp only [inner_sub_left] - refine Integrable.sub ?_ ?_ + · refine MemLp.integrable_inner ?_ ?_ · sorry · sorry · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y From 85d33b7c516bbc136e368e09379478d970509908 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 08:32:16 +0200 Subject: [PATCH 06/43] add integrability hypotheses --- .../Algorithms/GradientDescent.lean | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index e79da34d..f910754a 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -208,6 +208,14 @@ lemma sfdsf (hη : 0 < η) sorry _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] +lemma sfdsf' (hη : 0 < η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (y : E) (n : ℕ) : + ∫⁻ ω, ‖∇ (f n) (X n ω)‖ₑ ^ 2 ∂P ≤ ∫⁻ ω, ‖G n ω‖ₑ ^ 2 ∂P := by + simp_rw [← h_unbiased] + sorry + omit [IsProbabilityMeasure P] [InnerProductSpace ℝ E] [CompleteSpace E] [SecondCountableTopology E] in theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_two_norm_lt_top {f : Ω → E} @@ -238,8 +246,9 @@ lemma _root_.MeasureTheory.MemLp.integrable_inner {f g : Ω → E} lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by rw [sfdsf hη h h_unbiased y n] @@ -252,16 +261,18 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : P[fun ω ↦ ∑ i ∈ Finset.range n, f n (X n ω) - f n y] ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by sorry lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X n ω) - f y] ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + From 1917c7e253d166d95b9b1e2d0a908e6972367c3e Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 10:40:44 +0200 Subject: [PATCH 07/43] minor --- .../Optimization/Algorithms/GradientDescent.lean | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index f910754a..afb96d93 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -218,11 +218,11 @@ lemma sfdsf' (hη : 0 < η) omit [IsProbabilityMeasure P] [InnerProductSpace ℝ E] [CompleteSpace E] [SecondCountableTopology E] in -theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_two_norm_lt_top {f : Ω → E} - (hf : MemLp f 2 P) : - eLpNorm (fun x ↦ ‖f x‖ ^ (2 : ℝ)) 1 P < ∞ := by - simpa [eLpNorm_one_eq_lintegral_enorm] using - (hf.integrable_enorm_rpow (by simp) (by simp)).hasFiniteIntegral +theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_norm_lt_top {f : Ω → E} {p : ℝ≥0∞} + (hf : MemLp f p P) (hp_zero : p ≠ 0) (hp_top : p ≠ ∞) : + eLpNorm (fun x ↦ ‖f x‖ ^ p.toReal) 1 P < ∞ := by + simpa [eLpNorm_one_eq_lintegral_enorm, enorm_rpow_of_nonneg] using + (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral omit [IsProbabilityMeasure P] [CompleteSpace E] [SecondCountableTopology E] in lemma _root_.MeasureTheory.MemLp.integrable_inner {f g : Ω → E} @@ -242,7 +242,8 @@ lemma _root_.MeasureTheory.MemLp.integrable_inner {f g : Ω → E} · exact (hf.norm.aemeasurable.pow_const _).aestronglyMeasurable · exact (hg.norm.aemeasurable.pow_const _).aestronglyMeasurable rw [ENNReal.add_lt_top] - exact ⟨hf.eLpNorm_rpow_two_norm_lt_top, hg.eLpNorm_rpow_two_norm_lt_top⟩ + exact ⟨hf.eLpNorm_rpow_norm_lt_top (by simp) (by simp), + hg.eLpNorm_rpow_norm_lt_top (by simp) (by simp)⟩ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) From b6d8dda200a70c029d44f9528a9bf56ad583589a Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 12:02:58 +0200 Subject: [PATCH 08/43] work --- .../Algorithms/GradientDescent.lean | 79 ++++++++++++------- .../SequentialLearning/StationaryEnv.lean | 17 ++++ 2 files changed, 67 insertions(+), 29 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index afb96d93..debea358 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -190,32 +190,6 @@ lemma todo1 (hη : 0 < η) congr exact hω.1 -lemma sfdsf (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (y : E) (n : ℕ) : - P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by - let M n := MeasurableSpace.comap (IsAlgEnvSeq.hist X G n) inferInstance - calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] - _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | M n] ω] := by - sorry - _ = P[fun ω ↦ ⟪X n ω - y, P[G n | M n] ω⟫] := by - sorry - _ = P[fun ω ↦ ⟪X n ω - y, P[G n | MeasurableSpace.comap (X n) inferInstance] ω⟫] := by - sorry - _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by - refine integral_congr_ae ?_ - sorry - _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] - -lemma sfdsf' (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (y : E) (n : ℕ) : - ∫⁻ ω, ‖∇ (f n) (X n ω)‖ₑ ^ 2 ∂P ≤ ∫⁻ ω, ‖G n ω‖ₑ ^ 2 ∂P := by - simp_rw [← h_unbiased] - sorry - omit [IsProbabilityMeasure P] [InnerProductSpace ℝ E] [CompleteSpace E] [SecondCountableTopology E] in theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_norm_lt_top {f : Ω → E} {p : ℝ≥0∞} @@ -245,14 +219,61 @@ lemma _root_.MeasureTheory.MemLp.integrable_inner {f g : Ω → E} exact ⟨hf.eLpNorm_rpow_norm_lt_top (by simp) (by simp), hg.eLpNorm_rpow_norm_lt_top (by simp) (by simp)⟩ -lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) - (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) +theorem condExp_inner_of_stronglyMeasurable_left {Ω H : Type*} {m mΩ : MeasurableSpace Ω} + [NormedAddCommGroup H] [InnerProductSpace ℝ H] [CompleteSpace H] {μ : Measure Ω} {X g : Ω → H} + (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : + μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by + filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω + simpa [innerSL_apply_apply] using hω + +lemma memLp_X_sub (hη : 0 < η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : + MemLp (X n) 2 P := by + sorry + +lemma sfdsf (hη : 0 < η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (y : E) (n : ℕ) : + P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by + let M n := MeasurableSpace.comap (X n) inferInstance + have h_obl : HasCondDistrib (G n) (X n) (gradKernel n) P := h.hasCondDistrib_reward_obliviousEnv n + calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] + _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | M n] ω] := by + rw [integral_condExp] + exact (h.measurable_A _).comap_le + _ = P[fun ω ↦ ⟪X n ω - y, P[G n | M n] ω⟫] := by + refine integral_congr_ae ?_ + refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ + · refine StronglyMeasurable.sub ?_ (by fun_prop) + refine Measurable.stronglyMeasurable ?_ + rw [measurable_iff_comap_le] + · refine MemLp.integrable_inner (MemLp.sub ?_ (memLp_const _)) (h_memLp n) + exact memLp_X_sub hη h h_memLp n + · exact (h_memLp n).integrable (by simp) + _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by + have h_ae := h_obl.condDistrib_eq + refine integral_congr_ae ?_ + sorry + _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] + +lemma sfdsf' (hη : 0 < η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (y : E) (n : ℕ) : + ∫⁻ ω, ‖∇ (f n) (X n ω)‖ₑ ^ 2 ∂P ≤ ∫⁻ ω, ‖G n ω‖ₑ ^ 2 ∂P := by + simp_rw [← h_unbiased] + sorry + +lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by - rw [sfdsf hη h h_unbiased y n] + rw [sfdsf hη h h_unbiased h_memLp y n] gcongr · refine Integrable.sub ?_ (integrable_const _) sorry diff --git a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean index dbd7474c..be57a246 100644 --- a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean +++ b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean @@ -78,6 +78,17 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {A : ℕ → Ω → α} {R' : ℕ → Ω → R} {n N : ℕ} {ν : ℕ → Kernel α R} [∀ n, IsMarkovKernel (ν n)] +lemma hasCOndDistrib_reward_hist_action [IsObliviousEnv env] + (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : + HasCondDistrib (R' (n + 1)) (fun ω ↦ (IsAlgEnvSeq.hist A R' n ω, A (n + 1) ω)) + ((feedbackCondAction env (n + 1)).prodMkLeft _) P := by + have hA := h.measurable_A + have hR' := h.measurable_R + refine ⟨by fun_prop, by fun_prop, ?_⟩ + have h_eq := (h.hasCondDistrib_reward n).condDistrib_eq + rw [condDistrib_ae_eq_iff_measure_eq_compProd _ (by fun_prop)] at h_eq ⊢ + simpa only [feedback_eq_feedbackCondAction] using h_eq + lemma hasCondDistrib_reward [IsObliviousEnv env] (h : IsAlgEnvSeq A R' alg env P) (n : ℕ) : HasCondDistrib (R' n) (A n) (feedbackCondAction env n) P := by have hA := h.measurable_A @@ -198,6 +209,12 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} namespace IsAlgEnvSeq +/-- The conditional distribution of the reward at time `n` given the action at time `n` is `ν`. -/ +lemma hasCondDistrib_reward_obliviousEnv {ν : ℕ → Kernel α R} [∀ n, IsMarkovKernel (ν n)] + (h : IsAlgEnvSeq A R' alg (obliviousEnv ν) P) (n : ℕ) : + HasCondDistrib (R' n) (A n) (ν n) P := by + simpa using IsObliviousEnv.hasCondDistrib_reward h n + /-- The conditional distribution of the reward at time `n` given the action at time `n` is `ν`. -/ lemma hasCondDistrib_reward_stationaryEnv (h : IsAlgEnvSeq A R' alg (stationaryEnv ν) P) (n : ℕ) : From f6d53850d25e939d3018cb8344c0f5b3f895b89a Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 13:56:25 +0200 Subject: [PATCH 09/43] progress --- .../Algorithms/GradientDescent.lean | 54 +++++++++++++------ 1 file changed, 38 insertions(+), 16 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index debea358..eba07eff 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -226,14 +226,31 @@ theorem condExp_inner_of_stronglyMeasurable_left {Ω H : Type*} {m mΩ : Measura filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω simpa [innerSL_apply_apply] using hω -lemma memLp_X_sub (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) +lemma memLp_X (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : MemLp (X n) 2 P := by - sorry + induction n with + | zero => + have h0 : MemLp (fun _ ↦ x₀) 2 P := memLp_const _ + refine h0.ae_eq ?_ + filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω using hω.1.symm + | succ n hn => + have h_sub : MemLp (fun ω ↦ X n ω - η • G n ω) 2 P := hn.sub (MemLp.const_smul (h_memLp n) _) + refine h_sub.ae_eq ?_ + filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω using (hω.2 n).symm + +lemma condExp_reward_obliviousEnv_ae_eq_integral_id {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv ν) P) + (n : ℕ) (h_int : Integrable (G n) P) : + P[G n | MeasurableSpace.comap (X n) inferInstance] =ᵐ[P] fun ω ↦ (ν n (X n ω))[id] := by + have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_reward_obliviousEnv n + have h_ae := ae_of_ae_map (h.measurable_A n).aemeasurable h_obl.condDistrib_eq + have h_ae' := condExp_ae_eq_integral_condDistrib' (h.measurable_A n) h_int + filter_upwards [h_ae, h_ae'] with ω hω hω' + rw [hω', hω] + congr -lemma sfdsf (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) +lemma sfdsf (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (y : E) (n : ℕ) : P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by @@ -250,22 +267,27 @@ lemma sfdsf (hη : 0 < η) refine Measurable.stronglyMeasurable ?_ rw [measurable_iff_comap_le] · refine MemLp.integrable_inner (MemLp.sub ?_ (memLp_const _)) (h_memLp n) - exact memLp_X_sub hη h h_memLp n + exact memLp_X h h_memLp n · exact (h_memLp n).integrable (by simp) _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by - have h_ae := h_obl.condDistrib_eq + have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n + ((h_memLp n).integrable (by simp)) refine integral_congr_ae ?_ - sorry + filter_upwards [h_ae] with ω hω using by rw [hω] _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] -lemma sfdsf' (hη : 0 < η) +lemma memLp_gradient (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) - (y : E) (n : ℕ) : - ∫⁻ ω, ‖∇ (f n) (X n ω)‖ₑ ^ 2 ∂P ≤ ∫⁻ ω, ‖G n ω‖ₑ ^ 2 ∂P := by + (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : + MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by + let M n := MeasurableSpace.comap (X n) inferInstance + have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) + have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n + ((h_memLp n).integrable (by simp)) + refine h_lp.ae_eq <| h_ae.trans ?_ simp_rw [← h_unbiased] - sorry + rfl lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) @@ -273,13 +295,13 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by - rw [sfdsf hη h h_unbiased h_memLp y n] + rw [sfdsf h h_unbiased h_memLp y n] gcongr · refine Integrable.sub ?_ (integrable_const _) sorry · refine MemLp.integrable_inner ?_ ?_ - · sorry - · sorry + · exact (memLp_X h h_memLp n).sub (memLp_const _) + · exact memLp_gradient h h_unbiased h_memLp n · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) From 20336363aa0b5bb4cff08ecd5b2842fd9deedbf4 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 14:14:24 +0200 Subject: [PATCH 10/43] progress --- .../Algorithms/GradientDescent.lean | 53 ++++++++++++++----- 1 file changed, 39 insertions(+), 14 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index eba07eff..001027e4 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -179,17 +179,6 @@ section Stochastic variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] --- use the deterministic equality wrt any sequence -lemma todo1 (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (y : E) (n : ℕ) : - ∀ᵐ ω ∂P, ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫ ≤ - (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 := by - filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω - refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ - congr - exact hω.1 - omit [IsProbabilityMeasure P] [InnerProductSpace ℝ E] [CompleteSpace E] [SecondCountableTopology E] in theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_norm_lt_top {f : Ω → E} {p : ℝ≥0∞} @@ -304,14 +293,50 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable · exact memLp_gradient h h_unbiased h_memLp n · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y -lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hη : 0 < η) +-- use the deterministic equality wrt any sequence +lemma todo1 (hη : 0 < η) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (y : E) (n : ℕ) : + ∀ᵐ ω ∂P, ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫ ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 := by + filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω + refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ + congr + exact hω.1 + +lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : - P[fun ω ↦ ∑ i ∈ Finset.range n, f n (X n ω) - f n y] ≤ + P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - sorry + calc P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] + _ ≤ P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + rw [integral_finset_sum, integral_finset_sum] + rotate_left + · intro i hi + refine MemLp.integrable_inner ?_ (h_memLp i) + exact (memLp_X h h_memLp i).sub (memLp_const _) + · sorry + refine Finset.sum_le_sum fun i hi ↦ ?_ + exact qfqgs hf hdf hη h_unbiased h_memLp h y i + _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 ∂P := by + refine integral_mono_ae ?_ ?_ (todo1 hη h y n) + · refine integrable_finset_sum _ fun i hi ↦ ?_ + refine MemLp.integrable_inner ?_ (h_memLp i) + exact (memLp_X h h_memLp i).sub (memLp_const _) + · refine Integrable.add (integrable_const _) (Integrable.const_mul ?_ _) + refine integrable_finset_sum _ fun i hi ↦ ?_ + exact (h_memLp i).integrable_norm_pow (by simp) + _ = (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + rw [integral_add, integral_const_mul, integral_const_mul, integral_finset_sum] + · simp + · exact fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) + · exact integrable_const _ + · refine Integrable.const_mul ?_ _ + refine integrable_finset_sum _ fun i hi ↦ ?_ + exact (h_memLp i).integrable_norm_pow (by simp) lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) From c84984ef17633371ce047ec50e7110d6b971fdc3 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 14:42:28 +0200 Subject: [PATCH 11/43] sorry-free --- .../Algorithms/GradientDescent.lean | 46 +++++++++++++++---- 1 file changed, 38 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 001027e4..0190568e 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -118,6 +118,20 @@ lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ . rw [add_comm] exact hf.le_add_inner_gradient hfx y +omit [CompleteSpace E] [SecondCountableTopology E] in +lemma todo'3 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by + calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y + _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by + simp_rw [smul_sum] + grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] + _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by + simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] + rw [← sum_mul] + simp + field + omit [SecondCountableTopology E] in lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : @@ -278,16 +292,17 @@ lemma memLp_gradient simp_rw [← h_unbiased] rfl -lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) +lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption (y : E) (n : ℕ) : P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by rw [sfdsf h h_unbiased h_memLp y n] gcongr · refine Integrable.sub ?_ (integrable_const _) - sorry + exact h_int n · refine MemLp.integrable_inner ?_ ?_ · exact (memLp_X h h_memLp n).sub (memLp_const _) · exact memLp_gradient h h_unbiased h_memLp n @@ -308,6 +323,7 @@ lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentia (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by @@ -318,9 +334,9 @@ lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentia · intro i hi refine MemLp.integrable_inner ?_ (h_memLp i) exact (memLp_X h h_memLp i).sub (memLp_const _) - · sorry + · exact fun i hi ↦ (h_int i).sub (integrable_const _) refine Finset.sum_le_sum fun i hi ↦ ?_ - exact qfqgs hf hdf hη h_unbiased h_memLp h y i + exact qfqgs hf hdf h_unbiased h_memLp h h_int y i _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 ∂P := by refine integral_mono_ae ?_ ?_ (todo1 hη h y n) · refine integrable_finset_sum _ fun i hi ↦ ?_ @@ -338,15 +354,29 @@ lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentia refine integrable_finset_sum _ fun i hi ↦ ?_ exact (h_memLp i).integrable_norm_pow (by simp) -lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hη : 0 < η) +lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (y : E) (n : ℕ) : - P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X n ω) - f y] ≤ + (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) + (y : E) (n : ℕ) (hn : n ≠ 0) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - sorry + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by + rw [← integral_const_mul] + gcongr + · exact h_int_avg.sub (integrable_const _) + · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ + exact (h_int i).sub (integrable_const _) + exact fun ω ↦ todo'3 hf _ y n hn + _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + + (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + grw [qsfqqfqgs (fun _ ↦ hf) (fun _ ↦ hdf) hη h_unbiased h_memLp h h_int y n] + refine le_of_eq ?_ + field end Stochastic From 123af7f60094f4de3e518e60b979a5fbcf97dccd Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 15:16:25 +0200 Subject: [PATCH 12/43] minor --- .../Algorithms/GradientDescent.lean | 29 ++++++++++++++----- 1 file changed, 21 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 0190568e..cf9e1669 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -118,6 +118,15 @@ lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ . rw [add_comm] exact hf.le_add_inner_gradient hfx y +omit [SecondCountableTopology E] in +lemma onlineRegret_le_onlineRegret_linearizedLoss + (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) + (x : ℕ → E) (y : E) (n : ℕ) : + onlineRegret f y x n ≤ onlineRegret (linearizedLoss f x) y x n := by + simp only [onlineRegret, linearizedLoss, ← inner_sub_left] + gcongr with i hi + exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ + omit [CompleteSpace E] [SecondCountableTopology E] in lemma todo'3 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : @@ -137,14 +146,7 @@ lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by - simp_rw [smul_sum] - grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] - _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by - simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] - rw [← sum_mul] - simp - field + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by gcongr exact hf.sub_le_inner_gradient hdf.differentiableAt y @@ -354,6 +356,17 @@ lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentia refine integrable_finset_sum _ fun i hi ↦ ?_ exact (h_memLp i).integrable_norm_pow (by simp) +lemma integral_onlineRegret_le + (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) + (y : E) (n : ℕ) : + P[fun ω ↦ onlineRegret f y (X · ω) n] ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := + qsfqqfqgs hf hdf hη h_unbiased h_memLp h h_int y n + lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) From bb01cc001494ae1b16ad46e45e175051cc2be4ff Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 2 May 2026 18:12:31 +0200 Subject: [PATCH 13/43] reorganize --- .../Algorithms/GradientDescent.lean | 279 ++++++++++-------- 1 file changed, 154 insertions(+), 125 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index cf9e1669..37f4bdcd 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -19,50 +19,105 @@ public import Mathlib open MeasureTheory ProbabilityTheory Filter Real Finset open scoped Gradient ENNReal NNReal RealInnerProductSpace +section Aux + +variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} {P : Measure Ω} + [NormedAddCommGroup E] + +theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_norm_lt_top + {f : Ω → E} {p : ℝ≥0∞} + (hf : MemLp f p P) (hp_zero : p ≠ 0) (hp_top : p ≠ ∞) : + eLpNorm (fun x ↦ ‖f x‖ ^ p.toReal) 1 P < ∞ := by + simpa [eLpNorm_one_eq_lintegral_enorm, enorm_rpow_of_nonneg] using + (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral + +lemma _root_.MeasureTheory.MemLp.integrable_inner [InnerProductSpace ℝ E] + {f g : Ω → E} + (hf : MemLp f 2 P) (hg : MemLp g 2 P) : + Integrable (fun ω ↦ ⟪f ω, g ω⟫) P := by + rw [← memLp_one_iff_integrable] + constructor + · exact hf.aestronglyMeasurable.inner hg.aestronglyMeasurable + have h x : ‖⟪f x, g x⟫‖ ≤ ‖‖f x‖ ^ (2 : ℝ) + ‖g x‖ ^ (2 : ℝ)‖ := by + norm_cast + calc ‖⟪f x, g x⟫‖ ≤ ‖f x‖ * ‖g x‖ := norm_inner_le_norm _ _ + _ ≤ 2 * ‖f x‖ * ‖g x‖ := by + gcongr + exact le_mul_of_one_le_left (norm_nonneg _) one_le_two + _ ≤ ‖‖f x‖ ^ 2 + ‖g x‖ ^ 2‖ := (two_mul_le_add_sq _ _).trans (le_abs_self _) + refine (eLpNorm_mono h).trans_lt ((eLpNorm_add_le ?_ ?_ le_rfl).trans_lt ?_) + · exact (hf.norm.aemeasurable.pow_const _).aestronglyMeasurable + · exact (hg.norm.aemeasurable.pow_const _).aestronglyMeasurable + rw [ENNReal.add_lt_top] + exact ⟨hf.eLpNorm_rpow_norm_lt_top (by simp) (by simp), + hg.eLpNorm_rpow_norm_lt_top (by simp) (by simp)⟩ + +theorem condExp_inner_of_stronglyMeasurable_left {Ω : Type*} {m mΩ : MeasurableSpace Ω} + [InnerProductSpace ℝ E] [CompleteSpace E] {μ : Measure Ω} {X g : Ω → E} + (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : + μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by + filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω + simpa [innerSL_apply_apply] using hω + +end Aux + namespace Learning variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] - [NormedAddCommGroup E] [InnerProductSpace ℝ E] [CompleteSpace E] [BorelSpace E] - [MeasurableSub₂ E] [SecondCountableTopology E] - {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} {x x₀ : E} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [BorelSpace E] + [MeasurableSub₂ E] {x x₀ : E} {g : ℕ → E → E} {hg : ∀ n, Measurable (g n)} {env : Environment E E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} --- todo: write a process version? with `X : ℕ → Ω → E`, as a `ℕ → Ω → F` -def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := - ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) +section Linear -noncomputable def linearizedLoss (f : ℕ → E → ℝ) (x : ℕ → E) : ℕ → E → ℝ := - fun n y ↦ ⟪y, ∇ (f n) (x n)⟫ +lemma todo'' (x y g : E) (hη : 0 < η) : + ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by + have hsub : (x - η • g) - y = (x - y) - η • g := by abel + rw [hsub, norm_sub_sq_real (x - y) (η • g)] + simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] + field -/-- Online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`, -without projection. +lemma todo (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + + (γ i / 2) * ‖g i‖ ^ 2) := by + gcongr with i hi + rw [todo'' (x i) (y i) (g i) (hγ i)] -It is an algorithm that chooses actions in `E` and gets feedback in `E` (gradient of the function at -the queried point). -/ -noncomputable -def gradientDescent (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := - detAlgorithm (fun n hist ↦ (hist ⟨n, by grind⟩).1 - γ n • (hist ⟨n, by grind⟩).2) (by fun_prop) x₀ +lemma todo_sfsq (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) + (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := + (todo x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] -lemma action_gradientDescent_ae_eq (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) env P) (n : ℕ) : - X (n + 1) =ᵐ[P] X n - γ n • G n := h_seq.action_detAlgorithm_ae_eq n +section ConstantStep -lemma action_gradientDescent_ae_all_eq (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) env P) : - ∀ᵐ ω ∂P, X 0 ω = x₀ ∧ ∀ n, X (n + 1) ω = X n ω - γ n • G n ω := - h_seq.action_detAlgorithm_ae_all_eq +lemma todo''' (x g : ℕ → E) (y : E) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + refine (todo_sfsq x g y (fun _ ↦ hη) hx n).trans_eq ?_ + rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] -lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientDescent γ x₀) env P) (n : ℕ) : - X n =ᵐ[P] fun ω ↦ x₀ - ∑ i ∈ Finset.range n, γ i • G i ω := by - filter_upwards [h_seq.action_detAlgorithm_ae_all_eq] with ω ⟨hω0, hω⟩ - induction n with - | zero => simpa - | succ n ih => rw [hω n, Finset.sum_range_succ, ← sub_sub]; congr +lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + grw [todo''' x g y hη hx n] + gcongr + exact sub_le_self _ (sq_nonneg _) + +end ConstantStep + +end Linear section Convex -omit [CompleteSpace E] [SecondCountableTopology E] in lemma _root_.ConvexOn.fderiv_sub_le_sub {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : fderiv ℝ f x (y - x) ≤ f y - f x := by @@ -85,15 +140,13 @@ lemma _root_.ConvexOn.fderiv_sub_le_sub {f : E → ℝ} (hf : ConvexOn ℝ .univ simp [inv_mul_le_iff₀ ht.1] grind -omit [CompleteSpace E] [SecondCountableTopology E] in lemma _root_.ConvexOn.add_fderiv_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x + fderiv ℝ f x (y - x) ≤ f y := by suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind exact hf.fderiv_sub_le_sub hfx y -omit [SecondCountableTopology E] in -lemma _root_.ConvexOn.add_inner_gradient_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma _root_.ConvexOn.add_inner_gradient_le [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x + ⟪y - x, ∇ f x⟫ ≤ f y := by have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by @@ -101,8 +154,7 @@ lemma _root_.ConvexOn.add_inner_gradient_le {f : E → ℝ} (hf : ConvexOn ℝ . rw [← hfderiv] exact hf.add_fderiv_le hfx y -omit [SecondCountableTopology E] in -lemma _root_.ConvexOn.le_add_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma _root_.ConvexOn.le_add_inner_gradient [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x ≤ f y + ⟪x - y, ∇ f x⟫ := by have h_add_le := hf.add_inner_gradient_le hfx y @@ -110,24 +162,13 @@ lemma _root_.ConvexOn.le_add_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ . rw [show x - y = -(y - x) by abel, inner_neg_left] grind -omit [SecondCountableTopology E] in -lemma _root_.ConvexOn.sub_le_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma _root_.ConvexOn.sub_le_inner_gradient [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x - f y ≤ ⟪x - y, ∇ f x⟫ := by simp only [tsub_le_iff_right] rw [add_comm] exact hf.le_add_inner_gradient hfx y -omit [SecondCountableTopology E] in -lemma onlineRegret_le_onlineRegret_linearizedLoss - (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) - (x : ℕ → E) (y : E) (n : ℕ) : - onlineRegret f y x n ≤ onlineRegret (linearizedLoss f x) y x n := by - simp only [onlineRegret, linearizedLoss, ← inner_sub_left] - gcongr with i hi - exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ - -omit [CompleteSpace E] [SecondCountableTopology E] in lemma todo'3 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by @@ -141,8 +182,7 @@ lemma todo'3 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) simp field -omit [SecondCountableTopology E] in -lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) +lemma todo'2 [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y @@ -151,101 +191,90 @@ lemma todo'2 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable gcongr exact hf.sub_le_inner_gradient hdf.differentiableAt y -omit [CompleteSpace E] [SecondCountableTopology E] in -lemma todo'' (x y g : E) (hη : 0 < η) : - ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by - have hsub : (x - η • g) - y = (x - y) - η • g := by abel - rw [hsub, norm_sub_sq_real (x - y) (η • g)] - simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] - field +end Convex -omit [CompleteSpace E] [SecondCountableTopology E] in -lemma todo (x y g : ℕ → E) (hη : ∀ n, 0 < γ n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + - (γ i / 2) * ‖g i‖ ^ 2) := by +section OnlineRegret + +def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := + ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) + +noncomputable def linearizedLoss [CompleteSpace E] (f : ℕ → E → ℝ) (x : ℕ → E) : ℕ → E → ℝ := + fun n y ↦ ⟪y, ∇ (f n) (x n)⟫ + +lemma onlineRegret_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : ℕ → E → ℝ} + (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) + (x : ℕ → E) (y : E) (n : ℕ) : + onlineRegret f y x n ≤ onlineRegret (linearizedLoss f x) y x n := by + simp only [onlineRegret, linearizedLoss, ← inner_sub_left] gcongr with i hi - rw [todo'' (x i) (y i) (g i) (hη i)] + exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ -omit [CompleteSpace E] [SecondCountableTopology E] in -lemma todo''' (x g : ℕ → E) (y : E) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo x (fun _ ↦ y) g (fun _ ↦ hη) n] - rw [sum_add_distrib, ← mul_sum, ← mul_sum] - gcongr - refine le_of_eq ?_ - simp_rw [← hx] - exact Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n +lemma apply_avg_sub_le_onlineRegret {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := + todo'3 hf x y n hn -omit [CompleteSpace E] [SecondCountableTopology E] in -lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) +lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + onlineRegret (fun n x ↦ ⟪x, g n⟫) y x n ≤ (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo''' x g y hη hx n] - gcongr - exact sub_le_self _ (sq_nonneg _) + simpa [onlineRegret, inner_sub_left] using lem14dot1 x g y η hη hx n -end Convex +end OnlineRegret -section Stochastic +variable [SecondCountableTopology E] [CompleteSpace E] + {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} -variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] +/-- Online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`, +without projection. -omit [IsProbabilityMeasure P] [InnerProductSpace ℝ E] [CompleteSpace E] - [SecondCountableTopology E] in -theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_norm_lt_top {f : Ω → E} {p : ℝ≥0∞} - (hf : MemLp f p P) (hp_zero : p ≠ 0) (hp_top : p ≠ ∞) : - eLpNorm (fun x ↦ ‖f x‖ ^ p.toReal) 1 P < ∞ := by - simpa [eLpNorm_one_eq_lintegral_enorm, enorm_rpow_of_nonneg] using - (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral +It is an algorithm that chooses actions in `E` and gets feedback in `E` (gradient of the function at +the queried point). -/ +noncomputable +def gradientStep (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := + detAlgorithm (fun n hist ↦ (hist ⟨n, by grind⟩).1 - γ n • (hist ⟨n, by grind⟩).2) (by fun_prop) x₀ -omit [IsProbabilityMeasure P] [CompleteSpace E] [SecondCountableTopology E] in -lemma _root_.MeasureTheory.MemLp.integrable_inner {f g : Ω → E} - (hf : MemLp f 2 P) (hg : MemLp g 2 P) : - Integrable (fun ω ↦ ⟪f ω, g ω⟫) P := by - rw [← memLp_one_iff_integrable] - constructor - · exact hf.aestronglyMeasurable.inner hg.aestronglyMeasurable - have h x : ‖⟪f x, g x⟫‖ ≤ ‖‖f x‖ ^ (2 : ℝ) + ‖g x‖ ^ (2 : ℝ)‖ := by - norm_cast - calc ‖⟪f x, g x⟫‖ ≤ ‖f x‖ * ‖g x‖ := norm_inner_le_norm _ _ - _ ≤ 2 * ‖f x‖ * ‖g x‖ := by - gcongr - exact le_mul_of_one_le_left (norm_nonneg _) one_le_two - _ ≤ ‖‖f x‖ ^ 2 + ‖g x‖ ^ 2‖ := (two_mul_le_add_sq _ _).trans (le_abs_self _) - refine (eLpNorm_mono h).trans_lt ((eLpNorm_add_le ?_ ?_ le_rfl).trans_lt ?_) - · exact (hf.norm.aemeasurable.pow_const _).aestronglyMeasurable - · exact (hg.norm.aemeasurable.pow_const _).aestronglyMeasurable - rw [ENNReal.add_lt_top] - exact ⟨hf.eLpNorm_rpow_norm_lt_top (by simp) (by simp), - hg.eLpNorm_rpow_norm_lt_top (by simp) (by simp)⟩ +lemma action_gradientStep_ae_eq (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P) (n : ℕ) : + X (n + 1) =ᵐ[P] X n - γ n • G n := h_seq.action_detAlgorithm_ae_eq n -theorem condExp_inner_of_stronglyMeasurable_left {Ω H : Type*} {m mΩ : MeasurableSpace Ω} - [NormedAddCommGroup H] [InnerProductSpace ℝ H] [CompleteSpace H] {μ : Measure Ω} {X g : Ω → H} - (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : - μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by - filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω - simpa [innerSL_apply_apply] using hω +lemma action_gradientStep_ae_all_eq (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P) : + ∀ᵐ ω ∂P, X 0 ω = x₀ ∧ ∀ n, X (n + 1) ω = X n ω - γ n • G n ω := + h_seq.action_detAlgorithm_ae_all_eq + +lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P) (n : ℕ) : + X n =ᵐ[P] fun ω ↦ x₀ - ∑ i ∈ Finset.range n, γ i • G i ω := by + filter_upwards [h_seq.action_detAlgorithm_ae_all_eq] with ω ⟨hω0, hω⟩ + induction n with + | zero => simpa + | succ n ih => rw [hω n, Finset.sum_range_succ, ← sub_sub]; congr + +omit [SecondCountableTopology E] in +lemma apply_avg_sub_le_onlineRegret_linearizedLoss {f : E → ℝ} + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ + (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by + simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using todo'2 hf hdf x y n hn + +section Stochastic + +variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] -lemma memLp_X (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) +lemma memLp_X (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : MemLp (X n) 2 P := by induction n with | zero => have h0 : MemLp (fun _ ↦ x₀) 2 P := memLp_const _ refine h0.ae_eq ?_ - filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω using hω.1.symm + filter_upwards [action_gradientStep_ae_all_eq h] with ω hω using hω.1.symm | succ n hn => have h_sub : MemLp (fun ω ↦ X n ω - η • G n ω) 2 P := hn.sub (MemLp.const_smul (h_memLp n) _) refine h_sub.ae_eq ?_ - filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω using (hω.2 n).symm + filter_upwards [action_gradientStep_ae_all_eq h] with ω hω using (hω.2 n).symm lemma condExp_reward_obliviousEnv_ae_eq_integral_id {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv ν) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv ν) P) (n : ℕ) (h_int : Integrable (G n) P) : P[G n | MeasurableSpace.comap (X n) inferInstance] =ᵐ[P] fun ω ↦ (ν n (X n ω))[id] := by have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_reward_obliviousEnv n @@ -255,7 +284,7 @@ lemma condExp_reward_obliviousEnv_ae_eq_integral_id {ν : ℕ → Kernel E E} [ rw [hω', hω] congr -lemma sfdsf (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) +lemma sfdsf (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (y : E) (n : ℕ) : P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by @@ -282,7 +311,7 @@ lemma sfdsf (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviou _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] lemma memLp_gradient - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by @@ -297,7 +326,7 @@ lemma memLp_gradient lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption (y : E) (n : ℕ) : P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by @@ -312,11 +341,11 @@ lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable -- use the deterministic equality wrt any sequence lemma todo1 (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : ∀ᵐ ω ∂P, ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫ ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 := by - filter_upwards [action_gradientDescent_ae_all_eq h] with ω hω + filter_upwards [action_gradientStep_ae_all_eq h] with ω hω refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ congr exact hω.1 @@ -324,7 +353,7 @@ lemma todo1 (hη : 0 < η) lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ @@ -360,7 +389,7 @@ lemma integral_onlineRegret_le (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : P[fun ω ↦ onlineRegret f y (X · ω) n] ≤ @@ -370,7 +399,7 @@ lemma integral_onlineRegret_le lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G (gradientDescent (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) (y : E) (n : ℕ) (hn : n ≠ 0) (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : From 150ca459a3f58f712ce7b9438bdcd1192a8dc986 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sun, 3 May 2026 10:01:02 +0200 Subject: [PATCH 14/43] reorganize --- .../Algorithms/GradientDescent.lean | 189 +++++++++++------- 1 file changed, 118 insertions(+), 71 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 37f4bdcd..5109f5fb 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -195,6 +195,8 @@ end Convex section OnlineRegret +/-- The regret of a sequence `x : ℕ → E` compared to a point `y : E` in an online learning task +with losses `ℓ : ℕ → E → F`. -/ def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) @@ -225,6 +227,19 @@ end OnlineRegret variable [SecondCountableTopology E] [CompleteSpace E] {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} +omit [MeasurableSub₂ E] in +lemma condExp_reward_obliviousEnv_ae_eq_integral_id {alg : Algorithm E E} + {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (n : ℕ) (h_int : Integrable (G n) P) : + P[G n | MeasurableSpace.comap (X n) inferInstance] =ᵐ[P] fun ω ↦ (ν n (X n ω))[id] := by + have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_reward_obliviousEnv n + have h_ae := ae_of_ae_map (h.measurable_A n).aemeasurable h_obl.condDistrib_eq + have h_ae' := condExp_ae_eq_integral_condDistrib' (h.measurable_A n) h_int + filter_upwards [h_ae, h_ae'] with ω hω hω' + rw [hω', hω] + congr + /-- Online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`, without projection. @@ -273,35 +288,69 @@ lemma memLp_X (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (oblivious refine h_sub.ae_eq ?_ filter_upwards [action_gradientStep_ae_all_eq h] with ω hω using (hω.2 n).symm -lemma condExp_reward_obliviousEnv_ae_eq_integral_id {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] - (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv ν) P) - (n : ℕ) (h_int : Integrable (G n) P) : - P[G n | MeasurableSpace.comap (X n) inferInstance] =ᵐ[P] fun ω ↦ (ν n (X n ω))[id] := by - have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_reward_obliviousEnv n - have h_ae := ae_of_ae_map (h.measurable_A n).aemeasurable h_obl.condDistrib_eq - have h_ae' := condExp_ae_eq_integral_condDistrib' (h.measurable_A n) h_int - filter_upwards [h_ae, h_ae'] with ω hω hω' - rw [hω', hω] - congr +omit [MeasurableSub₂ E] in +lemma memLp_gradient {alg : Algorithm E E} + (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : + MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by + let M n := MeasurableSpace.comap (X n) inferInstance + have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) + have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n + ((h_memLp n).integrable (by simp)) + refine h_lp.ae_eq <| h_ae.trans ?_ + simp_rw [← h_unbiased] + rfl -lemma sfdsf (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) +section Linear + +lemma qsfqqfqgs'' (hη : 0 < η) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : + P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + calc P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] + _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 ∂P := by + refine integral_mono_ae ?_ ?_ ?_ + · refine integrable_finset_sum _ fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) + exact (memLp_X h h_memLp i).sub (memLp_const _) + · refine Integrable.add (integrable_const _) (Integrable.const_mul ?_ _) + exact integrable_finset_sum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) + · filter_upwards [action_gradientStep_ae_all_eq h] with ω hω + refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ + congr + exact hω.1 + _ = (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + rw [integral_add, integral_const_mul, integral_const_mul, integral_finset_sum] + · simp + · exact fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) + · exact integrable_const _ + · refine Integrable.const_mul ?_ _ + exact integrable_finset_sum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) + +end Linear + +section OnlineToBatch + +variable {alg : Algorithm E E} + +omit [MeasurableSub₂ E] in +lemma sfdsf (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by let M n := MeasurableSpace.comap (X n) inferInstance have h_obl : HasCondDistrib (G n) (X n) (gradKernel n) P := h.hasCondDistrib_reward_obliviousEnv n calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | M n] ω] := by - rw [integral_condExp] - exact (h.measurable_A _).comap_le + rw [integral_condExp (h.measurable_A _).comap_le] _ = P[fun ω ↦ ⟪X n ω - y, P[G n | M n] ω⟫] := by refine integral_congr_ae ?_ refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ · refine StronglyMeasurable.sub ?_ (by fun_prop) refine Measurable.stronglyMeasurable ?_ rw [measurable_iff_comap_le] - · refine MemLp.integrable_inner (MemLp.sub ?_ (memLp_const _)) (h_memLp n) - exact memLp_X h h_memLp n + · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) · exact (h_memLp n).integrable (by simp) _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n @@ -310,45 +359,63 @@ lemma sfdsf (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEn filter_upwards [h_ae] with ω hω using by rw [hω] _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] -lemma memLp_gradient - (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : - MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by - let M n := MeasurableSpace.comap (X n) inferInstance - have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) - have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n - ((h_memLp n).integrable (by simp)) - refine h_lp.ae_eq <| h_ae.trans ?_ - simp_rw [← h_unbiased] - rfl - +omit [MeasurableSub₂ E] in lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption (y : E) (n : ℕ) : P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by - rw [sfdsf h h_unbiased h_memLp y n] + rw [sfdsf h h_unbiased h_memLp hX_lp y n] gcongr - · refine Integrable.sub ?_ (integrable_const _) - exact h_int n + · exact (h_int n).sub (integrable_const _) · refine MemLp.integrable_inner ?_ ?_ - · exact (memLp_X h h_memLp n).sub (memLp_const _) + · exact (hX_lp n).sub (memLp_const _) · exact memLp_gradient h h_unbiased h_memLp n · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y --- use the deterministic equality wrt any sequence -lemma todo1 (hη : 0 < η) - (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) +omit [MeasurableSub₂ E] in +lemma qsfqqfqgs' (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : - ∀ᵐ ω ∂P, ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫ ≤ - (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 := by - filter_upwards [action_gradientStep_ae_all_eq h] with ω hω - refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ - congr - exact hω.1 + P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ + P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + rw [integral_finset_sum, integral_finset_sum] + rotate_left + · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) + exact (hX_lp i).sub (memLp_const _) + · exact fun i hi ↦ (h_int i).sub (integrable_const _) + refine Finset.sum_le_sum fun i hi ↦ ?_ + exact qfqgs hf hdf h_unbiased h_memLp h hX_lp h_int y i + +omit [MeasurableSub₂ E] in +lemma qsfqqfqgs'''' {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) + (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) + (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) + (y : E) (n : ℕ) (hn : n ≠ 0) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ + (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by + rw [← integral_const_mul] + gcongr + · exact h_int_avg.sub (integrable_const _) + · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ + exact (h_int i).sub (integrable_const _) + exact fun ω ↦ todo'3 hf _ y n hn + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + grw [qsfqqfqgs' (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h hX_lp h_int y n] + +end OnlineToBatch lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) @@ -359,31 +426,10 @@ lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentia P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by calc P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] - _ ≤ P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - rw [integral_finset_sum, integral_finset_sum] - rotate_left - · intro i hi - refine MemLp.integrable_inner ?_ (h_memLp i) - exact (memLp_X h h_memLp i).sub (memLp_const _) - · exact fun i hi ↦ (h_int i).sub (integrable_const _) - refine Finset.sum_le_sum fun i hi ↦ ?_ - exact qfqgs hf hdf h_unbiased h_memLp h h_int y i - _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 ∂P := by - refine integral_mono_ae ?_ ?_ (todo1 hη h y n) - · refine integrable_finset_sum _ fun i hi ↦ ?_ - refine MemLp.integrable_inner ?_ (h_memLp i) - exact (memLp_X h h_memLp i).sub (memLp_const _) - · refine Integrable.add (integrable_const _) (Integrable.const_mul ?_ _) - refine integrable_finset_sum _ fun i hi ↦ ?_ - exact (h_memLp i).integrable_norm_pow (by simp) - _ = (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - rw [integral_add, integral_const_mul, integral_const_mul, integral_finset_sum] - · simp - · exact fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) - · exact integrable_const _ - · refine Integrable.const_mul ?_ _ - refine integrable_finset_sum _ fun i hi ↦ ?_ - exact (h_memLp i).integrable_norm_pow (by simp) + _ ≤ P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := + qsfqqfqgs' hf hdf h_unbiased h_memLp h (memLp_X h h_memLp) h_int y n + _ ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := + qsfqqfqgs'' hη h_memLp h y n lemma integral_onlineRegret_le (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) @@ -396,7 +442,8 @@ lemma integral_onlineRegret_le (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := qsfqqfqgs hf hdf hη h_unbiased h_memLp h h_int y n -lemma qsfqgzr {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (hη : 0 < η) +lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) From 57d3b05737bb304d7e64fe6b043e187be87dfa2d Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 9 May 2026 15:59:17 +0200 Subject: [PATCH 15/43] fix --- .../Optimization/Algorithms/GradientDescent.lean | 11 ++++++----- .../SequentialLearning/StationaryEnv.lean | 10 +++++----- 2 files changed, 11 insertions(+), 10 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 5109f5fb..597d46b7 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -233,9 +233,9 @@ lemma condExp_reward_obliviousEnv_ae_eq_integral_id {alg : Algorithm E E} (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) (n : ℕ) (h_int : Integrable (G n) P) : P[G n | MeasurableSpace.comap (X n) inferInstance] =ᵐ[P] fun ω ↦ (ν n (X n ω))[id] := by - have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_reward_obliviousEnv n - have h_ae := ae_of_ae_map (h.measurable_A n).aemeasurable h_obl.condDistrib_eq - have h_ae' := condExp_ae_eq_integral_condDistrib' (h.measurable_A n) h_int + have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_feedback_obliviousEnv n + have h_ae := ae_of_ae_map (h.measurable_action n).aemeasurable h_obl.condDistrib_eq + have h_ae' := condExp_ae_eq_integral_condDistrib' (h.measurable_action n) h_int filter_upwards [h_ae, h_ae'] with ω hω hω' rw [hω', hω] congr @@ -340,10 +340,11 @@ lemma sfdsf (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by let M n := MeasurableSpace.comap (X n) inferInstance - have h_obl : HasCondDistrib (G n) (X n) (gradKernel n) P := h.hasCondDistrib_reward_obliviousEnv n + have h_obl : HasCondDistrib (G n) (X n) (gradKernel n) P := + h.hasCondDistrib_feedback_obliviousEnv n calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | M n] ω] := by - rw [integral_condExp (h.measurable_A _).comap_le] + rw [integral_condExp (h.measurable_action _).comap_le] _ = P[fun ω ↦ ⟪X n ω - y, P[G n | M n] ω⟫] := by refine integral_congr_ae ?_ refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ diff --git a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean index 9bdfaea1..85a78317 100644 --- a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean +++ b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean @@ -78,14 +78,14 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {A : ℕ → Ω → 𝓐} {Y : ℕ → Ω → 𝓨} {n N : ℕ} {ν : ℕ → Kernel 𝓐 𝓨} [∀ n, IsMarkovKernel (ν n)] -lemma hasCOndDistrib_reward_hist_action [IsObliviousEnv env] +lemma hasCondDistrib_feedback_hist_action [IsObliviousEnv env] (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) : HasCondDistrib (Y (n + 1)) (fun ω ↦ (IsAlgEnvSeq.hist A Y n ω, A (n + 1) ω)) ((feedbackCondAction env (n + 1)).prodMkLeft _) P := by have hA := h.measurable_action have hR' := h.measurable_feedback refine ⟨by fun_prop, by fun_prop, ?_⟩ - have h_eq := (h.hasCondDistrib_reward n).condDistrib_eq + have h_eq := (h.hasCondDistrib_feedback n).condDistrib_eq rw [condDistrib_ae_eq_iff_measure_eq_compProd _ (by fun_prop)] at h_eq ⊢ simpa only [feedback_eq_feedbackCondAction] using h_eq @@ -209,11 +209,11 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} namespace IsAlgEnvSeq -/-- The conditional distribution of the reward at time `n` given the action at time `n` is `ν`. -/ -lemma hasCondDistrib_reward_obliviousEnv {ν : ℕ → Kernel α R} [∀ n, IsMarkovKernel (ν n)] +/-- The conditional distribution of the feedback at time `n` given the action at time `n` is `ν`. -/ +lemma hasCondDistrib_feedback_obliviousEnv {ν : ℕ → Kernel 𝓐 𝓨} [∀ n, IsMarkovKernel (ν n)] (h : IsAlgEnvSeq A Y alg (obliviousEnv ν) P) (n : ℕ) : HasCondDistrib (Y n) (A n) (ν n) P := by - simpa using IsObliviousEnv.hasCondDistrib_reward h n + simpa using IsObliviousEnv.hasCondDistrib_feedback h n /-- The conditional distribution of the feedback at time `n` given the action at time `n` is `ν`. -/ lemma hasCondDistrib_feedback_stationaryEnv From 21f1ee5dbd8f489dde1036c6933f2428b32bc730 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 11 May 2026 09:21:22 +0200 Subject: [PATCH 16/43] move lemmas --- .../ConditionalExpectation/PullOut.lean | 30 +++++++++++ .../MeasureTheory/Function/L2Space.lean | 50 +++++++++++++++++++ .../Algorithms/GradientDescent.lean | 44 +--------------- 3 files changed, 82 insertions(+), 42 deletions(-) create mode 100644 LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean create mode 100644 LeanMachineLearning/MeasureTheory/Function/L2Space.lean diff --git a/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean b/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean new file mode 100644 index 00000000..b60cc88b --- /dev/null +++ b/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean @@ -0,0 +1,30 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import Mathlib.MeasureTheory.Function.ConditionalExpectation.PullOut + +/-! +# Integrability of inner products +-/ + +@[expose] public section + +open scoped ENNReal RealInnerProductSpace + +namespace MeasureTheory + +variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {mE : MeasurableSpace E} {P : Measure Ω} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [CompleteSpace E] + +theorem condExp_inner_of_stronglyMeasurable_left {Ω : Type*} {m mΩ : MeasurableSpace Ω} + {μ : Measure Ω} {X g : Ω → E} + (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : + μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by + filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω + simpa [innerSL_apply_apply] using hω + +end MeasureTheory diff --git a/LeanMachineLearning/MeasureTheory/Function/L2Space.lean b/LeanMachineLearning/MeasureTheory/Function/L2Space.lean new file mode 100644 index 00000000..e7b10f5b --- /dev/null +++ b/LeanMachineLearning/MeasureTheory/Function/L2Space.lean @@ -0,0 +1,50 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import Mathlib.MeasureTheory.Function.L2Space + +/-! +# Integrability of inner products +-/ + +@[expose] public section + +open scoped ENNReal RealInnerProductSpace + +namespace MeasureTheory + +variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {mE : MeasurableSpace E} {P : Measure Ω} + +lemma MemLp.eLpNorm_rpow_norm_lt_top [SeminormedAddCommGroup E] + {f : Ω → E} {p : ℝ≥0∞} + (hf : MemLp f p P) (hp_zero : p ≠ 0) (hp_top : p ≠ ∞) : + eLpNorm (fun x ↦ ‖f x‖ ^ p.toReal) 1 P < ∞ := by + simpa [eLpNorm_one_eq_lintegral_enorm, Real.enorm_rpow_of_nonneg] using + (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral + +lemma MemLp.integrable_inner [NormedAddCommGroup E] [InnerProductSpace ℝ E] + {f g : Ω → E} + (hf : MemLp f 2 P) (hg : MemLp g 2 P) : + Integrable (fun ω ↦ ⟪f ω, g ω⟫) P := by + rw [← memLp_one_iff_integrable] + constructor + · exact hf.aestronglyMeasurable.inner hg.aestronglyMeasurable + have h x : ‖⟪f x, g x⟫‖ ≤ ‖‖f x‖ ^ (2 : ℝ) + ‖g x‖ ^ (2 : ℝ)‖ := by + norm_cast + calc ‖⟪f x, g x⟫‖ ≤ ‖f x‖ * ‖g x‖ := norm_inner_le_norm _ _ + _ ≤ 2 * ‖f x‖ * ‖g x‖ := by + gcongr + exact le_mul_of_one_le_left (norm_nonneg _) one_le_two + _ ≤ ‖‖f x‖ ^ 2 + ‖g x‖ ^ 2‖ := (two_mul_le_add_sq _ _).trans (le_abs_self _) + refine (eLpNorm_mono h).trans_lt ((eLpNorm_add_le ?_ ?_ le_rfl).trans_lt ?_) + · exact (hf.norm.aemeasurable.pow_const _).aestronglyMeasurable + · exact (hg.norm.aemeasurable.pow_const _).aestronglyMeasurable + rw [ENNReal.add_lt_top] + exact ⟨hf.eLpNorm_rpow_norm_lt_top (by simp) (by simp), + hg.eLpNorm_rpow_norm_lt_top (by simp) (by simp)⟩ + +end MeasureTheory diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 597d46b7..8afe025e 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -5,6 +5,8 @@ Authors: Rémy Degenne -/ module +public import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut +public import LeanMachineLearning.MeasureTheory.Function.L2Space public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.SequentialLearning.EvaluationEnv public import Mathlib @@ -19,48 +21,6 @@ public import Mathlib open MeasureTheory ProbabilityTheory Filter Real Finset open scoped Gradient ENNReal NNReal RealInnerProductSpace -section Aux - -variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} {P : Measure Ω} - [NormedAddCommGroup E] - -theorem _root_.MeasureTheory.MemLp.eLpNorm_rpow_norm_lt_top - {f : Ω → E} {p : ℝ≥0∞} - (hf : MemLp f p P) (hp_zero : p ≠ 0) (hp_top : p ≠ ∞) : - eLpNorm (fun x ↦ ‖f x‖ ^ p.toReal) 1 P < ∞ := by - simpa [eLpNorm_one_eq_lintegral_enorm, enorm_rpow_of_nonneg] using - (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral - -lemma _root_.MeasureTheory.MemLp.integrable_inner [InnerProductSpace ℝ E] - {f g : Ω → E} - (hf : MemLp f 2 P) (hg : MemLp g 2 P) : - Integrable (fun ω ↦ ⟪f ω, g ω⟫) P := by - rw [← memLp_one_iff_integrable] - constructor - · exact hf.aestronglyMeasurable.inner hg.aestronglyMeasurable - have h x : ‖⟪f x, g x⟫‖ ≤ ‖‖f x‖ ^ (2 : ℝ) + ‖g x‖ ^ (2 : ℝ)‖ := by - norm_cast - calc ‖⟪f x, g x⟫‖ ≤ ‖f x‖ * ‖g x‖ := norm_inner_le_norm _ _ - _ ≤ 2 * ‖f x‖ * ‖g x‖ := by - gcongr - exact le_mul_of_one_le_left (norm_nonneg _) one_le_two - _ ≤ ‖‖f x‖ ^ 2 + ‖g x‖ ^ 2‖ := (two_mul_le_add_sq _ _).trans (le_abs_self _) - refine (eLpNorm_mono h).trans_lt ((eLpNorm_add_le ?_ ?_ le_rfl).trans_lt ?_) - · exact (hf.norm.aemeasurable.pow_const _).aestronglyMeasurable - · exact (hg.norm.aemeasurable.pow_const _).aestronglyMeasurable - rw [ENNReal.add_lt_top] - exact ⟨hf.eLpNorm_rpow_norm_lt_top (by simp) (by simp), - hg.eLpNorm_rpow_norm_lt_top (by simp) (by simp)⟩ - -theorem condExp_inner_of_stronglyMeasurable_left {Ω : Type*} {m mΩ : MeasurableSpace Ω} - [InnerProductSpace ℝ E] [CompleteSpace E] {μ : Measure Ω} {X g : Ω → E} - (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : - μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by - filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω - simpa [innerSL_apply_apply] using hω - -end Aux - namespace Learning variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} From b518cefdbe4ea2b87b5bb6d43aa6c9a956d92c91 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 11 May 2026 10:31:26 +0200 Subject: [PATCH 17/43] move a lemma --- .../Algorithms/GradientDescent.lean | 17 ++--------------- .../SequentialLearning/StationaryEnv.lean | 15 +++++++++++++++ 2 files changed, 17 insertions(+), 15 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 8afe025e..cb7efed3 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -187,19 +187,6 @@ end OnlineRegret variable [SecondCountableTopology E] [CompleteSpace E] {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} -omit [MeasurableSub₂ E] in -lemma condExp_reward_obliviousEnv_ae_eq_integral_id {alg : Algorithm E E} - {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (n : ℕ) (h_int : Integrable (G n) P) : - P[G n | MeasurableSpace.comap (X n) inferInstance] =ᵐ[P] fun ω ↦ (ν n (X n ω))[id] := by - have h_obl : HasCondDistrib (G n) (X n) (ν n) P := h.hasCondDistrib_feedback_obliviousEnv n - have h_ae := ae_of_ae_map (h.measurable_action n).aemeasurable h_obl.condDistrib_eq - have h_ae' := condExp_ae_eq_integral_condDistrib' (h.measurable_action n) h_int - filter_upwards [h_ae, h_ae'] with ω hω hω' - rw [hω', hω] - congr - /-- Online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`, without projection. @@ -256,7 +243,7 @@ lemma memLp_gradient {alg : Algorithm E E} MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by let M n := MeasurableSpace.comap (X n) inferInstance have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) - have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n + have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n ((h_memLp n).integrable (by simp)) refine h_lp.ae_eq <| h_ae.trans ?_ simp_rw [← h_unbiased] @@ -314,7 +301,7 @@ lemma sfdsf (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) · exact (h_memLp n).integrable (by simp) _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by - have h_ae := condExp_reward_obliviousEnv_ae_eq_integral_id h n + have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n ((h_memLp n).integrable (by simp)) refine integral_congr_ae ?_ filter_upwards [h_ae] with ω hω using by rw [hω] diff --git a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean index 85a78317..c5f12d58 100644 --- a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean +++ b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean @@ -227,6 +227,21 @@ lemma condDistrib_feedback_stationaryEnv condDistrib (Y n) (A n) P =ᵐ[P.map (A n)] ν := (hasCondDistrib_feedback_stationaryEnv h n).condDistrib_eq +-- todo: generalize to IsObliviousEnv +lemma condExp_feedback_obliviousEnv_ae_eq_integral_id {E : Type*} + [NormedAddCommGroup E] [NormedSpace ℝ E] [SecondCountableTopology E] [CompleteSpace E] + {mE : MeasurableSpace E} [BorelSpace E] + {alg : Algorithm 𝓐 E} {Y : ℕ → Ω → E} + {ν : ℕ → Kernel 𝓐 E} [∀ n, IsMarkovKernel (ν n)] + (h : IsAlgEnvSeq A Y alg (obliviousEnv ν) P) (n : ℕ) (h_int : Integrable (Y n) P) : + P[Y n | m𝓐.comap (A n)] =ᵐ[P] fun ω ↦ (ν n (A n ω))[id] := by + have h_obl : HasCondDistrib (Y n) (A n) (ν n) P := h.hasCondDistrib_feedback_obliviousEnv n + have h_ae : ∀ᵐ ω ∂P, 𝓛[Y n | A n; P] (A n ω) = ν n (A n ω) := + ae_of_ae_map (h.measurable_action n).aemeasurable h_obl.condDistrib_eq + have h_ae' : P[Y n | m𝓐.comap (A n)] =ᵐ[P] fun ω ↦ ∫ y, y ∂𝓛[Y n | A n; P] (A n ω) := + condExp_ae_eq_integral_condDistrib' (h.measurable_action n) h_int + filter_upwards [h_ae, h_ae'] with ω hω hω' using by simp [hω', hω] + /-- The feedback at time `n + 1` is conditionally independent of the history up to time `n` given the action at time `n + 1`. -/ lemma condIndepFun_feedback_hist_action [StandardBorelSpace Ω] From 031d32db4dc329c22b56671aae74ad78b9eab3dd Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 11 May 2026 11:03:41 +0200 Subject: [PATCH 18/43] move --- .../Algorithms/GradientDescent.lean | 365 +++++++++--------- 1 file changed, 185 insertions(+), 180 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index cb7efed3..d82020d4 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -21,64 +21,12 @@ public import Mathlib open MeasureTheory ProbabilityTheory Filter Real Finset open scoped Gradient ENNReal NNReal RealInnerProductSpace -namespace Learning - -variable {E Ω : Type*} {mE : MeasurableSpace E} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - [NormedAddCommGroup E] [InnerProductSpace ℝ E] [BorelSpace E] - [MeasurableSub₂ E] {x x₀ : E} - {g : ℕ → E → E} {hg : ∀ n, Measurable (g n)} - {env : Environment E E} - {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} - -section Linear - -lemma todo'' (x y g : E) (hη : 0 < η) : - ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by - have hsub : (x - η • g) - y = (x - y) - η • g := by abel - rw [hsub, norm_sub_sq_real (x - y) (η • g)] - simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] - field - -lemma todo (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + - (γ i / 2) * ‖g i‖ ^ 2) := by - gcongr with i hi - rw [todo'' (x i) (y i) (g i) (hγ i)] - -lemma todo_sfsq (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) - (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := - (todo x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] - -section ConstantStep - -lemma todo''' (x g : ℕ → E) (y : E) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - refine (todo_sfsq x g y (fun _ ↦ hη) hx n).trans_eq ?_ - rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] - -lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo''' x g y hη hx n] - gcongr - exact sub_le_self _ (sq_nonneg _) - -end ConstantStep +-- todo: move to another file +namespace ConvexOn -end Linear +variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x : E} -section Convex - -lemma _root_.ConvexOn.fderiv_sub_le_sub {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : fderiv ℝ f x (y - x) ≤ f y - f x := by have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : @@ -100,13 +48,13 @@ lemma _root_.ConvexOn.fderiv_sub_le_sub {f : E → ℝ} (hf : ConvexOn ℝ .univ simp [inv_mul_le_iff₀ ht.1] grind -lemma _root_.ConvexOn.add_fderiv_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x + fderiv ℝ f x (y - x) ≤ f y := by suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind exact hf.fderiv_sub_le_sub hfx y -lemma _root_.ConvexOn.add_inner_gradient_le [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x + ⟪y - x, ∇ f x⟫ ≤ f y := by have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by @@ -114,7 +62,7 @@ lemma _root_.ConvexOn.add_inner_gradient_le [CompleteSpace E] {f : E → ℝ} (h rw [← hfderiv] exact hf.add_fderiv_le hfx y -lemma _root_.ConvexOn.le_add_inner_gradient [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x ≤ f y + ⟪x - y, ∇ f x⟫ := by have h_add_le := hf.add_inner_gradient_le hfx y @@ -122,14 +70,14 @@ lemma _root_.ConvexOn.le_add_inner_gradient [CompleteSpace E] {f : E → ℝ} (h rw [show x - y = -(y - x) by abel, inner_neg_left] grind -lemma _root_.ConvexOn.sub_le_inner_gradient [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) (hfx : DifferentiableAt ℝ f x) (y : E) : f x - f y ≤ ⟪x - y, ∇ f x⟫ := by simp only [tsub_le_iff_right] rw [add_comm] exact hf.le_add_inner_gradient hfx y -lemma todo'3 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) +lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y @@ -142,8 +90,8 @@ lemma todo'3 {f : E → ℝ} (hf : ConvexOn ℝ .univ f) simp field -lemma todo'2 [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : +lemma todo'2 [InnerProductSpace ℝ E] [CompleteSpace E] + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn @@ -151,7 +99,161 @@ lemma todo'2 [CompleteSpace E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf gcongr exact hf.sub_le_inner_gradient hdf.differentiableAt y -end Convex +end ConvexOn + +namespace Learning + +variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] + +section OnlineToBatch + +variable {E : Type*} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [SecondCountableTopology E] [CompleteSpace E] + {mE : MeasurableSpace E} [BorelSpace E] + {X G : ℕ → Ω → E} {alg : Algorithm E E} + {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] + {f : ℕ → E → ℝ} + +-- todo: name +lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : + MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by + let M n := MeasurableSpace.comap (X n) inferInstance + have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) + have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n + ((h_memLp n).integrable (by simp)) + refine h_lp.ae_eq <| h_ae.trans ?_ + simp_rw [← h_unbiased] + rfl + +lemma integral_inner_eq_integral_inner_gradient + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : + P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by + have h_obl : HasCondDistrib (G n) (X n) (ν n) P := + h.hasCondDistrib_feedback_obliviousEnv n + calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] + _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | mE.comap (X n)] ω] := by + rw [integral_condExp (h.measurable_action _).comap_le] + _ = P[fun ω ↦ ⟪X n ω - y, P[G n | mE.comap (X n)] ω⟫] := by + refine integral_congr_ae ?_ + refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ + · refine StronglyMeasurable.sub ?_ (by fun_prop) + refine Measurable.stronglyMeasurable ?_ + rw [measurable_iff_comap_le] + · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) + · exact (h_memLp n).integrable (by simp) + _ = P[fun ω ↦ ⟪X n ω - y, (ν n (X n ω))[id]⟫] := by + refine integral_congr_ae ?_ + filter_upwards [h.condExp_feedback_obliviousEnv_ae_eq_integral_id n + ((h_memLp n).integrable (by simp))] with ω hω using by rw [hω] + _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] + +lemma integral_sub_le_integral_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption + (y : E) (n : ℕ) : + P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by + rw [integral_inner_eq_integral_inner_gradient h h_unbiased h_memLp hX_lp y n] + gcongr + · exact (h_int n).sub (integrable_const _) + · refine MemLp.integrable_inner ?_ ?_ + · exact (hX_lp n).sub (memLp_const _) + · exact memLp_gradient h h_unbiased h_memLp n + · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y + +lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) + (y : E) (n : ℕ) : + P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ + P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + rw [integral_finset_sum, integral_finset_sum] + rotate_left + · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) + exact (hX_lp i).sub (memLp_const _) + · exact fun i hi ↦ (h_int i).sub (integrable_const _) + refine Finset.sum_le_sum fun i hi ↦ ?_ + exact integral_sub_le_integral_inner hf hdf h_unbiased h_memLp h hX_lp h_int y i + +lemma integral_apply_avg_sub_le_integral_sum_sub + {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) + (y : E) (n : ℕ) (hn : n ≠ 0) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ + (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by + rw [← integral_const_mul] + gcongr + · exact h_int_avg.sub (integrable_const _) + · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ + exact (h_int i).sub (integrable_const _) + exact fun ω ↦ hf.todo'3 _ y n hn + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + grw [integral_sum_sub_le_integral_sum_inner (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h + hX_lp h_int y n] + +end OnlineToBatch + +variable {E : Type*} {mE : MeasurableSpace E} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [BorelSpace E] + {x x₀ : E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} + +section Linear + +lemma todo'' (x y g : E) (hη : 0 < η) : + ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by + have hsub : (x - η • g) - y = (x - y) - η • g := by abel + rw [hsub, norm_sub_sq_real (x - y) (η • g)] + simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] + field + +lemma todo (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + + (γ i / 2) * ‖g i‖ ^ 2) := by + gcongr with i hi + rw [todo'' (x i) (y i) (g i) (hγ i)] + +lemma todo_sfsq (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) + (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := + (todo x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] + +section ConstantStep + +lemma todo''' (x g : ℕ → E) (y : E) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + refine (todo_sfsq x g y (fun _ ↦ hη) hx n).trans_eq ?_ + rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] + +lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + grw [todo''' x g y hη hx n] + gcongr + exact sub_le_self _ (sq_nonneg _) + +end ConstantStep + +end Linear section OnlineRegret @@ -174,7 +276,7 @@ lemma onlineRegret_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : ℕ → lemma apply_avg_sub_le_onlineRegret {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := - todo'3 hf x y n hn + hf.todo'3 x y n hn lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : @@ -187,6 +289,10 @@ end OnlineRegret variable [SecondCountableTopology E] [CompleteSpace E] {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} +section Definition + +variable {env : Environment E E} + /-- Online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`, without projection. @@ -210,15 +316,17 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P | zero => simpa | succ n ih => rw [hω n, Finset.sum_range_succ, ← sub_sub]; congr +end Definition + omit [SecondCountableTopology E] in lemma apply_avg_sub_le_onlineRegret_linearizedLoss {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by - simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using todo'2 hf hdf x y n hn + simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using hf.todo'2 hdf x y n hn -section Stochastic +namespace GradientStep variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] @@ -235,23 +343,9 @@ lemma memLp_X (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (oblivious refine h_sub.ae_eq ?_ filter_upwards [action_gradientStep_ae_all_eq h] with ω hω using (hω.2 n).symm -omit [MeasurableSub₂ E] in -lemma memLp_gradient {alg : Algorithm E E} - (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : - MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by - let M n := MeasurableSpace.comap (X n) inferInstance - have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) - have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n - ((h_memLp n).integrable (by simp)) - refine h_lp.ae_eq <| h_ae.trans ?_ - simp_rw [← h_unbiased] - rfl - section Linear -lemma qsfqqfqgs'' (hη : 0 < η) (h_memLp : ∀ n, MemLp (G n) 2 P) +lemma integral_sum_inner_le (hη : 0 < η) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] ≤ @@ -277,107 +371,18 @@ lemma qsfqqfqgs'' (hη : 0 < η) (h_memLp : ∀ n, MemLp (G n) 2 P) end Linear -section OnlineToBatch - -variable {alg : Algorithm E E} - -omit [MeasurableSub₂ E] in -lemma sfdsf (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : - P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by - let M n := MeasurableSpace.comap (X n) inferInstance - have h_obl : HasCondDistrib (G n) (X n) (gradKernel n) P := - h.hasCondDistrib_feedback_obliviousEnv n - calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] - _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | M n] ω] := by - rw [integral_condExp (h.measurable_action _).comap_le] - _ = P[fun ω ↦ ⟪X n ω - y, P[G n | M n] ω⟫] := by - refine integral_congr_ae ?_ - refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ - · refine StronglyMeasurable.sub ?_ (by fun_prop) - refine Measurable.stronglyMeasurable ?_ - rw [measurable_iff_comap_le] - · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) - · exact (h_memLp n).integrable (by simp) - _ = P[fun ω ↦ ⟪X n ω - y, (gradKernel n (X n ω))[id]⟫] := by - have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n - ((h_memLp n).integrable (by simp)) - refine integral_congr_ae ?_ - filter_upwards [h_ae] with ω hω using by rw [hω] - _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] - -omit [MeasurableSub₂ E] in -lemma qfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) +lemma integral_sum_sub_le (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) + (hη : 0 < η) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) - (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption - (y : E) (n : ℕ) : - P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by - rw [sfdsf h h_unbiased h_memLp hX_lp y n] - gcongr - · exact (h_int n).sub (integrable_const _) - · refine MemLp.integrable_inner ?_ ?_ - · exact (hX_lp n).sub (memLp_const _) - · exact memLp_gradient h h_unbiased h_memLp n - · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y - -omit [MeasurableSub₂ E] in -lemma qsfqqfqgs' (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) - (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) - (y : E) (n : ℕ) : - P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ - P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - rw [integral_finset_sum, integral_finset_sum] - rotate_left - · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) - exact (hX_lp i).sub (memLp_const _) - · exact fun i hi ↦ (h_int i).sub (integrable_const _) - refine Finset.sum_le_sum fun i hi ↦ ?_ - exact qfqgs hf hdf h_unbiased h_memLp h hX_lp h_int y i - -omit [MeasurableSub₂ E] in -lemma qsfqqfqgs'''' {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) - (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv gradKernel) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) - (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) - (y : E) (n : ℕ) (hn : n ≠ 0) - (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : - P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ - (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by - rw [← integral_const_mul] - gcongr - · exact h_int_avg.sub (integrable_const _) - · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ - exact (h_int i).sub (integrable_const _) - exact fun ω ↦ todo'3 hf _ y n hn - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - grw [qsfqqfqgs' (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h hX_lp h_int y n] - -end OnlineToBatch - -lemma qsfqqfqgs (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) - (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) - (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) - (y : E) (n : ℕ) : + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by calc P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] _ ≤ P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := - qsfqqfqgs' hf hdf h_unbiased h_memLp h (memLp_X h h_memLp) h_int y n + integral_sum_sub_le_integral_sum_inner hf hdf h_unbiased h_memLp h (memLp_X h h_memLp) h_int y n _ ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := - qsfqqfqgs'' hη h_memLp h y n + integral_sum_inner_le hη h_memLp h y n lemma integral_onlineRegret_le (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (hη : 0 < η) @@ -388,7 +393,7 @@ lemma integral_onlineRegret_le (y : E) (n : ℕ) : P[fun ω ↦ onlineRegret f y (X · ω) n] ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := - qsfqqfqgs hf hdf hη h_unbiased h_memLp h h_int y n + integral_sum_sub_le hf hdf hη h_unbiased h_memLp h h_int y n lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (hη : 0 < η) @@ -408,13 +413,13 @@ lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : D · exact h_int_avg.sub (integrable_const _) · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ exact (h_int i).sub (integrable_const _) - exact fun ω ↦ todo'3 hf _ y n hn + exact fun ω ↦ hf.todo'3 _ y n hn _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - grw [qsfqqfqgs (fun _ ↦ hf) (fun _ ↦ hdf) hη h_unbiased h_memLp h h_int y n] + grw [integral_sum_sub_le (fun _ ↦ hf) (fun _ ↦ hdf) hη h_unbiased h_memLp h h_int y n] refine le_of_eq ?_ field -end Stochastic +end GradientStep end Learning From 84cfe5d155d1243f74cce282252b0970ffee3273 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 11 May 2026 11:05:49 +0200 Subject: [PATCH 19/43] move --- .../Optimization/Algorithms/GradientDescent.lean | 15 +++++++-------- 1 file changed, 7 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index d82020d4..dea82ec4 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -284,6 +284,13 @@ lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by simpa [onlineRegret, inner_sub_left] using lem14dot1 x g y η hη hx n +lemma apply_avg_sub_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : E → ℝ} + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ + (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by + simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using hf.todo'2 hdf x y n hn + end OnlineRegret variable [SecondCountableTopology E] [CompleteSpace E] @@ -318,14 +325,6 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P end Definition -omit [SecondCountableTopology E] in -lemma apply_avg_sub_le_onlineRegret_linearizedLoss {f : E → ℝ} - (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ - (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by - simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using hf.todo'2 hdf x y n hn - namespace GradientStep variable {gradKernel : ℕ → Kernel E E} [∀ n, IsMarkovKernel (gradKernel n)] From 64eb80063db77b2bc31751dd4fc5494ef4ca17b9 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 11 May 2026 11:36:27 +0200 Subject: [PATCH 20/43] add sqrt bound --- .../Algorithms/GradientDescent.lean | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index dea82ec4..0dda6c7b 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -419,6 +419,47 @@ lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : D refine le_of_eq ?_ field +lemma integral_apply_avg_const_div_sqrt {f : E → ℝ} + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) + {D L : ℝ} (hD_pos : 0 < D) (hL_pos : 0 < L) + {y : E} (hxy_le : ‖x₀ - y‖ ≤ D) (hG_le : ∀ n ω, ‖G n ω‖ ≤ L) + (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) + {n : ℕ} (hn : n ≠ 0) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) + (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ D / (L * √n)) x₀) (obliviousEnv gradKernel) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ D * L / √n := by + let η := D / (L * √n) + have hG_lp n : MemLp (G n) 2 P := by + refine MemLp.mono (g := fun _ ↦ L) (memLp_const _) + (have := h.measurable_feedback; by fun_prop) (ae_of_all _ fun ω ↦ ?_) + simpa [abs_of_nonneg hL_pos.le] using hG_le n ω + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] + _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + + (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + refine integral_apply_avg_le hf hdf ?_ h_unbiased hG_lp h h_int y n hn h_int_avg + positivity + _ ≤ (2 * η * n)⁻¹ * D ^ 2 + (η / 2) * L ^ 2 := by + gcongr 1 + · gcongr + · field_simp + rw [mul_assoc] + gcongr + calc ∑ x ∈ range n, ∫ ω, ‖G x ω‖ ^ 2 ∂P + _ ≤ ∑ x ∈ range n, ∫ ω, L ^ 2 ∂P := by + gcongr with i hi + · exact (hG_lp i).integrable_norm_pow (by simp) + · simp + intro ω + mono + positivity + _ = n * L ^ 2 := by simp + _ = D * L / √n := by + simp only [mul_inv_rev, inv_div, η] + field_simp + rw [Real.sq_sqrt (by positivity)] + ring + end GradientStep end Learning From 282b1f0da4192b8c78db626f52842074930b280b Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 11 May 2026 20:48:52 +0200 Subject: [PATCH 21/43] minor --- .../Optimization/Algorithms/GradientDescent.lean | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 0dda6c7b..cfc589a4 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -124,8 +124,7 @@ lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n ((h_memLp n).integrable (by simp)) refine h_lp.ae_eq <| h_ae.trans ?_ - simp_rw [← h_unbiased] - rfl + simp [← h_unbiased] lemma integral_inner_eq_integral_inner_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) From bc2161a178f9555062649af830d866069944cc2b Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Tue, 9 Jun 2026 15:04:34 +0200 Subject: [PATCH 22/43] fix --- LeanMachineLearning.lean | 2 ++ .../Optimization/Algorithms/GradientDescent.lean | 14 +++++++------- 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index e4947194..15e6b58a 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -2,6 +2,8 @@ module -- shake: keep-all public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel +public import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut +public import LeanMachineLearning.MeasureTheory.Function.L2Space public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.UCB diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index cfc589a4..bbaf8590 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -174,7 +174,7 @@ lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f (y : E) (n : ℕ) : P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - rw [integral_finset_sum, integral_finset_sum] + rw [integral_finsetSum, integral_finsetSum] rotate_left · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) exact (hX_lp i).sub (memLp_const _) @@ -196,7 +196,7 @@ lemma integral_apply_avg_sub_le_integral_sum_sub rw [← integral_const_mul] gcongr · exact h_int_avg.sub (integrable_const _) - · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ + · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ exact (h_int i).sub (integrable_const _) exact fun ω ↦ hf.todo'3 _ y n hn _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by @@ -351,21 +351,21 @@ lemma integral_sum_inner_le (hη : 0 < η) (h_memLp : ∀ n, MemLp (G n) 2 P) calc P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 ∂P := by refine integral_mono_ae ?_ ?_ ?_ - · refine integrable_finset_sum _ fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) + · refine integrable_finsetSum _ fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) exact (memLp_X h h_memLp i).sub (memLp_const _) · refine Integrable.add (integrable_const _) (Integrable.const_mul ?_ _) - exact integrable_finset_sum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) + exact integrable_finsetSum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) · filter_upwards [action_gradientStep_ae_all_eq h] with ω hω refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ congr exact hω.1 _ = (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - rw [integral_add, integral_const_mul, integral_const_mul, integral_finset_sum] + rw [integral_add, integral_const_mul, integral_const_mul, integral_finsetSum] · simp · exact fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) · exact integrable_const _ · refine Integrable.const_mul ?_ _ - exact integrable_finset_sum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) + exact integrable_finsetSum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) end Linear @@ -409,7 +409,7 @@ lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : D rw [← integral_const_mul] gcongr · exact h_int_avg.sub (integrable_const _) - · refine Integrable.const_mul (integrable_finset_sum _ fun i hi ↦ ?_) _ + · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ exact (h_int i).sub (integrable_const _) exact fun ω ↦ hf.todo'3 _ y n hn _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + From 9cb9174d03d09d491b4bf25c4d4b95e7019c1dfb Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Tue, 9 Jun 2026 15:19:09 +0200 Subject: [PATCH 23/43] fix --- .../Function/ConditionalExpectation/PullOut.lean | 4 +++- LeanMachineLearning/MeasureTheory/Function/L2Space.lean | 5 +++-- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean b/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean index b60cc88b..bf95f787 100644 --- a/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean +++ b/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean @@ -25,6 +25,8 @@ theorem condExp_inner_of_stronglyMeasurable_left {Ω : Type*} {m mΩ : Measurabl (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by filter_upwards [condExp_bilin_of_stronglyMeasurable_left (innerSL ℝ) hX hXg hg] with ω hω - simpa [innerSL_apply_apply] using hω + convert hω + · rfl + · rfl end MeasureTheory diff --git a/LeanMachineLearning/MeasureTheory/Function/L2Space.lean b/LeanMachineLearning/MeasureTheory/Function/L2Space.lean index e7b10f5b..fec73192 100644 --- a/LeanMachineLearning/MeasureTheory/Function/L2Space.lean +++ b/LeanMachineLearning/MeasureTheory/Function/L2Space.lean @@ -23,8 +23,9 @@ lemma MemLp.eLpNorm_rpow_norm_lt_top [SeminormedAddCommGroup E] {f : Ω → E} {p : ℝ≥0∞} (hf : MemLp f p P) (hp_zero : p ≠ 0) (hp_top : p ≠ ∞) : eLpNorm (fun x ↦ ‖f x‖ ^ p.toReal) 1 P < ∞ := by - simpa [eLpNorm_one_eq_lintegral_enorm, Real.enorm_rpow_of_nonneg] using - (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral + simp only [eLpNorm_one_eq_lintegral_enorm, norm_nonneg, ENNReal.toReal_nonneg, + Real.enorm_rpow_of_nonneg, enorm_norm] + exact (hf.integrable_enorm_rpow hp_zero hp_top).hasFiniteIntegral lemma MemLp.integrable_inner [NormedAddCommGroup E] [InnerProductSpace ℝ E] {f g : Ω → E} From b5f90315bcbfc99f63347990f82598f135fd3781 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Thu, 18 Jun 2026 10:56:02 +0200 Subject: [PATCH 24/43] fix --- LeanMachineLearning.lean | 3 +++ LeanMachineLearning/SequentialLearning/StationaryEnv.lean | 4 ++-- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 9f11b08c..15c05291 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -18,6 +18,8 @@ public import LeanMachineLearning.ForMathlib.Probability.Kernel.IonescuTulcea.Tr public import LeanMachineLearning.ForMathlib.Probability.Kernel.KernelSub public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity +public import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut +public import LeanMachineLearning.MeasureTheory.Function.L2Space public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace @@ -25,6 +27,7 @@ public import LeanMachineLearning.Online.Bandit.BayesRegret public import LeanMachineLearning.Online.Bandit.Regret public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.Optimization.Algorithms.GradientDescent public import LeanMachineLearning.SequentialLearning.Algorithm public import LeanMachineLearning.SequentialLearning.AlgorithmDensity public import LeanMachineLearning.SequentialLearning.AlgorithmDensityBayes diff --git a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean index 0d9f2710..a6651e74 100644 --- a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean +++ b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean @@ -78,9 +78,9 @@ variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {A : ℕ → Ω → 𝓐} {Y : ℕ → Ω → 𝓨} {n N : ℕ} {ν : ℕ → Kernel 𝓐 𝓨} [∀ n, IsMarkovKernel (ν n)] -lemma hasCondDistrib_feedback_hist_action [IsObliviousEnv env] +lemma hasCondDistrib_feedback_history_action [IsObliviousEnv env] (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) : - HasCondDistrib (Y (n + 1)) (fun ω ↦ (IsAlgEnvSeq.hist A Y n ω, A (n + 1) ω)) + HasCondDistrib (Y (n + 1)) (fun ω ↦ (history A Y n ω, A (n + 1) ω)) ((feedbackCondAction env (n + 1)).prodMkLeft _) P := by have hA := h.measurable_action have hR' := h.measurable_feedback From 4650f7f286aef8f1981b3dbc6266d24cc1ce93f6 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Thu, 18 Jun 2026 11:14:19 +0200 Subject: [PATCH 25/43] split file --- LeanMachineLearning/Online/OnlineRegret.lean | 298 ++++++++++++++++++ .../Algorithms/GradientDescent.lean | 285 +---------------- 2 files changed, 308 insertions(+), 275 deletions(-) create mode 100644 LeanMachineLearning/Online/OnlineRegret.lean diff --git a/LeanMachineLearning/Online/OnlineRegret.lean b/LeanMachineLearning/Online/OnlineRegret.lean new file mode 100644 index 00000000..21ceebf7 --- /dev/null +++ b/LeanMachineLearning/Online/OnlineRegret.lean @@ -0,0 +1,298 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import LeanMachineLearning.SequentialLearning.StationaryEnv +public import Mathlib.Analysis.Calculus.Gradient.Basic + +import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut +import LeanMachineLearning.MeasureTheory.Function.L2Space +import Mathlib.Analysis.Calculus.Deriv.Comp +import Mathlib.Analysis.Calculus.Deriv.Mul +import Mathlib.Analysis.Calculus.Deriv.Slope + +/-! +# Online gradient descent + +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset +open scoped Gradient ENNReal NNReal RealInnerProductSpace + +-- todo: move to another file +namespace ConvexOn + +variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x : E} + +lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + fderiv ℝ f x (y - x) ≤ f y - f x := by + have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : + f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by + have h1 : x + t • (y - x) = (1 - t) • x + t • y := by module + have h2 : f ((1 - t) • x + t • y) ≤ (1 - t) • f x + t • f y := + hf.2 (Set.mem_univ x) (Set.mem_univ y) (by grind) (by grind) (by simp) + simp only [smul_eq_mul] at h2 + grind + have h_path_deriv : HasDerivAt (fun t : ℝ ↦ f (x + t • (y - x))) + (fderiv ℝ f x (y - x)) 0 := by + have h1 : HasDerivAt (fun t : ℝ ↦ x + t • (y - x)) (y - x) 0 := by + simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) + have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by + simpa using hfx.hasFDerivAt + exact h2.comp_hasDerivAt _ h1 + refine le_of_tendsto h_path_deriv.tendsto_slope_zero_right (Filter.eventually_of_mem + (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht ↦ ?_) + simp [inv_mul_le_iff₀ ht.1] + grind + +lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x + fderiv ℝ f x (y - x) ≤ f y := by + suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind + exact hf.fderiv_sub_le_sub hfx y + +lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x + ⟪y - x, ∇ f x⟫ ≤ f y := by + have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by + simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] + rw [← hfderiv] + exact hf.add_fderiv_le hfx y + +lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x ≤ f y + ⟪x - y, ∇ f x⟫ := by + have h_add_le := hf.add_inner_gradient_le hfx y + have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by + rw [show x - y = -(y - x) by abel, inner_neg_left] + grind + +lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x - f y ≤ ⟪x - y, ∇ f x⟫ := by + simp only [tsub_le_iff_right] + rw [add_comm] + exact hf.le_add_inner_gradient hfx y + +lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by + calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y + _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by + simp_rw [smul_sum] + grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] + _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by + simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] + rw [← sum_mul] + simp + field + +lemma todo'2 [InnerProductSpace ℝ E] [CompleteSpace E] + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by + calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by + gcongr + exact hf.sub_le_inner_gradient hdf.differentiableAt y + +end ConvexOn + +namespace Learning + +variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] + +section OnlineToBatch + +variable {E : Type*} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [SecondCountableTopology E] [CompleteSpace E] + {mE : MeasurableSpace E} [BorelSpace E] + {X G : ℕ → Ω → E} {alg : Algorithm E E} + {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] + {f : ℕ → E → ℝ} + +-- todo: name +lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : + MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by + let M n := MeasurableSpace.comap (X n) inferInstance + have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) + have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n + ((h_memLp n).integrable (by simp)) + refine h_lp.ae_eq <| h_ae.trans ?_ + simp [← h_unbiased] + +lemma integral_inner_eq_integral_inner_gradient + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : + P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by + have h_obl : HasCondDistrib (G n) (X n) (ν n) P := + h.hasCondDistrib_feedback_obliviousEnv n + calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] + _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | mE.comap (X n)] ω] := by + rw [integral_condExp (h.measurable_action _).comap_le] + _ = P[fun ω ↦ ⟪X n ω - y, P[G n | mE.comap (X n)] ω⟫] := by + refine integral_congr_ae ?_ + refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ + · refine StronglyMeasurable.sub ?_ (by fun_prop) + refine Measurable.stronglyMeasurable ?_ + rw [measurable_iff_comap_le] + · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) + · exact (h_memLp n).integrable (by simp) + _ = P[fun ω ↦ ⟪X n ω - y, (ν n (X n ω))[id]⟫] := by + refine integral_congr_ae ?_ + filter_upwards [h.condExp_feedback_obliviousEnv_ae_eq_integral_id n + ((h_memLp n).integrable (by simp))] with ω hω using by rw [hω] + _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] + +lemma integral_sub_le_integral_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption + (y : E) (n : ℕ) : + P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by + rw [integral_inner_eq_integral_inner_gradient h h_unbiased h_memLp hX_lp y n] + gcongr + · exact (h_int n).sub (integrable_const _) + · refine MemLp.integrable_inner ?_ ?_ + · exact (hX_lp n).sub (memLp_const _) + · exact memLp_gradient h h_unbiased h_memLp n + · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y + +lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) + (y : E) (n : ℕ) : + P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ + P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + rw [integral_finsetSum, integral_finsetSum] + rotate_left + · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) + exact (hX_lp i).sub (memLp_const _) + · exact fun i hi ↦ (h_int i).sub (integrable_const _) + refine Finset.sum_le_sum fun i hi ↦ ?_ + exact integral_sub_le_integral_inner hf hdf h_unbiased h_memLp h hX_lp h_int y i + +lemma integral_apply_avg_sub_le_integral_sum_sub + {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) + (y : E) (n : ℕ) (hn : n ≠ 0) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ + (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by + rw [← integral_const_mul] + gcongr + · exact h_int_avg.sub (integrable_const _) + · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ + exact (h_int i).sub (integrable_const _) + exact fun ω ↦ hf.todo'3 _ y n hn + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by + grw [integral_sum_sub_le_integral_sum_inner (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h + hX_lp h_int y n] + +end OnlineToBatch + +variable {E : Type*} {mE : MeasurableSpace E} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [BorelSpace E] + {x x₀ : E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} + +section Linear + +lemma todo'' (x y g : E) (hη : 0 < η) : + ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by + have hsub : (x - η • g) - y = (x - y) - η • g := by abel + rw [hsub, norm_sub_sq_real (x - y) (η • g)] + simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] + field + +lemma todo (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + + (γ i / 2) * ‖g i‖ ^ 2) := by + gcongr with i hi + rw [todo'' (x i) (y i) (g i) (hγ i)] + +lemma todo_sfsq (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) + (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := + (todo x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] + +section ConstantStep + +lemma todo''' (x g : ℕ → E) (y : E) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + refine (todo_sfsq x g y (fun _ ↦ hη) hx n).trans_eq ?_ + rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] + +lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + grw [todo''' x g y hη hx n] + gcongr + exact sub_le_self _ (sq_nonneg _) + +end ConstantStep + +end Linear + +section OnlineRegret + +/-- The regret of a sequence `x : ℕ → E` compared to a point `y : E` in an online learning task +with losses `ℓ : ℕ → E → F`. -/ +def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := + ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) + +noncomputable def linearizedLoss [CompleteSpace E] (f : ℕ → E → ℝ) (x : ℕ → E) : ℕ → E → ℝ := + fun n y ↦ ⟪y, ∇ (f n) (x n)⟫ + +lemma onlineRegret_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : ℕ → E → ℝ} + (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) + (x : ℕ → E) (y : E) (n : ℕ) : + onlineRegret f y x n ≤ onlineRegret (linearizedLoss f x) y x n := by + simp only [onlineRegret, linearizedLoss, ← inner_sub_left] + gcongr with i hi + exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ + +lemma apply_avg_sub_le_onlineRegret {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := + hf.todo'3 x y n hn + +lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + onlineRegret (fun n x ↦ ⟪x, g n⟫) y x n ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + simpa [onlineRegret, inner_sub_left] using lem14dot1 x g y η hη hx n + +lemma apply_avg_sub_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : E → ℝ} + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ + (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by + simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using hf.todo'2 hdf x y n hn + +end OnlineRegret + +end Learning diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index bbaf8590..015f05bf 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -5,11 +5,10 @@ Authors: Rémy Degenne -/ module -public import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut -public import LeanMachineLearning.MeasureTheory.Function.L2Space +public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.SequentialLearning.EvaluationEnv -public import Mathlib + +import LeanMachineLearning.MeasureTheory.Function.L2Space /-! # Online gradient descent @@ -21,278 +20,13 @@ public import Mathlib open MeasureTheory ProbabilityTheory Filter Real Finset open scoped Gradient ENNReal NNReal RealInnerProductSpace --- todo: move to another file -namespace ConvexOn - -variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x : E} - -lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - fderiv ℝ f x (y - x) ≤ f y - f x := by - have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : - f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by - have h1 : x + t • (y - x) = (1 - t) • x + t • y := by module - have h2 : f ((1 - t) • x + t • y) ≤ (1 - t) • f x + t • f y := - hf.2 (Set.mem_univ x) (Set.mem_univ y) (by grind) (by grind) (by simp) - simp only [smul_eq_mul] at h2 - grind - have h_path_deriv : HasDerivAt (fun t : ℝ ↦ f (x + t • (y - x))) - (fderiv ℝ f x (y - x)) 0 := by - have h1 : HasDerivAt (fun t : ℝ ↦ x + t • (y - x)) (y - x) 0 := by - simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) - have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by - simpa using hfx.hasFDerivAt - exact h2.comp_hasDerivAt _ h1 - refine le_of_tendsto h_path_deriv.tendsto_slope_zero_right (Filter.eventually_of_mem - (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht ↦ ?_) - simp [inv_mul_le_iff₀ ht.1] - grind - -lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x + fderiv ℝ f x (y - x) ≤ f y := by - suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind - exact hf.fderiv_sub_le_sub hfx y - -lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x + ⟪y - x, ∇ f x⟫ ≤ f y := by - have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by - simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] - rw [← hfderiv] - exact hf.add_fderiv_le hfx y - -lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x ≤ f y + ⟪x - y, ∇ f x⟫ := by - have h_add_le := hf.add_inner_gradient_le hfx y - have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by - rw [show x - y = -(y - x) by abel, inner_neg_left] - grind - -lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x - f y ≤ ⟪x - y, ∇ f x⟫ := by - simp only [tsub_le_iff_right] - rw [add_comm] - exact hf.le_add_inner_gradient hfx y - -lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by - calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by - simp_rw [smul_sum] - grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] - _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by - simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] - rw [← sum_mul] - simp - field - -lemma todo'2 [InnerProductSpace ℝ E] [CompleteSpace E] - (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by - calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by - gcongr - exact hf.sub_le_inner_gradient hdf.differentiableAt y - -end ConvexOn - namespace Learning -variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] - -section OnlineToBatch - -variable {E : Type*} - [NormedAddCommGroup E] [InnerProductSpace ℝ E] [SecondCountableTopology E] [CompleteSpace E] - {mE : MeasurableSpace E} [BorelSpace E] - {X G : ℕ → Ω → E} {alg : Algorithm E E} - {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] - {f : ℕ → E → ℝ} - --- todo: name -lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : - MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by - let M n := MeasurableSpace.comap (X n) inferInstance - have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) - have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n - ((h_memLp n).integrable (by simp)) - refine h_lp.ae_eq <| h_ae.trans ?_ - simp [← h_unbiased] - -lemma integral_inner_eq_integral_inner_gradient - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : - P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by - have h_obl : HasCondDistrib (G n) (X n) (ν n) P := - h.hasCondDistrib_feedback_obliviousEnv n - calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] - _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | mE.comap (X n)] ω] := by - rw [integral_condExp (h.measurable_action _).comap_le] - _ = P[fun ω ↦ ⟪X n ω - y, P[G n | mE.comap (X n)] ω⟫] := by - refine integral_congr_ae ?_ - refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ - · refine StronglyMeasurable.sub ?_ (by fun_prop) - refine Measurable.stronglyMeasurable ?_ - rw [measurable_iff_comap_le] - · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) - · exact (h_memLp n).integrable (by simp) - _ = P[fun ω ↦ ⟪X n ω - y, (ν n (X n ω))[id]⟫] := by - refine integral_congr_ae ?_ - filter_upwards [h.condExp_feedback_obliviousEnv_ae_eq_integral_id n - ((h_memLp n).integrable (by simp))] with ω hω using by rw [hω] - _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] - -lemma integral_sub_le_integral_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) - (hdf : ∀ n, Differentiable ℝ (f n)) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) - (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption - (y : E) (n : ℕ) : - P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by - rw [integral_inner_eq_integral_inner_gradient h h_unbiased h_memLp hX_lp y n] - gcongr - · exact (h_int n).sub (integrable_const _) - · refine MemLp.integrable_inner ?_ ?_ - · exact (hX_lp n).sub (memLp_const _) - · exact memLp_gradient h h_unbiased h_memLp n - · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y - -lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) - (hdf : ∀ n, Differentiable ℝ (f n)) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) - (y : E) (n : ℕ) : - P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ - P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - rw [integral_finsetSum, integral_finsetSum] - rotate_left - · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) - exact (hX_lp i).sub (memLp_const _) - · exact fun i hi ↦ (h_int i).sub (integrable_const _) - refine Finset.sum_le_sum fun i hi ↦ ?_ - exact integral_sub_le_integral_inner hf hdf h_unbiased h_memLp h hX_lp h_int y i - -lemma integral_apply_avg_sub_le_integral_sum_sub - {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) - (y : E) (n : ℕ) (hn : n ≠ 0) - (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : - P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ - (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by - rw [← integral_const_mul] - gcongr - · exact h_int_avg.sub (integrable_const _) - · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ - exact (h_int i).sub (integrable_const _) - exact fun ω ↦ hf.todo'3 _ y n hn - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - grw [integral_sum_sub_le_integral_sum_inner (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h - hX_lp h_int y n] - -end OnlineToBatch - -variable {E : Type*} {mE : MeasurableSpace E} - [NormedAddCommGroup E] [InnerProductSpace ℝ E] [BorelSpace E] +variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {mE : MeasurableSpace E} + [NormedAddCommGroup E] [InnerProductSpace ℝ E] + [SecondCountableTopology E] [CompleteSpace E] [BorelSpace E] + {P : Measure Ω} [IsProbabilityMeasure P] {x x₀ : E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} - -section Linear - -lemma todo'' (x y g : E) (hη : 0 < η) : - ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by - have hsub : (x - η • g) - y = (x - y) - η • g := by abel - rw [hsub, norm_sub_sq_real (x - y) (η • g)] - simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] - field - -lemma todo (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + - (γ i / 2) * ‖g i‖ ^ 2) := by - gcongr with i hi - rw [todo'' (x i) (y i) (g i) (hγ i)] - -lemma todo_sfsq (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) - (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := - (todo x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] - -section ConstantStep - -lemma todo''' (x g : ℕ → E) (y : E) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - refine (todo_sfsq x g y (fun _ ↦ hη) hx n).trans_eq ?_ - rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] - -lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo''' x g y hη hx n] - gcongr - exact sub_le_self _ (sq_nonneg _) - -end ConstantStep - -end Linear - -section OnlineRegret - -/-- The regret of a sequence `x : ℕ → E` compared to a point `y : E` in an online learning task -with losses `ℓ : ℕ → E → F`. -/ -def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := - ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) - -noncomputable def linearizedLoss [CompleteSpace E] (f : ℕ → E → ℝ) (x : ℕ → E) : ℕ → E → ℝ := - fun n y ↦ ⟪y, ∇ (f n) (x n)⟫ - -lemma onlineRegret_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : ℕ → E → ℝ} - (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) - (x : ℕ → E) (y : E) (n : ℕ) : - onlineRegret f y x n ≤ onlineRegret (linearizedLoss f x) y x n := by - simp only [onlineRegret, linearizedLoss, ← inner_sub_left] - gcongr with i hi - exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ - -lemma apply_avg_sub_le_onlineRegret {f : E → ℝ} (hf : ConvexOn ℝ .univ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := - hf.todo'3 x y n hn - -lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - onlineRegret (fun n x ↦ ⟪x, g n⟫) y x n ≤ - (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - simpa [onlineRegret, inner_sub_left] using lem14dot1 x g y η hη hx n - -lemma apply_avg_sub_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : E → ℝ} - (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ - (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by - simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using hf.todo'2 hdf x y n hn - -end OnlineRegret - -variable [SecondCountableTopology E] [CompleteSpace E] {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} section Definition @@ -450,8 +184,9 @@ lemma integral_apply_avg_const_div_sqrt {f : E → ℝ} · exact (hG_lp i).integrable_norm_pow (by simp) · simp intro ω - mono - positivity + simp only + gcongr + exact hG_le _ _ _ = n * L ^ 2 := by simp _ = D * L / √n := by simp only [mul_inv_rev, inv_div, η] From 7dca3b5ae22f9718c68f83030a6dae87b8709dcb Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Thu, 18 Jun 2026 11:21:55 +0200 Subject: [PATCH 26/43] split file --- LeanMachineLearning.lean | 2 + .../Analysis/Calculus/Deriv/Slope.lean | 101 ++++++++++++++++++ LeanMachineLearning/Online/OnlineRegret.lean | 84 +-------------- .../Algorithms/GradientDescent.lean | 3 +- 4 files changed, 106 insertions(+), 84 deletions(-) create mode 100644 LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 15c05291..f29ae759 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,5 +1,6 @@ module -- shake: keep-all --deprecated_module: ignore +public import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.ForMathlib.MeasureTheory.Measurable @@ -27,6 +28,7 @@ public import LeanMachineLearning.Online.Bandit.BayesRegret public import LeanMachineLearning.Online.Bandit.Regret public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.Optimization.Algorithms.GradientDescent public import LeanMachineLearning.SequentialLearning.Algorithm public import LeanMachineLearning.SequentialLearning.AlgorithmDensity diff --git a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean new file mode 100644 index 00000000..db972e65 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean @@ -0,0 +1,101 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import Mathlib.Analysis.Calculus.Gradient.Basic + +import Mathlib.Analysis.Calculus.Deriv.Comp +import Mathlib.Analysis.Calculus.Deriv.Mul +import Mathlib.Analysis.Calculus.Deriv.Slope + +/-! +# Convexity lemmas + +-/ + +@[expose] public section + +open Finset +open scoped Gradient RealInnerProductSpace + +namespace ConvexOn + +variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x : E} + +lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + fderiv ℝ f x (y - x) ≤ f y - f x := by + have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : + f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by + have h1 : x + t • (y - x) = (1 - t) • x + t • y := by module + have h2 : f ((1 - t) • x + t • y) ≤ (1 - t) • f x + t • f y := + hf.2 (Set.mem_univ x) (Set.mem_univ y) (by grind) (by grind) (by simp) + simp only [smul_eq_mul] at h2 + grind + have h_path_deriv : HasDerivAt (fun t : ℝ ↦ f (x + t • (y - x))) + (fderiv ℝ f x (y - x)) 0 := by + have h1 : HasDerivAt (fun t : ℝ ↦ x + t • (y - x)) (y - x) 0 := by + simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) + have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by + simpa using hfx.hasFDerivAt + exact h2.comp_hasDerivAt _ h1 + refine le_of_tendsto h_path_deriv.tendsto_slope_zero_right (Filter.eventually_of_mem + (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht ↦ ?_) + simp [inv_mul_le_iff₀ ht.1] + grind + +lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x + fderiv ℝ f x (y - x) ≤ f y := by + suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind + exact hf.fderiv_sub_le_sub hfx y + +lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x + ⟪y - x, ∇ f x⟫ ≤ f y := by + have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by + simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] + rw [← hfderiv] + exact hf.add_fderiv_le hfx y + +lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x ≤ f y + ⟪x - y, ∇ f x⟫ := by + have h_add_le := hf.add_inner_gradient_le hfx y + have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by + rw [show x - y = -(y - x) by abel, inner_neg_left] + grind + +lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) + (hfx : DifferentiableAt ℝ f x) (y : E) : + f x - f y ≤ ⟪x - y, ∇ f x⟫ := by + simp only [tsub_le_iff_right] + rw [add_comm] + exact hf.le_add_inner_gradient hfx y + +lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by + calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y + _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by + simp_rw [smul_sum] + grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] + _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by + simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] + rw [← sum_mul] + simp + field + +lemma todo'2 [InnerProductSpace ℝ E] [CompleteSpace E] + (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by + calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by + gcongr + exact hf.sub_le_inner_gradient hdf.differentiableAt y + +end ConvexOn diff --git a/LeanMachineLearning/Online/OnlineRegret.lean b/LeanMachineLearning/Online/OnlineRegret.lean index 21ceebf7..6c9c9dc3 100644 --- a/LeanMachineLearning/Online/OnlineRegret.lean +++ b/LeanMachineLearning/Online/OnlineRegret.lean @@ -8,11 +8,9 @@ module public import LeanMachineLearning.SequentialLearning.StationaryEnv public import Mathlib.Analysis.Calculus.Gradient.Basic +import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut import LeanMachineLearning.MeasureTheory.Function.L2Space -import Mathlib.Analysis.Calculus.Deriv.Comp -import Mathlib.Analysis.Calculus.Deriv.Mul -import Mathlib.Analysis.Calculus.Deriv.Slope /-! # Online gradient descent @@ -24,86 +22,6 @@ import Mathlib.Analysis.Calculus.Deriv.Slope open MeasureTheory ProbabilityTheory Filter Real Finset open scoped Gradient ENNReal NNReal RealInnerProductSpace --- todo: move to another file -namespace ConvexOn - -variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x : E} - -lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - fderiv ℝ f x (y - x) ≤ f y - f x := by - have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : - f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by - have h1 : x + t • (y - x) = (1 - t) • x + t • y := by module - have h2 : f ((1 - t) • x + t • y) ≤ (1 - t) • f x + t • f y := - hf.2 (Set.mem_univ x) (Set.mem_univ y) (by grind) (by grind) (by simp) - simp only [smul_eq_mul] at h2 - grind - have h_path_deriv : HasDerivAt (fun t : ℝ ↦ f (x + t • (y - x))) - (fderiv ℝ f x (y - x)) 0 := by - have h1 : HasDerivAt (fun t : ℝ ↦ x + t • (y - x)) (y - x) 0 := by - simpa using (hasDerivAt_id (0 : ℝ)).smul_const (y - x) - have h2 : HasFDerivAt f (fderiv ℝ f x) (x + (0 : ℝ) • (y - x)) := by - simpa using hfx.hasFDerivAt - exact h2.comp_hasDerivAt _ h1 - refine le_of_tendsto h_path_deriv.tendsto_slope_zero_right (Filter.eventually_of_mem - (Ioo_mem_nhdsGT_of_mem ⟨le_rfl, zero_lt_one⟩) fun t ht ↦ ?_) - simp [inv_mul_le_iff₀ ht.1] - grind - -lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x + fderiv ℝ f x (y - x) ≤ f y := by - suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind - exact hf.fderiv_sub_le_sub hfx y - -lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x + ⟪y - x, ∇ f x⟫ ≤ f y := by - have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by - simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] - rw [← hfderiv] - exact hf.add_fderiv_le hfx y - -lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x ≤ f y + ⟪x - y, ∇ f x⟫ := by - have h_add_le := hf.add_inner_gradient_le hfx y - have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by - rw [show x - y = -(y - x) by abel, inner_neg_left] - grind - -lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : - f x - f y ≤ ⟪x - y, ∇ f x⟫ := by - simp only [tsub_le_iff_right] - rw [add_comm] - exact hf.le_add_inner_gradient hfx y - -lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by - calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by - simp_rw [smul_sum] - grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] - _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by - simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] - rw [← sum_mul] - simp - field - -lemma todo'2 [InnerProductSpace ℝ E] [CompleteSpace E] - (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by - calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by - gcongr - exact hf.sub_le_inner_gradient hdf.differentiableAt y - -end ConvexOn - namespace Learning variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 015f05bf..57a0ef9f 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -8,6 +8,7 @@ module public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.SequentialLearning.Deterministic +import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope import LeanMachineLearning.MeasureTheory.Function.L2Space /-! @@ -152,7 +153,7 @@ lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : D refine le_of_eq ?_ field -lemma integral_apply_avg_const_div_sqrt {f : E → ℝ} +theorem integral_apply_avg_const_div_sqrt {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) {D L : ℝ} (hD_pos : 0 < D) (hL_pos : 0 < L) From 2b7960d975f00de4f836fff64f36d1071d2250dd Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Thu, 18 Jun 2026 11:47:00 +0200 Subject: [PATCH 27/43] move files --- LeanMachineLearning.lean | 4 ++-- .../Function/ConditionalExpectation/PullOut.lean | 0 .../{ => ForMathlib}/MeasureTheory/Function/L2Space.lean | 0 LeanMachineLearning/Online/OnlineRegret.lean | 4 ++-- .../Optimization/Algorithms/GradientDescent.lean | 2 +- 5 files changed, 5 insertions(+), 5 deletions(-) rename LeanMachineLearning/{ => ForMathlib}/MeasureTheory/Function/ConditionalExpectation/PullOut.lean (100%) rename LeanMachineLearning/{ => ForMathlib}/MeasureTheory/Function/L2Space.lean (100%) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index f29ae759..1961d96d 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -3,6 +3,8 @@ module -- shake: keep-all --deprecated_module: ignore public import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.Polish.StandardBorel +public import LeanMachineLearning.ForMathlib.MeasureTheory.Function.ConditionalExpectation.PullOut +public import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space public import LeanMachineLearning.ForMathlib.MeasureTheory.Measurable public import LeanMachineLearning.ForMathlib.MeasureTheory.Measure.AbsolutelyContinuous public import LeanMachineLearning.ForMathlib.MeasureTheory.OuterMeasure.Basic @@ -19,8 +21,6 @@ public import LeanMachineLearning.ForMathlib.Probability.Kernel.IonescuTulcea.Tr public import LeanMachineLearning.ForMathlib.Probability.Kernel.KernelSub public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity -public import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut -public import LeanMachineLearning.MeasureTheory.Function.L2Space public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace diff --git a/LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean b/LeanMachineLearning/ForMathlib/MeasureTheory/Function/ConditionalExpectation/PullOut.lean similarity index 100% rename from LeanMachineLearning/MeasureTheory/Function/ConditionalExpectation/PullOut.lean rename to LeanMachineLearning/ForMathlib/MeasureTheory/Function/ConditionalExpectation/PullOut.lean diff --git a/LeanMachineLearning/MeasureTheory/Function/L2Space.lean b/LeanMachineLearning/ForMathlib/MeasureTheory/Function/L2Space.lean similarity index 100% rename from LeanMachineLearning/MeasureTheory/Function/L2Space.lean rename to LeanMachineLearning/ForMathlib/MeasureTheory/Function/L2Space.lean diff --git a/LeanMachineLearning/Online/OnlineRegret.lean b/LeanMachineLearning/Online/OnlineRegret.lean index 6c9c9dc3..2b56ad2c 100644 --- a/LeanMachineLearning/Online/OnlineRegret.lean +++ b/LeanMachineLearning/Online/OnlineRegret.lean @@ -9,8 +9,8 @@ public import LeanMachineLearning.SequentialLearning.StationaryEnv public import Mathlib.Analysis.Calculus.Gradient.Basic import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope -import LeanMachineLearning.MeasureTheory.Function.ConditionalExpectation.PullOut -import LeanMachineLearning.MeasureTheory.Function.L2Space +import LeanMachineLearning.ForMathlib.MeasureTheory.Function.ConditionalExpectation.PullOut +import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space /-! # Online gradient descent diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 57a0ef9f..dd1040cb 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -9,7 +9,7 @@ public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.SequentialLearning.Deterministic import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope -import LeanMachineLearning.MeasureTheory.Function.L2Space +import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space /-! # Online gradient descent From dc2a24036001343c8c65b02c26ea8913e333d0be Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Thu, 18 Jun 2026 13:52:25 +0200 Subject: [PATCH 28/43] minor --- .../MeasureTheory/Function/ConditionalExpectation/PullOut.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/ForMathlib/MeasureTheory/Function/ConditionalExpectation/PullOut.lean b/LeanMachineLearning/ForMathlib/MeasureTheory/Function/ConditionalExpectation/PullOut.lean index bf95f787..4d9a8079 100644 --- a/LeanMachineLearning/ForMathlib/MeasureTheory/Function/ConditionalExpectation/PullOut.lean +++ b/LeanMachineLearning/ForMathlib/MeasureTheory/Function/ConditionalExpectation/PullOut.lean @@ -20,7 +20,7 @@ namespace MeasureTheory variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {mE : MeasurableSpace E} {P : Measure Ω} [NormedAddCommGroup E] [InnerProductSpace ℝ E] [CompleteSpace E] -theorem condExp_inner_of_stronglyMeasurable_left {Ω : Type*} {m mΩ : MeasurableSpace Ω} +lemma condExp_inner_of_stronglyMeasurable_left {Ω : Type*} {m mΩ : MeasurableSpace Ω} {μ : Measure Ω} {X g : Ω → E} (hX : StronglyMeasurable[m] X) (hXg : Integrable (fun ω ↦ ⟪X ω, g ω⟫) μ) (hg : Integrable g μ) : μ[fun ω ↦ ⟪X ω, g ω⟫ | m] =ᵐ[μ] fun ω ↦ ⟪X ω, μ[g | m] ω⟫ := by From 899aa17a0efa73f71d8b5afd9682adc4886edc22 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Fri, 19 Jun 2026 18:22:18 +0200 Subject: [PATCH 29/43] docstring --- .../Optimization/Algorithms/GradientDescent.lean | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index dd1040cb..feb0f9f4 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -38,7 +38,11 @@ variable {env : Environment E E} without projection. It is an algorithm that chooses actions in `E` and gets feedback in `E` (gradient of the function at -the queried point). -/ +the queried point). +The point `x (n + 1)` is defined as `x (n + 1) = x n - γ n • g n`, where `g n` is the feedback +received at step `n`. +Since the algorithm is expressed as a function of the history `hist : ℕ → Iic n → E × E`, +we write `(hist ⟨n, …⟩).1` for `x n` and `(hist ⟨n, …⟩).2` for `g n`. -/ noncomputable def gradientStep (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := detAlgorithm (fun n hist ↦ (hist ⟨n, by grind⟩).1 - γ n • (hist ⟨n, by grind⟩).2) (by fun_prop) x₀ From 31d551da2cd0e938285ffba75146724ebf12d88d Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 27 Jun 2026 11:00:31 +0200 Subject: [PATCH 30/43] fix --- LeanMachineLearning/SequentialLearning/StationaryEnv.lean | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean index 811be007..68ce0fa8 100644 --- a/LeanMachineLearning/SequentialLearning/StationaryEnv.lean +++ b/LeanMachineLearning/SequentialLearning/StationaryEnv.lean @@ -83,9 +83,8 @@ lemma hasCondDistrib_feedback_history_action [IsObliviousEnv env] ((feedbackCondAction env (n + 1)).prodMkLeft _) P := by have hA := h.measurable_action have hR' := h.measurable_feedback - refine ⟨by fun_prop, by fun_prop, ?_⟩ - have h_eq := (h.hasCondDistrib_feedback n).condDistrib_eq - rw [condDistrib_ae_eq_iff_measure_eq_compProd _ (by fun_prop)] at h_eq ⊢ + refine ⟨by fun_prop, ?_⟩ + have h_eq := (h.hasCondDistrib_feedback n).map_eq simpa only [feedback_eq_feedbackCondAction] using h_eq lemma hasCondDistrib_feedback [IsObliviousEnv env] (h : IsAlgEnvSeq A Y alg env P) (n : ℕ) : From 32be14962ce3367ef672f4349e4c1d00fd101290 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 27 Jun 2026 14:26:47 +0200 Subject: [PATCH 31/43] reorganize --- .../Analysis/Calculus/Deriv/Slope.lean | 6 +- LeanMachineLearning/Online/OnlineRegret.lean | 192 ++---------------- LeanMachineLearning/Online/OnlineToBatch.lean | 123 +++++++++++ .../Algorithms/GradientDescent.lean | 113 ++++++++--- 4 files changed, 227 insertions(+), 207 deletions(-) create mode 100644 LeanMachineLearning/Online/OnlineToBatch.lean diff --git a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean index db972e65..0b6693d6 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean @@ -76,7 +76,7 @@ lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : Co rw [add_comm] exact hf.le_add_inner_gradient hfx y -lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) +lemma apply_avg_sub_le_avg_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y @@ -89,11 +89,11 @@ lemma todo'3 [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) simp field -lemma todo'2 [InnerProductSpace ℝ E] [CompleteSpace E] +lemma apply_avg_sub_le_avg_inner [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := todo'3 hf x y n hn + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := apply_avg_sub_le_avg_sub hf x y n hn _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by gcongr exact hf.sub_le_inner_gradient hdf.differentiableAt y diff --git a/LeanMachineLearning/Online/OnlineRegret.lean b/LeanMachineLearning/Online/OnlineRegret.lean index 2b56ad2c..ca54536d 100644 --- a/LeanMachineLearning/Online/OnlineRegret.lean +++ b/LeanMachineLearning/Online/OnlineRegret.lean @@ -5,211 +5,53 @@ Authors: Rémy Degenne -/ module -public import LeanMachineLearning.SequentialLearning.StationaryEnv public import Mathlib.Analysis.Calculus.Gradient.Basic import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope -import LeanMachineLearning.ForMathlib.MeasureTheory.Function.ConditionalExpectation.PullOut -import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space /-! -# Online gradient descent +# Online regret -/ @[expose] public section -open MeasureTheory ProbabilityTheory Filter Real Finset +open Filter Real Finset open scoped Gradient ENNReal NNReal RealInnerProductSpace namespace Learning -variable {Ω : Type*} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] - -section OnlineToBatch - -variable {E : Type*} - [NormedAddCommGroup E] [InnerProductSpace ℝ E] [SecondCountableTopology E] [CompleteSpace E] - {mE : MeasurableSpace E} [BorelSpace E] - {X G : ℕ → Ω → E} {alg : Algorithm E E} - {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] - {f : ℕ → E → ℝ} - --- todo: name -lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) - (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : - MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by - let M n := MeasurableSpace.comap (X n) inferInstance - have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) - have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n - ((h_memLp n).integrable (by simp)) - refine h_lp.ae_eq <| h_ae.trans ?_ - simp [← h_unbiased] - -lemma integral_inner_eq_integral_inner_gradient - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : - P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by - have h_obl : HasCondDistrib (G n) (X n) (ν n) P := - h.hasCondDistrib_feedback_obliviousEnv n - calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] - _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | mE.comap (X n)] ω] := by - rw [integral_condExp (h.measurable_action _).comap_le] - _ = P[fun ω ↦ ⟪X n ω - y, P[G n | mE.comap (X n)] ω⟫] := by - refine integral_congr_ae ?_ - refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ - · refine StronglyMeasurable.sub ?_ (by fun_prop) - refine Measurable.stronglyMeasurable ?_ - rw [measurable_iff_comap_le] - · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) - · exact (h_memLp n).integrable (by simp) - _ = P[fun ω ↦ ⟪X n ω - y, (ν n (X n ω))[id]⟫] := by - refine integral_congr_ae ?_ - filter_upwards [h.condExp_feedback_obliviousEnv_ae_eq_integral_id n - ((h_memLp n).integrable (by simp))] with ω hω using by rw [hω] - _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] - -lemma integral_sub_le_integral_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) - (hdf : ∀ n, Differentiable ℝ (f n)) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) - (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption - (y : E) (n : ℕ) : - P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by - rw [integral_inner_eq_integral_inner_gradient h h_unbiased h_memLp hX_lp y n] - gcongr - · exact (h_int n).sub (integrable_const _) - · refine MemLp.integrable_inner ?_ ?_ - · exact (hX_lp n).sub (memLp_const _) - · exact memLp_gradient h h_unbiased h_memLp n - · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y - -lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) - (hdf : ∀ n, Differentiable ℝ (f n)) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) - (y : E) (n : ℕ) : - P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ - P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - rw [integral_finsetSum, integral_finsetSum] - rotate_left - · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) - exact (hX_lp i).sub (memLp_const _) - · exact fun i hi ↦ (h_int i).sub (integrable_const _) - refine Finset.sum_le_sum fun i hi ↦ ?_ - exact integral_sub_le_integral_inner hf hdf h_unbiased h_memLp h hX_lp h_int y i - -lemma integral_apply_avg_sub_le_integral_sum_sub - {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) - (h_unbiased : ∀ n x, (ν n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) - (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) - (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) - (y : E) (n : ℕ) (hn : n ≠ 0) - (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : - P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ - (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by - rw [← integral_const_mul] - gcongr - · exact h_int_avg.sub (integrable_const _) - · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ - exact (h_int i).sub (integrable_const _) - exact fun ω ↦ hf.todo'3 _ y n hn - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := by - grw [integral_sum_sub_le_integral_sum_inner (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h - hX_lp h_int y n] - -end OnlineToBatch - -variable {E : Type*} {mE : MeasurableSpace E} - [NormedAddCommGroup E] [InnerProductSpace ℝ E] [BorelSpace E] - {x x₀ : E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} - -section Linear - -lemma todo'' (x y g : E) (hη : 0 < η) : - ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by - have hsub : (x - η • g) - y = (x - y) - η • g := by abel - rw [hsub, norm_sub_sq_real (x - y) (η • g)] - simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] - field - -lemma todo (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + - (γ i / 2) * ‖g i‖ ^ 2) := by - gcongr with i hi - rw [todo'' (x i) (y i) (g i) (hγ i)] - -lemma todo_sfsq (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) - (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := - (todo x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] - -section ConstantStep - -lemma todo''' (x g : ℕ → E) (y : E) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - refine (todo_sfsq x g y (fun _ ↦ hη) hx n).trans_eq ?_ - rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] - -lemma lem14dot1 (x g : ℕ → E) (y : E) (η : ℝ) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [todo''' x g y hη hx n] - gcongr - exact sub_le_self _ (sq_nonneg _) - -end ConstantStep - -end Linear +variable {E : Type*} [NormedAddCommGroup E] section OnlineRegret /-- The regret of a sequence `x : ℕ → E` compared to a point `y : E` in an online learning task with losses `ℓ : ℕ → E → F`. -/ def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) (x : ℕ → E) (n : ℕ) : F := - ∑ i ∈ Finset.range n, (ℓ i (x i) - ℓ i y) + ∑ i ∈ range n, (ℓ i (x i) - ℓ i y) -noncomputable def linearizedLoss [CompleteSpace E] (f : ℕ → E → ℝ) (x : ℕ → E) : ℕ → E → ℝ := - fun n y ↦ ⟪y, ∇ (f n) (x n)⟫ +lemma apply_avg_sub_le_onlineRegret [NormedSpace ℝ E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) + (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := + hf.apply_avg_sub_le_avg_sub x y n hn -lemma onlineRegret_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : ℕ → E → ℝ} +variable [InnerProductSpace ℝ E] [CompleteSpace E] + +lemma onlineRegret_le_onlineRegret_inner_gradient {f : ℕ → E → ℝ} (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) (x : ℕ → E) (y : E) (n : ℕ) : - onlineRegret f y x n ≤ onlineRegret (linearizedLoss f x) y x n := by - simp only [onlineRegret, linearizedLoss, ← inner_sub_left] + onlineRegret f y x n ≤ onlineRegret (fun n y ↦ ⟪y, ∇ (f n) (x n)⟫) y x n := by + simp only [onlineRegret, ← inner_sub_left] gcongr with i hi exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ -lemma apply_avg_sub_le_onlineRegret {f : E → ℝ} (hf : ConvexOn ℝ .univ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : - f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := - hf.todo'3 x y n hn - -lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) - (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - onlineRegret (fun n x ↦ ⟪x, g n⟫) y x n ≤ - (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - simpa [onlineRegret, inner_sub_left] using lem14dot1 x g y η hη hx n - -lemma apply_avg_sub_le_onlineRegret_linearizedLoss [CompleteSpace E] {f : E → ℝ} +lemma apply_avg_sub_le_onlineRegret_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ - (n : ℝ)⁻¹ * (onlineRegret (linearizedLoss (fun _ ↦ f) x) y x n) := by - simpa [onlineRegret, linearizedLoss, ← inner_sub_left] using hf.todo'2 hdf x y n hn + (n : ℝ)⁻¹ * (onlineRegret (fun n y ↦ ⟪y, ∇ f (x n)⟫) y x n) := by + simpa [onlineRegret, ← inner_sub_left] using + hf.apply_avg_sub_le_avg_inner hdf x y n hn end OnlineRegret diff --git a/LeanMachineLearning/Online/OnlineToBatch.lean b/LeanMachineLearning/Online/OnlineToBatch.lean new file mode 100644 index 00000000..ad4b7542 --- /dev/null +++ b/LeanMachineLearning/Online/OnlineToBatch.lean @@ -0,0 +1,123 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import LeanMachineLearning.SequentialLearning.StationaryEnv +public import Mathlib.Analysis.Calculus.Gradient.Basic + +import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope +import LeanMachineLearning.ForMathlib.MeasureTheory.Function.ConditionalExpectation.PullOut +import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space + +/-! +# Online to batch conversion + +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset +open scoped Gradient ENNReal NNReal RealInnerProductSpace + +namespace Learning + +variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {P : Measure Ω} [IsProbabilityMeasure P] + [NormedAddCommGroup E] [InnerProductSpace ℝ E] [SecondCountableTopology E] [CompleteSpace E] + {mE : MeasurableSpace E} [BorelSpace E] + {X G : ℕ → Ω → E} {alg : Algorithm E E} + {ν : ℕ → Kernel E E} [∀ n, IsMarkovKernel (ν n)] + {f : ℕ → E → ℝ} + +-- todo: name +lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) + (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : + MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by + let M n := MeasurableSpace.comap (X n) inferInstance + have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) + have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n + ((h_memLp n).integrable (by simp)) + refine h_lp.ae_eq <| h_ae.trans ?_ + simp [← h_unbiased] + +lemma integral_inner_eq_integral_inner_gradient + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (y : E) (n : ℕ) : + P[fun ω ↦ ⟪X n ω - y, G n ω⟫] = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by + have h_obl : HasCondDistrib (G n) (X n) (ν n) P := + h.hasCondDistrib_feedback_obliviousEnv n + calc P[fun ω ↦ ⟪X n ω - y, G n ω⟫] + _ = P[fun ω ↦ P[fun ω' ↦ ⟪X n ω' - y, G n ω'⟫ | mE.comap (X n)] ω] := by + rw [integral_condExp (h.measurable_action _).comap_le] + _ = P[fun ω ↦ ⟪X n ω - y, P[G n | mE.comap (X n)] ω⟫] := by + refine integral_congr_ae ?_ + refine condExp_inner_of_stronglyMeasurable_left ?_ ?_ ?_ + · refine StronglyMeasurable.sub ?_ (by fun_prop) + refine Measurable.stronglyMeasurable ?_ + rw [measurable_iff_comap_le] + · exact MemLp.integrable_inner ((hX_lp n).sub (memLp_const _)) (h_memLp n) + · exact (h_memLp n).integrable (by simp) + _ = P[fun ω ↦ ⟪X n ω - y, (ν n (X n ω))[id]⟫] := by + refine integral_congr_ae ?_ + filter_upwards [h.condExp_feedback_obliviousEnv_ae_eq_integral_id n + ((h_memLp n).integrable (by simp))] with ω hω using by rw [hω] + _ = P[fun ω ↦ ⟪X n ω - y, ∇ (f n) (X n ω)⟫] := by simp_rw [h_unbiased n] + +lemma integral_sub_le_integral_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) + (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) -- todo: discuss this assumption + (y : E) (n : ℕ) : + P[fun ω ↦ f n (X n ω) - f n y] ≤ P[fun ω ↦ ⟪X n ω - y, G n ω⟫] := by + rw [integral_inner_eq_integral_inner_gradient h h_unbiased h_memLp hX_lp y n] + gcongr + · exact (h_int n).sub (integrable_const _) + · refine MemLp.integrable_inner ?_ ?_ + · exact (hX_lp n).sub (memLp_const _) + · exact memLp_gradient h h_unbiased h_memLp n + · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y + +lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) + (hdf : ∀ n, Differentiable ℝ (f n)) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) + (y : E) (n : ℕ) : + P[fun ω ↦ ∑ i ∈ range n, (f i (X i ω) - f i y)] ≤ + P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] := by + rw [integral_finsetSum, integral_finsetSum] + rotate_left + · refine fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) + exact (hX_lp i).sub (memLp_const _) + · exact fun i hi ↦ (h_int i).sub (integrable_const _) + refine sum_le_sum fun i hi ↦ ?_ + exact integral_sub_le_integral_inner hf hdf h_unbiased h_memLp h hX_lp h_int y i + +lemma integral_apply_avg_sub_le_integral_sum_sub + {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) + (h_unbiased : ∀ n x, (ν n x)[id] = ∇ f x) (h_memLp : ∀ n, MemLp (G n) 2 P) + (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) + (hX_lp : ∀ n, MemLp (X n) 2 P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) + (y : E) (n : ℕ) (hn : n ≠ 0) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω)) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω) - f y] ≤ + (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] := by + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω) - f y] + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ range n, (f (X i ω) - f y)] := by + rw [← integral_const_mul] + gcongr + · exact h_int_avg.sub (integrable_const _) + · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ + exact (h_int i).sub (integrable_const _) + exact fun ω ↦ hf.apply_avg_sub_le_avg_sub _ y n hn + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] := by + grw [integral_sum_sub_le_integral_sum_inner (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h + hX_lp h_int y n] + +end Learning diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index feb0f9f4..3bb4d428 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -7,12 +7,14 @@ module public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.SequentialLearning.StationaryEnv import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space +import LeanMachineLearning.Online.OnlineToBatch /-! -# Online gradient descent +# Online and stochastic gradient descent -/ @@ -25,9 +27,62 @@ namespace Learning variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {mE : MeasurableSpace E} [NormedAddCommGroup E] [InnerProductSpace ℝ E] - [SecondCountableTopology E] [CompleteSpace E] [BorelSpace E] {P : Measure Ω} [IsProbabilityMeasure P] {x x₀ : E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} + +section Linear + +lemma inner_eq_add (x y g : E) (hη : 0 < η) : + ⟪x - y, g⟫ = (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖(x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by + have hsub : (x - η • g) - y = (x - y) - η • g := by abel + rw [hsub, norm_sub_sq_real (x - y) (η • g)] + simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] + field + +lemma sum_inner_le_sum' (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + + (γ i / 2) * ‖g i‖ ^ 2) := by + gcongr with i hi + rw [inner_eq_add (x i) (y i) (g i) (hγ i)] + +lemma sum_inner_le_sum (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) + (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + ∑ i ∈ Finset.range n, + ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := + (sum_inner_le_sum' x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] + +section ConstantStep + +lemma sum_inner_le_add (x g : ℕ → E) (y : E) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + refine (sum_inner_le_sum x g y (fun _ ↦ hη) hx n).trans_eq ?_ + rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] + +/-- Lemma 14.1 in Understanding Machine Learning: From Theory to Algorithms. -/ +lemma gradient_descent_linear_regret (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by + grw [sum_inner_le_add x g y hη hx n] + gcongr + exact sub_le_self _ (sq_nonneg _) + +end ConstantStep + +lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : + onlineRegret (fun n x ↦ ⟪x, g n⟫) y x n ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := by + simpa [onlineRegret, inner_sub_left] using gradient_descent_linear_regret x g y η hη hx n + +end Linear + +variable [SecondCountableTopology E] [CompleteSpace E] [BorelSpace E] {f : ℕ → E → ℝ} {hf : ∀ n, Measurable (∇ (f n))} section Definition @@ -55,11 +110,11 @@ lemma action_gradientStep_ae_all_eq (h_seq : IsAlgEnvSeq X G (gradientStep γ x h_seq.action_detAlgorithm_ae_all_eq lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P) (n : ℕ) : - X n =ᵐ[P] fun ω ↦ x₀ - ∑ i ∈ Finset.range n, γ i • G i ω := by + X n =ᵐ[P] fun ω ↦ x₀ - ∑ i ∈ range n, γ i • G i ω := by filter_upwards [h_seq.action_detAlgorithm_ae_all_eq] with ω ⟨hω0, hω⟩ induction n with | zero => simpa - | succ n ih => rw [hω n, Finset.sum_range_succ, ← sub_sub]; congr + | succ n ih => rw [hω n, sum_range_succ, ← sub_sub]; congr end Definition @@ -85,24 +140,24 @@ section Linear lemma integral_sum_inner_le (hη : 0 < η) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (y : E) (n : ℕ) : - P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] ≤ - (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - calc P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] - _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖G i ω‖ ^ 2 ∂P := by + P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + calc P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] + _ ≤ ∫ ω, (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖G i ω‖ ^ 2 ∂P := by refine integral_mono_ae ?_ ?_ ?_ · refine integrable_finsetSum _ fun i hi ↦ MemLp.integrable_inner ?_ (h_memLp i) exact (memLp_X h h_memLp i).sub (memLp_const _) · refine Integrable.add (integrable_const _) (Integrable.const_mul ?_ _) exact integrable_finsetSum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) · filter_upwards [action_gradientStep_ae_all_eq h] with ω hω - refine (lem14dot1 _ _ y η hη hω.2 n).trans_eq ?_ + refine (gradient_descent_linear_regret _ _ y η hη hω.2 n).trans_eq ?_ congr exact hω.1 - _ = (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + _ = (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by rw [integral_add, integral_const_mul, integral_const_mul, integral_finsetSum] · simp · exact fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) - · exact integrable_const _ + · fun_prop · refine Integrable.const_mul ?_ _ exact integrable_finsetSum _ fun i hi ↦ (h_memLp i).integrable_norm_pow (by simp) @@ -113,12 +168,12 @@ lemma integral_sum_sub_le (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, D (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ (f n) x) (h_memLp : ∀ n, MemLp (G n) 2 P) (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : - P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] ≤ - (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - calc P[fun ω ↦ ∑ i ∈ Finset.range n, (f i (X i ω) - f i y)] - _ ≤ P[fun ω ↦ ∑ i ∈ Finset.range n, ⟪X i ω - y, G i ω⟫] := + P[fun ω ↦ ∑ i ∈ range n, (f i (X i ω) - f i y)] ≤ + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + calc P[fun ω ↦ ∑ i ∈ range n, (f i (X i ω) - f i y)] + _ ≤ P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] := integral_sum_sub_le_integral_sum_inner hf hdf h_unbiased h_memLp h (memLp_X h h_memLp) h_int y n - _ ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := + _ ≤ (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := integral_sum_inner_le hη h_memLp h y n lemma integral_onlineRegret_le @@ -129,7 +184,7 @@ lemma integral_onlineRegret_le (h_int : ∀ n, Integrable (fun ω ↦ f n (X n ω)) P) (y : E) (n : ℕ) : P[fun ω ↦ onlineRegret f y (X · ω) n] ≤ - (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := + (2 * η)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := integral_sum_sub_le hf hdf hη h_unbiased h_memLp h h_int y n lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) @@ -139,42 +194,42 @@ lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : D (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ η) x₀) (obliviousEnv gradKernel) P) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) (y : E) (n : ℕ) (hn : n ≠ 0) - (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) : - P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω)) P) : + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω) - f y] ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + - (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by - calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] - _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ Finset.range n, (f (X i ω) - f y)] := by + (η / (2 * n)) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω) - f y] + _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ range n, (f (X i ω) - f y)] := by rw [← integral_const_mul] gcongr · exact h_int_avg.sub (integrable_const _) · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ exact (h_int i).sub (integrable_const _) - exact fun ω ↦ hf.todo'3 _ y n hn + exact fun ω ↦ hf.apply_avg_sub_le_avg_sub _ y n hn _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + - (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + (η / (2 * n)) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by grw [integral_sum_sub_le (fun _ ↦ hf) (fun _ ↦ hdf) hη h_unbiased h_memLp h h_int y n] refine le_of_eq ?_ field -theorem integral_apply_avg_const_div_sqrt {f : E → ℝ} +theorem integral_apply_avg_le_const_div_sqrt {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (h_unbiased : ∀ n x, (gradKernel n x)[id] = ∇ f x) {D L : ℝ} (hD_pos : 0 < D) (hL_pos : 0 < L) {y : E} (hxy_le : ‖x₀ - y‖ ≤ D) (hG_le : ∀ n ω, ‖G n ω‖ ≤ L) (h_int : ∀ n, Integrable (fun ω ↦ f (X n ω)) P) {n : ℕ} (hn : n ≠ 0) - (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω)) P) + (h_int_avg : Integrable (fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω)) P) (h : IsAlgEnvSeq X G (gradientStep (fun _ ↦ D / (L * √n)) x₀) (obliviousEnv gradKernel) P) : - P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] ≤ D * L / √n := by + P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω) - f y] ≤ D * L / √n := by let η := D / (L * √n) have hG_lp n : MemLp (G n) 2 P := by refine MemLp.mono (g := fun _ ↦ L) (memLp_const _) (have := h.measurable_feedback; by fun_prop) (ae_of_all _ fun ω ↦ ?_) simpa [abs_of_nonneg hL_pos.le] using hG_le n ω - calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ Finset.range n, X i ω) - f y] + calc P[fun ω ↦ f ((n : ℝ)⁻¹ • ∑ i ∈ range n, X i ω) - f y] _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + - (η / (2 * n)) * ∑ i ∈ Finset.range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by + (η / (2 * n)) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by refine integral_apply_avg_le hf hdf ?_ h_unbiased hG_lp h h_int y n hn h_int_avg positivity _ ≤ (2 * η * n)⁻¹ * D ^ 2 + (η / 2) * L ^ 2 := by From 0e7fb86943bc1788c67cadbf39ee02df58f68fcd Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 27 Jun 2026 14:27:19 +0200 Subject: [PATCH 32/43] mk_all --- LeanMachineLearning.lean | 1 + 1 file changed, 1 insertion(+) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 85428d94..225677f5 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -32,6 +32,7 @@ public import LeanMachineLearning.Online.Bandit.Regret public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.Online.OnlineRegret +public import LeanMachineLearning.Online.OnlineToBatch public import LeanMachineLearning.Optimization.Algorithms.GradientDescent public import LeanMachineLearning.SequentialLearning.Algorithm public import LeanMachineLearning.SequentialLearning.AlgorithmDensity From ced056cb7aa44ed1c3fcab2a0848bbb368dadb47 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sun, 28 Jun 2026 06:31:16 +0200 Subject: [PATCH 33/43] work towards projected gradient descent --- LeanMachineLearning.lean | 1 + .../Analysis/Calculus/Deriv/Slope.lean | 111 ++++++++++----- LeanMachineLearning/Online/OnlineRegret.lean | 10 +- LeanMachineLearning/Online/OnlineToBatch.lean | 4 +- LeanMachineLearning/Online/Projection.lean | 129 ++++++++++++++++++ .../Algorithms/GradientDescent.lean | 2 +- 6 files changed, 218 insertions(+), 39 deletions(-) create mode 100644 LeanMachineLearning/Online/Projection.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 225677f5..002d2d12 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -33,6 +33,7 @@ public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.Online.OnlineToBatch +public import LeanMachineLearning.Online.Projection public import LeanMachineLearning.Optimization.Algorithms.GradientDescent public import LeanMachineLearning.SequentialLearning.Algorithm public import LeanMachineLearning.SequentialLearning.AlgorithmDensity diff --git a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean index 0b6693d6..1b24f702 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean @@ -10,6 +10,7 @@ public import Mathlib.Analysis.Calculus.Gradient.Basic import Mathlib.Analysis.Calculus.Deriv.Comp import Mathlib.Analysis.Calculus.Deriv.Mul import Mathlib.Analysis.Calculus.Deriv.Slope +import Mathlib.Analysis.Calculus.LocalExtr.Basic /-! # Convexity lemmas @@ -18,21 +19,21 @@ import Mathlib.Analysis.Calculus.Deriv.Slope @[expose] public section -open Finset -open scoped Gradient RealInnerProductSpace +open Finset Filter +open scoped Gradient RealInnerProductSpace Topology namespace ConvexOn -variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x : E} +variable {E : Type*} [NormedAddCommGroup E] {f : E → ℝ} {x y : E} {s : Set E} -lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : +lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (y : E) (hx : x ∈ s) (hy : y ∈ s) : fderiv ℝ f x (y - x) ≤ f y - f x := by have h_convex t (ht : t ∈ Set.Ioo (0 : ℝ) 1) : f (x + t • (y - x)) ≤ t * f y + (1 - t) * f x := by have h1 : x + t • (y - x) = (1 - t) • x + t • y := by module have h2 : f ((1 - t) • x + t • y) ≤ (1 - t) • f x + t • f y := - hf.2 (Set.mem_univ x) (Set.mem_univ y) (by grind) (by grind) (by simp) + hf.2 hx hy (by grind) (by grind) (by simp) simp only [smul_eq_mul] at h2 grind have h_path_deriv : HasDerivAt (fun t : ℝ ↦ f (x + t • (y - x))) @@ -47,42 +48,89 @@ lemma fderiv_sub_le_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) simp [inv_mul_le_iff₀ ht.1] grind -lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : +lemma add_fderiv_le [NormedSpace ℝ E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (hy : y ∈ s) : f x + fderiv ℝ f x (y - x) ≤ f y := by suffices fderiv ℝ f x (y - x) ≤ f y - f x by grind - exact hf.fderiv_sub_le_sub hfx y + exact hf.fderiv_sub_le_sub hfx y hx hy -lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : +lemma add_inner_gradient_le [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (hy : y ∈ s) : f x + ⟪y - x, ∇ f x⟫ ≤ f y := by - have hfderiv : (fderiv ℝ f x) (y - x) = ⟪y - x, ∇ f x⟫ := by - simp [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] - rw [← hfderiv] - exact hf.add_fderiv_le hfx y + rw [gradient, real_inner_comm, InnerProductSpace.toDual_symm_apply] + exact hf.add_fderiv_le hfx hx hy -lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : +lemma le_add_fderiv [NormedSpace ℝ E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (hy : y ∈ s) : + f x ≤ f y + fderiv ℝ f x (x - y) := by + have h_add_le := hf.add_fderiv_le hfx hx hy + grind + +lemma le_add_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (hy : y ∈ s) : f x ≤ f y + ⟪x - y, ∇ f x⟫ := by - have h_add_le := hf.add_inner_gradient_le hfx y - have h_neg : ⟪x - y, ∇ f x⟫ = -⟪y - x, ∇ f x⟫ := by - rw [show x - y = -(y - x) by abel, inner_neg_left] + rw [gradient, real_inner_comm, InnerProductSpace.toDual_symm_apply] + exact le_add_fderiv hf hfx hx hy + +lemma sub_le_fderiv [NormedSpace ℝ E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (hy : y ∈ s) : + f x - f y ≤ fderiv ℝ f x (x - y) := by + have h_le := hf.le_add_fderiv hfx hx hy grind -lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ .univ f) - (hfx : DifferentiableAt ℝ f x) (y : E) : +lemma sub_le_inner_gradient [InnerProductSpace ℝ E] [CompleteSpace E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (hy : y ∈ s) : f x - f y ≤ ⟪x - y, ∇ f x⟫ := by - simp only [tsub_le_iff_right] - rw [add_comm] - exact hf.le_add_inner_gradient hfx y + rw [gradient, real_inner_comm, InnerProductSpace.toDual_symm_apply] + exact sub_le_fderiv hf hfx hx hy + +lemma isMinOn_of_fderiv_nonneg [NormedSpace ℝ E] (hf : ConvexOn ℝ s f) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (h_nonneg : ∀ y ∈ s, 0 ≤ fderiv ℝ f x (y - x)) : + IsMinOn f s x := by + intro y hy + have h_le := hf.le_add_fderiv hfx hx hy + specialize h_nonneg y hy + grind -lemma apply_avg_sub_le_avg_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) - (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : +lemma isMinOn_of_inner_gradient_nonneg [InnerProductSpace ℝ E] [CompleteSpace E] + (hf : ConvexOn ℝ s f) (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) + (h_nonneg : ∀ y ∈ s, 0 ≤ ⟪y - x, ∇ f x⟫) : + IsMinOn f s x := by + refine isMinOn_of_fderiv_nonneg hf hfx hx fun y hy ↦ ?_ + convert h_nonneg y hy + rw [gradient, ← InnerProductSpace.toDual_symm_apply, real_inner_comm] + +lemma fderiv_nonneg_of_isMinOn [NormedSpace ℝ E] (hs : Convex ℝ s) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (h_min : IsMinOn f s x) {y : E} (hy : y ∈ s) : + 0 ≤ fderiv ℝ f x (y - x) := by + refine IsLocalMinOn.hasFDerivWithinAt_nonneg h_min.localize (y := y - x) + hfx.hasFDerivAt.hasFDerivWithinAt ?_ + refine sub_mem_posTangentConeAt_of_openSegment_subset ?_ + exact StarConvex.openSegment_subset (hs hx) hy + +lemma inner_gradient_nonneg_of_isMinOn [InnerProductSpace ℝ E] [CompleteSpace E] (hs : Convex ℝ s) + (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) (h_min : IsMinOn f s x) {y : E} (hy : y ∈ s) : + 0 ≤ ⟪y - x, ∇ f x⟫ := by + rw [gradient, real_inner_comm, InnerProductSpace.toDual_symm_apply] + exact fderiv_nonneg_of_isMinOn hs hfx hx h_min hy + +lemma isMinOn_iff_fderiv_nonneg [NormedSpace ℝ E] (hs : Convex ℝ s) + (hf : ConvexOn ℝ s f) (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) : + IsMinOn f s x ↔ ∀ y ∈ s, 0 ≤ fderiv ℝ f x (y - x) := + ⟨fderiv_nonneg_of_isMinOn hs hfx hx, hf.isMinOn_of_fderiv_nonneg hfx hx⟩ + +lemma isMinOn_iff_inner_gradient_nonneg [InnerProductSpace ℝ E] [CompleteSpace E] (hs : Convex ℝ s) + (hf : ConvexOn ℝ s f) (hfx : DifferentiableAt ℝ f x) (hx : x ∈ s) : + IsMinOn f s x ↔ ∀ y ∈ s, 0 ≤ ⟪y - x, ∇ f x⟫ := + ⟨inner_gradient_nonneg_of_isMinOn hs hfx hx, hf.isMinOn_of_inner_gradient_nonneg hfx hx⟩ + +lemma apply_avg_sub_le_avg_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ s f) + {x : ℕ → E} (hx : ∀ i, x i ∈ s) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, (f (x i) - f y) := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y _ ≤ (n : ℝ)⁻¹ • ∑ i ∈ range n, f (x i) - f y := by simp_rw [smul_sum] - grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (by simp)] + grw [hf.map_sum_le (fun _ _ ↦ by positivity) (by simp; field) (fun i _ ↦ hx i)] _ = (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := by simp_rw [smul_eq_mul, mul_sum, mul_sub, sum_sub_distrib] rw [← sum_mul] @@ -90,12 +138,13 @@ lemma apply_avg_sub_le_avg_sub [NormedSpace ℝ E] (hf : ConvexOn ℝ .univ f) field lemma apply_avg_sub_le_avg_inner [InnerProductSpace ℝ E] [CompleteSpace E] - (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : + (hf : ConvexOn ℝ s f) (hdf : Differentiable ℝ f) {x : ℕ → E} (hx : ∀ i, x i ∈ s) + (hy : y ∈ s) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by calc f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y - _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := apply_avg_sub_le_avg_sub hf x y n hn + _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, (f (x i) - f y) := apply_avg_sub_le_avg_sub hf hx y n hn _ ≤ (n : ℝ)⁻¹ * ∑ i ∈ range n, ⟪x i - y, ∇ f (x i)⟫ := by gcongr - exact hf.sub_le_inner_gradient hdf.differentiableAt y + exact hf.sub_le_inner_gradient hdf.differentiableAt (hx i) hy end ConvexOn diff --git a/LeanMachineLearning/Online/OnlineRegret.lean b/LeanMachineLearning/Online/OnlineRegret.lean index ca54536d..4f2f1af8 100644 --- a/LeanMachineLearning/Online/OnlineRegret.lean +++ b/LeanMachineLearning/Online/OnlineRegret.lean @@ -16,8 +16,8 @@ import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope @[expose] public section -open Filter Real Finset -open scoped Gradient ENNReal NNReal RealInnerProductSpace +open Filter Real Finset +open scoped Gradient RealInnerProductSpace namespace Learning @@ -33,7 +33,7 @@ def onlineRegret {E F : Type*} [AddCommGroup F] (ℓ : ℕ → E → F) (y : E) lemma apply_avg_sub_le_onlineRegret [NormedSpace ℝ E] {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (x : ℕ → E) (y : E) (n : ℕ) (hn : n ≠ 0) : f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ • onlineRegret (fun _ ↦ f) y x n := - hf.apply_avg_sub_le_avg_sub x y n hn + hf.apply_avg_sub_le_avg_sub (by simp) _ n hn variable [InnerProductSpace ℝ E] [CompleteSpace E] @@ -43,7 +43,7 @@ lemma onlineRegret_le_onlineRegret_inner_gradient {f : ℕ → E → ℝ} onlineRegret f y x n ≤ onlineRegret (fun n y ↦ ⟪y, ∇ (f n) (x n)⟫) y x n := by simp only [onlineRegret, ← inner_sub_left] gcongr with i hi - exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt _ + exact (hf i).sub_le_inner_gradient (hdf i).differentiableAt (by simp) (by simp) lemma apply_avg_sub_le_onlineRegret_inner_gradient {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : Differentiable ℝ f) @@ -51,7 +51,7 @@ lemma apply_avg_sub_le_onlineRegret_inner_gradient {f : E → ℝ} f ((n : ℝ)⁻¹ • ∑ i ∈ range n, x i) - f y ≤ (n : ℝ)⁻¹ * (onlineRegret (fun n y ↦ ⟪y, ∇ f (x n)⟫) y x n) := by simpa [onlineRegret, ← inner_sub_left] using - hf.apply_avg_sub_le_avg_inner hdf x y n hn + hf.apply_avg_sub_le_avg_inner hdf (by simp) (by simp) n hn end OnlineRegret diff --git a/LeanMachineLearning/Online/OnlineToBatch.lean b/LeanMachineLearning/Online/OnlineToBatch.lean index ad4b7542..388c0ffb 100644 --- a/LeanMachineLearning/Online/OnlineToBatch.lean +++ b/LeanMachineLearning/Online/OnlineToBatch.lean @@ -81,7 +81,7 @@ lemma integral_sub_le_integral_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) · refine MemLp.integrable_inner ?_ ?_ · exact (hX_lp n).sub (memLp_const _) · exact memLp_gradient h h_unbiased h_memLp n - · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt y + · exact fun ω ↦ (hf n).sub_le_inner_gradient (hdf n).differentiableAt (by simp) (by simp) lemma integral_sum_sub_le_integral_sum_inner (hf : ∀ n, ConvexOn ℝ .univ (f n)) (hdf : ∀ n, Differentiable ℝ (f n)) @@ -115,7 +115,7 @@ lemma integral_apply_avg_sub_le_integral_sum_sub · exact h_int_avg.sub (integrable_const _) · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ exact (h_int i).sub (integrable_const _) - exact fun ω ↦ hf.apply_avg_sub_le_avg_sub _ y n hn + exact fun ω ↦ hf.apply_avg_sub_le_avg_sub (by simp) y n hn _ ≤ (n : ℝ)⁻¹ * P[fun ω ↦ ∑ i ∈ range n, ⟪X i ω - y, G i ω⟫] := by grw [integral_sum_sub_le_integral_sum_inner (fun _ ↦ hf) (fun _ ↦ hdf) h_unbiased h_memLp h hX_lp h_int y n] diff --git a/LeanMachineLearning/Online/Projection.lean b/LeanMachineLearning/Online/Projection.lean new file mode 100644 index 00000000..ae5c8412 --- /dev/null +++ b/LeanMachineLearning/Online/Projection.lean @@ -0,0 +1,129 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import Mathlib + +import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope + +/-! +# Projection on a nonempty closed convex set in an inner product space + +-/ + +@[expose] public section + +open Real Finset Metric +open scoped RealInnerProductSpace + +namespace Learning + +section Definition + +variable {E : Type*} [PseudoMetricSpace E] {s : Set E} + +open Classical in +/-- Projection on a set: closest point to `x` in the set `s`, taking an arbitrary value if there is +no such point. -/ +noncomputable +def proj [Zero E] (s : Set E) (x : E) : E := + if h : ∃ y ∈ s, IsMinOn (dist x) s y then h.choose else 0 + +lemma _root_.IsClosed.exists_isMinOn_dist [ProperSpace E] + (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : + ∃ y ∈ s, IsMinOn (dist x) s y := by + have h_cont : Continuous (dist x) := by fun_prop + obtain ⟨z, hz⟩ := h_nonempty + have h_compact : IsCompact (closedBall x (dist x z) ∩ s) := + IsCompact.inter_right (isCompact_closedBall _ _) h_closed + have h3 : (closedBall x (dist x z) ∩ s).Nonempty := ⟨z, ⟨by simp [dist_comm], hz⟩⟩ + obtain ⟨y, hy, hy_min⟩ := h_compact.exists_isMinOn h3 h_cont.continuousOn + refine ⟨y, hy.2, ?_⟩ + intro u hu + simp only [Set.mem_setOf_eq] + by_cases h1 : u ∈ closedBall x (dist x z) + · specialize hy_min ⟨h1, hu⟩ + grind + · simp only [mem_closedBall, not_le, Set.mem_inter_iff] at h1 hy + grind [dist_comm] + +lemma _root_.IsClosed.proj_mem [Zero E] [ProperSpace E] + (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : + proj s x ∈ s := by + have h := h_closed.exists_isMinOn_dist h_nonempty x + rw [proj, dif_pos h] + exact h.choose_spec.1 + +lemma _root_.IsClosed.isMinOn_proj [Zero E] [ProperSpace E] + (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : + IsMinOn (dist x) s (proj s x) := by + have h := h_closed.exists_isMinOn_dist h_nonempty x + rw [proj, dif_pos h] + exact h.choose_spec.2 + +end Definition + +lemma _root_.IsClosed.isMinOn_norm_sq_proj {E : Type*} [NormedAddCommGroup E] [ProperSpace E] + {s : Set E} (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : + IsMinOn (fun y ↦ ‖y - x‖ ^ 2) s (proj s x) := by + intro y hy + simp only [Set.mem_setOf_eq, sq_le_sq, abs_dist, ← dist_eq_norm] + have h_min := h_closed.isMinOn_proj h_nonempty x hy + simp_rw [dist_comm _ x] + exact h_min + +section Convex + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] [FiniteDimensional ℝ E] + {s : Set E} + +lemma gradient_norm_sub_sq (x y : E) : gradient (fun z ↦ ‖z - x‖ ^ 2) y = 2 • (y - x) := by + have h := ((hasFDerivAt_id y).sub_const x).norm_sq.hasGradientAt.gradient + simp only [id_eq, map_sub, ContinuousLinearMap.comp_id, map_nsmul] at h + rw [h] + congr + · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ + · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ + +lemma gradient_dist_sq (x y : E) : gradient (fun z ↦ dist x z ^ 2) y = 2 • (y - x) := by + simp only [dist_eq_norm, norm_sub_rev x] + exact gradient_norm_sub_sq x y + +lemma dist_proj_le (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (x : E) {y : E} (hy : y ∈ s) : + dist (proj s x) y ≤ dist x y := by + suffices dist (proj s x) y ^ 2 ≤ dist x y ^ 2 by simpa [sq_le_sq] using this + calc dist (proj s x) y ^ 2 + _ = ‖proj s x - y‖ ^ 2 := by simp [dist_eq_norm] + _ ≤ ‖proj s x - y‖ ^ 2 + ‖proj s x - x‖ ^ 2 + 2 * ⟪proj s x - x, y - proj s x⟫ := by + rw [add_assoc] + refine le_add_of_nonneg_right ?_ + suffices 0 ≤ 2 * ⟪proj s x - x, y - proj s x⟫ by positivity + have h_min := h_closed.isMinOn_norm_sq_proj h_nonempty x hy + have h_inner := ConvexOn.inner_gradient_nonneg_of_isMinOn h_convex ?_ ?_ ?_ (x := proj s x) + (y := y) (f := fun z ↦ ‖z - x‖ ^ 2) hy + · rw [real_inner_comm, ← inner_smul_right] + convert h_inner using 2 + symm + convert gradient_norm_sub_sq x (proj s x) + exact ofNat_smul_eq_nsmul ℝ 2 (proj s x - x) + · suffices DifferentiableAt ℝ (fun z ↦ ‖z - x‖ ^ (2 : ℝ)) (proj s x) by + convert this + simp + refine Differentiable.differentiableAt ?_ + refine Differentiable.norm_rpow ?_ (by simp) + fun_prop + · exact h_closed.proj_mem h_nonempty x + · exact h_closed.isMinOn_norm_sq_proj h_nonempty x + _ = ‖y - proj s x + (proj s x - x)‖ ^ 2 := by + rw [norm_add_sq (𝕜 := ℝ), RCLike.re_to_real, norm_sub_rev, add_assoc, add_comm _ (2 * _), + ← add_assoc, real_inner_comm] + _ = ‖y - x‖ ^ 2 := by congr 2; simp + _ = dist x y ^ 2 := by simp [dist_eq_norm, norm_sub_rev] + +end Convex + +end Learning diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 3bb4d428..9a9865f4 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -205,7 +205,7 @@ lemma integral_apply_avg_le {f : E → ℝ} (hf : ConvexOn ℝ .univ f) (hdf : D · exact h_int_avg.sub (integrable_const _) · refine Integrable.const_mul (integrable_finsetSum _ fun i hi ↦ ?_) _ exact (h_int i).sub (integrable_const _) - exact fun ω ↦ hf.apply_avg_sub_le_avg_sub _ y n hn + exact fun ω ↦ hf.apply_avg_sub_le_avg_sub (by simp) y n hn _ ≤ (2 * η * n)⁻¹ * ‖x₀ - y‖ ^ 2 + (η / (2 * n)) * ∑ i ∈ range n, P[fun ω ↦ ‖G i ω‖ ^ 2] := by grw [integral_sum_sub_le (fun _ ↦ hf) (fun _ ↦ hdf) hη h_unbiased h_memLp h h_int y n] From 29319fdcfd051be27ec60259774e5f984bd4050f Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sun, 28 Jun 2026 08:09:21 +0200 Subject: [PATCH 34/43] projected GD def --- LeanMachineLearning/Online/Projection.lean | 147 +++++++++++++----- .../Algorithms/GradientDescent.lean | 35 ++++- 2 files changed, 142 insertions(+), 40 deletions(-) diff --git a/LeanMachineLearning/Online/Projection.lean b/LeanMachineLearning/Online/Projection.lean index ae5c8412..b678d1e6 100644 --- a/LeanMachineLearning/Online/Projection.lean +++ b/LeanMachineLearning/Online/Projection.lean @@ -5,7 +5,8 @@ Authors: Rémy Degenne -/ module -public import Mathlib +public import Mathlib.Analysis.Calculus.Gradient.Basic +public import Mathlib.Analysis.InnerProductSpace.NormPow import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope @@ -17,7 +18,7 @@ import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope @[expose] public section open Real Finset Metric -open scoped RealInnerProductSpace +open scoped RealInnerProductSpace Gradient namespace Learning @@ -32,6 +33,7 @@ noncomputable def proj [Zero E] (s : Set E) (x : E) : E := if h : ∃ y ∈ s, IsMinOn (dist x) s y then h.choose else 0 +/-- If the set is closed and nonempty, then the projection exists. -/ lemma _root_.IsClosed.exists_isMinOn_dist [ProperSpace E] (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : ∃ y ∈ s, IsMinOn (dist x) s y := by @@ -50,6 +52,7 @@ lemma _root_.IsClosed.exists_isMinOn_dist [ProperSpace E] · simp only [mem_closedBall, not_le, Set.mem_inter_iff] at h1 hy grind [dist_comm] +/-- If the set is closed and nonempty, then the projection belongs to the set. -/ lemma _root_.IsClosed.proj_mem [Zero E] [ProperSpace E] (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : proj s x ∈ s := by @@ -57,30 +60,42 @@ lemma _root_.IsClosed.proj_mem [Zero E] [ProperSpace E] rw [proj, dif_pos h] exact h.choose_spec.1 -lemma _root_.IsClosed.isMinOn_proj [Zero E] [ProperSpace E] - (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : +lemma isMinOn_proj_of_exists [Zero E] {x : E} (h : ∃ y ∈ s, IsMinOn (dist x) s y) : IsMinOn (dist x) s (proj s x) := by - have h := h_closed.exists_isMinOn_dist h_nonempty x rw [proj, dif_pos h] exact h.choose_spec.2 +/-- If the set is closed and nonempty, then the projection is a minimizer of the distance. -/ +lemma _root_.IsClosed.isMinOn_proj [Zero E] [ProperSpace E] + (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : + IsMinOn (dist x) s (proj s x) := + isMinOn_proj_of_exists (h_closed.exists_isMinOn_dist h_nonempty x) + +/-- If `x ∈ s`, then the projection of `x` onto `s` is `x`. -/ +@[simp] +lemma proj_of_mem {E : Type*} [MetricSpace E] [Zero E] {s : Set E} {x : E} + (hx : x ∈ s) : + proj s x = x := by + have h_min : IsMinOn (dist x) s x := fun y hy ↦ by simp + have h_min_proj := isMinOn_proj_of_exists ⟨x, hx, h_min⟩ hx + symm + simpa using h_min_proj + end Definition lemma _root_.IsClosed.isMinOn_norm_sq_proj {E : Type*} [NormedAddCommGroup E] [ProperSpace E] {s : Set E} (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : IsMinOn (fun y ↦ ‖y - x‖ ^ 2) s (proj s x) := by intro y hy - simp only [Set.mem_setOf_eq, sq_le_sq, abs_dist, ← dist_eq_norm] - have h_min := h_closed.isMinOn_proj h_nonempty x hy - simp_rw [dist_comm _ x] - exact h_min + simp only [Set.mem_setOf_eq, sq_le_sq, abs_dist, ← dist_eq_norm, dist_comm _ x] + exact h_closed.isMinOn_proj h_nonempty x hy section Convex variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] [FiniteDimensional ℝ E] {s : Set E} -lemma gradient_norm_sub_sq (x y : E) : gradient (fun z ↦ ‖z - x‖ ^ 2) y = 2 • (y - x) := by +lemma gradient_norm_sub_sq (x y : E) : ∇ (fun z ↦ ‖z - x‖ ^ 2) y = 2 • (y - x) := by have h := ((hasFDerivAt_id y).sub_const x).norm_sq.hasGradientAt.gradient simp only [id_eq, map_sub, ContinuousLinearMap.comp_id, map_nsmul] at h rw [h] @@ -88,41 +103,95 @@ lemma gradient_norm_sub_sq (x y : E) : gradient (fun z ↦ ‖z - x‖ ^ 2) y = · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ -lemma gradient_dist_sq (x y : E) : gradient (fun z ↦ dist x z ^ 2) y = 2 • (y - x) := by +lemma gradient_dist_sq (x y : E) : ∇ (fun z ↦ dist x z ^ 2) y = 2 • (y - x) := by simp only [dist_eq_norm, norm_sub_rev x] exact gradient_norm_sub_sq x y +section +-- Mathlib.Analysis.InnerProductSpace.NormPow + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] +variable {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] + +theorem Differentiable.norm_pow {f : F → E} (hf : Differentiable ℝ f) {p : ℕ} (hp : 1 < p) : + Differentiable ℝ (fun x ↦ ‖f x‖ ^ p) := by + suffices Differentiable ℝ (fun x ↦ ‖f x‖ ^ (p : ℝ)) by + convert this using 1 + simp + exact hf.norm_rpow (by simp [hp]) + +end + +lemma inner_proj_nonpos (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (x : E) {y : E} (hy : y ∈ s) : + ⟪proj s x - x, proj s x - y⟫ ≤ 0 := by + suffices 0 ≤ 2 * ⟪proj s x - x, y - proj s x⟫ by + simp only [Nat.ofNat_pos, mul_nonneg_iff_of_pos_left] at this + rwa [← neg_sub y, inner_neg_right, neg_nonpos] + have h_inner := ConvexOn.inner_gradient_nonneg_of_isMinOn h_convex ?_ ?_ ?_ (x := proj s x) + (y := y) (f := fun z ↦ ‖z - x‖ ^ 2) hy + · rw [real_inner_comm, ← inner_smul_right] + convert h_inner using 2 + symm + convert gradient_norm_sub_sq x (proj s x) + exact ofNat_smul_eq_nsmul ℝ 2 (proj s x - x) + · refine Differentiable.differentiableAt ?_ + refine Differentiable.norm_pow ?_ (by simp) + fun_prop + · exact h_closed.proj_mem h_nonempty x + · exact h_closed.isMinOn_norm_sq_proj h_nonempty x + +lemma inner_proj_nonneg (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (x : E) {y : E} (hy : y ∈ s) : + 0 ≤ ⟪proj s x - x, y - proj s x⟫ := by + rw [← neg_sub _ y, inner_neg_right, neg_nonneg] + exact inner_proj_nonpos h_closed h_convex h_nonempty x hy + +lemma dist_proj_proj_le (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (x y : E) : + dist (proj s x) (proj s y) ≤ dist x y := by + suffices dist (proj s x) (proj s y) ^ 2 ≤ dist x y ^ 2 by simpa [sq_le_sq] using this + have h_eq : ‖x - y‖ ^ 2 = ‖proj s x - proj s y‖ ^ 2 + ‖x - y + proj s y - proj s x‖ ^ 2 + + 2 * ⟪proj s x - x, proj s y - proj s x⟫ + 2 * ⟪proj s y - y, proj s x - proj s y⟫ := by + calc ‖x - y‖ ^ 2 + _ = ‖proj s x - proj s y + (x - y + proj s y - proj s x)‖ ^ 2 := by congr; abel + _ = ‖proj s x - proj s y‖ ^ 2 + ‖x - y + proj s y - proj s x‖ ^ 2 + + 2 * ⟪proj s x - proj s y, x - y + proj s y - proj s x⟫ := by + rw [norm_add_sq (𝕜 := ℝ)] + simp only [RCLike.re_to_real] + ring + _ = ‖proj s x - proj s y‖ ^ 2 + ‖x - y + proj s y - proj s x‖ ^ 2 + + 2 * ⟪proj s x - x, proj s y - proj s x⟫ + 2 * ⟪proj s y - y, proj s x - proj s y⟫ := by + simp_rw [add_assoc] + congr + rw [← mul_add, ← neg_sub _ (proj s y), inner_neg_right, add_comm (- _), ← sub_eq_add_neg, + ← inner_sub_left, real_inner_comm] + congr 2 + abel + simp_rw [dist_eq_norm, h_eq, add_assoc] + refine le_add_of_nonneg_right ?_ + have h1 : 0 ≤ ⟪proj s x - x, proj s y - proj s x⟫ := + inner_proj_nonneg h_closed h_convex h_nonempty x (h_closed.proj_mem h_nonempty y) + have h2 : 0 ≤ ⟪proj s y - y, proj s x - proj s y⟫ := + inner_proj_nonneg h_closed h_convex h_nonempty y (h_closed.proj_mem h_nonempty x) + positivity + lemma dist_proj_le (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x : E) {y : E} (hy : y ∈ s) : dist (proj s x) y ≤ dist x y := by - suffices dist (proj s x) y ^ 2 ≤ dist x y ^ 2 by simpa [sq_le_sq] using this - calc dist (proj s x) y ^ 2 - _ = ‖proj s x - y‖ ^ 2 := by simp [dist_eq_norm] - _ ≤ ‖proj s x - y‖ ^ 2 + ‖proj s x - x‖ ^ 2 + 2 * ⟪proj s x - x, y - proj s x⟫ := by - rw [add_assoc] - refine le_add_of_nonneg_right ?_ - suffices 0 ≤ 2 * ⟪proj s x - x, y - proj s x⟫ by positivity - have h_min := h_closed.isMinOn_norm_sq_proj h_nonempty x hy - have h_inner := ConvexOn.inner_gradient_nonneg_of_isMinOn h_convex ?_ ?_ ?_ (x := proj s x) - (y := y) (f := fun z ↦ ‖z - x‖ ^ 2) hy - · rw [real_inner_comm, ← inner_smul_right] - convert h_inner using 2 - symm - convert gradient_norm_sub_sq x (proj s x) - exact ofNat_smul_eq_nsmul ℝ 2 (proj s x - x) - · suffices DifferentiableAt ℝ (fun z ↦ ‖z - x‖ ^ (2 : ℝ)) (proj s x) by - convert this - simp - refine Differentiable.differentiableAt ?_ - refine Differentiable.norm_rpow ?_ (by simp) - fun_prop - · exact h_closed.proj_mem h_nonempty x - · exact h_closed.isMinOn_norm_sq_proj h_nonempty x - _ = ‖y - proj s x + (proj s x - x)‖ ^ 2 := by - rw [norm_add_sq (𝕜 := ℝ), RCLike.re_to_real, norm_sub_rev, add_assoc, add_comm _ (2 * _), - ← add_assoc, real_inner_comm] - _ = ‖y - x‖ ^ 2 := by congr 2; simp - _ = dist x y ^ 2 := by simp [dist_eq_norm, norm_sub_rev] + nth_rw 1 [← proj_of_mem hy] + exact dist_proj_proj_le h_closed h_convex h_nonempty x y + +lemma lipschitz_proj {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : + LipschitzWith 1 (proj s) := by + intro x y + simp only [ENNReal.coe_one, one_mul, edist_dist] + grw [dist_proj_proj_le h_closed h_convex h_nonempty x y] + +lemma continuous_proj {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : + Continuous (proj s) := (lipschitz_proj h_closed h_convex h_nonempty).continuous end Convex diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 9a9865f4..18f61e35 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -6,6 +6,7 @@ Authors: Rémy Degenne module public import LeanMachineLearning.Online.OnlineRegret +public import LeanMachineLearning.Online.Projection public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.SequentialLearning.StationaryEnv @@ -39,6 +40,14 @@ lemma inner_eq_add (x y g : E) (hη : 0 < η) : simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] field +lemma inner_le_add_proj [FiniteDimensional ℝ E] {s : Set E} (h_closed : IsClosed s) + (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x g : E) {y : E} (hy : y ∈ s) (hη : 0 < η) : + ⟪x - y, g⟫ ≤ (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖proj s (x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by + rw [inner_eq_add x y g hη] + gcongr + rw [← dist_eq_norm, ← dist_eq_norm] + exact dist_proj_le h_closed h_convex h_nonempty (x - η • g) hy + lemma sum_inner_le_sum' (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ ∑ i ∈ Finset.range n, @@ -100,7 +109,9 @@ Since the algorithm is expressed as a function of the history `hist : ℕ → Ii we write `(hist ⟨n, …⟩).1` for `x n` and `(hist ⟨n, …⟩).2` for `g n`. -/ noncomputable def gradientStep (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := - detAlgorithm (fun n hist ↦ (hist ⟨n, by grind⟩).1 - γ n • (hist ⟨n, by grind⟩).2) (by fun_prop) x₀ + let xn := fun (n : ℕ) (hist : Iic n → E × E) ↦ (hist ⟨n, by grind⟩).1 + let gn := fun (n : ℕ) (hist : Iic n → E × E) ↦ (hist ⟨n, by grind⟩).2 + detAlgorithm (fun n hist ↦ xn n hist - γ n • gn n hist) (by fun_prop) x₀ lemma action_gradientStep_ae_eq (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P) (n : ℕ) : X (n + 1) =ᵐ[P] X n - γ n • G n := h_seq.action_detAlgorithm_ae_eq n @@ -116,6 +127,28 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P | zero => simpa | succ n ih => rw [hω n, sum_range_succ, ← sub_sub]; congr +omit [SecondCountableTopology E] [CompleteSpace E] in +lemma measurable_proj [FiniteDimensional ℝ E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : + Measurable (proj s) := (continuous_proj h_closed h_convex h_nonempty).measurable + +omit [SecondCountableTopology E] [CompleteSpace E] in +protected +lemma _root_.Measurable.proj [FiniteDimensional ℝ E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + {f : Ω → E} (hf : Measurable f) : + Measurable (fun ω ↦ proj s (f ω)) := + (measurable_proj h_closed h_convex h_nonempty).comp hf + +noncomputable +def projGradStep [FiniteDimensional ℝ E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := + let xn := fun (n : ℕ) (hist : Iic n → E × E) ↦ (hist ⟨n, by grind⟩).1 + let gn := fun (n : ℕ) (hist : Iic n → E × E) ↦ (hist ⟨n, by grind⟩).2 + detAlgorithm (fun n hist ↦ proj s (xn n hist - γ n • gn n hist)) + (fun _ ↦ Measurable.proj h_closed h_convex h_nonempty (by fun_prop)) x₀ + end Definition namespace GradientStep From a5613a3e1f0a090c090df2a779c144aec3c4fad6 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sun, 28 Jun 2026 09:25:12 +0200 Subject: [PATCH 35/43] docstring --- LeanMachineLearning/Online/Projection.lean | 8 +++----- .../Optimization/Algorithms/GradientDescent.lean | 8 ++++++++ 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/Online/Projection.lean b/LeanMachineLearning/Online/Projection.lean index b678d1e6..8e4ec5f4 100644 --- a/LeanMachineLearning/Online/Projection.lean +++ b/LeanMachineLearning/Online/Projection.lean @@ -182,16 +182,14 @@ lemma dist_proj_le (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty nth_rw 1 [← proj_of_mem hy] exact dist_proj_proj_le h_closed h_convex h_nonempty x y -lemma lipschitz_proj {s : Set E} - (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : +lemma lipschitzWith_proj (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : LipschitzWith 1 (proj s) := by intro x y simp only [ENNReal.coe_one, one_mul, edist_dist] grw [dist_proj_proj_le h_closed h_convex h_nonempty x y] -lemma continuous_proj {s : Set E} - (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : - Continuous (proj s) := (lipschitz_proj h_closed h_convex h_nonempty).continuous +lemma continuous_proj (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : + Continuous (proj s) := (lipschitzWith_proj h_closed h_convex h_nonempty).continuous end Convex diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 18f61e35..35cfab5d 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -140,6 +140,14 @@ lemma _root_.Measurable.proj [FiniteDimensional ℝ E] {s : Set E} Measurable (fun ω ↦ proj s (f ω)) := (measurable_proj h_closed h_convex h_nonempty).comp hf +/-- Projected online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`. + +It is an algorithm that chooses actions in `E` and gets feedback in `E` (gradient of the function at +the queried point). +The point `x (n + 1)` is defined as `x (n + 1) = proj s (x n - γ n • g n)`, where `g n` is +the feedback received at step `n` and `proj s` is the projection onto `s`. +Since the algorithm is expressed as a function of the history `hist : ℕ → Iic n → E × E`, +we write `(hist ⟨n, …⟩).1` for `x n` and `(hist ⟨n, …⟩).2` for `g n`. -/ noncomputable def projGradStep [FiniteDimensional ℝ E] {s : Set E} (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) From f7506ebb288859b6061dc5333161490000a99180 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Wed, 1 Jul 2026 17:54:09 +0200 Subject: [PATCH 36/43] wip --- .../Algorithms/GradientDescent.lean | 107 ++++++++++++++---- 1 file changed, 86 insertions(+), 21 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 35cfab5d..88e40e08 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -40,6 +40,17 @@ lemma inner_eq_add (x y g : E) (hη : 0 < η) : simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos hη] field +lemma inner_eq_add' (x y g : ℕ → E) (hx : ∀ n, x (n + 1) = x n - γ n • g n) + (hγ : ∀ n, 0 < γ n) (i : ℕ) : + ⟪x i - y i, g i⟫ = + (2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖x (i + 1) - y i‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2 := by + have hsub : (x i - γ i • g i) - y i = (x i - y i) - γ i • g i := by abel + simp only [hx] + rw [hsub, norm_sub_sq_real (x i - y i) (γ i • g i)] + simp only [inner_smul_right, norm_smul, Real.norm_eq_abs, abs_of_pos (hγ i)] + specialize hγ i + field + lemma inner_le_add_proj [FiniteDimensional ℝ E] {s : Set E} (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x g : E) {y : E} (hy : y ∈ s) (hη : 0 < η) : ⟪x - y, g⟫ ≤ (2 * η)⁻¹ * (‖x - y‖ ^ 2 - ‖proj s (x - η • g) - y‖ ^ 2) + (η / 2) * ‖g‖ ^ 2 := by @@ -48,38 +59,83 @@ lemma inner_le_add_proj [FiniteDimensional ℝ E] {s : Set E} (h_closed : IsClos rw [← dist_eq_norm, ← dist_eq_norm] exact dist_proj_le h_closed h_convex h_nonempty (x - η • g) hy -lemma sum_inner_le_sum' (x y g : ℕ → E) (hγ : ∀ n, 0 < γ n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y i, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖(x i - γ i • g i) - y i‖ ^ 2) + - (γ i / 2) * ‖g i‖ ^ 2) := by - gcongr with i hi - rw [inner_eq_add (x i) (y i) (g i) (hγ i)] - +lemma inner_le_add_proj' [FiniteDimensional ℝ E] {s : Set E} (h_closed : IsClosed s) + (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x g : ℕ → E) + (hx : ∀ n, x (n + 1) = proj s (x n - γ n • g n)) {y : ℕ → E} (hy : ∀ i, y i ∈ s) + (hγ : ∀ i, 0 < γ i) (i : ℕ) : + ⟪x i - y i, g i⟫ ≤ + (2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖x (i + 1) - y i‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2 := by + grw [inner_le_add_proj h_closed h_convex h_nonempty (x i) (g i) (hy i) (hγ i)] + simp [hx] + +lemma todo (x y g : ℕ → E) + (h : ∀ i, ⟪x i - y i, g i⟫ ≤ + (2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖x (i + 1) - y i‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) (n : ℕ) : + ∑ i ∈ range n, ⟪x i - y i, g i⟫ ≤ + ∑ i ∈ range n, + ((2 * γ i)⁻¹ * (‖x i - y i‖ ^ 2 - ‖x (i + 1) - y i‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := by + grw [h] + +-- todo: change `y` to `ℕ → E` ? lemma sum_inner_le_sum (x g : ℕ → E) (y : E) (hγ : ∀ n, 0 < γ n) (hx : ∀ n, x (n + 1) = x n - γ n • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - ∑ i ∈ Finset.range n, - ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := - (sum_inner_le_sum' x (fun _ ↦ y) g hγ n).trans_eq <| by simp [hx] + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + ∑ i ∈ range n, + ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := by + refine todo x (fun _ ↦ y) g (fun i ↦ ?_) n + grw [inner_eq_add' x (fun _ ↦ y) g hx hγ] + +lemma sum_inner_le_sum_proj [FiniteDimensional ℝ E] {s : Set E} (h_closed : IsClosed s) + (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x g : ℕ → E) (y : E) (hys : y ∈ s) + (hγ : ∀ n, 0 < γ n) (hx : ∀ n, x (n + 1) = proj s (x n - γ n • g n)) (n : ℕ) : + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + ∑ i ∈ range n, + ((2 * γ i)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (γ i / 2) * ‖g i‖ ^ 2) := by + refine todo x (fun _ ↦ y) g (fun i ↦ ?_) n + grw [inner_le_add_proj' h_closed h_convex h_nonempty x g hx (fun _ ↦ hys) hγ] section ConstantStep +lemma todo' (x g : ℕ → E) (y : E) + (h : ∀ i, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (η / 2) * ‖g i‖ ^ 2) (n : ℕ) : + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := by + refine (todo x (fun _ ↦ y) g h n).trans_eq ?_ + rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] + lemma sum_inner_le_add (x g : ℕ → E) (y : E) (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - refine (sum_inner_le_sum x g y (fun _ ↦ hη) hx n).trans_eq ?_ - rw [sum_add_distrib, ← mul_sum, ← mul_sum, Finset.sum_range_sub' (fun i ↦ ‖x i - y‖ ^ 2) n] + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x 0 - y‖ ^ 2 - ‖x n - y‖ ^ 2) + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := + todo' x g y (fun i ↦ (inner_eq_add' x (fun _ ↦ y) g hx (fun _ ↦ hη) i).le) n + +lemma todo'' (x g : ℕ → E) (y : E) + (h : ∀ i, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * (‖x i - y‖ ^ 2 - ‖x (i + 1) - y‖ ^ 2) + (η / 2) * ‖g i‖ ^ 2) + (hη : 0 < η) (n : ℕ) : + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := by + grw [todo' x g y h n] + gcongr + exact sub_le_self _ (sq_nonneg _) /-- Lemma 14.1 in Understanding Machine Learning: From Theory to Algorithms. -/ lemma gradient_descent_linear_regret (x g : ℕ → E) (y : E) (η : ℝ) (hη : 0 < η) (hx : ∀ n, x (n + 1) = x n - η • g n) (n : ℕ) : - ∑ i ∈ Finset.range n, ⟪x i - y, g i⟫ ≤ - (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ Finset.range n, ‖g i‖ ^ 2 := by - grw [sum_inner_le_add x g y hη hx n] - gcongr - exact sub_le_self _ (sq_nonneg _) + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := + todo'' x g y (fun i ↦ (inner_eq_add' x (fun _ ↦ y) g hx (fun _ ↦ hη) i).le) hη n + +lemma proj_gradient_descent_linear_regret [FiniteDimensional ℝ E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (x g : ℕ → E) {y : E} (hys : y ∈ s) (hη : 0 < η) + (hx : ∀ n, x (n + 1) = proj s (x n - η • g n)) (n : ℕ) : + ∑ i ∈ range n, ⟪x i - y, g i⟫ ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := + todo'' x g y + (fun i ↦ inner_le_add_proj' h_closed h_convex h_nonempty x g hx (fun _ ↦ hys) (fun _ ↦ hη) i) + hη n end ConstantStep @@ -89,6 +145,15 @@ lemma onlineRegret_gradientStep_le (x g : ℕ → E) (y : E) (η : ℝ) (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := by simpa [onlineRegret, inner_sub_left] using gradient_descent_linear_regret x g y η hη hx n +lemma onlineRegret_projGradStep_le [FiniteDimensional ℝ E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + (x g : ℕ → E) {y : E} (hys : y ∈ s) (η : ℝ) + (hη : 0 < η) (hx : ∀ n, x (n + 1) = proj s (x n - η • g n)) (n : ℕ) : + onlineRegret (fun n x ↦ ⟪x, g n⟫) y x n ≤ + (2 * η)⁻¹ * ‖x 0 - y‖ ^ 2 + (η / 2) * ∑ i ∈ range n, ‖g i‖ ^ 2 := by + simpa [onlineRegret, inner_sub_left] using + proj_gradient_descent_linear_regret h_closed h_convex h_nonempty x g hys hη hx n + end Linear variable [SecondCountableTopology E] [CompleteSpace E] [BorelSpace E] From 781859e26cdb25f1ca06bb7c5529f3b96f7a5cc5 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Wed, 19 Aug 2026 17:48:36 +0200 Subject: [PATCH 37/43] fix --- LeanMachineLearning/Online/OnlineToBatch.lean | 2 +- LeanMachineLearning/Online/Projection.lean | 8 ++++---- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/Online/OnlineToBatch.lean b/LeanMachineLearning/Online/OnlineToBatch.lean index 388c0ffb..401af470 100644 --- a/LeanMachineLearning/Online/OnlineToBatch.lean +++ b/LeanMachineLearning/Online/OnlineToBatch.lean @@ -37,7 +37,7 @@ lemma memLp_gradient (h : IsAlgEnvSeq X G alg (obliviousEnv ν) P) (h_memLp : ∀ n, MemLp (G n) 2 P) (n : ℕ) : MemLp (fun ω ↦ ∇ (f n) (X n ω)) 2 P := by let M n := MeasurableSpace.comap (X n) inferInstance - have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) + have h_lp : MemLp P[G n | M n] 2 P := (h_memLp n).condExp (m := M n) (by simp) have h_ae := h.condExp_feedback_obliviousEnv_ae_eq_integral_id n ((h_memLp n).integrable (by simp)) refine h_lp.ae_eq <| h_ae.trans ?_ diff --git a/LeanMachineLearning/Online/Projection.lean b/LeanMachineLearning/Online/Projection.lean index 8e4ec5f4..529a883e 100644 --- a/LeanMachineLearning/Online/Projection.lean +++ b/LeanMachineLearning/Online/Projection.lean @@ -45,7 +45,7 @@ lemma _root_.IsClosed.exists_isMinOn_dist [ProperSpace E] obtain ⟨y, hy, hy_min⟩ := h_compact.exists_isMinOn h3 h_cont.continuousOn refine ⟨y, hy.2, ?_⟩ intro u hu - simp only [Set.mem_setOf_eq] + simp only [Set.mem_ofPred_eq] by_cases h1 : u ∈ closedBall x (dist x z) · specialize hy_min ⟨h1, hu⟩ grind @@ -57,12 +57,12 @@ lemma _root_.IsClosed.proj_mem [Zero E] [ProperSpace E] (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : proj s x ∈ s := by have h := h_closed.exists_isMinOn_dist h_nonempty x - rw [proj, dif_pos h] + rw [proj, dite_eq_left h] exact h.choose_spec.1 lemma isMinOn_proj_of_exists [Zero E] {x : E} (h : ∃ y ∈ s, IsMinOn (dist x) s y) : IsMinOn (dist x) s (proj s x) := by - rw [proj, dif_pos h] + rw [proj, dite_eq_left h] exact h.choose_spec.2 /-- If the set is closed and nonempty, then the projection is a minimizer of the distance. -/ @@ -87,7 +87,7 @@ lemma _root_.IsClosed.isMinOn_norm_sq_proj {E : Type*} [NormedAddCommGroup E] [P {s : Set E} (h_closed : IsClosed s) (h_nonempty : s.Nonempty) (x : E) : IsMinOn (fun y ↦ ‖y - x‖ ^ 2) s (proj s x) := by intro y hy - simp only [Set.mem_setOf_eq, sq_le_sq, abs_dist, ← dist_eq_norm, dist_comm _ x] + simp only [Set.mem_ofPred_eq, sq_le_sq, abs_dist, ← dist_eq_norm, dist_comm _ x] exact h_closed.isMinOn_proj h_nonempty x hy section Convex From b8a99159535aa1c4347a915cc60373fc116fd624 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Wed, 19 Aug 2026 17:53:36 +0200 Subject: [PATCH 38/43] move file --- LeanMachineLearning.lean | 2 +- .../Analysis/InnerProductSpace}/Projection.lean | 0 .../Optimization/Algorithms/GradientDescent.lean | 2 +- 3 files changed, 2 insertions(+), 2 deletions(-) rename LeanMachineLearning/{Online => ForMathlib/Analysis/InnerProductSpace}/Projection.lean (100%) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 6c8c3dae..28f5f811 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,6 +1,7 @@ module -- shake: keep-all --deprecated_module: ignore public import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope +public import LeanMachineLearning.ForMathlib.Analysis.InnerProductSpace.Projection public import LeanMachineLearning.ForMathlib.MeasureTheory.Function.ConditionalExpectation.PullOut public import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space public import LeanMachineLearning.ForMathlib.MeasureTheory.Measurable @@ -34,7 +35,6 @@ public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.Online.OnlineRegret public import LeanMachineLearning.Online.OnlineToBatch -public import LeanMachineLearning.Online.Projection public import LeanMachineLearning.Optimization.Algorithms.GradientDescent public import LeanMachineLearning.SequentialLearning.Algorithm public import LeanMachineLearning.SequentialLearning.AlgorithmDensity diff --git a/LeanMachineLearning/Online/Projection.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean similarity index 100% rename from LeanMachineLearning/Online/Projection.lean rename to LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 88e40e08..217d813a 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -5,8 +5,8 @@ Authors: Rémy Degenne -/ module +public import LeanMachineLearning.ForMathlib.Analysis.InnerProductSpace.Projection public import LeanMachineLearning.Online.OnlineRegret -public import LeanMachineLearning.Online.Projection public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.SequentialLearning.StationaryEnv From b9cd07fcd4055b9fb3c98b45d7beeb6a524b0f19 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Wed, 19 Aug 2026 18:22:23 +0200 Subject: [PATCH 39/43] move --- LeanMachineLearning.lean | 1 + .../Analysis/InnerProductSpace/NormPow.lean | 27 +++++++++++++++++++ .../InnerProductSpace/Projection.lean | 17 +----------- 3 files changed, 29 insertions(+), 16 deletions(-) create mode 100644 LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 28f5f811..7b4b215a 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,6 +1,7 @@ module -- shake: keep-all --deprecated_module: ignore public import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope +public import LeanMachineLearning.ForMathlib.Analysis.InnerProductSpace.NormPow public import LeanMachineLearning.ForMathlib.Analysis.InnerProductSpace.Projection public import LeanMachineLearning.ForMathlib.MeasureTheory.Function.ConditionalExpectation.PullOut public import LeanMachineLearning.ForMathlib.MeasureTheory.Function.L2Space diff --git a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean new file mode 100644 index 00000000..75c82796 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean @@ -0,0 +1,27 @@ +/- +Copyright (c) 2026 Rémy Degenne. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Rémy Degenne +-/ +module + +public import Mathlib.Analysis.InnerProductSpace.NormPow + +import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope + +/-! +# Differentiability of the norm to a power + +-/ + +@[expose] public section + +variable {E F : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + [NormedAddCommGroup F] [NormedSpace ℝ F] + +theorem Differentiable.norm_pow {f : F → E} (hf : Differentiable ℝ f) {p : ℕ} (hp : 1 < p) : + Differentiable ℝ (fun x ↦ ‖f x‖ ^ p) := by + suffices Differentiable ℝ (fun x ↦ ‖f x‖ ^ (p : ℝ)) by + convert this using 1 + simp + exact hf.norm_rpow (by simp [hp]) diff --git a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean index 529a883e..25ea4041 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean @@ -6,7 +6,7 @@ Authors: Rémy Degenne module public import Mathlib.Analysis.Calculus.Gradient.Basic -public import Mathlib.Analysis.InnerProductSpace.NormPow +public import LeanMachineLearning.ForMathlib.Analysis.InnerProductSpace.NormPow import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope @@ -107,21 +107,6 @@ lemma gradient_dist_sq (x y : E) : ∇ (fun z ↦ dist x z ^ 2) y = 2 • (y - x simp only [dist_eq_norm, norm_sub_rev x] exact gradient_norm_sub_sq x y -section --- Mathlib.Analysis.InnerProductSpace.NormPow - -variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] -variable {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] - -theorem Differentiable.norm_pow {f : F → E} (hf : Differentiable ℝ f) {p : ℕ} (hp : 1 < p) : - Differentiable ℝ (fun x ↦ ‖f x‖ ^ p) := by - suffices Differentiable ℝ (fun x ↦ ‖f x‖ ^ (p : ℝ)) by - convert this using 1 - simp - exact hf.norm_rpow (by simp [hp]) - -end - lemma inner_proj_nonpos (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x : E) {y : E} (hy : y ∈ s) : ⟪proj s x - x, proj s x - y⟫ ≤ 0 := by From bad994e301129177bca74b2bcaa906a46d98d2db Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Wed, 19 Aug 2026 18:28:28 +0200 Subject: [PATCH 40/43] move --- .../Analysis/InnerProductSpace/NormPow.lean | 18 +++++++++++++++++- .../Analysis/InnerProductSpace/Projection.lean | 12 ------------ 2 files changed, 17 insertions(+), 13 deletions(-) diff --git a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean index 75c82796..9b126f08 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean @@ -5,6 +5,7 @@ Authors: Rémy Degenne -/ module +public import Mathlib.Analysis.Calculus.Gradient.Basic public import Mathlib.Analysis.InnerProductSpace.NormPow import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope @@ -16,12 +17,27 @@ import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope @[expose] public section +open scoped Gradient + variable {E F : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] [NormedAddCommGroup F] [NormedSpace ℝ F] -theorem Differentiable.norm_pow {f : F → E} (hf : Differentiable ℝ f) {p : ℕ} (hp : 1 < p) : +lemma Differentiable.norm_pow {f : F → E} (hf : Differentiable ℝ f) {p : ℕ} (hp : 1 < p) : Differentiable ℝ (fun x ↦ ‖f x‖ ^ p) := by suffices Differentiable ℝ (fun x ↦ ‖f x‖ ^ (p : ℝ)) by convert this using 1 simp exact hf.norm_rpow (by simp [hp]) + +lemma gradient_norm_sub_sq [CompleteSpace E] (x y : E) : + ∇ (fun z ↦ ‖z - x‖ ^ 2) y = 2 • (y - x) := by + have h := ((hasFDerivAt_id y).sub_const x).norm_sq.hasGradientAt.gradient + simp only [id_eq, map_sub, ContinuousLinearMap.comp_id, map_nsmul] at h + rw [h] + congr + · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ + · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ + +lemma gradient_dist_sq [CompleteSpace E] (x y : E) : ∇ (fun z ↦ dist x z ^ 2) y = 2 • (y - x) := by + simp only [dist_eq_norm, norm_sub_rev x] + exact gradient_norm_sub_sq x y diff --git a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean index 25ea4041..c2ba9cc2 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean @@ -95,18 +95,6 @@ section Convex variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] [FiniteDimensional ℝ E] {s : Set E} -lemma gradient_norm_sub_sq (x y : E) : ∇ (fun z ↦ ‖z - x‖ ^ 2) y = 2 • (y - x) := by - have h := ((hasFDerivAt_id y).sub_const x).norm_sq.hasGradientAt.gradient - simp only [id_eq, map_sub, ContinuousLinearMap.comp_id, map_nsmul] at h - rw [h] - congr - · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ - · exact (InnerProductSpace.toDual ℝ E).symm_apply_apply _ - -lemma gradient_dist_sq (x y : E) : ∇ (fun z ↦ dist x z ^ 2) y = 2 • (y - x) := by - simp only [dist_eq_norm, norm_sub_rev x] - exact gradient_norm_sub_sq x y - lemma inner_proj_nonpos (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) (x : E) {y : E} (hy : y ∈ s) : ⟪proj s x - x, proj s x - y⟫ ≤ 0 := by From 4f0e2a87c45f984b268a9dd9d2ffba4c44d6feae Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Wed, 19 Aug 2026 18:30:02 +0200 Subject: [PATCH 41/43] minor --- .../ForMathlib/Analysis/Calculus/Deriv/Slope.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean index 1b24f702..435ff861 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Calculus/Deriv/Slope.lean @@ -13,7 +13,7 @@ import Mathlib.Analysis.Calculus.Deriv.Slope import Mathlib.Analysis.Calculus.LocalExtr.Basic /-! -# Convexity lemmas +# Convexity lemmas about derivatives and gradients -/ From 2e40ccbfe870f0837fd24c308f766b4ee08adf83 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sun, 23 Aug 2026 07:51:56 +0200 Subject: [PATCH 42/43] fix imports --- .../ForMathlib/Analysis/InnerProductSpace/NormPow.lean | 2 +- .../ForMathlib/Analysis/InnerProductSpace/Projection.lean | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean index 9b126f08..92c7c829 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/NormPow.lean @@ -6,9 +6,9 @@ Authors: Rémy Degenne module public import Mathlib.Analysis.Calculus.Gradient.Basic -public import Mathlib.Analysis.InnerProductSpace.NormPow import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope +import Mathlib.Analysis.InnerProductSpace.NormPow /-! # Differentiability of the norm to a power diff --git a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean index c2ba9cc2..dc09fc5e 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/InnerProductSpace/Projection.lean @@ -5,8 +5,9 @@ Authors: Rémy Degenne -/ module -public import Mathlib.Analysis.Calculus.Gradient.Basic public import LeanMachineLearning.ForMathlib.Analysis.InnerProductSpace.NormPow +public import Mathlib.Analysis.Calculus.Gradient.Basic +public import Mathlib.Analysis.InnerProductSpace.NormPow import LeanMachineLearning.ForMathlib.Analysis.Calculus.Deriv.Slope From 589f2ea21685e31e73e82f7809f45b2dc831c58b Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 24 Aug 2026 13:53:34 +0200 Subject: [PATCH 43/43] minor --- .../Algorithms/GradientDescent.lean | 38 ++++++++++++------- 1 file changed, 24 insertions(+), 14 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean index 217d813a..7c5d53b0 100644 --- a/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean +++ b/LeanMachineLearning/Optimization/Algorithms/GradientDescent.lean @@ -31,6 +31,16 @@ variable {Ω E : Type*} {mΩ : MeasurableSpace Ω} {mE : MeasurableSpace E} {P : Measure Ω} [IsProbabilityMeasure P] {x x₀ : E} {X G : ℕ → Ω → E} {γ : ℕ → ℝ} {η : ℝ} +lemma measurable_proj [FiniteDimensional ℝ E] [BorelSpace E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : + Measurable (proj s) := (continuous_proj h_closed h_convex h_nonempty).measurable + +protected lemma _root_.Measurable.proj [FiniteDimensional ℝ E] [BorelSpace E] {s : Set E} + (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) + {f : Ω → E} (hf : Measurable f) : + Measurable (fun ω ↦ proj s (f ω)) := + (measurable_proj h_closed h_convex h_nonempty).comp hf + section Linear lemma inner_eq_add (x y g : E) (hη : 0 < η) : @@ -192,18 +202,8 @@ lemma action_ae_eq_sub_sum (h_seq : IsAlgEnvSeq X G (gradientStep γ x₀) env P | zero => simpa | succ n ih => rw [hω n, sum_range_succ, ← sub_sub]; congr -omit [SecondCountableTopology E] [CompleteSpace E] in -lemma measurable_proj [FiniteDimensional ℝ E] {s : Set E} - (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) : - Measurable (proj s) := (continuous_proj h_closed h_convex h_nonempty).measurable - -omit [SecondCountableTopology E] [CompleteSpace E] in -protected -lemma _root_.Measurable.proj [FiniteDimensional ℝ E] {s : Set E} - (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) - {f : Ω → E} (hf : Measurable f) : - Measurable (fun ω ↦ proj s (f ω)) := - (measurable_proj h_closed h_convex h_nonempty).comp hf +variable [FiniteDimensional ℝ E] + {s : Set E} {h_closed : IsClosed s} {h_convex : Convex ℝ s} {h_nonempty : s.Nonempty} /-- Projected online gradient descent with step sizes `γ : ℕ → ℝ` and initial point `x₀ : E`. @@ -214,14 +214,24 @@ the feedback received at step `n` and `proj s` is the projection onto `s`. Since the algorithm is expressed as a function of the history `hist : ℕ → Iic n → E × E`, we write `(hist ⟨n, …⟩).1` for `x n` and `(hist ⟨n, …⟩).2` for `g n`. -/ noncomputable -def projGradStep [FiniteDimensional ℝ E] {s : Set E} +def projGradStep (s : Set E) (h_closed : IsClosed s) (h_convex : Convex ℝ s) (h_nonempty : s.Nonempty) - (γ : ℕ → ℝ) (x₀ : E) : Algorithm E E := + (γ : ℕ → ℝ) (x₀ : E) : + Algorithm E E := let xn := fun (n : ℕ) (hist : Iic n → E × E) ↦ (hist ⟨n, by grind⟩).1 let gn := fun (n : ℕ) (hist : Iic n → E × E) ↦ (hist ⟨n, by grind⟩).2 detAlgorithm (fun n hist ↦ proj s (xn n hist - γ n • gn n hist)) (fun _ ↦ Measurable.proj h_closed h_convex h_nonempty (by fun_prop)) x₀ +lemma action_projGradStep_ae_eq + (h_seq : IsAlgEnvSeq X G (projGradStep s h_closed h_convex h_nonempty γ x₀) env P) {n : ℕ} : + X (n + 1) =ᵐ[P] fun ω ↦ proj s (X n ω - γ n • G n ω) := h_seq.action_detAlgorithm_ae_eq n + +lemma action_projGradStep_ae_all_eq + (h_seq : IsAlgEnvSeq X G (projGradStep s h_closed h_convex h_nonempty γ x₀) env P) : + ∀ᵐ ω ∂P, X 0 ω = x₀ ∧ ∀ n, X (n + 1) ω = proj s (X n ω - γ n • G n ω) := + h_seq.action_detAlgorithm_ae_all_eq + end Definition namespace GradientStep