From fbbfe2af1d595743f5ddecf414a0aabb19ffb2d3 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 29 May 2026 10:10:11 -0400 Subject: [PATCH 01/82] feat : initial algorithm foundation for linUCB --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 7d539fc3..67c48acc 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -4,6 +4,7 @@ public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.Measura public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From cce2458373a3320f1833e1453f2115a27c1e0853 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 29 May 2026 10:10:11 -0400 Subject: [PATCH 02/82] feat : initial algorithm foundation for linUCB --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 3dbb4310..7d6d4a9b 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -4,6 +4,7 @@ public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.Measura public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 7b654c310a82c41a5ce8abd95c3cf406fbb66eb9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 3 Jun 2026 15:45:44 -0400 Subject: [PATCH 03/82] feat(linUCB): proving generic one-step bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 7c45c27b..adf9767a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -225,6 +225,41 @@ lemma forall_index_le_index_arm [Nonempty (Fin K)] end AlgorithmBehavior +omit [IsMarkovKernel ν] in +/-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the +selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected +arm's LinUCB width. -/ +lemma mean_sub_mean_arm_le_two_mul_width (a : Fin K) + (h_best : (ν a)[id] ≤ index A R reg β x a n ω) + (h_arm : estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (h_le : index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω) : + (ν a)[id] - (ν (A n ω))[id] ≤ + 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + rw [sub_le_iff_le_add'] + calc + (ν a)[id] ≤ index A R reg β x a n ω := h_best + _ ≤ index A R reg β x (A n ω) n ω := h_le + _ ≤ (ν (A n ω))[id] + + 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + rw [index, two_mul, ← add_assoc] + gcongr + rwa [sub_le_iff_le_add] at h_arm + +omit [IsMarkovKernel ν] in +/-- The gap of the selected arm is bounded by twice its LinUCB bonus whenever the usual confidence +inequalities hold and the selected arm has maximal LinUCB index. -/ +lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] + (h_best : (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (h_le : index A R reg β x (bestArm ν) n ω ≤ + index A R reg β x (A n ω) n ω) : + gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + rw [gap_eq_bestArm_sub] + exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) (a := bestArm ν) h_best h_arm h_le + end LinUCB end Bandits From 394750eb24342d2e37ba7df24bb9fc1935fc9492 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 4 Jun 2026 15:35:38 -0400 Subject: [PATCH 04/82] feat(linUCB): almost-sure instantaneous regret/gap bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index adf9767a..596cfebc 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -260,6 +260,21 @@ lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (a := bestArm ν) h_best h_arm h_le +/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus whenever the usual +confidence inequalities hold almost surely. -/ +lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (hn : n ≠ 0) + (h_best : ∀ᵐ ω ∂P, (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, + gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + filter_upwards [h_best, h_arm, index_le_index_arm h (bestArm ν) hn] with + ω h_bestω h_armω h_leω + exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) h_bestω h_armω h_leω + end LinUCB end Bandits From c27ce9be436e452c02539a61b6a6fcfd8bf78e70 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 5 Jun 2026 13:18:40 -0400 Subject: [PATCH 05/82] feat(linUCB): all-positive-times instantaneous gap bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 596cfebc..4eb0e7a2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -275,6 +275,23 @@ lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h_bestω h_armω h_leω +/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus at every positive +time whenever the usual confidence inequalities hold almost surely at every positive time. -/ +lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + filter_upwards [h_best, h_arm, forall_index_le_index_arm h (bestArm ν)] with + ω h_bestω h_armω h_leω + intro n hn + exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) + end LinUCB end Bandits From 9cde1581fb12f5bac428b3bb6c8eda195c2c07bb Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 8 Jun 2026 16:51:40 -0400 Subject: [PATCH 06/82] feat(linUCB): cumulative regret bridge --- .../Online/Bandit/Algorithms/LinUCB.lean | 48 +++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 4eb0e7a2..77e33432 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -292,6 +292,54 @@ lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) +omit [IsMarkovKernel ν] in +/-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded +by the corresponding sum of pointwise bounds. -/ +lemma regret_le_sum_of_gap_bound (B : ℕ → ℝ) + (hB : ∀ t, t ∈ range n → gap ν (A t ω) ≤ B t) : + regret ν A n ω ≤ ∑ t ∈ range n, B t := by + rw [regret_eq_sum_gap] + exact sum_le_sum hB + +omit [IsMarkovKernel ν] in +/-- A pathwise cumulative-regret bound obtained by summing the positive-time LinUCB width bound. + +The time-zero gap is left unchanged because the current LinUCB max-index theorem applies only at +positive times. -/ +lemma regret_le_sum_width_of_forall_gap_le + (h_gap : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) : + regret ν A n ω ≤ + ∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by + refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) + (B := fun t ↦ + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simpa [ht0] using h_gap t ht ht0 + +/-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the +usual confidence inequalities hold almost surely at every positive time. -/ +lemma regret_ae_le_sum_width [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + ∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by + filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm] with ω h_gapω + exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 + end LinUCB end Bandits From af7a7185596e991b161c7b9947084983b79038f8 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 9 Jun 2026 11:25:23 -0400 Subject: [PATCH 07/82] sum of widths inequality using Cauchy-Schwarz from mathlib --- .../Online/Bandit/Algorithms/LinUCB.lean | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 77e33432..0ae161df 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -322,6 +322,34 @@ lemma regret_le_sum_width_of_forall_gap_le · simp [ht0] · simpa [ht0] using h_gap t ht ht0 +omit [IsMarkovKernel ν] in +/-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ +lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ≤ + 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + calc + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) + = 2 * ∑ t ∈ range n, + (if t = 0 then 0 else √(β (t + 1))) * + (if t = 0 then 0 else width A reg x (A t ω) t ω) := by + rw [mul_sum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + gcongr + exact Real.sum_mul_le_sqrt_mul_sqrt (range n) + (fun t ↦ if t = 0 then 0 else √(β (t + 1))) + (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) + /-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the usual confidence inequalities hold almost surely at every positive time. -/ lemma regret_ae_le_sum_width [Nonempty (Fin K)] @@ -340,6 +368,40 @@ lemma regret_ae_le_sum_width [Nonempty (Fin K)] exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 +/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound on +the positive-time LinUCB width terms. -/ +lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + filter_upwards [regret_ae_le_sum_width h h_best h_arm] with ω h_regret + refine h_regret.trans ?_ + have hsplit : + (∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) = + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + ∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by + rw [← sum_add_distrib] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + rw [hsplit] + exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) + (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ + end LinUCB end Bandits From 93b64a8ac9c8403edf8aad5516784bfbdc1b5a7d Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 9 Jun 2026 13:32:14 -0400 Subject: [PATCH 08/82] feat(linUCB): simplify the Cauchy beta factor --- .../Online/Bandit/Algorithms/LinUCB.lean | 30 +++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 0ae161df..26312bda 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -350,6 +350,17 @@ lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : (fun t ↦ if t = 0 then 0 else √(β (t + 1))) (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) +/-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is +nonnegative. -/ +lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : + (∑ t ∈ range n, if t = 0 then 0 else √(β (t + 1)) ^ 2) = + ∑ t ∈ range n, if t = 0 then 0 else β (t + 1) := by + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0, Real.sq_sqrt (hβ t)] + /-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the usual confidence inequalities hold almost surely at every positive time. -/ lemma regret_ae_le_sum_width [Nonempty (Fin K)] @@ -402,6 +413,25 @@ lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ +/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound whose +beta factor has been simplified using nonnegativity of the confidence schedule. -/ +lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + filter_upwards [regret_ae_le_initial_add_cauchy (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret + simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret + end LinUCB end Bandits From 2db39aa49ae4537e10c82be132140de438feb7f1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 12:00:40 -0400 Subject: [PATCH 09/82] squared LinUCB width sum is bounded by W, then the regret bound can use square root W --- .../Online/Bandit/Algorithms/LinUCB.lean | 43 +++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 26312bda..d9af312c 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -432,6 +432,49 @@ lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret +omit [IsMarkovKernel ν] in +/-- If the squared LinUCB widths are bounded by `W`, then the Cauchy-Schwarz regret bound can use +`√W` in place of the square root of the realized squared-width sum. -/ +lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) + (h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) + (hW : (∑ t ∈ range n, + (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (_hW_nonneg : 0 ≤ W) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by + refine h_regret.trans ?_ + gcongr + +/-- Almost surely, cumulative regret is bounded by the initial gap plus +`2 * √(sum beta terms) * √W` whenever the squared LinUCB widths are almost surely bounded by `W`. + +This is the interface expected from a future elliptical-potential bound. -/ +lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) + (hW : ∀ᵐ ω ∂P, + (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by + filter_upwards [regret_ae_le_initial_add_cauchy_simplified (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with + ω h_regret hWω + exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω hW_nonneg + end LinUCB end Bandits From f813c857cb01534ce7732bf884a6da43a7445073 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 12:14:34 -0400 Subject: [PATCH 10/82] =?UTF-8?q?feat(linUCB):=20moving=20closer=20to=20te?= =?UTF-8?q?xt=20book=20version=20regret=20=E2=89=A4=20initial=20gap=20+=20?= =?UTF-8?q?2=20*=20=E2=88=9AB=20*=20=E2=88=9AW?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index d9af312c..f86001dd 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -475,6 +475,46 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω hW_nonneg +omit [IsMarkovKernel ν] in +/-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square +root of the beta sum. -/ +lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) + (h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) + (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) + (_hB_nonneg : 0 ≤ B) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by + refine h_regret.trans ?_ + gcongr + +/-- Almost surely, cumulative regret is bounded by the initial gap plus +`2 * √B * √W` whenever the beta sum is bounded by `B` and the squared LinUCB widths are almost +surely bounded by `W`. -/ +lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) + (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) + (hB_nonneg : 0 ≤ B) + (hW : ∀ᵐ ω ∂P, + (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by + filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW + hW_nonneg] with ω h_regret + exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) + (n := n) (ω := ω) B W h_regret hB hB_nonneg + end LinUCB end Bandits From bf90410b75c453b62cd855c88cbc42e06b202655 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 15:24:37 -0400 Subject: [PATCH 11/82] =?UTF-8?q?feat(linUCB):=20moving=20closer=20to=20te?= =?UTF-8?q?xt=20book=20version=20regret=20=E2=89=A4=20initial=20gap=20+=20?= =?UTF-8?q?2=20*=20=E2=88=9A(n=20*=20=CE=B2=20n)=20*=20=E2=88=9AW?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 53 +++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f86001dd..a9f55340 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -515,6 +515,59 @@ lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) (n := n) (ω := ω) B W h_regret hB hB_nonneg +/-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is +bounded by the horizon times the terminal beta value. -/ +lemma beta_sum_le_nat_mul_of_monotone + (hβ_mono : Monotone β) (hβ : ∀ t, 0 ≤ β (t + 1)) : + (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ (n : ℝ) * β n := by + calc + (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) + ≤ ∑ _t ∈ range n, β n := by + refine sum_le_sum ?_ + intro t ht + by_cases ht0 : t = 0 + · rw [if_pos ht0] + have hn_pos : 0 < n := by + simpa [ht0] using mem_range.mp ht + have hn_beta : 0 ≤ β n := by + have htime : n - 1 + 1 = n := by grind + simpa [htime] using hβ (n - 1) + exact hn_beta + · rw [if_neg ht0] + exact hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) + _ = (n : ℝ) * β n := by + simp [sum_const, nsmul_eq_mul] + +/-- Almost surely, cumulative regret is bounded by the initial gap plus +`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` +is nonnegative and monotone. -/ +lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (hW : ∀ᵐ ω ∂P, + (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √W) := by + refine regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W + (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) ?_ + hW hW_nonneg + by_cases hn : n = 0 + · simp [hn] + · have hn_pos : 0 < n := Nat.pos_of_ne_zero hn + have hn_beta : 0 ≤ β n := by + have htime : n - 1 + 1 = n := by grind + simpa [htime] using hβ (n - 1) + positivity + end LinUCB end Bandits From d8bcffb54dd219219784b74136a154bc213c6001 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 16:03:36 -0400 Subject: [PATCH 12/82] feat(linUCB): remaining initial gap sum + cleanup --- .../Online/Bandit/Algorithms/LinUCB.lean | 47 +++++++++++++++---- 1 file changed, 39 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a9f55340..c10aabe6 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -127,6 +127,11 @@ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) +/-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ +noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) @@ -441,12 +446,12 @@ lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) - (hW : (∑ t ∈ range n, - (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : widthSqSum A reg x n ω ≤ W) (_hW_nonneg : 0 ≤ W) : regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by + rw [widthSqSum] at hW refine h_regret.trans ?_ gcongr @@ -462,8 +467,7 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, - (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) (hW_nonneg : 0 ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -503,8 +507,7 @@ lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) (hB_nonneg : 0 ≤ B) - (hW : ∀ᵐ ω ∂P, - (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) (hW_nonneg : 0 ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -538,6 +541,14 @@ lemma beta_sum_le_nat_mul_of_monotone _ = (n : ℝ) * β n := by simp [sum_const, nsmul_eq_mul] +omit [IsMarkovKernel ν] in +/-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the +horizon is zero. -/ +lemma initial_gap_sum_eq : + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) = + if n = 0 then 0 else gap ν (A 0 ω) := by + cases n <;> simp + /-- Almost surely, cumulative regret is bounded by the initial gap plus `2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` is nonnegative and monotone. -/ @@ -549,8 +560,7 @@ lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, - (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) (hW_nonneg : 0 ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -568,6 +578,27 @@ lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] simpa [htime] using hβ (n - 1) positivity +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` +is nonnegative and monotone. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW + hW_nonneg] with ω h_regret + simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret + end LinUCB end Bandits From 2eaf4356ad252f232a53ff37941117bef25f18a1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 09:39:02 -0400 Subject: [PATCH 13/82] feat(linUCB): widthSqSum as quadtratic with proof --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index c10aabe6..0e24b1e2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -127,6 +127,15 @@ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) +/-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that +quadratic form is nonnegative. -/ +lemma width_sq_eq_quadratic_form (a : Fin K) + (h_nonneg : 0 ≤ + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) : + width A reg x a n ω ^ 2 = + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) := by + simp [width, Real.sq_sqrt h_nonneg] + /-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 32d15ef48925cc224e8d9238bba0caf3280df5e7 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 09:44:09 -0400 Subject: [PATCH 14/82] feat(linUCB): widthSqSum as quadtratic with proof --- .../Online/Bandit/Algorithms/LinUCB.lean | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 0e24b1e2..a95a966b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -141,6 +141,27 @@ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 +/-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive +time quadratic form is nonnegative. -/ +lemma widthSqSum_eq_sum_quadratic_form + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) : + widthSqSum A reg x n ω = + ∑ t ∈ range n, + if t = 0 then 0 else + dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) := by + rw [widthSqSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0] + rw [if_neg ht0] + exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) + (n := t) (ω := ω) (h_nonneg t ht ht0) + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) From 7bfa214b8892ff431b2bfadf7c458400bc21e183 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 10:56:51 -0400 Subject: [PATCH 15/82] feat(linUCB): bridge from a quadratic-form sum bound to a widthSqSum bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a95a966b..858473e7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -162,6 +162,22 @@ lemma widthSqSum_eq_sum_quadratic_form exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) (n := t) (ω := ω) (h_nonneg t ht ht0) +/-- A quadratic-form sum bound implies the corresponding bound on `widthSqSum`. This is the shape +expected from a later elliptical-potential argument. -/ +lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) + (h_quad_le : + (∑ t ∈ range n, + if t = 0 then 0 else + dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + exact h_quad_le + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) From ae8310916cb51d91f3808c010f44c56add8bd5ba Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 11:00:55 -0400 Subject: [PATCH 16/82] feat(linUCB): clean up repeated term with quadraticWidthSum --- .../Online/Bandit/Algorithms/LinUCB.lean | 22 +++++++++---------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 858473e7..e058569e 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -141,18 +141,22 @@ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 +/-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ +noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else + dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ dotProduct (x (A t ω)) (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) : - widthSqSum A reg x n ω = - ∑ t ∈ range n, - if t = 0 then 0 else - dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) := by - rw [widthSqSum] + widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by + rw [widthSqSum, quadraticWidthSum] refine sum_congr rfl ?_ intro t ht by_cases ht0 : t = 0 @@ -168,11 +172,7 @@ lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ dotProduct (x (A t ω)) (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) - (h_quad_le : - (∑ t ∈ range n, - if t = 0 then 0 else - dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) ≤ W) : + (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : widthSqSum A reg x n ω ≤ W := by rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg] From 5c922c7618f2864858bba59b0fa91de159bcbe21 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 11:27:25 -0400 Subject: [PATCH 17/82] feat(linUCB): lint clean up --- .../Online/Bandit/Algorithms/LinUCB.lean | 58 +++++++++---------- 1 file changed, 28 insertions(+), 30 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e058569e..d4d14d1b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -29,24 +29,30 @@ section Algorithm namespace LinUCB +/-- Feature vectors for finite-dimensional linear bandits. -/ abbrev Feature (d : ℕ) := Fin d → ℝ +/-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) +/-- History-level response vector for LinUCB. -/ noncomputable def responseVector' (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := ∑ s : Iic n, (h s).2 • x (h s).1 +/-- History-level regularized least-squares estimate. -/ noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) +/-- History-level estimated reward of an arm. -/ noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := dotProduct (thetaHat' reg x n h) (x a) +/-- History-level elliptical confidence width of an arm. -/ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) @@ -65,7 +71,6 @@ open Classical in /-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK measurableArgmax (fun h a ↦ index' reg β x n h a) h @@ -75,7 +80,7 @@ lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) (n : ℕ) : - Measurable (nextArm hK reg β x h_index n) := by + Measurable (nextArm hK reg β x n) := by have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact measurable_measurableArgmax fun a ↦ h_index n a @@ -86,7 +91,7 @@ noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → LinUCB.Feature d) (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : Algorithm (Fin K) ℝ := - detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ end Algorithm @@ -107,6 +112,12 @@ noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) +/-- The design matrix update after observing one additional action. -/ +lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designMatrix A reg x (n + 1) ω = + designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by + simp [designMatrix, sum_range_succ, add_assoc] + /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := @@ -237,7 +248,7 @@ lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (n : ℕ) : A (n + 1) =ᵐ[P] - fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact h.action_detAlgorithm_ae_eq n @@ -246,7 +257,7 @@ lemma arm_ae_all_eq [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : ∀ᵐ ω ∂P, ∀ n, A (n + 1) ω = - nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by simp_rw [ae_all_iff] exact fun n ↦ arm_ae_eq_linUCBNextArm h n @@ -493,7 +504,7 @@ lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) (hW : widthSqSum A reg x n ω ≤ W) - (_hW_nonneg : 0 ≤ W) : + : regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by @@ -513,8 +524,7 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + @@ -523,7 +533,7 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with ω h_regret hWω exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω hW_nonneg + (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω omit [IsMarkovKernel ν] in /-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square @@ -534,7 +544,7 @@ lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (_hB_nonneg : 0 ≤ B) : + : regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by refine h_regret.trans ?_ @@ -552,17 +562,15 @@ lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (hB_nonneg : 0 ≤ B) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW - hW_nonneg] with ω h_regret + ] with ω h_regret exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) - (n := n) (ω := ω) B W h_regret hB hB_nonneg + (n := n) (ω := ω) B W h_regret hB /-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is bounded by the horizon times the terminal beta value. -/ @@ -606,23 +614,14 @@ lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√((n : ℝ) * β n) * √W) := by - refine regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) + exact regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W - (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) ?_ - hW hW_nonneg - by_cases hn : n = 0 - · simp [hn] - · have hn_pos : 0 < n := Nat.pos_of_ne_zero hn - have hn_beta : 0 ≤ β n := by - have htime : n - 1 + 1 = n := by grind - simpa [htime] using hβ (n - 1) - positivity + (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) hW /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` @@ -635,14 +634,13 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW - hW_nonneg] with ω h_regret + ] with ω h_regret simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret end LinUCB From d2bc9e263c059e8ec3ffdd51eff02c0e10de3a15 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 13:09:21 -0400 Subject: [PATCH 18/82] feat(linUCB): design matrix is just the regularization matrix --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index d4d14d1b..dcd7988f 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -112,6 +112,11 @@ noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) +/-- The initial design matrix before any actions are included. -/ +lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + designMatrix A reg x 0 ω = reg • 1 := by + simp [designMatrix] + /-- The design matrix update after observing one additional action. -/ lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : designMatrix A reg x (n + 1) ω = From 0b0a9045fe7ccb7068c3f11f73f16142bbc6d12a Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 14:18:46 -0400 Subject: [PATCH 19/82] feat(linUCB): proof for initial response vector --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index dcd7988f..8cb15783 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -128,6 +128,12 @@ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := ∑ s ∈ range n, R s ω • x (A s ω) +/-- The initial response vector before any rewards are included. -/ +lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (ω : Ω) : + responseVector A R x 0 ω = 0 := by + simp [responseVector] + /-- The process-level regularized least-squares estimate. -/ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := From f1da71a20b6026161946f8ca61f3512f01b39377 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 15:18:40 -0400 Subject: [PATCH 20/82] feat(linUCB): responseVector is the LinUCB reward-feature accumulator --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 8cb15783..1f8a5872 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -134,6 +134,13 @@ lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) responseVector A R x 0 ω = 0 := by simp [responseVector] +/-- The response-vector update after observing one additional reward. -/ +lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + responseVector A R x (n + 1) ω = + responseVector A R x n ω + R n ω • x (A n ω) := by + simp [responseVector, sum_range_succ] + /-- The process-level regularized least-squares estimate. -/ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := From a96beab49491a715418f376251be0bad9ad806fc Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 10:19:15 -0400 Subject: [PATCH 21/82] feat(linUCB): estimated reward for any arm is zero --- .../Online/Bandit/Algorithms/LinUCB.lean | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 1f8a5872..fa354124 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -146,11 +146,24 @@ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) +/-- The initial least-squares estimate is zero because no reward-feature observations have been +included yet. -/ +lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + thetaHat A R reg x 0 ω = 0 := by + simp [thetaHat, responseVector_zero] + /-- The process-level estimated linear reward. -/ noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := dotProduct (thetaHat A R reg x n ω) (x a) +/-- The initial estimated reward is zero for every arm. -/ +lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + estimatedReward A R reg x a 0 ω = 0 := by + simp [estimatedReward, thetaHat_zero] + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := From 5c633f2898d723aaa4918f109d58eeff93fc7dd9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 10:33:01 -0400 Subject: [PATCH 22/82] =?UTF-8?q?feat(linUCB):=20at=20time=200,=20LinUCB?= =?UTF-8?q?=E2=80=99s=20optimistic=20index=20for=20an=20arm=20is=20just=20?= =?UTF-8?q?its=20uncertainty=20bonus?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index fa354124..373c8d05 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -226,6 +226,13 @@ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (n : ℕ) (ω : Ω) : ℝ := estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω +/-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is +zero. -/ +lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by + simp [index, estimatedReward_zero] + lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = From 523728e4b3d153076d8fd6d10e839235c1bc8c65 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 10:37:06 -0400 Subject: [PATCH 23/82] feat(linUCB): at 0 time, the LinUCB width is computed using only the initial design matrix --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 373c8d05..abd220d7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -169,6 +169,13 @@ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) +/-- The initial width is the quadratic form induced by the inverse regularized identity. -/ +lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + width A reg x a 0 ω = + √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by + simp [width, designMatrix_zero] + /-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that quadratic form is nonnegative. -/ lemma width_sq_eq_quadratic_form (a : Fin K) From c47c7e1f8362094ceff53fc71785a48b19233756 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:08:58 -0400 Subject: [PATCH 24/82] feat(linUCB): combining index_zero, width_zero --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index abd220d7..75c1153d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -240,6 +240,14 @@ lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by simp [index, estimatedReward_zero] +/-- At time zero, the LinUCB index is the confidence schedule times the initial quadratic-form +width. -/ +lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + index A R reg β x a 0 ω = + √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by + simp [index_zero, width_zero] + lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = From 84fd6fd677a71da0b5eb23a2054262c0bda50fc3 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:13:52 -0400 Subject: [PATCH 25/82] feat(linUCB): the accumulated squared-width term is zero at horizon 0 --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 75c1153d..eb1d2077 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -190,6 +190,12 @@ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 +/-- No positive-time widths are accumulated at horizon zero. -/ +lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + widthSqSum A reg x 0 ω = 0 := by + simp [widthSqSum] + /-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 33ee65ab07d6f1cfd4700734623d84eda9243d51 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:15:03 -0400 Subject: [PATCH 26/82] feat(linUCB): proof for quadraticWidthSum --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index eb1d2077..9849935e 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -204,6 +204,12 @@ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) dotProduct (x (A t ω)) (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) +/-- No positive-time quadratic width forms are accumulated at horizon zero. -/ +lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + quadraticWidthSum A reg x 0 ω = 0 := by + simp [quadraticWidthSum] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From fe9b9f1de5b969ae9f47cbfca6b5470c201e0ce2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:20:56 -0400 Subject: [PATCH 27/82] feat(linUCB): when the horizon advances from n to n + 1, the accumulated squared-width sum equals the old sum plus the new final term at time n --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 9849935e..979b4ad5 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -196,6 +196,14 @@ lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) widthSqSum A reg x 0 ω = 0 := by simp [widthSqSum] +/-- Advancing the horizon adds the next positive-time squared width term. -/ +lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + + (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by + simp [widthSqSum, sum_range_succ] + /-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 1d1725ea6085610e1cfab7292533f56746834686 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:28:05 -0400 Subject: [PATCH 28/82] feat(linUCB): when the horizon advances from n to n + 1, the accumulated quadratic squared-width sum equals the old quadratic sum plus the new final term at time n --- .../Online/Bandit/Algorithms/LinUCB.lean | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 979b4ad5..e9d9b603 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -218,6 +218,16 @@ lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) quadraticWidthSum A reg x 0 ω = 0 := by simp [quadraticWidthSum] +/-- Advancing the horizon adds the next positive-time quadratic width form. -/ +lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + if n = 0 then 0 else + dotProduct (x (A n ω)) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + simp [quadraticWidthSum, sum_range_succ] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From c86ad3e7ae68582bbbc7129e9c155d23c499eebf Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:31:23 -0400 Subject: [PATCH 29/82] feat(linUCB): proof for widthSqSum_succ when n not equal to 0 --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e9d9b603..e5bdf266 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -204,6 +204,13 @@ lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by simp [widthSqSum, sum_range_succ] +/-- At positive times, advancing the horizon adds the selected arm's squared width. -/ +lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + width A reg x (A n ω) n ω ^ 2 := by + simp [widthSqSum_succ, hn] + /-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 3f94ae7a2c472728de2bf4c33afaf5a242f338a3 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:34:19 -0400 Subject: [PATCH 30/82] feat(linUCB): proof for quadraticWidth_succ when n not equal to 0 --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e5bdf266..1824ab82 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -235,6 +235,15 @@ lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by simp [quadraticWidthSum, sum_range_succ] +/-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ +lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + dotProduct (x (A n ω)) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + simp [quadraticWidthSum_succ, hn] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From 94c32a3e07e4f62696c5ccb4a54de2234fafd68e Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:41:20 -0400 Subject: [PATCH 31/82] feat(linUCB): if the two accumulators agree at time n, then they also agree at time n + 1, as long as the new quadratic form is nonnegative --- .../Online/Bandit/Algorithms/LinUCB.lean | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 1824ab82..fd7a1679 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -244,6 +244,20 @@ lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by simp [quadraticWidthSum_succ, hn] +/-- If the squared-width and quadratic-form accumulators agree through a positive time and the +next quadratic form is nonnegative, then they still agree after adding the next term. -/ +lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) + (h_nonneg : 0 ≤ dotProduct (x (A n ω)) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω)))) : + widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by + rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, + quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hn, h_eq] + rw [width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A n ω) + (n := n) (ω := ω) h_nonneg] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From ee5c82bdbb63f2e8a14e92a99a933d25ec65e6ef Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:52:08 -0400 Subject: [PATCH 32/82] feat(linUCB): refactor --- .../Online/Bandit/Algorithms/LinUCB.lean | 46 ++++++++++--------- 1 file changed, 24 insertions(+), 22 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index fd7a1679..97396fe0 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -164,25 +164,35 @@ lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) estimatedReward A R reg x a 0 ω = 0 := by simp [estimatedReward, thetaHat_zero] +/-- The quadratic form `x_aᵀ V_n⁻¹ x_a` underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) + +/-- The initial width quadratic form is induced by the inverse regularized identity. -/ +lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + widthQuadraticForm A reg x a 0 ω = + dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by + simp [widthQuadraticForm, designMatrix_zero] + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + √(widthQuadraticForm A reg x a n ω) /-- The initial width is the quadratic form induced by the inverse regularized identity. -/ lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : width A reg x a 0 ω = √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [width, designMatrix_zero] + simp [width, widthQuadraticForm_zero] /-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that quadratic form is nonnegative. -/ lemma width_sq_eq_quadratic_form (a : Fin K) - (h_nonneg : 0 ≤ - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) : - width A reg x a n ω ^ 2 = - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) := by + (h_nonneg : 0 ≤ widthQuadraticForm A reg x a n ω) : + width A reg x a n ω ^ 2 = widthQuadraticForm A reg x a n ω := by simp [width, Real.sq_sqrt h_nonneg] /-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ @@ -215,9 +225,7 @@ lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, - if t = 0 then 0 else - dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) + if t = 0 then 0 else widthQuadraticForm A reg x (A t ω) t ω /-- No positive-time quadratic width forms are accumulated at horizon zero. -/ lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -230,18 +238,14 @@ lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : quadraticWidthSum A reg x (n + 1) ω = quadraticWidthSum A reg x n ω + - if n = 0 then 0 else - dotProduct (x (A n ω)) - (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + if n = 0 then 0 else widthQuadraticForm A reg x (A n ω) n ω := by simp [quadraticWidthSum, sum_range_succ] /-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - dotProduct (x (A n ω)) - (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by simp [quadraticWidthSum_succ, hn] /-- If the squared-width and quadratic-form accumulators agree through a positive time and the @@ -249,8 +253,7 @@ next quadratic form is nonnegative, then they still agree after adding the next lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) - (h_nonneg : 0 ≤ dotProduct (x (A n ω)) - (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω)))) : + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) @@ -262,8 +265,7 @@ lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) : + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by rw [widthSqSum, quadraticWidthSum] refine sum_congr rfl ?_ @@ -279,8 +281,7 @@ lemma widthSqSum_eq_sum_quadratic_form expected from a later elliptical-potential argument. -/ lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : widthSqSum A reg x n ω ≤ W := by rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) @@ -346,7 +347,8 @@ lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + simp [width, widthQuadraticForm, width', + designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : From e75d4ca5e0384dea8a60a8804960edac6bfde9c9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 11:34:38 -0400 Subject: [PATCH 33/82] feat(linUCB): history-level companion to widthQuadraticForm --- .../Online/Bandit/Algorithms/LinUCB.lean | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 97396fe0..4227a4d2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -52,10 +52,15 @@ noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := dotProduct (thetaHat' reg x n h) (x a) +/-- History-level quadratic form underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) + /-- History-level elliptical confidence width of an arm. -/ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + √(widthQuadraticForm' reg x n h a) /-- LinUCB optimistic index of an arm. @@ -344,11 +349,18 @@ lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] +lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthQuadraticForm A reg x a n ω = + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [widthQuadraticForm, widthQuadraticForm', + designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, widthQuadraticForm, width', - designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n + ω hn] lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : From 9c2a1cc5e16e6bb1256b2fa3dc9b624e4d7346ec Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:06:22 -0400 Subject: [PATCH 34/82] feat(linUCB): history-level analogue of the existing process-level square-width lemma --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 4227a4d2..9d018a2c 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -62,6 +62,14 @@ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := √(widthQuadraticForm' reg x n h a) +/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic +form is nonnegative. -/ +lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) + (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : + width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by + simp [width', Real.sq_sqrt h_nonneg] + /-- LinUCB optimistic index of an arm. The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` From 9aea4326f1f765fabf445a241f24c933238e2d38 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:09:56 -0400 Subject: [PATCH 35/82] feat(linUCB): transports the nonnegativity condition across the process/history bridge --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 9d018a2c..04e5facc 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -364,6 +364,14 @@ lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Featu simp [widthQuadraticForm, widthQuadraticForm', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] +/-- At positive process times, nonnegativity of the process-level width quadratic form is +equivalent to nonnegativity of the matching history-level width quadratic form. -/ +lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + 0 ≤ widthQuadraticForm A reg x a n ω ↔ + 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] + lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by From a943c01a84fd7e2da8956362117a189a4fa9993b Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:18:21 -0400 Subject: [PATCH 36/82] feat(linUCB): for positive process time n, the process-level squared width can be rewritten directly as the matching history-level quadratic form --- .../Online/Bandit/Algorithms/LinUCB.lean | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 04e5facc..77249f7f 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -378,6 +378,18 @@ lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] +/-- At positive process times, squaring the process-level width recovers the matching history-level +quadratic form when that history-level quadratic form is nonnegative. -/ +lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_nonneg : + 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a) : + width A reg x a n ω ^ 2 = + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + rw [width_eq_width' (A := A) (R := R) reg x a n ω hn] + exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a + h_nonneg + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = From 49ad7c992ac0528a68d83a86d20daa60f5dbf7c0 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:22:26 -0400 Subject: [PATCH 37/82] feat(linUCB): widthSqSum(n + 1)=widthSqSum(n)+history-level quadratic form for the selected arm --- .../Online/Bandit/Algorithms/LinUCB.lean | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 77249f7f..38d2f204 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -390,6 +390,18 @@ lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a h_nonneg +/-- At positive process times, advancing `widthSqSum` adds the matching history-level quadratic +form when that history-level quadratic form is nonnegative. -/ +lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_nonneg : + 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] + rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = From 54db96b4a2b9d1342b95968bfec1354a28f945a1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:29:15 -0400 Subject: [PATCH 38/82] feat(linUCB): sums the history-level quadratic forms --- .../Online/Bandit/Algorithms/LinUCB.lean | 119 ++++++++++++++++++ 1 file changed, 119 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 38d2f204..3dae62f2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -402,6 +402,103 @@ lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feat rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] +/-- At positive process times, advancing `quadraticWidthSum` adds the matching history-level +quadratic form. -/ +lemma quadraticWidthSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + rw [quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hn] + rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn] + +/-- The history-level quadratic-form accumulator aligned with process times. + +The term at process time `t = 0` is set to zero, matching the convention used by `widthSqSum` and +`quadraticWidthSum`. At positive process time `t`, the history available to LinUCB is +`IsAlgEnvSeq.hist A R (t - 1) ω`. -/ +noncomputable def historyQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) + +/-- No positive-time history-level quadratic width forms are accumulated at horizon zero. -/ +lemma historyQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + historyQuadraticWidthSum A R reg x 0 ω = 0 := by + simp [historyQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time history-level quadratic width form. -/ +lemma historyQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + historyQuadraticWidthSum A R reg x (n + 1) ω = + historyQuadraticWidthSum A R reg x n ω + + if n = 0 then 0 else + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + simp [historyQuadraticWidthSum, sum_range_succ] + +/-- At positive process times, advancing the history-level quadratic accumulator adds the selected +arm's history-level quadratic width form. -/ +lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + historyQuadraticWidthSum A R reg x (n + 1) ω = + historyQuadraticWidthSum A R reg x n ω + + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + simp [historyQuadraticWidthSum_succ, hn] + +/-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with +the same process times. -/ +lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) : + quadraticWidthSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by + rw [quadraticWidthSum, historyQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0 + +/-- The squared-width accumulator equals the history-level quadratic-form accumulator whenever the +positive-time history-level quadratic forms are nonnegative. -/ +lemma widthSqSum_eq_historyQuadraticWidthSum + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) : + widthSqSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by + have h_process_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + intro t ht ht0 + exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).2 (h_nonneg t ht ht0) + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_process_nonneg] + exact quadraticWidthSum_eq_historyQuadraticWidthSum (A := A) (R := R) reg x n ω + +/-- A bound on the history-level quadratic-form accumulator implies the corresponding bound on +`widthSqSum`, provided the positive-time history-level quadratic forms are nonnegative. -/ +lemma widthSqSum_le_of_history_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_hist_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_historyQuadraticWidthSum (A := A) (R := R) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + exact h_hist_le + +omit [IsProbabilityMeasure P] in +/-- Almost surely, a history-level quadratic-form bound gives the `widthSqSum` bound consumed by +the regret chain. -/ +lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_hist_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_nonneg, h_hist_le] with ω h_nonnegω h_hist_leω + exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = @@ -810,6 +907,28 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin ] with ω h_regret simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever a history-level quadratic-form bound supplies the future +elliptical-potential input. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (hW : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) + end LinUCB end Bandits From 2a6a0cd63a6f6a07e977c000faf295e4759d602c Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:36:04 -0400 Subject: [PATCH 39/82] feat(linUCB): history-level quadratic-width related lemmas --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 3dae62f2..be4953bb 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -499,6 +499,39 @@ lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The pointwise input expected from a future elliptical-potential argument. + +It packages the two facts needed to turn a history-level quadratic-width estimate into the +`widthSqSum` estimate used by the regret chain: + +* each positive-time quadratic width form is nonnegative; +* their history-level accumulated sum is bounded by `W`. -/ +def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := + (∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ + historyQuadraticWidthSum A R reg x n ω ≤ W + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the +regret chain. -/ +lemma widthSqSum_le_of_history_quadratic_width_bound {W : ℝ} + (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_bound.1 h_bound.2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged history-level quadratic-width input implies the `widthSqSum` bound +consumed by the regret chain. -/ +lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} + (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_bound] with ω h_boundω + exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) (W := W) h_boundω + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = @@ -929,6 +962,29 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [No (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the packaged history-level quadratic-width input holds almost +surely. + +This is the theorem a future elliptical-potential lemma should feed into directly. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) + end LinUCB end Bandits From 38e20e1b2fcfdc4be34486ae3006f62467520d43 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:40:01 -0400 Subject: [PATCH 40/82] feat(linUCB): capped version of the quadratic-width accumulator: --- .../Online/Bandit/Algorithms/LinUCB.lean | 121 ++++++++++++++++++ 1 file changed, 121 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index be4953bb..e7b53c71 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -448,6 +448,58 @@ lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by simp [historyQuadraticWidthSum_succ, hn] +/-- The capped history-level quadratic-form accumulator aligned with process times. + +This is the accumulator shape that commonly appears in elliptical-potential statements: +each positive-time quadratic width form is capped at `1`. -/ +noncomputable def historyCappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else + min 1 (widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + +/-- No positive-time capped history-level quadratic width forms are accumulated at horizon zero. -/ +lemma historyCappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + historyCappedQuadraticWidthSum A R reg x 0 ω = 0 := by + simp [historyCappedQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time capped history-level quadratic width form. -/ +lemma historyCappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + historyCappedQuadraticWidthSum A R reg x (n + 1) ω = + historyCappedQuadraticWidthSum A R reg x n ω + + if n = 0 then 0 else + min 1 + (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by + simp [historyCappedQuadraticWidthSum, sum_range_succ] + +/-- At positive process times, advancing the capped history-level quadratic accumulator adds the +selected arm's capped history-level quadratic width form. -/ +lemma historyCappedQuadraticWidthSum_succ_of_ne_zero + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + historyCappedQuadraticWidthSum A R reg x (n + 1) ω = + historyCappedQuadraticWidthSum A R reg x n ω + + min 1 + (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by + simp [historyCappedQuadraticWidthSum_succ, hn] + +/-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and +capped history-level accumulators agree. -/ +lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) : + historyQuadraticWidthSum A R reg x n ω = + historyCappedQuadraticWidthSum A R reg x n ω := by + rw [historyQuadraticWidthSum, historyCappedQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact (min_eq_right (h_le_one t ht ht0)).symm + /-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with the same process times. -/ lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) @@ -513,6 +565,75 @@ def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ historyQuadraticWidthSum A R reg x n ω ≤ W +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Build the packaged history-level quadratic-width input from its two component facts. -/ +lemma historyQuadraticWidthBound_of_nonneg_and_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_sum_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : + HistoryQuadraticWidthBound A R reg x n ω W := by + exact ⟨h_nonneg, h_sum_le⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged history-level quadratic-width input is monotone in the numeric bound. -/ +lemma historyQuadraticWidthBound_mono {W W' : ℝ} + (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : + HistoryQuadraticWidthBound A R reg x n ω W' := by + exact ⟨h_bound.1, h_bound.2.trans hW⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, build the packaged history-level quadratic-width input from its two component +facts. -/ +lemma historyQuadraticWidthBound_ae_of_nonneg_and_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_sum_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by + filter_upwards [h_nonneg, h_sum_le] with ω h_nonnegω h_sum_leω + exact historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_sum_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged history-level quadratic-width input is monotone in the numeric +bound. -/ +lemma historyQuadraticWidthBound_ae_mono {W W' : ℝ} + (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : + ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W' := by + filter_upwards [h_bound] with ω h_boundω + exact historyQuadraticWidthBound_mono (A := A) (R := R) (reg := reg) (x := x) + (n := n) (ω := ω) h_boundω hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped quadratic-width sum bound gives the packaged history-level input whenever every +positive-time quadratic width form is nonnegative and at most `1`. -/ +lemma historyQuadraticWidthBound_of_capped_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + HistoryQuadraticWidthBound A R reg x n ω W := by + refine historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg ?_ + rw [historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) h_le_one] + exact h_capped_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped quadratic-width sum bound gives the packaged history-level input +whenever every positive-time quadratic width form is almost surely nonnegative and at most `1`. -/ +lemma historyQuadraticWidthBound_ae_of_capped_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_capped_le] with + ω h_nonnegω h_le_oneω h_capped_leω + exact historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 1bb706d1e6b68329b1d65786785739115d1edb9b Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:42:30 -0400 Subject: [PATCH 41/82] feat(linUCB): bridge from the capped elliptical-potential-style quantity to the regret chain --- .../Online/Bandit/Algorithms/LinUCB.lean | 60 +++++++++++++++++++ 1 file changed, 60 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e7b53c71..cee58c4b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -653,6 +653,39 @@ lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) (x := x) (n := n) (ω := ω) (W := W) h_boundω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped history-level quadratic-width sum bound implies the `widthSqSum` bound consumed by +the regret chain, provided the positive-time quadratic width forms are nonnegative and at most +`1`. -/ +lemma widthSqSum_le_of_capped_history_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) (W := W) + (historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonneg h_le_one h_capped_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped history-level quadratic-width sum bound implies the `widthSqSum` bound +consumed by the regret chain, provided the positive-time quadratic width forms are almost surely +nonnegative and at most `1`. -/ +lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + h_capped_le) + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = @@ -1106,6 +1139,33 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bou (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever a capped history-level quadratic-width sum bound holds almost +surely and every positive-time quadratic width form is almost surely nonnegative and at most `1`. + +This is the direct interface for the common capped form of the elliptical-potential lemma. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (hW : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) + end LinUCB end Bandits From 24b5503407f99354b203f1e7728ce1cdb16bebc9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:55:05 -0400 Subject: [PATCH 42/82] feat(linUCB): process-level capped accumulator --- .../Online/Bandit/Algorithms/LinUCB.lean | 117 ++++++++++++++++++ 1 file changed, 117 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index cee58c4b..e2b02f96 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -261,6 +261,47 @@ lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by simp [quadraticWidthSum_succ, hn] +/-- The accumulated capped quadratic forms corresponding to the positive-time LinUCB widths. -/ +noncomputable def cappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω) + +/-- No positive-time capped quadratic width forms are accumulated at horizon zero. -/ +lemma cappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + cappedQuadraticWidthSum A reg x 0 ω = 0 := by + simp [cappedQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time capped quadratic width form. -/ +lemma cappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + cappedQuadraticWidthSum A reg x (n + 1) ω = + cappedQuadraticWidthSum A reg x n ω + + if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by + simp [cappedQuadraticWidthSum, sum_range_succ] + +/-- At positive times, advancing the horizon adds the selected arm's capped quadratic width form. -/ +lemma cappedQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + cappedQuadraticWidthSum A reg x (n + 1) ω = + cappedQuadraticWidthSum A reg x n ω + min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by + simp [cappedQuadraticWidthSum_succ, hn] + +/-- If every positive-time process-level quadratic width form is at most `1`, then the uncapped +and capped process-level quadratic-width accumulators agree. -/ +lemma quadraticWidthSum_eq_cappedQuadraticWidthSum + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + quadraticWidthSum A reg x n ω = cappedQuadraticWidthSum A reg x n ω := by + rw [quadraticWidthSum, cappedQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact (min_eq_right (h_le_one t ht ht0)).symm + /-- If the squared-width and quadratic-form accumulators agree through a positive time and the next quadratic form is nonnegative, then they still agree after adding the next term. -/ lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -301,6 +342,38 @@ lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} (n := n) (ω := ω) h_nonneg] exact h_quad_le +/-- A capped process-level quadratic-form sum bound implies the corresponding bound on +`widthSqSum`, provided the positive-time process-level quadratic forms are nonnegative and at most +`1`. -/ +lemma widthSqSum_le_of_capped_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_capped_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + rw [quadraticWidthSum_eq_cappedQuadraticWidthSum (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_le_one] + exact h_capped_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped process-level quadratic-form sum bound implies the corresponding bound +on `widthSqSum`, provided the positive-time process-level quadratic forms are almost surely +nonnegative and at most `1`. -/ +lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_nonneg, h_le_one, h_capped_le] with + ω h_nonnegω h_le_oneω h_capped_leω + exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) @@ -485,6 +558,21 @@ lemma historyCappedQuadraticWidthSum_succ_of_ne_zero (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by simp [historyCappedQuadraticWidthSum_succ, hn] +/-- The process-level capped quadratic-width accumulator equals the history-level capped +accumulator aligned with the same process times. -/ +lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + cappedQuadraticWidthSum A reg x n ω = + historyCappedQuadraticWidthSum A R reg x n ω := by + rw [cappedQuadraticWidthSum, historyCappedQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact congrArg (fun q : ℝ ↦ min 1 q) + (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) + /-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and capped history-level accumulators agree. -/ lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum @@ -1166,6 +1254,35 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bo (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever a capped process-level quadratic-width sum bound holds almost +surely and every positive-time process-level quadratic width form is almost surely nonnegative and +at most `1`. + +This is the direct interface for an elliptical-potential lemma stated using the process-level design +matrices `designMatrix A reg x t ω`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) + end LinUCB end Bandits From c511de3ae55f53ceb0a3e6a636a68d800ea00c9e Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 09:29:42 -0400 Subject: [PATCH 43/82] feat(linUCB): process/history transport lemmas --- .../Online/Bandit/Algorithms/LinUCB.lean | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e2b02f96..5533b4ea 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -445,6 +445,78 @@ lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] +/-- At positive process times, the process-level quadratic width form is at most `1` iff the +matching history-level quadratic width form is at most `1`. -/ +lemma widthQuadraticForm_le_one_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthQuadraticForm A reg x a n ω ≤ 1 ↔ + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a ≤ 1 := by + rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] + +/-- The all-positive-times process-level nonnegativity assumption is equivalent to the matching +history-level nonnegativity assumption. -/ +lemma widthQuadraticForm_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) : + (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ + ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by + constructor + · intro h t ht ht0 + exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).1 (h t ht ht0) + · intro h t ht ht0 + exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).2 (h t ht ht0) + +/-- The all-positive-times process-level `≤ 1` assumption is equivalent to the matching +history-level `≤ 1` assumption. -/ +lemma widthQuadraticForm_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) : + (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ + ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by + constructor + · intro h t ht ht0 + exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).1 (h t ht ht0) + · intro h t ht ht0 + exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).2 (h t ht ht0) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, process-level all-positive-times nonnegativity is equivalent to the matching +history-level nonnegativity assumption. -/ +lemma widthQuadraticForm_ae_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) : + (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by + constructor + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).1 hω + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).2 hω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the process-level all-positive-times `≤ 1` assumption is equivalent to the +matching history-level `≤ 1` assumption. -/ +lemma widthQuadraticForm_ae_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) : + (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by + constructor + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).1 hω + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).2 hω + lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by @@ -573,6 +645,32 @@ lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) exact congrArg (fun q : ℝ ↦ min 1 q) (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) +/-- A process-level capped quadratic-width sum bound is equivalent to the matching history-level +capped quadratic-width sum bound. -/ +lemma cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : + cappedQuadraticWidthSum A reg x n ω ≤ W ↔ + historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by + rw [cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) + reg x n ω] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a process-level capped quadratic-width sum bound is equivalent to the matching +history-level capped quadratic-width sum bound. -/ +lemma cappedQuadraticWidthSum_ae_le_iff_historyCappedQuadraticWidthSum_ae_le + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (W : ℝ) : + (∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) ↔ + ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by + constructor + · intro h + filter_upwards [h] with ω hω + exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le + (A := A) (R := R) reg x n ω W).1 hω + · intro h + filter_upwards [h] with ω hω + exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le + (A := A) (R := R) reg x n ω W).2 hω + /-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and capped history-level accumulators agree. -/ lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum From 85c4425a07627455d7b5650d882a5c106f2628fd Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 09:34:21 -0400 Subject: [PATCH 44/82] feat(linUCB): compact regret theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 100 ++++++++++++++++++ 1 file changed, 100 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 5533b4ea..45657e6d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -374,6 +374,83 @@ lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The process-level capped quadratic-width input expected from an elliptical-potential argument. + +It packages the three facts needed to turn a capped process-level quadratic-width estimate into the +`widthSqSum` estimate used by the regret chain: + +* each positive-time process-level quadratic width form is nonnegative; +* each positive-time process-level quadratic width form is at most `1`; +* their capped process-level accumulated sum is bounded by `W`. -/ +def CappedQuadraticWidthBound (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := + (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ∧ + (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ∧ + cappedQuadraticWidthSum A reg x n ω ≤ W + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Build the packaged process-level capped quadratic-width input from its component facts. -/ +lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_sum_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : + CappedQuadraticWidthBound A reg x n ω W := by + exact ⟨h_nonneg, h_le_one, h_sum_le⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ +lemma cappedQuadraticWidthBound_mono {W W' : ℝ} + (h_bound : CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : + CappedQuadraticWidthBound A reg x n ω W' := by + exact ⟨h_bound.1, h_bound.2.1, h_bound.2.2.trans hW⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, build the packaged process-level capped quadratic-width input from its component +facts. -/ +lemma cappedQuadraticWidthBound_ae_of_nonneg_le_one_and_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_sum_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_sum_le] with + ω h_nonnegω h_le_oneω h_sum_leω + exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_sum_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged process-level capped quadratic-width input is monotone in the +numeric bound. -/ +lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W' := by + filter_upwards [h_bound] with ω h_boundω + exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) h_boundω hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed +by the regret chain. -/ +lemma widthSqSum_le_of_capped_quadratic_width_bound {W : ℝ} + (h_bound : CappedQuadraticWidthBound A reg x n ω W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_bound.1 h_bound.2.1 h_bound.2.2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged process-level capped quadratic-width input implies the `widthSqSum` +bound consumed by the regret chain. -/ +lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_bound] with ω h_boundω + exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) (W := W) h_boundω + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) @@ -1381,6 +1458,29 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the packaged process-level capped quadratic-width input holds +almost surely. + +This is the compact theorem a process-level elliptical-potential lemma should feed into directly. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) + (x := x) (n := n) (P := P) (W := W) h_bound) + end LinUCB end Bandits From 2a7e5b101bf3420a472148c8d4c582891c1252cd Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:15:42 -0400 Subject: [PATCH 45/82] feat(linUCB): direct regret theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 107 ++++++++++++++++++ 1 file changed, 107 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 45657e6d..f0e087a7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -374,6 +374,46 @@ lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Determinant of the process-level LinUCB design matrix. -/ +noncomputable def designDet (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + Matrix.det (designMatrix A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial design determinant is the determinant of the regularized identity. -/ +lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by + simp [designDet, designMatrix_zero] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ +noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + designDet A reg x n ω / designDet A reg x 0 ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the determinant ratio is `1` when the initial design determinant is nonzero. -/ +lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + designDetRatio A reg x 0 ω = 1 := by + simp [designDetRatio, hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The log-determinant expression that appears in the elliptical-potential lemma. -/ +noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + 2 * Real.log (designDetRatio A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the log-determinant potential is zero when the initial design determinant is +nonzero. -/ +lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + ellipticalPotential A reg x 0 ω = 0 := by + simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -432,6 +472,38 @@ lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_boundω hW +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped-sum bound by the log-determinant potential, together with a constant bound on that +potential, gives the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_of_ellipticalPotential_le_bound {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ellipticalPotential A reg x n ω ≤ W) : + CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonneg h_le_one (h_elliptical.trans h_potential_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped-sum bound by the log-determinant potential and an almost-sure constant +bound on that potential give the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_elliptical, h_potential_le] with + ω h_nonnegω h_le_oneω h_ellipticalω h_potential_leω + exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ @@ -1481,6 +1553,41 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the +log-determinant elliptical potential and that potential is bounded by `W`. + +This is the first theorem whose assumptions match the two matrix-analysis steps of the actual +elliptical-potential argument: + +* prove `cappedQuadraticWidthSum ≤ ellipticalPotential`; +* prove `ellipticalPotential ≤ W`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono W + (cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one + h_elliptical h_potential_le) + end LinUCB end Bandits From d568d50d95be7eea25c0adf455e2555321423b11 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:22:52 -0400 Subject: [PATCH 46/82] feat(linUCB): base case for the log-determinant elliptical-potential path --- .../Online/Bandit/Algorithms/LinUCB.lean | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f0e087a7..f6b75248 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -414,6 +414,17 @@ lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) ellipticalPotential A reg x 0 ω = 0 := by simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the log-determinant elliptical-potential inequality. At horizon zero there are no +positive-time capped quadratic width forms, and the log-determinant potential is zero when the +initial design determinant is nonzero. -/ +lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) + (hdet : designDet A reg x 0 ω ≠ 0) : + cappedQuadraticWidthSum A reg x 0 ω ≤ ellipticalPotential A reg x 0 ω := by + rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), + ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -440,6 +451,34 @@ lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} CappedQuadraticWidthBound A reg x n ω W := by exact ⟨h_nonneg, h_le_one, h_sum_le⟩ +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the packaged process-level capped quadratic-width input. At horizon zero, the +nonnegativity and `≤ 1` side conditions are vacuous, and the capped sum is zero. -/ +lemma cappedQuadraticWidthBound_zero {W : ℝ} (hW : 0 ≤ W) : + CappedQuadraticWidthBound A reg x 0 ω W := by + refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ + · intro t ht _ + simp at ht + · intro t ht _ + simp at ht + · simpa [cappedQuadraticWidthSum_zero] using hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the packaged process-level capped quadratic-width input when the constant bound +is supplied through the log-determinant potential. -/ +lemma cappedQuadraticWidthBound_zero_of_ellipticalPotential_le_bound {W : ℝ} + (hdet : designDet A reg x 0 ω ≠ 0) (h_potential_le : ellipticalPotential A reg x 0 ω ≤ W) : + CappedQuadraticWidthBound A reg x 0 ω W := by + refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ + · intro t ht _ + simp at ht + · intro t ht _ + simp at ht + · exact (cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) + (x := x) (ω := ω) hdet).trans h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ lemma cappedQuadraticWidthBound_mono {W W' : ℝ} From 9766e16289de5bec74a618ed8c0dcc001b654a96 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:26:54 -0400 Subject: [PATCH 47/82] feat(linUCB): per-step potential-increment shell --- .../Online/Bandit/Algorithms/LinUCB.lean | 80 +++++++++++++++++++ 1 file changed, 80 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f6b75248..7fa0aa2a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -425,6 +425,66 @@ lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step increment of the log-determinant elliptical potential. -/ +noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the next capped quadratic width term is bounded by the next log-determinant potential +increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ +lemma cappedQuadraticWidthSum_succ_le_ellipticalPotential + (h_prev : cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_step : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialIncrement A reg x n ω) : + cappedQuadraticWidthSum A reg x (n + 1) ω ≤ ellipticalPotential A reg x (n + 1) ω := by + rw [cappedQuadraticWidthSum_succ (A := A) (reg := reg) (x := x) (n := n) (ω := ω)] + calc + cappedQuadraticWidthSum A reg x n ω + + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) + ≤ ellipticalPotential A reg x n ω + ellipticalPotentialIncrement A reg x n ω := by + exact add_le_add h_prev h_step + _ = ellipticalPotential A reg x (n + 1) ω := by + simp [ellipticalPotentialIncrement] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A per-step bound by log-determinant potential increments implies the cumulative +elliptical-potential inequality. This is the induction shell for the future determinant-update +proof. -/ +lemma cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le + (hdet : designDet A reg x 0 ω ≠ 0) : + (∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) → + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + induction n with + | zero => + intro _ + exact cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) + (x := x) (ω := ω) hdet + | succ n ih => + intro h_step + refine cappedQuadraticWidthSum_succ_le_ellipticalPotential (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) ?_ ?_ + · exact ih fun t ht ↦ h_step t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + · exact h_step n (by simp) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a per-step bound by log-determinant potential increments implies the cumulative +elliptical-potential inequality. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + filter_upwards [hdet, h_step] with ω hdetω h_stepω + exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetω h_stepω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -543,6 +603,26 @@ lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by log-determinant potential increments and a final constant +bound on the potential give the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet h_step) + h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 9d1125484a236eb30d8f7e28c4e1c0b73a47ac08 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:38:23 -0400 Subject: [PATCH 48/82] feat(linUCB): one-step determinant ratio related lemmas --- .../Online/Bandit/Algorithms/LinUCB.lean | 70 +++++++++++++++++++ 1 file changed, 70 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 7fa0aa2a..c9ffb5c8 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -400,12 +400,31 @@ lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) designDetRatio A reg x 0 ω = 1 := by simp [designDetRatio, hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. + +This is the determinant-ratio target used by the matrix-determinant part of the elliptical +potential lemma. -/ +noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + designDet A reg x (n + 1) ω / designDet A reg x n ω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := 2 * Real.log (designDetRatio A reg x n ω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. + +The future determinant-update proof should naturally establish the capped quadratic-width term is +bounded by this quantity. A separate log/telescoping bridge then connects this one-step quantity to +`ellipticalPotentialIncrement`. -/ +noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + 2 * Real.log (designDetStepRatio A reg x n ω) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- At horizon zero, the log-determinant potential is zero when the initial design determinant is nonzero. -/ @@ -485,6 +504,30 @@ lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdetω h_stepω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the +cumulative capped-sum/log-det inequality, provided the one-step determinant-ratio potential is +bounded by the corresponding cumulative-potential increment. + +This separates the future elliptical-potential proof into two local obligations: + +* a matrix-determinant update bounding the selected arm's capped quadratic form by + `ellipticalPotentialStep`; +* a log/telescoping bridge from `ellipticalPotentialStep` to `ellipticalPotentialIncrement`. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet ?_ + filter_upwards [h_step, h_step_le_increment] with ω h_stepω h_step_le_incrementω + intro t ht + exact (h_stepω t ht).trans (h_step_le_incrementω t ht) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -623,6 +666,33 @@ lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : (reg := reg) (x := x) (n := n) (P := P) hdet h_step) h_potential_le +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, one-step determinant-ratio potential bounds, their bridge to cumulative +potential increments, and a final constant bound on the potential give the packaged process-level +capped quadratic-width input. + +This is the packaged form of the determinant-update interface: once the true matrix determinant +lemma proves the `h_step` assumption and the log/telescoping algebra proves +`h_step_le_increment`, the existing regret chain can consume the resulting bound. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) + h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 70916c42d2fdf726b293adcb51900bf3ae6415d0 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 14:02:08 -0400 Subject: [PATCH 49/82] feat(linUCB): elliptical-potential chain related changes --- .../Online/Bandit/Algorithms/LinUCB.lean | 484 ++++++++++++++++++ 1 file changed, 484 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index c9ffb5c8..272c7bfc 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -8,6 +8,8 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.LinearAlgebra.Matrix.SchurComplement public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse /-! @@ -387,6 +389,22 @@ lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by simp [designDet, designMatrix_zero] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial design determinant is `reg ^ d`. -/ +lemma designDet_zero_eq_reg_pow (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + designDet A reg x 0 ω = reg ^ d := by + rw [designDet_zero] + simp + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A nonzero regularization parameter gives a nonzero initial design determinant. -/ +lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hreg : reg ≠ 0) : + designDet A reg x 0 ω ≠ 0 := by + rw [designDet_zero_eq_reg_pow] + exact pow_ne_zero d hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -400,6 +418,15 @@ lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) designDetRatio A reg x 0 ω = 1 := by simp [designDetRatio, hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the determinant ratio is positive when the initial design determinant is +nonzero. -/ +lemma designDetRatio_zero_pos (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + 0 < designDetRatio A reg x 0 ω := by + rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + norm_num + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. @@ -409,12 +436,210 @@ noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := designDet A reg x (n + 1) ω / designDet A reg x n ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The scalar determinant appearing in the rank-one determinant update is the quadratic form +`uᵀ M u`. -/ +lemma det_one_add_replicateRow_mul_matrix_mul_replicateCol + (M : Matrix (Fin d) (Fin d) ℝ) (u : Feature d) : + (1 + Matrix.replicateRow Unit u * M * Matrix.replicateCol Unit u).det = + 1 + dotProduct u (Matrix.mulVec M u) := by + have hsum : + (∑ j, (∑ i, u i * M i j) * u j) = + ∑ i, u i * ∑ j, M i j * u j := by + calc + (∑ j, (∑ i, u i * M i j) * u j) + = ∑ j, ∑ i, (u i * M i j) * u j := by + simp [Finset.sum_mul] + _ = ∑ i, ∑ j, (u i * M i j) * u j := by + rw [Finset.sum_comm] + _ = ∑ i, u i * ∑ j, M i j * u j := by + refine Finset.sum_congr rfl ?_ + intro i _ + rw [Finset.mul_sum] + refine Finset.sum_congr rfl ?_ + intro j _ + ring + rw [Matrix.det_unique] + simpa [Matrix.mul_apply, Matrix.replicateRow, Matrix.replicateCol, Matrix.mulVec, + dotProduct] using hsum + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Process-level matrix determinant update for the LinUCB design matrix. + +If `V_n` has nonzero determinant, then the rank-one update +`V_{n+1} = V_n + x_{A_n} x_{A_n}ᵀ` satisfies +`det(V_{n+1}) = det(V_n) * (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ +lemma designDet_succ_eq_mul_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDet A reg x (n + 1) ω = + designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + have hM : IsUnit (designMatrix A reg x n ω).det := by + simpa [designDet] using (isUnit_iff_ne_zero.mpr hdet) + calc + designDet A reg x (n + 1) ω = + (designMatrix A reg x n ω + + Matrix.vecMulVec (x (A n ω)) (x (A n ω))).det := by + simp [designDet, designMatrix_succ] + _ = (designMatrix A reg x n ω + + Matrix.replicateCol Unit (x (A n ω)) * Matrix.replicateRow Unit (x (A n ω))).det := by + rw [Matrix.vecMulVec_eq Unit] + _ = (designMatrix A reg x n ω).det * + (1 + Matrix.replicateRow Unit (x (A n ω)) * + (designMatrix A reg x n ω)⁻¹ * Matrix.replicateCol Unit (x (A n ω))).det := by + exact Matrix.det_add_replicateCol_mul_replicateRow (A := designMatrix A reg x n ω) + (ι := Unit) hM (x (A n ω)) (x (A n ω)) + _ = designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + rw [designDet] + congr 1 + exact det_one_add_replicateRow_mul_matrix_mul_replicateCol + (M := (designMatrix A reg x n ω)⁻¹) (u := x (A n ω)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `det(V_n)` is nonzero and the selected quadratic form is nonnegative, then +`det(V_{n+1})` is nonzero. -/ +lemma designDet_succ_ne_zero_of_widthQuadraticForm_nonneg + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + designDet A reg x (n + 1) ω ≠ 0 := by + rw [designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + exact mul_ne_zero hdet (ne_of_gt (by linarith)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms preserve +nonzero design determinants up to any fixed time. -/ +lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt + (m : ℕ) (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t < m → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + designDet A reg x m ω ≠ 0 := by + induction m with + | zero => exact hdet0 + | succ m ih => + exact designDet_succ_ne_zero_of_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := m) (ω := ω) + (ih fun t ht ↦ h_nonneg t (Nat.lt_trans ht (Nat.lt_succ_self m))) + (h_nonneg m (Nat.lt_succ_self m)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms imply that +all design determinants through horizon `n` are nonzero. -/ +lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by + intro t ht + exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := t) (ω := ω) hdet0 fun s hs ↦ + h_nonneg s (mem_range.mpr (Nat.lt_of_lt_of_le hs (Nat.le_of_lt_succ (mem_range.mp ht)))) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant and nonnegative selected quadratic forms imply +that all design determinants through horizon `n` are nonzero. -/ +lemma designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω + exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `det(V_n) ≠ 0`, then the one-step determinant ratio is +`1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n}`. -/ +lemma designDetStepRatio_eq_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDetStepRatio A reg x n ω = + 1 + widthQuadraticForm A reg x (A n ω) n ω := by + simp [designDetStepRatio, + designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet, hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The cumulative determinant ratio advances by multiplying by the one-step determinant ratio. -/ +lemma designDetRatio_succ_eq_mul_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDetRatio A reg x (n + 1) ω = + designDetRatio A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + rw [designDetRatio, designDetRatio, + designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms make the +cumulative determinant ratio positive. -/ +lemma designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + 0 < designDetRatio A reg x n ω := by + induction n with + | zero => + exact designDetRatio_zero_pos (A := A) (reg := reg) (x := x) (ω := ω) hdet0 + | succ n ih => + have hdetn : designDet A reg x n ω ≠ 0 := + designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ + h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) + rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetn] + exact mul_pos + (ih fun t ht ↦ h_nonneg t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) + (by linarith [h_nonneg n (by simp)]) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, starting from a nonzero initial determinant, nonnegative selected quadratic +forms make the cumulative determinant ratio positive. -/ +lemma designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by + filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω + exact designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter and nonnegative selected quadratic forms make +the cumulative determinant ratio positive. -/ +lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by + refine designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := 2 * Real.log (designDetRatio A reg x n ω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A positive determinant ratio bounded by `D` gives the corresponding log-determinant potential +bound. -/ +lemma ellipticalPotential_le_two_mul_log_of_designDetRatio_le {D : ℝ} + (h_ratio_pos : 0 < designDetRatio A reg x n ω) + (h_ratio_le : designDetRatio A reg x n ω ≤ D) : + ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by + rw [ellipticalPotential] + exact mul_le_mul_of_nonneg_left (Real.log_le_log h_ratio_pos h_ratio_le) (by norm_num) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a positive determinant ratio bounded by `D` gives the corresponding +log-determinant potential bound. -/ +lemma ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le {D : ℝ} + (h_ratio_pos : ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by + filter_upwards [h_ratio_pos, h_ratio_le] with ω h_ratio_posω h_ratio_leω + exact ellipticalPotential_le_two_mul_log_of_designDetRatio_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_ratio_posω h_ratio_leω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. @@ -425,6 +650,70 @@ noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := 2 * Real.log (designDetStepRatio A reg x n ω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing, the one-step log-determinant potential is +`2 * log (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ +lemma ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + ellipticalPotentialStep A reg x n ω = + 2 * Real.log (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + simp [ellipticalPotentialStep, + designDetStepRatio_eq_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar log inequality used in the elliptical-potential proof: for `0 ≤ q ≤ 1`, +`min 1 q ≤ 2 * log (1 + q)`. -/ +lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} + (hq_nonneg : 0 ≤ q) (hq_le_one : q ≤ 1) : + min 1 q ≤ 2 * Real.log (1 + q) := by + have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := + Real.le_log_one_add_of_nonneg hq_nonneg + have hq_add_two_pos : 0 < q + 2 := by linarith + have hq_le_two : q ≤ 2 := by linarith + have hq_le_log_lower : q ≤ 2 * (2 * q / (q + 2)) := by + rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] + rw [le_div_iff₀ hq_add_two_pos] + nlinarith + rw [min_eq_right hq_le_one] + exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the +single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ +lemma cappedWidthTerm_le_ellipticalPotentialStep + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) + (h_le_one : n ≠ 0 → widthQuadraticForm A reg x (A n ω) n ω ≤ 1) : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialStep A reg x n ω := by + by_cases hn : n = 0 + · rw [if_pos hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) + · rw [if_neg hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply +the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ +lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω := by + filter_upwards [hdet, h_nonneg, h_le_one] with ω hdetω h_nonnegω h_le_oneω + intro t ht + exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- At horizon zero, the log-determinant potential is zero when the initial design determinant is nonzero. -/ @@ -450,6 +739,34 @@ noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The one-step determinant-ratio potential equals the increment of the cumulative +log-determinant potential, provided the relevant design determinants are nonzero. -/ +lemma ellipticalPotentialStep_eq_increment + (hdet0 : designDet A reg x 0 ω ≠ 0) + (hdetn : designDet A reg x n ω ≠ 0) + (hdet_succ : designDet A reg x (n + 1) ω ≠ 0) : + ellipticalPotentialStep A reg x n ω = ellipticalPotentialIncrement A reg x n ω := by + simp [ellipticalPotentialStep, designDetStepRatio, ellipticalPotentialIncrement, + ellipticalPotential, designDetRatio, Real.log_div hdet_succ hdetn, + Real.log_div hdet_succ hdet0, Real.log_div hdetn hdet0] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the one-step determinant-ratio potential equals the increment of the cumulative +log-determinant potential throughout the finite horizon, provided all determinants up to that +horizon are nonzero almost surely. -/ +lemma ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω = ellipticalPotentialIncrement A reg x t ω := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact ellipticalPotentialStep_eq_increment (A := A) (reg := reg) (x := x) (n := t) + (ω := ω) (hdetω 0 (by simp)) + (hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) + (hdetω (t + 1) (mem_range.mpr (Nat.succ_lt_succ (mem_range.mp ht)))) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- If the next capped quadratic width term is bounded by the next log-determinant potential increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ @@ -528,6 +845,29 @@ lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le intro t ht exact (h_stepω t ht).trans (h_step_le_incrementω t ht) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the +cumulative capped-sum/log-det inequality when all design determinants up to the horizon are nonzero +almost surely. + +Compared with `cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le`, this +version discharges the log/telescoping bridge automatically from determinant nonvanishing. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + have hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + exact hdetω 0 (by simp) + refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_step ?_ + filter_upwards [ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet] with ω h_eq + intro t ht + rw [h_eq t ht] + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -693,6 +1033,150 @@ lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bo (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) h_potential_le +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, one-step determinant-ratio potential bounds, determinant nonvanishing up to the +horizon, and a final constant bound on the potential give the packaged process-level capped +quadratic-width input. + +This is the determinant-nonvanishing version of the one-step interface: the remaining hard +elliptical-potential work is to prove the one-step matrix inequality and the final +log-determinant bound. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero + {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet h_step) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the determinant-update step, determinant nonvanishing up to the horizon, and a +final constant bound on the log-determinant potential give the packaged capped quadratic-width +input used by the regret chain. + +The assumptions now match the concrete obligations left for a full elliptical-potential proof: + +* prove all relevant design determinants are nonzero; +* prove selected quadratic forms are nonnegative and at most `1` at positive times; +* prove the final log-determinant potential is at most `W`. -/ +lemma cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound {W : ℝ} + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + have h_nonneg_positive : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + filter_upwards [h_nonneg] with ω h_nonnegω + intro t ht _ + exact h_nonnegω t ht + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) + h_nonneg_positive h_le_one hdet + (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero (A := A) (reg := reg) + (x := x) (n := n) (P := P) hdet_range_n h_nonneg h_le_one) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant, the determinant-update step, and a final constant +bound on the log-determinant potential give the packaged capped quadratic-width input used by the +regret chain. + +This removes the need to assume determinant nonvanishing at every time: it is derived inductively +from `det(V_0) ≠ 0` and nonnegative selected quadratic forms. -/ +lemma cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound {W : ℝ} + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) + h_nonneg h_le_one h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter, the determinant-update step, and a final +constant bound on the log-determinant potential give the packaged capped quadratic-width input used +by the regret chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential_le_bound {W : ℝ} + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + refine cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) ?_ h_nonneg h_le_one + h_potential_le + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a +determinant-ratio upper bound, and the determinant-update step give the packaged capped +quadratic-width input used by the regret chain. + +This version accepts the determinant-ratio bound directly and converts it into the +`ellipticalPotential ≤ 2 * log D` bound internally. -/ +lemma cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound {D : ℝ} + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by + exact cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := 2 * Real.log D) + hdet0 h_nonneg h_le_one + (ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) + h_ratio_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter, nonnegative selected quadratic forms, a +determinant-ratio upper bound, and the determinant-update step give the packaged capped +quadratic-width input used by the regret chain. + +This is the most direct interface for the final determinant-bound part of the finite-action +elliptical-potential argument: after proving `designDetRatio ≤ D`, the theorem supplies the +`CappedQuadraticWidthBound` with bound `2 * log D`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound {D : ℝ} + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by + refine cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one h_ratio_le + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 88d52b661ae64c83d4ce1dfa09f45316f48db7d1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 29 May 2026 10:10:11 -0400 Subject: [PATCH 50/82] feat : initial algorithm foundation for linUCB --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 3dbb4310..7d6d4a9b 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -4,6 +4,7 @@ public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.Measura public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 56cf08f3a23ecae5a56379395162981d12be97b2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 09:54:29 -0400 Subject: [PATCH 51/82] feat(linUCB):D = 2 ^ n intermediate bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 93 +++++++++++++++++++ 1 file changed, 93 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 272c7bfc..39de230a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -613,6 +613,77 @@ lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, the cumulative determinant ratio is the finite +product of the per-round determinant-update factors. -/ +lemma designDetRatio_eq_prod_one_add_widthQuadraticForm + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + designDetRatio A reg x n ω = + ∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω) := by + induction n with + | zero => + rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet0] + simp + | succ n ih => + have hdetn : designDet A reg x n ω ≠ 0 := + designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ + h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) + rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetn] + rw [ih fun t ht ↦ h_nonneg t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))] + simp [Finset.prod_range_succ] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If every selected quadratic form is in `[0, 1]`, the cumulative determinant ratio is at most +`2 ^ n`. -/ +lemma designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + rw [designDetRatio_eq_prod_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0 h_nonneg] + calc + (∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω)) + ≤ ∏ _t ∈ range n, (2 : ℝ) := by + exact Finset.prod_le_prod + (fun t ht ↦ by linarith [h_nonneg t ht]) + (fun t ht ↦ by linarith [h_le_one t ht]) + _ = (2 : ℝ) ^ n := by + simp + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, if every selected quadratic form is in `[0, 1]`, the cumulative determinant +ratio is at most `2 ^ n`. -/ +lemma designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + filter_upwards [hdet0, h_nonneg, h_le_one] with ω hdet0ω h_nonnegω h_le_oneω + exact designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one (A := A) + (reg := reg) (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω h_le_oneω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter and selected quadratic forms in `[0, 1]` +imply the cumulative determinant ratio is at most `2 ^ n`. -/ +lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + refine designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one + (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1177,6 +1248,28 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_b exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A simple explicit determinant-ratio bound for the capped quadratic-width input. + +If `reg ≠ 0` and every selected quadratic form is almost surely in `[0, 1]`, then the determinant +ratio is at most `2 ^ n`, so the existing determinant-update/elliptical-potential chain gives the +packaged capped-width bound with budget `2 * log (2 ^ n)`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω (2 * Real.log ((2 : ℝ) ^ n)) := by + refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (2 : ℝ) ^ n) + hreg h_nonneg ?_ ?_ + · filter_upwards [h_le_one] with ω h_le_oneω + exact fun t ht _ ↦ h_le_oneω t ht + · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From feff54daaacdc767e170e95bdbfbc47db6c083ff Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:10:56 -0400 Subject: [PATCH 52/82] feat(linUCB):trace/determinant-budget layer --- .../Online/Bandit/Algorithms/LinUCB.lean | 78 +++++++++++++++++++ 1 file changed, 78 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 39de230a..da5bdbed 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -34,6 +34,11 @@ namespace LinUCB /-- Feature vectors for finite-dimensional linear bandits. -/ abbrev Feature (d : ℕ) := Fin d → ℝ +/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product +`x_aᵀ x_a`. -/ +def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := + dotProduct (x a) (x a) + /-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := @@ -138,6 +143,55 @@ lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by simp [designMatrix, sum_range_succ, add_assoc] +/-- Trace of the process-level regularized design matrix. -/ +noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + Matrix.trace (designMatrix A reg x n ω) + +/-- Before any observations, the design trace is the trace of `reg • I_d`, namely `reg * d`. -/ +lemma designTrace_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + designTrace A reg x 0 ω = reg * (d : ℝ) := by + simp [designTrace, designMatrix_zero] + +/-- Updating the design matrix by `x_a x_aᵀ` increases the trace by `x_aᵀ x_a`. -/ +lemma designTrace_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designTrace A reg x (n + 1) ω = + designTrace A reg x n ω + featureSqNorm x (A n ω) := by + simp [designTrace, designMatrix_succ, featureSqNorm, Matrix.trace_vecMulVec] + +/-- Closed form for the design trace: initial regularization trace plus accumulated squared +feature norms. -/ +lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designTrace A reg x n ω = + reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by + simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] + +/-- If every selected feature vector has squared norm at most `L2`, then the trace of the design +matrix is at most `reg * d + n * L2`. -/ +lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by + rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] + gcongr + calc + (∑ t ∈ range n, featureSqNorm x (A t ω)) ≤ ∑ _t ∈ range n, L2 := by + exact sum_le_sum fun t ht ↦ hL2 t ht + _ = (n : ℝ) * L2 := by + simp [nsmul_eq_mul] + +omit [IsProbabilityMeasure P] in +/-- Almost surely, bounded selected feature norms give the corresponding deterministic trace +budget `reg * d + n * L2`. -/ +lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : + ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by + filter_upwards [hL2] with ω hL2ω + exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) L2 hL2ω + /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := @@ -1270,6 +1324,30 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Trace-budget interface for the determinant part of the finite-action elliptical-potential +argument. + +The future spectral/AM-GM determinant theorem should prove the hypothesis +`designDetRatio ≤ (T / (reg * d)) ^ d`, where `T` is an upper bound on `trace(V_n)`. This theorem +then feeds that determinant-ratio bound into the already-proved determinant-update and +elliptical-potential chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (T : ℝ) + (h_ratio_le : ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * Real.log ((T / (reg * (d : ℝ))) ^ d)) := by + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 0a81fe7433925aaf8404f627e9da8a7078e5dd2a Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:15:17 -0400 Subject: [PATCH 53/82] feat(linUCB):trace/determinant-budget layer --- .../Online/Bandit/Algorithms/LinUCB.lean | 72 ------------------- 1 file changed, 72 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 2d21d954..da5bdbed 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -8,11 +8,8 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax -<<<<<<< HEAD public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.LinearAlgebra.Matrix.SchurComplement -======= ->>>>>>> main public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse /-! @@ -34,7 +31,6 @@ section Algorithm namespace LinUCB -<<<<<<< HEAD /-- Feature vectors for finite-dimensional linear bandits. -/ abbrev Feature (d : ℕ) := Fin d → ℝ @@ -44,39 +40,25 @@ def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := dotProduct (x a) (x a) /-- History-level regularized design matrix for LinUCB. -/ -======= -abbrev Feature (d : ℕ) := Fin d → ℝ - ->>>>>>> main noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) -<<<<<<< HEAD /-- History-level response vector for LinUCB. -/ -======= ->>>>>>> main noncomputable def responseVector' (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := ∑ s : Iic n, (h s).2 • x (h s).1 -<<<<<<< HEAD /-- History-level regularized least-squares estimate. -/ -======= ->>>>>>> main noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) -<<<<<<< HEAD /-- History-level estimated reward of an arm. -/ -======= ->>>>>>> main noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := dotProduct (thetaHat' reg x n h) (x a) -<<<<<<< HEAD /-- History-level quadratic form underlying the LinUCB confidence width. -/ noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := @@ -94,11 +76,6 @@ lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by simp [width', Real.sq_sqrt h_nonneg] -======= -noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) ->>>>>>> main /-- LinUCB optimistic index of an arm. @@ -114,10 +91,6 @@ open Classical in /-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) -<<<<<<< HEAD -======= - (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) ->>>>>>> main (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK measurableArgmax (fun h a ↦ index' reg β x n h a) h @@ -127,11 +100,7 @@ lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) (n : ℕ) : -<<<<<<< HEAD Measurable (nextArm hK reg β x n) := by -======= - Measurable (nextArm hK reg β x h_index n) := by ->>>>>>> main have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact measurable_measurableArgmax fun a ↦ h_index n a @@ -142,11 +111,7 @@ noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → LinUCB.Feature d) (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : Algorithm (Fin K) ℝ := -<<<<<<< HEAD detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ -======= - detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ ->>>>>>> main end Algorithm @@ -167,7 +132,6 @@ noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) -<<<<<<< HEAD /-- The initial design matrix before any actions are included. -/ lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : designMatrix A reg x 0 ω = reg • 1 := by @@ -228,14 +192,11 @@ lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) L2 hL2ω -======= ->>>>>>> main /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := ∑ s ∈ range n, R s ω • x (A s ω) -<<<<<<< HEAD /-- The initial response vector before any rewards are included. -/ lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (ω : Ω) : @@ -249,14 +210,11 @@ lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) responseVector A R x n ω + R n ω • x (A n ω) := by simp [responseVector, sum_range_succ] -======= ->>>>>>> main /-- The process-level regularized least-squares estimate. -/ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) -<<<<<<< HEAD /-- The initial least-squares estimate is zero because no reward-feature observations have been included yet. -/ lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) @@ -264,14 +222,11 @@ lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) thetaHat A R reg x 0 ω = 0 := by simp [thetaHat, responseVector_zero] -======= ->>>>>>> main /-- The process-level estimated linear reward. -/ noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := dotProduct (thetaHat A R reg x n ω) (x a) -<<<<<<< HEAD /-- The initial estimated reward is zero for every arm. -/ lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : @@ -1411,12 +1366,6 @@ lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} filter_upwards [h_bound] with ω h_boundω exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) (W := W) h_boundω -======= -/-- The process-level elliptical confidence width. -/ -noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) ->>>>>>> main /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) @@ -1424,7 +1373,6 @@ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (n : ℕ) (ω : Ω) : ℝ := estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω -<<<<<<< HEAD /-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is zero. -/ lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) @@ -1440,8 +1388,6 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by simp [index_zero, width_zero] -======= ->>>>>>> main lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = @@ -1477,7 +1423,6 @@ lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] -<<<<<<< HEAD lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : widthQuadraticForm A reg x a n ω = @@ -1919,12 +1864,6 @@ lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one h_capped_le) -======= -lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] ->>>>>>> main lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : @@ -1939,11 +1878,7 @@ lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (n : ℕ) : A (n + 1) =ᵐ[P] -<<<<<<< HEAD fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by -======= - fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by ->>>>>>> main have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact h.action_detAlgorithm_ae_eq n @@ -1952,11 +1887,7 @@ lemma arm_ae_all_eq [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : ∀ᵐ ω ∂P, ∀ n, A (n + 1) ω = -<<<<<<< HEAD nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by -======= - nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by ->>>>>>> main simp_rw [ae_all_iff] exact fun n ↦ arm_ae_eq_linUCBNextArm h n @@ -1986,7 +1917,6 @@ lemma forall_index_le_index_arm [Nonempty (Fin K)] end AlgorithmBehavior -<<<<<<< HEAD omit [IsMarkovKernel ν] in /-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected @@ -2502,8 +2432,6 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one h_elliptical h_potential_le) -======= ->>>>>>> main end LinUCB end Bandits From 51d73c5357a02db14c1e8c41f0874c8c73599c29 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:28:56 -0400 Subject: [PATCH 54/82] feat(linUCB):packaging step which allows formal slot where the hard matrix inequality will plug in --- .../Online/Bandit/Algorithms/LinUCB.lean | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index da5bdbed..65cd1781 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -738,6 +738,39 @@ lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_o exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Converts an almost-sure trace bound into the determinant-ratio bound expected from a future +trace/determinant comparison theorem. -/ +lemma designDetRatio_ae_le_trace_budget_of_designTrace_ae_le + (T : ℝ) + (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ T → + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + filter_upwards [h_trace_le] with ω h_traceω + exact h_ratio_of_trace ω h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms give the concrete trace budget +`reg * d + n * L2`; a future trace/determinant comparison then gives the corresponding +determinant-ratio bound. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + exact designDetRatio_ae_le_trace_budget_of_designTrace_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) (T := reg * (d : ℝ) + (n : ℝ) * L2) + (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2) + h_ratio_of_trace + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1348,6 +1381,35 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface for the determinant part of the finite-action +elliptical-potential argument. + +If selected feature vectors have squared norm at most `L2`, then `trace(V_n) ≤ reg * d + n * L2`. +Given a future deterministic trace/determinant comparison that turns this trace budget into the +determinant-ratio bound, this theorem supplies the packaged capped-width input with the explicit +budget `2 * log (((reg * d + n * L2) / (reg * d)) ^ d)`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d)) := by + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg h_nonneg h_le_one + (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 13b222e59b067f1da87b415cc51ae3cdee1f454f Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:41:10 -0400 Subject: [PATCH 55/82] feat(linUCB): formal bound shaped like the standard elliptical-potential term used in LinUCB regret proofs --- .../Online/Bandit/Algorithms/LinUCB.lean | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 65cd1781..a1b5e49c 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -1410,6 +1410,45 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budge (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The explicit feature-norm determinant budget can be rewritten in the common +`d * log(1 + n L² / (reg d))` form. -/ +lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add + (L2 : ℝ) (hden : reg * (d : ℝ) ≠ 0) : + 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) = + 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by + have hbase : + (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) = + 1 + (n : ℝ) * L2 / (reg * (d : ℝ)) := by + exact same_add_div hden + rw [Real.log_pow, hbase] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface with the log term rewritten in the standard +`2 * d * log(1 + n L² / (reg d))` shape. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (hreg : reg ≠ 0) (hd : d ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg (by exact_mod_cast hd) + rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 + h_ratio_of_trace + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From c01cb041d2a74aaec422b8dde15f1797639325aa Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:44:10 -0400 Subject: [PATCH 56/82] feat(linUCB): end-to-end regret-facing step --- .../Online/Bandit/Algorithms/LinUCB.lean | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a1b5e49c..1cf9bcea 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2498,6 +2498,46 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the +feature-budget elliptical-potential term +`2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. + +The remaining matrix-analysis input is isolated in `h_ratio_of_trace`: a future determinant/trace +comparison theorem should prove that the trace budget implies the displayed determinant-ratio +bound. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) + (hreg : reg ≠ 0) (hd : d ≠ 0) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg hd h_quad_nonneg + h_quad_le_one L2 hL2 h_ratio_of_trace) + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the log-determinant elliptical potential and that potential is bounded by `W`. From c6e66069778944b25f515b96488627bb61a8c6d9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:48:39 -0400 Subject: [PATCH 57/82] feat(linUCB): proof chain no longer needs the future matrix theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 123 ++++++++++++++++++ 1 file changed, 123 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 1cf9bcea..82dd5743 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -771,6 +771,67 @@ lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (x := x) (n := n) (P := P) L2 hL2) h_ratio_of_trace +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A determinant upper bound for `V_n` implies the corresponding determinant-ratio bound, using +`det(V_0) = reg ^ d`. -/ +lemma designDetRatio_le_trace_budget_of_designDet_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hdet_le : designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + rw [designDetRatio, designDet_zero_eq_reg_pow] + have hreg_pow_nonneg : 0 ≤ reg ^ d := (pow_pos hreg_pos d).le + have hdiv : designDet A reg x n ω / reg ^ d ≤ (T / (d : ℝ)) ^ d / reg ^ d := by + exact div_le_div_of_nonneg_right hdet_le hreg_pow_nonneg + refine hdiv.trans_eq ?_ + rw [← div_pow] + congr 1 + field_simp [hreg_pos.ne', by exact_mod_cast hd] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a determinant upper bound for `V_n` implies the corresponding determinant-ratio +bound. -/ +lemma designDetRatio_ae_le_trace_budget_of_designDet_ae_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hdet_le : ∀ᵐ ω ∂P, designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + filter_upwards [hdet_le] with ω hdetω + exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) T hreg_pos hd hdetω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Converts an almost-sure trace bound plus a future determinant/trace comparison for `det(V_n)` +into the determinant-ratio bound used by the elliptical-potential chain. -/ +lemma designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ T → designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + refine designDetRatio_ae_le_trace_budget_of_designDet_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) T hreg_pos hd ?_ + filter_upwards [h_trace_le] with ω h_traceω + exact hdet_of_trace ω h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms reduce the determinant-ratio goal to the determinant upper bound +`det(V_n) ≤ ((reg * d + n * L2) / d) ^ d`. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + exact designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd + (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2) + hdet_of_trace + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1449,6 +1510,32 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budge (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 h_ratio_of_trace +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface with the determinant/trace comparison stated as a determinant +upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd h_nonneg + h_le_one L2 hL2 ?_ + intro ω h_traceω + exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd + (hdet_of_trace ω h_traceω) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ @@ -2538,6 +2625,42 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg hd h_quad_nonneg h_quad_le_one L2 hL2 h_ratio_of_trace) +/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term +when the remaining matrix-analysis input is stated as the determinant upper bound +`det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_nonneg + h_quad_le_one L2 hL2 hdet_of_trace) + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the log-determinant elliptical potential and that potential is bounded by `W`. From 5c3eb5c7194fe2b8ba39fb34ee30ce2363580b51 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 11:13:11 -0400 Subject: [PATCH 58/82] feat(linUCB): small clean up --- .../Online/Bandit/Algorithms/LinUCB.lean | 199 ++++++++++++++++-- 1 file changed, 184 insertions(+), 15 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 82dd5743..a3b08e01 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -9,6 +9,8 @@ public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.Data.Real.StarOrdered +public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse @@ -39,6 +41,12 @@ abbrev Feature (d : ℕ) := Fin d → ℝ def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := dotProduct (x a) (x a) +/-- The squared feature norm is nonnegative. -/ +lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : + 0 ≤ featureSqNorm x a := by + rw [featureSqNorm, dotProduct] + exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) + /-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := @@ -143,6 +151,16 @@ lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by simp [designMatrix, sum_range_succ, add_assoc] +/-- With nonnegative regularization, the process-level design matrix is positive semidefinite. -/ +lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : + (designMatrix A reg x n ω).PosSemidef := by + unfold designMatrix + apply Matrix.PosSemidef.add + · exact Matrix.PosSemidef.smul Matrix.PosSemidef.one hreg_nonneg + · refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -167,6 +185,14 @@ lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] +/-- With nonnegative regularization, the design trace is nonnegative. -/ +lemma designTrace_nonneg (hreg_nonneg : 0 ≤ reg) : + 0 ≤ designTrace A reg x n ω := by + rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] + exact add_nonneg + (mul_nonneg hreg_nonneg (Nat.cast_nonneg d)) + (sum_nonneg fun t _ ↦ featureSqNorm_nonneg x (A t ω)) + /-- If every selected feature vector has squared norm at most `L2`, then the trace of the design matrix is at most `reg * d + n * L2`. -/ lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound @@ -245,6 +271,42 @@ lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by simp [widthQuadraticForm, designMatrix_zero] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Nonnegative regularization makes every LinUCB width quadratic form nonnegative. + +The reason is purely matrix-theoretic: `V_n` is positive semidefinite, the nonsingular inverse of a +positive semidefinite matrix is positive semidefinite in mathlib, and every quadratic form induced +by a positive semidefinite matrix is nonnegative. -/ +lemma widthQuadraticForm_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) (a : Fin K) : + 0 ≤ widthQuadraticForm A reg x a n ω := by + simpa [widthQuadraticForm] using + ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg).inv.dotProduct_mulVec_nonneg (x a)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, nonnegative regularization gives nonnegative selected quadratic width forms +through any finite horizon. -/ +lemma widthQuadraticForm_ae_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + exact Filter.Eventually.of_forall fun ω t _ht ↦ + widthQuadraticForm_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) hreg_nonneg (A t ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive-time version of `widthQuadraticForm_ae_nonneg_of_reg_nonneg`, matching the side +condition shape used by the regret/width-sum bridge lemmas. -/ +lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + filter_upwards [widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) + (x := x) (n := n) (P := P) hreg_nonneg] with ω h_nonnegω + intro t ht _ht0 + exact h_nonnegω t ht + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := @@ -832,6 +894,61 @@ lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le (x := x) (n := n) (P := P) L2 hL2) hdet_of_trace +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix-level determinant/trace comparison needed for the finite-dimensional +elliptical-potential bound. + +For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality +`det(M) ≤ (trace(M) / d) ^ d`. -/ +def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := + ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the +design matrix is positive semidefinite. -/ +lemma designDet_le_trace_average_pow_of_matrix_det_trace_bound + (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) : + designDet A reg x n ω ≤ (designTrace A reg x n ω / (d : ℝ)) ^ d := by + simpa [designDet, designTrace] using + hdet_trace (designMatrix A reg x n ω) + (designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Combining `det(M) ≤ (trace(M)/d)^d` with a trace budget gives the determinant upper bound +`det(V_n) ≤ (T/d)^d`. -/ +lemma designDet_le_trace_budget_of_matrix_det_trace_bound + (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) + (hd : d ≠ 0) (T : ℝ) (h_trace_le : designTrace A reg x n ω ≤ T) : + designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d := by + have hd_pos : 0 < (d : ℝ) := by + exact_mod_cast Nat.pos_of_ne_zero hd + have hbase_nonneg : 0 ≤ designTrace A reg x n ω / (d : ℝ) := + div_nonneg (designTrace_nonneg (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg) hd_pos.le + have hbase_le : designTrace A reg x n ω / (d : ℝ) ≤ T / (d : ℝ) := + (div_le_div_iff_of_pos_right hd_pos).mpr h_trace_le + exact (designDet_le_trace_average_pow_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet_trace hreg_nonneg).trans + (pow_le_pow_left₀ hbase_nonneg hbase_le d) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms and the matrix-level determinant/trace comparison give the +determinant-ratio bound used by the elliptical-potential chain. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + refine designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 ?_ + intro ω h_traceω + exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) + hdet_trace hreg_pos.le hd h_traceω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1515,8 +1632,6 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (L2 : ℝ) @@ -1529,13 +1644,37 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo CappedQuadraticWidthBound A reg x n ω (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd h_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) h_le_one L2 hL2 ?_ intro ω h_traceω exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd (hdet_of_trace ω h_traceω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface where the remaining matrix-analysis input is the reusable +positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + refine + cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_le_one + L2 hL2 ?_ + intro ω h_traceω + exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd + (reg * (d : ℝ) + (n : ℝ) * L2) h_traceω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ @@ -2601,9 +2740,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg : reg ≠ 0) (hd : d ≠ 0) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hreg_pos : 0 < reg) (hd : d ≠ 0) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (L2 : ℝ) @@ -2622,7 +2759,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg hd h_quad_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) h_quad_le_one L2 hL2 h_ratio_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term @@ -2638,8 +2777,6 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (L2 : ℝ) @@ -2658,8 +2795,39 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_nonneg - h_quad_le_one L2 hL2 hdet_of_trace) + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one + L2 hL2 hdet_of_trace) + +/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term +when the remaining hard input is the reusable PSD matrix determinant/trace comparison +`det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one + L2 hL2 hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the @@ -2679,8 +2847,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hreg_nonneg : 0 ≤ reg) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (h_elliptical : ∀ᵐ ω ∂P, @@ -2693,8 +2860,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W (cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one - h_elliptical h_potential_le) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg (A := A) (reg := reg) + (x := x) (n := n) (P := P) hreg_nonneg) + h_quad_le_one h_elliptical h_potential_le) end LinUCB From c45b44629f4e0a25fb25bbbfbd8760cb77e97271 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 11:19:29 -0400 Subject: [PATCH 59/82] feat(linUCB): positive regularization now proves the design matrix is positive definite --- .../Online/Bandit/Algorithms/LinUCB.lean | 71 +++++++++++++++---- 1 file changed, 56 insertions(+), 15 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a3b08e01..cbf2b6b3 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -161,6 +161,16 @@ lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : intro t _ simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) +/-- Positive regularization makes the process-level design matrix positive definite. -/ +lemma designMatrix_posDef (hreg_pos : 0 < reg) : + (designMatrix A reg x n ω).PosDef := by + unfold designMatrix + apply Matrix.PosDef.add_posSemidef + · exact Matrix.PosDef.smul Matrix.PosDef.one hreg_pos + · refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -521,6 +531,26 @@ lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : rw [designDet_zero_eq_reg_pow] exact pow_ne_zero d hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization makes every process-level design determinant nonzero. -/ +lemma designDet_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : + designDet A reg x n ω ≠ 0 := by + have hunit : IsUnit (designMatrix A reg x n ω) := + (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).isUnit + have hdet_unit : IsUnit (designMatrix A reg x n ω).det := + (Matrix.isUnit_iff_isUnit_det (A := designMatrix A reg x n ω)).mp hunit + exact hdet_unit.ne_zero + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, positive regularization makes all design determinants in a finite horizon +nonzero. -/ +lemma designDet_ae_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + exact Filter.Eventually.of_forall fun ω t _ht ↦ + designDet_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) (ω := ω) + hreg_pos + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1468,6 +1498,23 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization discharges the determinant-nonvanishing and quadratic-form +nonnegativity obligations in the log-determinant elliptical-potential chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound {W : ℝ} + (hreg_pos : 0 < reg) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) + (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) + (n := n + 1) (P := P) hreg_pos) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + h_le_one h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a determinant-ratio upper bound, and the determinant-update step give the packaged capped @@ -2830,14 +2877,12 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound L2 hL2 hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the -log-determinant elliptical potential and that potential is bounded by `W`. - -This is the first theorem whose assumptions match the two matrix-analysis steps of the actual -elliptical-potential argument: +`2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final +log-determinant potential bound hold. -* prove `cappedQuadraticWidthSum ≤ ellipticalPotential`; -* prove `ellipticalPotential ≤ W`. -/ +The capped-sum/log-determinant part of the elliptical-potential argument is now proved internally: +positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while +`h_quad_le_one` supplies the cap needed for `min 1 q ≤ 2 * log (1 + q)`. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) @@ -2847,11 +2892,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hreg_nonneg : 0 ≤ reg) + (hreg_pos : 0 < reg) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -2859,11 +2902,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg (A := A) (reg := reg) - (x := x) (n := n) (P := P) hreg_nonneg) - h_quad_le_one h_elliptical h_potential_le) + (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) hreg_pos + h_quad_le_one h_potential_le) end LinUCB From c79afb07592fb5c4e9f3d52d12c1d949c7f1601f Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 13:15:17 -0400 Subject: [PATCH 60/82] feat(linUCB): linear-algebra bridge for the elliptical-potential regret theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 106 ++++++++++++++---- 1 file changed, 83 insertions(+), 23 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index cbf2b6b3..90e6693d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -317,6 +317,61 @@ lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg intro t ht _ht0 exact h_nonnegω t ht +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The matrix comparison needed to turn bounded feature vectors into the positive-time LinUCB +width cap. + +Mathematically, this says `x_aᵀ V_t⁻¹ x_a ≤ ‖x_a‖² / reg`. A later matrix-order proof should +derive it from `reg > 0` and `V_t = reg I + ∑ x_s x_sᵀ`. Keeping it as a named property makes the +remaining linear-algebra obligation precise and reusable. -/ +def WidthQuadraticFormLeFeatureSqNormDivReg + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (a : Fin K) (n : ℕ) (ω : Ω), + widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the +quadratic form is at most one. -/ +lemma widthQuadraticForm_le_one_of_featureSqNorm_le_reg + (a : Fin K) + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) + (h_feature_le : featureSqNorm x a ≤ reg) : + widthQuadraticForm A reg x a n ω ≤ 1 := by + refine (h_width a n ω).trans ?_ + rwa [div_le_one hreg_pos] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure positive-time width cap from the matrix comparison and an almost-sure +`featureSqNorm ≤ reg` bound along the selected actions. -/ +lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) + (h_feature_le : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by + filter_upwards [h_feature_le] with ω h_feature_leω + intro t ht _ht0 + exact widthQuadraticForm_le_one_of_featureSqNorm_le_reg + (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (A t ω) h_width hreg_pos + (h_feature_leω t ht) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure positive-time width cap from the matrix comparison and a selected-feature budget +`featureSqNorm ≤ L2`, when `L2 ≤ reg`. -/ +lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) {L2 : ℝ} + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by + refine widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg + (A := A) (reg := reg) (x := x) (n := n) (P := P) h_width hreg_pos ?_ + filter_upwards [hL2] with ω hL2ω + intro t ht + exact (hL2ω t ht).trans hL2_le_reg + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := @@ -1679,10 +1734,10 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_of_trace : ∀ ω, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → designDet A reg x n ω ≤ @@ -1694,7 +1749,9 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) - h_le_one L2 hL2 ?_ + (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + L2 hL2 ?_ intro ω h_traceω exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd @@ -1705,18 +1762,18 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_trace : MatrixDetLeTraceAveragePow d) : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by refine cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_le_one - L2 hL2 ?_ + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + h_width_le_feature L2 hL2 hL2_le_reg ?_ intro ω h_traceω exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd @@ -2775,9 +2832,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. -The remaining matrix-analysis input is isolated in `h_ratio_of_trace`: a future determinant/trace -comparison theorem should prove that the trace budget implies the displayed determinant-ratio -bound. -/ +The remaining matrix-analysis inputs are isolated as named hypotheses: `h_width_le_feature` should +come from the inverse-design comparison `xᵀV⁻¹x ≤ ‖x‖² / reg`, and `h_ratio_of_trace` should come +from a determinant/trace comparison proving that the trace budget implies the displayed +determinant-ratio bound. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) @@ -2788,10 +2846,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (h_ratio_of_trace : ∀ ω, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → designDetRatio A reg x n ω ≤ @@ -2809,10 +2867,12 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) - h_quad_le_one L2 hL2 h_ratio_of_trace) + (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + L2 hL2 h_ratio_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the remaining matrix-analysis input is stated as the determinant upper bound +when the determinant/trace input is stated as the determinant upper bound `det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le [Nonempty (Fin K)] @@ -2824,10 +2884,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_of_trace : ∀ ω, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → designDet A reg x n ω ≤ @@ -2842,11 +2902,11 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one - L2 hL2 hdet_of_trace) + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + h_width_le_feature L2 hL2 hL2_le_reg hdet_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the remaining hard input is the reusable PSD matrix determinant/trace comparison +when the determinant/trace input is the reusable PSD matrix determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound [Nonempty (Fin K)] @@ -2858,10 +2918,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_trace : MatrixDetLeTraceAveragePow d) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -2873,8 +2933,8 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one - L2 hL2 hdet_trace) + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + h_width_le_feature L2 hL2 hL2_le_reg hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 0fb3ae1b5e9f6e46a52a9cd0af22f1e76e2d87ca Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 13:21:17 -0400 Subject: [PATCH 61/82] feat(linUCB): convert matrix inequalities into scalar quadratic-form inequalities --- .../Online/Bandit/Algorithms/LinUCB.lean | 33 ++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 90e6693d..b8adacc6 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -9,6 +9,7 @@ public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.Analysis.Matrix.Order public import Mathlib.Data.Real.StarOrdered public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement @@ -23,7 +24,7 @@ Chapter 19 of *Bandit Algorithms*: open MeasureTheory ProbabilityTheory Filter Real Finset Learning -open scoped ENNReal NNReal Matrix +open scoped ENNReal NNReal Matrix MatrixOrder namespace Bandits @@ -171,6 +172,36 @@ lemma designMatrix_posDef (hreg_pos : 0 < reg) : intro t _ simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design matrix dominates its regularization part: after subtracting `reg • I`, what remains +is the sum of observed rank-one feature matrices, hence positive semidefinite. -/ +lemma designMatrix_sub_reg_smul_one_posSemidef : + (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by + have hsum : + (∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω))).PosSemidef := by + refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + simpa [designMatrix, add_sub_cancel_left] using hsum + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix-order form of `designMatrix_sub_reg_smul_one_posSemidef`: `reg • I ≤ V_n`. -/ +lemma reg_smul_one_le_designMatrix : + reg • (1 : Matrix (Fin d) (Fin d) ℝ) ≤ designMatrix A reg x n ω := by + rw [Matrix.le_iff] + exact designMatrix_sub_reg_smul_one_posSemidef (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix order is preserved by evaluating the quadratic form against a fixed feature vector. -/ +lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} + (hMN : M ≤ N) (u : Feature d) : + dotProduct u (M *ᵥ u) ≤ dotProduct u (N *ᵥ u) := by + have h_nonneg : 0 ≤ dotProduct u ((N - M) *ᵥ u) := by + simpa using (Matrix.le_iff.mp hMN).dotProduct_mulVec_nonneg u + rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg + exact sub_nonneg.mp h_nonneg + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From e4ac7701ac93c1256753eaa17f5b5436eb744201 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 13:31:06 -0400 Subject: [PATCH 62/82] feat(linUCB): width comparison needed by the regret theorem path --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index b8adacc6..99a2929d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -202,6 +202,31 @@ lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg exact sub_nonneg.mp h_nonneg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The inverse of the regularized identity is the reciprocal-scaled identity. -/ +lemma reg_smul_one_inv (hreg : reg ≠ 0) : + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ = + reg⁻¹ • (1 : Matrix (Fin d) (Fin d) ℝ) := by + rw [Matrix.inv_eq_left_inv] + simp [smul_smul, hreg] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The quadratic form induced by `(reg • I)⁻¹` is the squared norm divided by `reg`. -/ +lemma dotProduct_reg_smul_one_inv_mulVec (hreg : reg ≠ 0) (u : Feature d) : + dotProduct u (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ u) = + dotProduct u u / reg := by + rw [reg_smul_one_inv (reg := reg) (d := d) hreg] + simp [Matrix.smul_mulVec, div_eq_inv_mul, mul_comm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Arm-specific form of `dotProduct_reg_smul_one_inv_mulVec`. -/ +lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div + (hreg : reg ≠ 0) (a : Fin K) : + dotProduct (x a) (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) = + featureSqNorm x a / reg := by + simpa [featureSqNorm] using + dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -360,6 +385,37 @@ def WidthQuadraticFormLeFeatureSqNormDivReg ∀ (a : Fin K) (n : ℕ) (ω : Ω), widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the inverse design matrix is bounded by the inverse regularized identity, then the LinUCB +quadratic width is bounded by `featureSqNorm / reg` for one arm, time, and sample point. -/ +lemma widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le + (a : Fin K) + (h_inv : (designMatrix A reg x n ω)⁻¹ ≤ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) + (hreg : reg ≠ 0) : + widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg := by + calc + widthQuadraticForm A reg x a n ω = + dotProduct (x a) (((designMatrix A reg x n ω)⁻¹) *ᵥ (x a)) := rfl + _ ≤ dotProduct (x a) + (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) := + dotProduct_mulVec_le_of_matrix_le h_inv (x a) + _ = featureSqNorm x a / reg := + dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div + (reg := reg) (x := x) hreg a + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A pointwise inverse-order comparison for all times and sample points gives the reusable +`WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ +lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le + (hreg : reg ≠ 0) + (h_inv : ∀ n ω, (designMatrix A reg x n ω)⁻¹ ≤ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) : + WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by + intro a n ω + exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le + (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a (h_inv n ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the quadratic form is at most one. -/ From 6c16639090f29498ed55819b731b70af0999e539 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 14:17:59 -0400 Subject: [PATCH 63/82] feat(linUCB): regret theorem now using updated matrix theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++------ 1 file changed, 40 insertions(+), 16 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 99a2929d..99214ae9 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -227,6 +227,24 @@ lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div simpa [featureSqNorm] using dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret +route. + +Mathematically, this should follow from `reg • I ≤ V_t` and positive regularization: inversion +reverses the positive-definite matrix order, so `V_t⁻¹ ≤ (reg • I)⁻¹`. -/ +def DesignMatrixInvLeRegInv + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (n : ℕ) (ω : Ω), + (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection lemma for the named inverse-order obligation. -/ +lemma DesignMatrixInvLeRegInv.apply + (h_inv : DesignMatrixInvLeRegInv A reg x) (n : ℕ) (ω : Ω) : + (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := + h_inv n ω + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -409,12 +427,12 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in `WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (hreg : reg ≠ 0) - (h_inv : ∀ n ω, (designMatrix A reg x n ω)⁻¹ ≤ - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) : + (h_inv : DesignMatrixInvLeRegInv A reg x) : WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by intro a n ω exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a (h_inv n ω) hreg + (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a + (h_inv.apply n ω) hreg omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the @@ -1821,7 +1839,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1837,7 +1855,10 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + (x := x) (n := n) (P := P) + (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) + (x := x) hreg_pos.ne' h_inv_le_reg) + hreg_pos hL2 hL2_le_reg) L2 hL2 ?_ intro ω h_traceω exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) @@ -1849,7 +1870,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1860,7 +1881,7 @@ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound refine cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_width_le_feature L2 hL2 hL2_le_reg ?_ + h_inv_le_reg L2 hL2 hL2_le_reg ?_ intro ω h_traceω exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd @@ -2919,9 +2940,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_width_le_feature` should -come from the inverse-design comparison `xᵀV⁻¹x ≤ ‖x‖² / reg`, and `h_ratio_of_trace` should come -from a determinant/trace comparison proving that the trace budget implies the displayed +The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_le_reg` is the +inverse-design comparison `V_t⁻¹ ≤ (reg I)⁻¹`, and `h_ratio_of_trace` should come from a +determinant/trace comparison proving that the trace budget implies the displayed determinant-ratio bound. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound [Nonempty (Fin K)] @@ -2933,7 +2954,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -2955,7 +2976,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + (x := x) (n := n) (P := P) + (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) + (x := x) hreg_pos.ne' h_inv_le_reg) + hreg_pos hL2 hL2_le_reg) L2 hL2 h_ratio_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term @@ -2971,7 +2995,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -2990,7 +3014,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_width_le_feature L2 hL2 hL2_le_reg hdet_of_trace) + h_inv_le_reg L2 hL2 hL2_le_reg hdet_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term when the determinant/trace input is the reusable PSD matrix determinant/trace comparison @@ -3005,7 +3029,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -3021,7 +3045,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_width_le_feature L2 hL2 hL2_le_reg hdet_trace) + h_inv_le_reg L2 hL2 hL2_le_reg hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 92c3df161b2b82431d68c7c368ced811ec3da352 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 14:21:53 -0400 Subject: [PATCH 64/82] feat(linUCB): regret theorem dependent on reusable linear-algebra theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 51 ++++++++++++++----- 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 99214ae9..5955c6a1 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -227,6 +227,14 @@ lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div simpa [featureSqNorm] using dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Reusable matrix-analysis theorem needed for the LinUCB width comparison. + +It states the usual inverse anti-monotonicity of positive-definite matrices in the PSD order: +if `M` is positive definite and `M ≤ N`, then inversion reverses the order. -/ +def MatrixInvAntiMonoOnPosDef (d : ℕ) : Prop := + ∀ M N : Matrix (Fin d) (Fin d) ℝ, M.PosDef → M ≤ N → N⁻¹ ≤ M⁻¹ + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret route. @@ -245,6 +253,19 @@ lemma DesignMatrixInvLeRegInv.apply (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := h_inv n ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The reusable positive-definite inverse anti-monotonicity theorem implies the LinUCB-specific +inverse-design comparison. -/ +lemma DesignMatrixInvLeRegInv.of_matrix_inv_antitone + (hreg_pos : 0 < reg) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) : + DesignMatrixInvLeRegInv A reg x := by + intro n ω + exact h_inv_antitone (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) + (designMatrix A reg x n ω) + (Matrix.PosDef.smul Matrix.PosDef.one hreg_pos) + (reg_smul_one_le_designMatrix (A := A) (reg := reg) (x := x) (n := n) (ω := ω)) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -1839,7 +1860,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1857,7 +1878,9 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) (x := x) (n := n) (P := P) (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' h_inv_le_reg) + (x := x) hreg_pos.ne' + (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) + (x := x) hreg_pos h_inv_antitone)) hreg_pos hL2 hL2_le_reg) L2 hL2 ?_ intro ω h_traceω @@ -1870,7 +1893,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1881,7 +1904,7 @@ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound refine cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_le_reg L2 hL2 hL2_le_reg ?_ + h_inv_antitone L2 hL2 hL2_le_reg ?_ intro ω h_traceω exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd @@ -2940,9 +2963,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_le_reg` is the -inverse-design comparison `V_t⁻¹ ≤ (reg I)⁻¹`, and `h_ratio_of_trace` should come from a -determinant/trace comparison proving that the trace budget implies the displayed +The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_antitone` is the +generic inverse anti-monotonicity theorem for positive-definite matrices, and `h_ratio_of_trace` +should come from a determinant/trace comparison proving that the trace budget implies the displayed determinant-ratio bound. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound [Nonempty (Fin K)] @@ -2954,7 +2977,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -2978,7 +3001,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) (x := x) (n := n) (P := P) (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' h_inv_le_reg) + (x := x) hreg_pos.ne' + (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) + (x := x) hreg_pos h_inv_antitone)) hreg_pos hL2 hL2_le_reg) L2 hL2 h_ratio_of_trace) @@ -2995,7 +3020,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -3014,7 +3039,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_le_reg L2 hL2 hL2_le_reg hdet_of_trace) + h_inv_antitone L2 hL2 hL2_le_reg hdet_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term when the determinant/trace input is the reusable PSD matrix determinant/trace comparison @@ -3029,7 +3054,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -3045,7 +3070,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_le_reg L2 hL2 hL2_le_reg hdet_trace) + h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 5fb147e6b855e514c4d4fa295a44c7317f428d97 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 09:03:25 -0400 Subject: [PATCH 65/82] feat(linUCB): remaining deterministic/regret-shell steps --- .../Online/Bandit/Algorithms/LinUCB.lean | 474 +++++++++++++++++- 1 file changed, 472 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 5955c6a1..71e93bf7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -48,6 +48,13 @@ lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : rw [featureSqNorm, dotProduct] exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) +/-- Uniform squared feature-norm bound for finite-action LinUCB. + +This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared +form as `‖x_a‖₂² ≤ L2` for every action. -/ +def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := + ∀ a, featureSqNorm x a ≤ L2 + /-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := @@ -323,6 +330,14 @@ lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) L2 hL2ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A uniform finite-action feature bound implies the selected-action feature bound through any +finite horizon. -/ +lemma featureSqNorm_ae_le_of_featureSqNormBound + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2 := + Filter.Eventually.of_forall fun ω t _ht ↦ hL2 (A t ω) + /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := @@ -1225,6 +1240,25 @@ lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} rw [min_eq_right hq_le_one] exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar log inequality used in the textbook elliptical-potential proof: for `0 ≤ q`, +`min 1 q ≤ 2 * log (1 + q)`. -/ +lemma min_one_le_two_mul_log_one_add_of_nonneg {q : ℝ} + (hq_nonneg : 0 ≤ q) : + min 1 q ≤ 2 * Real.log (1 + q) := by + by_cases hq_le_one : q ≤ 1 + · exact min_one_le_two_mul_log_one_add_of_nonneg_le_one hq_nonneg hq_le_one + · have hq_one : 1 ≤ q := by linarith + have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := + Real.le_log_one_add_of_nonneg hq_nonneg + have hq_add_two_pos : 0 < q + 2 := by linarith + have hone_le_log_lower : 1 ≤ 2 * (2 * q / (q + 2)) := by + rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] + rw [le_div_iff₀ hq_add_two_pos] + nlinarith + rw [min_eq_left hq_one] + exact hone_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ @@ -1244,6 +1278,25 @@ lemma cappedWidthTerm_le_ellipticalPotentialStep (x := x) (n := n) (ω := ω) hdet] exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing and nonnegativity of the selected quadratic form, the single +capped quadratic-width term is bounded by the one-step log-determinant potential. This is the +textbook form; no separate `q ≤ 1` assumption is needed because the term is already capped. -/ +lemma cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialStep A reg x n ω := by + by_cases hn : n = 0 + · rw [if_pos hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) + · rw [if_neg hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact min_one_le_two_mul_log_one_add_of_nonneg h_nonneg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ @@ -1261,6 +1314,21 @@ lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the +per-step one-step-potential bound for the capped quadratic-width term. -/ +lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω := by + filter_upwards [hdet, h_nonneg] with ω hdetω h_nonnegω + intro t ht + exact cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg (A := A) (reg := reg) + (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- At horizon zero, the log-determinant potential is zero when the initial design determinant is nonzero. -/ @@ -1415,6 +1483,39 @@ lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_o intro t ht rw [h_eq t ht] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the +capped-sum/log-determinant elliptical-potential bound. + +This is the capped form used in the textbook proof of LinUCB: the quadratic forms do not need to +be bounded by `1`, because the accumulated quantity is `min 1 q_t`. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet + (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet_range_n h_nonneg) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization discharges determinant nonvanishing and nonnegativity, yielding the +capped-sum/log-determinant elliptical-potential bound directly. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos + (hreg_pos : 0 < reg) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) + (n := n + 1) (P := P) hreg_pos) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -1830,6 +1931,42 @@ lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add rw [Real.log_pow, hbase] ring +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Textbook capped elliptical-potential budget from bounded selected feature norms and the +matrix-level determinant/trace comparison. + +Unlike `cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound`, this theorem bounds the capped +quadratic-width sum directly and does not assume the individual quadratic forms are at most `1`. -/ +lemma cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ + 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) + rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] + have h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := + widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le + have h_potential_le : ∀ᵐ ω ∂P, + ellipticalPotential A reg x n ω ≤ + 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) := by + exact ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' h_nonneg) + (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 + hdet_trace) + filter_upwards [cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos, h_potential_le] with + ω h_capped_le h_potentialω + exact h_capped_le.trans h_potentialω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Feature-norm-budget interface with the log term rewritten in the standard `2 * d * log(1 + n L² / (reg d))` shape. -/ @@ -1950,6 +2087,72 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by simp [index_zero, width_zero] +/-- The pointwise LinUCB confidence event used by the finite-action regret proof. + +For every positive process time, the best arm's true mean lies below its optimistic index, and the +selected arm's pessimistic index lies below its true mean. On this event, the max-index property of +LinUCB turns optimism into an instantaneous regret bound. -/ +def LinUCBConfidenceEvent [Nonempty (Fin K)] + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] + +omit [IsMarkovKernel ν] in +/-- Uniform bound on arm gaps, used as the finite-action analogue of the textbook bounded +instantaneous-regret assumption. -/ +def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := + ∀ a, gap ν a ≤ G + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ +lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → gap ν (A t ω) ≤ G := + Filter.Eventually.of_forall fun ω t _ht ↦ hG (A t ω) + +omit [IsMarkovKernel ν] in +/-- First projection from the packaged LinUCB confidence event: optimism for the best arm. -/ +lemma LinUCBConfidenceEvent.best [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by + intro t ht + exact (h_conf t ht).1 + +omit [IsMarkovKernel ν] in +/-- Second projection from the packaged LinUCB confidence event: validity of the selected arm's +lower confidence inequality. -/ +lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ t, t ≠ 0 → + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by + intro t ht + exact (h_conf t ht).2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ +lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by + filter_upwards [h_conf] with ω h_confω + exact h_confω.best + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure projection of the packaged confidence event to the selected arm's lower confidence +inequality. -/ +lemma linUCBConfidenceEvent_ae_arm [Nonempty (Fin K)] + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by + filter_upwards [h_conf] with ω h_confω + exact h_confω.arm + lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = @@ -2546,6 +2749,43 @@ lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) +omit [IsMarkovKernel ν] in +/-- Pointwise capped LinUCB regret bound for one positive time. + +If the instantaneous gap is bounded by `2`, and the confidence/max-index argument gives the usual +`2 * sqrt(β_t) * width_t` bound, then monotonicity up to the terminal `β n` gives the textbook +capped form `2 * sqrt(β n) * sqrt(min 1 q_t)`, where `q_t` is the width quadratic form. -/ +lemma gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm + (t : ℕ) + (h_gap_two : gap ν (A t ω) ≤ 2) + (h_gap_width : gap ν (A t ω) ≤ + 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) + (hβ_le : β (t + 1) ≤ β n) + (hβn_one : 1 ≤ β n) : + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + by_cases hq_le_one : widthQuadraticForm A reg x (A t ω) t ω ≤ 1 + · have hwidth_nonneg : 0 ≤ width A reg x (A t ω) t ω := Real.sqrt_nonneg _ + have hsqrt_le : √(β (t + 1)) ≤ √(β n) := Real.sqrt_le_sqrt hβ_le + have hbonus_le : + 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) ≤ + 2 * (√(β n) * width A reg x (A t ω) t ω) := by + exact mul_le_mul_of_nonneg_left + (mul_le_mul_of_nonneg_right hsqrt_le hwidth_nonneg) (by norm_num) + have hmin : + √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)) = + width A reg x (A t ω) t ω := by + rw [min_eq_right hq_le_one, width] + simpa [hmin] using h_gap_width.trans hbonus_le + · have hq_one : 1 ≤ widthQuadraticForm A reg x (A t ω) t ω := by linarith + have hsqrt_one : 1 ≤ √(β n) := by + simpa using (Real.one_le_sqrt).2 hβn_one + have htwo_le : + 2 ≤ 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + rw [min_eq_left hq_one, Real.sqrt_one] + nlinarith + exact h_gap_two.trans htwo_le + omit [IsMarkovKernel ν] in /-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded by the corresponding sum of pointwise bounds. -/ @@ -2576,6 +2816,26 @@ lemma regret_le_sum_width_of_forall_gap_le · simp [ht0] · simpa [ht0] using h_gap t ht ht0 +omit [IsMarkovKernel ν] in +/-- A pathwise cumulative-regret bound obtained by summing the positive-time capped LinUCB width +bound. -/ +lemma regret_le_sum_sqrt_capped_width_of_forall_gap_le + (h_gap : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : + regret ν A n ω ≤ + ∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) + (B := fun t ↦ + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simpa [ht0] using h_gap t ht ht0 + omit [IsMarkovKernel ν] in /-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : @@ -2604,6 +2864,112 @@ lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : (fun t ↦ if t = 0 then 0 else √(β (t + 1))) (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) +omit [IsMarkovKernel ν] in +/-- Cauchy-Schwarz bound for the positive-time capped LinUCB bonus sum. -/ +lemma sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum + (hβn_nonneg : 0 ≤ β n) + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ≤ + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by + calc + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) + = 2 * ∑ t ∈ range n, + (if t = 0 then 0 else √(β n)) * + (if t = 0 then 0 + else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + rw [mul_sum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) * + √(∑ t ∈ range n, + (if t = 0 then 0 + else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) ^ 2)) := by + gcongr + exact Real.sum_mul_le_sqrt_mul_sqrt (range n) + (fun t ↦ if t = 0 then 0 else √(β n)) + (fun t ↦ if t = 0 then 0 + else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) + _ ≤ 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by + gcongr + · calc + (∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) + ≤ ∑ _t ∈ range n, β n := by + refine sum_le_sum ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0, hβn_nonneg] + · simp [ht0, Real.sq_sqrt hβn_nonneg] + _ = (n : ℝ) * β n := by + simp [sum_const, nsmul_eq_mul] + · rw [cappedQuadraticWidthSum] + refine le_of_eq ?_ + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · have hmin_nonneg : + 0 ≤ min 1 (widthQuadraticForm A reg x (A t ω) t ω) := by + exact le_min zero_le_one (h_nonneg t ht ht0) + simp [ht0, Real.sq_sqrt hmin_nonneg] + +omit [IsMarkovKernel ν] in +/-- Pathwise cumulative-regret bound using the textbook capped quadratic-width sum. -/ +lemma regret_le_initial_add_sqrt_nat_mul_beta_capped_sum + (hβn_nonneg : 0 ≤ β n) + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_gap : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by + refine (regret_le_sum_sqrt_capped_width_of_forall_gap_le (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) h_gap).trans ?_ + have hsplit : + (∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) = + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + ∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β n) * + √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + rw [← sum_add_distrib] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + rw [hsplit] + exact add_le_add le_rfl + (sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum + (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω) + hβn_nonneg h_nonneg) + +omit [IsMarkovKernel ν] in +/-- If the capped quadratic-width sum is bounded by `W`, the pathwise capped regret bound can use +`√W` in place of the realized capped-sum square root. -/ +lemma regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (W : ℝ) + (h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω))) + (hW : cappedQuadraticWidthSum A reg x n ω ≤ W) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √W) := by + refine h_regret.trans ?_ + gcongr + /-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is nonnegative. -/ lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : @@ -2959,6 +3325,61 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is almost surely bounded +by `W`. + +This version follows the proof structure of *Bandit Algorithms*, Theorem 19.2: optimism gives the +width bound, bounded instantaneous gaps give the cap, monotonicity of `β` moves all confidence +radii to `β n`, and Cauchy-Schwarz turns the sum into the square root of the capped quadratic-width +sum. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) + (hβ_nonneg : ∀ t, 0 ≤ β t) + (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] + with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω + have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + intro t ht _ht0 + exact h_quad_nonnegω t ht + have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + intro t ht ht0 + have hβ_le : β (t + 1) ≤ β n := + hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) + have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 + have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) + have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos + have hβn_one : 1 ≤ β n := hβ_one.trans (hβ_mono hn_one) + exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) + (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one + have h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := + regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (hβ_nonneg n) h_quad_pos + h_gap_capped + simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using + regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. @@ -3072,13 +3493,62 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) +/-- Textbook-shaped finite-action LinUCB regret theorem. + +This theorem is the same deterministic regret skeleton as the theorem above, but with assumptions +packaged in the way the finite-action linear-bandit proof is usually read: + +* `h_conf` is the high-probability confidence event for all positive times; +* `h_gap_bound` is the bounded instantaneous-regret/gap assumption; +* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`; +* `hdet_trace` is the reusable determinant/trace matrix-analysis obligation. + +The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression +`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this +formalization lets the deterministic algorithm play its default initial arm at time zero. -/ +lemma regret_ae_le_textbook_finite_action + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) + (h_gap_bound : GapBound (K := K) ν 2) + (hβ_nonneg : ∀ t, 0 ≤ β t) + (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by + filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) + 2 h_gap_bound] with ω h_gapω + intro t ht _ht0 + exact h_gapω t ht + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + h_gap_two hβ_nonneg hβ_one hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 + (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) + (P := P) L2 hL2) + hdet_trace) + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final log-determinant potential bound hold. -The capped-sum/log-determinant part of the elliptical-potential argument is now proved internally: +The capped-sum/log-determinant part of the elliptical-potential argument is proved internally: positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while -`h_quad_le_one` supplies the cap needed for `min 1 q ≤ 2 * log (1 + q)`. -/ +`h_quad_le_one` lets this older theorem feed the uncapped `widthSqSum` regret route. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) From 2057e7326e2844c3fca0ab737c6d9b375cd13ee2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:00:07 -0400 Subject: [PATCH 66/82] feat(linUCB): the matrix determinant/trace assumption --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++++++-- 1 file changed, 51 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 71e93bf7..0ce917b4 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -8,6 +8,7 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.Analysis.MeanInequalities public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.Analysis.Matrix.Order public import Mathlib.Data.Real.StarOrdered @@ -1129,6 +1130,53 @@ For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar AM-GM in the form used for PSD matrix eigenvalues: +the product of nonnegative entries is bounded by the arithmetic mean to the `card` power. -/ +lemma prod_le_average_pow_of_nonneg {ι : Type*} [Fintype ι] [Nonempty ι] + (z : ι → ℝ) (hz : ∀ i, 0 ≤ z i) : + (∏ i, z i) ≤ ((∑ i, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by + classical + have hN_pos : 0 < (Fintype.card ι : ℝ) := by + exact_mod_cast Fintype.card_pos_iff.mpr inferInstance + have hweights_pos : 0 < ∑ i : ι, (1 : ℝ) := by + simpa using hN_pos + have h_amgm := Real.geom_mean_le_arith_mean (s := Finset.univ) + (w := fun _ : ι ↦ (1 : ℝ)) (z := z) + (by intro i hi; norm_num) hweights_pos (by intro i hi; exact hz i) + have h_amgm' : + (∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹) ≤ + (∑ i : ι, z i) / (Fintype.card ι : ℝ) := by + simpa using h_amgm + have hprod_nonneg : 0 ≤ ∏ i : ι, z i := by + exact Finset.prod_nonneg fun i _ ↦ hz i + have hraise := Real.rpow_le_rpow (Real.rpow_nonneg hprod_nonneg _) h_amgm' hN_pos.le + have hleft : + ((∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹)) ^ (Fintype.card ι : ℝ) = + ∏ i : ι, z i := by + rw [← Real.rpow_mul hprod_nonneg] + rw [inv_mul_cancel₀ hN_pos.ne'] + simp + have hright : + ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ (Fintype.card ι : ℝ) = + ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by + rw [Real.rpow_natCast] + simpa [hleft, hright] using hraise + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- PSD matrix determinant/trace comparison from AM-GM over eigenvalues: +`det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma matrixDetLeTraceAveragePow : MatrixDetLeTraceAveragePow d := by + intro M hM + by_cases hd : d = 0 + · subst d + simp + · haveI : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp (Nat.pos_of_ne_zero hd) + rw [hM.1.det_eq_prod_eigenvalues, hM.1.trace_eq_sum_eigenvalues] + simpa using prod_le_average_pow_of_nonneg + (z := fun i : Fin d ↦ hM.1.eigenvalues i) + (fun i ↦ hM.eigenvalues_nonneg i) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the design matrix is positive semidefinite. -/ @@ -3500,8 +3548,7 @@ packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; * `h_gap_bound` is the bounded instantaneous-regret/gap assumption; -* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`; -* `hdet_trace` is the reusable determinant/trace matrix-analysis obligation. +* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression `2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this @@ -3514,8 +3561,7 @@ lemma regret_ae_le_textbook_finite_action (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + @@ -3540,7 +3586,7 @@ lemma regret_ae_le_textbook_finite_action (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) (P := P) L2 hL2) - hdet_trace) + matrixDetLeTraceAveragePow) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From fdf56f92161bff5136ffeefdb38a10b108e97dcc Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:06:29 -0400 Subject: [PATCH 67/82] =?UTF-8?q?feat(linUCB):=20regret=20theorem=20no=20l?= =?UTF-8?q?onger=20requires=20the=20assumption=20hd=20:=20d=20=E2=89=A0=20?= =?UTF-8?q?0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 125 +++++++++++++++--- 1 file changed, 104 insertions(+), 21 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 0ce917b4..ca2af36b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2135,6 +2135,36 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by simp [index_zero, width_zero] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every least-squares reward estimate is zero. -/ +lemma estimatedReward_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + estimatedReward A R reg x a n ω = 0 := by + subst d + simp [estimatedReward, dotProduct] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB quadratic width form is zero. -/ +lemma widthQuadraticForm_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + widthQuadraticForm A reg x a n ω = 0 := by + subst d + simp [widthQuadraticForm, dotProduct] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB width is zero. -/ +lemma width_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + width A reg x a n ω = 0 := by + simp [width, widthQuadraticForm_eq_zero_of_dim_eq_zero (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hd a] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB index is zero. -/ +lemma index_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + index A R reg β x a n ω = 0 := by + simp [index, estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) hd a, + width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hd a] + /-- The pointwise LinUCB confidence event used by the finite-action regret proof. For every positive process time, the best arm's true mean lies below its optimistic index, and the @@ -2181,6 +2211,28 @@ lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] intro t ht exact (h_conf t ht).2 +omit [IsMarkovKernel ν] in +/-- In zero feature dimension, the confidence event forces every positive-time selected gap to be +nonpositive. The best-arm index is zero, and the selected-arm pessimistic index is also zero. -/ +lemma gap_nonpos_of_confidence_dim_eq_zero [Nonempty (Fin K)] + (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) + (t : ℕ) (ht : t ≠ 0) : + gap ν (A t ω) ≤ 0 := by + have hbest := LinUCBConfidenceEvent.best (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (ω := ω) h_conf t ht + have harm := LinUCBConfidenceEvent.arm (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (ω := ω) h_conf t ht + rw [gap_eq_bestArm_sub] + have hbest0 : (ν (bestArm ν))[id] ≤ 0 := by + simpa [index_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) (β := β) + (x := x) (n := t) (ω := ω) hd (bestArm ν)] using hbest + have harm0 : 0 ≤ (ν (A t ω))[id] := by + simpa [estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) + (x := x) (n := t) (ω := ω) hd (A t ω), + width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := t) + (ω := ω) hd (A t ω)] using harm + linarith + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] @@ -3209,6 +3261,32 @@ lemma initial_gap_sum_eq : if n = 0 then 0 else gap ν (A 0 ω) := by cases n <;> simp +omit [IsMarkovKernel ν] in +/-- In zero feature dimension, the confidence event bounds cumulative regret by the initial gap. +There is no positive-time width contribution because all widths are zero. -/ +lemma regret_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] + (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by + refine (regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) + (B := fun t ↦ if t = 0 then gap ν (A 0 ω) else 0) ?_).trans ?_ + · intro t _ht + by_cases ht0 : t = 0 + · simp [ht0] + · simpa [ht0] using + gap_nonpos_of_confidence_dim_eq_zero (A := A) (R := R) (reg := reg) + (β := β) (x := x) (ν := ν) (ω := ω) hd h_conf t ht0 + · rw [initial_gap_sum_eq] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure zero-dimensional version of the finite-action LinUCB regret skeleton. -/ +lemma regret_ae_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] + (hd : d = 0) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ᵐ ω ∂P, regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by + filter_upwards [h_conf] with ω h_confω + exact regret_le_initial_gap_of_confidence_dim_eq_zero (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hd h_confω + /-- Almost surely, cumulative regret is bounded by the initial gap plus `2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` is nonnegative and monotone. -/ @@ -3560,33 +3638,38 @@ lemma regret_ae_le_textbook_finite_action (h_gap_bound : GapBound (K := K) ν 2) (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by - filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 h_gap_bound] with ω h_gapω - intro t ht _ht0 - exact h_gapω t ht - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) - (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) - h_gap_two hβ_nonneg hβ_one hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 - (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) - (P := P) L2 hL2) - matrixDetLeTraceAveragePow) + by_cases hd : d = 0 + · subst d + simpa using regret_ae_le_initial_gap_of_confidence_dim_eq_zero + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + (P := P) (d := 0) rfl h_conf + · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by + filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) + 2 h_gap_bound] with ω h_gapω + intro t ht _ht0 + exact h_gapω t ht + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + h_gap_two hβ_nonneg hβ_one hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 + (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) + (P := P) L2 hL2) + matrixDetLeTraceAveragePow) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 2ca2a1b173d006db48e0f0506fcabd91f99612d5 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:10:33 -0400 Subject: [PATCH 68/82] =?UTF-8?q?feat(linUCB):=20final=20theorem=20no=20lo?= =?UTF-8?q?nger=20exposes=20the=20raw=20GapBound=20=CE=BD=202=20assumption?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 33 +++++++++++++++++-- 1 file changed, 30 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index ca2af36b..82c63fef 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2185,6 +2185,32 @@ instantaneous-regret assumption. -/ def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := ∀ a, gap ν a ≤ G +omit [IsMarkovKernel ν] in +/-- Uniform bound on arm means. For finite-action linear bandits this is a convenient way to state +the usual bounded expected-reward assumption, for example `(ν a)[id] ∈ [-1, 1]`. -/ +def MeanRewardBound (ν : Kernel (Fin K) ℝ) (lo hi : ℝ) : Prop := + ∀ a, lo ≤ (ν a)[id] ∧ (ν a)[id] ≤ hi + +omit [IsMarkovKernel ν] in +/-- If every arm mean lies in `[lo, hi]`, then every arm gap is at most `hi - lo`. -/ +lemma gap_le_of_meanRewardBound [Nonempty (Fin K)] {lo hi : ℝ} + (hμ : MeanRewardBound ν lo hi) (a : Fin K) : + gap ν a ≤ hi - lo := by + rw [gap_eq_bestArm_sub] + have hbest_le : (ν (bestArm ν))[id] ≤ hi := (hμ (bestArm ν)).2 + have ha_ge : lo ≤ (ν a)[id] := (hμ a).1 + linarith + +omit [IsMarkovKernel ν] in +/-- Arm means in `[-1, 1]` imply the gap cap `gap ≤ 2` used by the capped regret argument. -/ +lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] + (hμ : MeanRewardBound ν (-1) 1) : + GapBound (K := K) ν 2 := by + intro a + have hgap := gap_le_of_meanRewardBound (ν := ν) (lo := -1) (hi := 1) hμ a + norm_num at hgap + exact hgap + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : @@ -3625,7 +3651,7 @@ This theorem is the same deterministic regret skeleton as the theorem above, but packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; -* `h_gap_bound` is the bounded instantaneous-regret/gap assumption; +* `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression @@ -3635,7 +3661,7 @@ lemma regret_ae_le_textbook_finite_action [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_gap_bound : GapBound (K := K) ν 2) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) @@ -3652,7 +3678,8 @@ lemma regret_ae_le_textbook_finite_action (P := P) (d := 0) rfl h_conf · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 h_gap_bound] with ω h_gapω + 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with + ω h_gapω intro t ht _ht0 exact h_gapω t ht exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound From 36fa51a421e776cec1b91577fa0ef42075fa10c1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:03:11 -0400 Subject: [PATCH 69/82] =?UTF-8?q?feat(linUCB):=20remove=20h=CE=B2=5Fnonneg?= =?UTF-8?q?=20assumption?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 22 +++++++++++++++---- 1 file changed, 18 insertions(+), 4 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 82c63fef..74746837 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3279,6 +3279,15 @@ lemma beta_sum_le_nat_mul_of_monotone _ = (n : ℝ) * β n := by simp [sum_const, nsmul_eq_mul] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive +horizon. -/ +lemma beta_nonneg_of_one_le_of_monotone + (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) {n : ℕ} (hn : n ≠ 0) : + 0 ≤ β n := by + have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) + exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) + omit [IsMarkovKernel ν] in /-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the horizon is zero. -/ @@ -3494,7 +3503,6 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (W : ℝ) (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) @@ -3502,6 +3510,11 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + by_cases hn : n = 0 + · subst n + exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] + have hβn_nonneg : 0 ≤ β n := + beta_nonneg_of_one_le_of_monotone (β := β) hβ_one hβ_mono hn filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → @@ -3526,7 +3539,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (hβ_nonneg n) h_quad_pos + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos h_gap_capped simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) @@ -3652,6 +3665,8 @@ packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; * `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; +* `hβ_one` and `hβ_mono` state that the confidence-radius schedule starts at least at one and is + monotone; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression @@ -3662,7 +3677,6 @@ lemma regret_ae_le_textbook_finite_action (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : @@ -3688,7 +3702,7 @@ lemma regret_ae_le_textbook_finite_action (x := x) (ν := ν) (P := P) h_conf) (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (P := P) h_conf) - h_gap_two hβ_nonneg hβ_one hβ_mono + h_gap_two hβ_one hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) From 3ad7f30c4c6a66408d3662ae6ba527c597298d74 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:09:10 -0400 Subject: [PATCH 70/82] feat(linUCB): theorem statement closer to textbook --- .../Online/Bandit/Algorithms/LinUCB.lean | 37 +++++++++++++++---- 1 file changed, 29 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 74746837..e1b9a9ab 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3279,6 +3279,22 @@ lemma beta_sum_le_nat_mul_of_monotone _ = (n : ℝ) * β n := by simp [sum_const, nsmul_eq_mul] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Minimal confidence-radius schedule assumptions used by the capped finite-action LinUCB regret +chain: the schedule starts at least at one and is monotone in time. -/ +def BetaSchedule (β : ℕ → ℝ) : Prop := + 1 ≤ β 1 ∧ Monotone β + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection from `BetaSchedule`: the confidence-radius schedule starts at least at one. -/ +lemma BetaSchedule.one (hβ : BetaSchedule β) : 1 ≤ β 1 := + hβ.1 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection from `BetaSchedule`: the confidence-radius schedule is monotone. -/ +lemma BetaSchedule.monotone (hβ : BetaSchedule β) : Monotone β := + hβ.2 + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive horizon. -/ @@ -3288,6 +3304,12 @@ lemma beta_nonneg_of_one_le_of_monotone have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A `BetaSchedule` is nonnegative at every positive horizon. -/ +lemma BetaSchedule.nonneg_of_ne_zero (hβ : BetaSchedule β) {n : ℕ} (hn : n ≠ 0) : + 0 ≤ β n := + beta_nonneg_of_one_le_of_monotone (β := β) hβ.one hβ.monotone hn + omit [IsMarkovKernel ν] in /-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the horizon is zero. -/ @@ -3503,7 +3525,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (W : ℝ) + (hβ_schedule : BetaSchedule β) (W : ℝ) (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : @@ -3514,7 +3536,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound · subst n exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] have hβn_nonneg : 0 ≤ β n := - beta_nonneg_of_one_le_of_monotone (β := β) hβ_one hβ_mono hn + hβ_schedule.nonneg_of_ne_zero hn filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → @@ -3526,11 +3548,11 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by intro t ht ht0 have hβ_le : β (t + 1) ≤ β n := - hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) + hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_one.trans (hβ_mono hn_one) + have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one @@ -3665,8 +3687,7 @@ packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; * `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; -* `hβ_one` and `hβ_mono` state that the confidence-radius schedule starts at least at one and is - monotone; +* `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression @@ -3677,7 +3698,7 @@ lemma regret_ae_le_textbook_finite_action (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) + (hβ_schedule : BetaSchedule β) (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, @@ -3702,7 +3723,7 @@ lemma regret_ae_le_textbook_finite_action (x := x) (ν := ν) (P := P) h_conf) (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (P := P) h_conf) - h_gap_two hβ_one hβ_mono + h_gap_two hβ_schedule (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) From be7acde83f358b6d0aeb794db6dbe9d6ff012837 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:20:24 -0400 Subject: [PATCH 71/82] feat(linUCB): theorem statement closer to textbook --- .../Online/Bandit/Algorithms/LinUCB.lean | 120 +++++++++++++++--- 1 file changed, 102 insertions(+), 18 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e1b9a9ab..947877f4 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3567,6 +3567,66 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω +/-- Almost surely, on the LinUCB confidence event, cumulative regret is bounded by the simplified +initial-gap term plus `2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is +almost surely bounded by `W`. + +This is the good-event form of the deterministic regret argument. It separates the algorithmic +regret proof from the future concentration theorem: a later self-normalized concentration result +should prove that `LinUCBConfidenceEvent` holds with high probability, and this theorem converts +that event into the regret bound. -/ +lemma regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) + (hβ_schedule : BetaSchedule β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + LinUCBConfidenceEvent A R reg β x ν ω → + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + by_cases hn : n = 0 + · subst n + exact Filter.Eventually.of_forall fun ω _h_confω ↦ by simp [regret] + have hβn_nonneg : 0 ≤ β n := + hβ_schedule.nonneg_of_ne_zero hn + filter_upwards [forall_index_le_index_arm h (bestArm ν), h_gap_two, h_quad_nonneg, hW] with + ω h_indexω h_gap_twoω h_quad_nonnegω hWω h_confω + have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + intro t ht _ht0 + exact h_quad_nonnegω t ht + have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + intro t ht ht0 + have h_gap_width : + gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := + gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (n := t) (ω := ω) (h_confω.best t ht0) + (h_confω.arm t ht0) (h_indexω t ht0) + have hβ_le : β (t + 1) ≤ β n := + hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) + have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 + have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) + have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos + have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) + exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) + (h_gap_twoω t ht ht0) h_gap_width hβ_le hβn_one + have h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := + regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos + h_gap_capped + simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using + regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. @@ -3680,49 +3740,52 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) -/-- Textbook-shaped finite-action LinUCB regret theorem. +/-- Textbook-shaped finite-action LinUCB regret theorem on the confidence event. -This theorem is the same deterministic regret skeleton as the theorem above, but with assumptions -packaged in the way the finite-action linear-bandit proof is usually read: +This theorem is the good-event form closest to the finite-action LinUCB proof in +*Bandit Algorithms*: after the deterministic algorithm/max-index argument and the elliptical +potential bound are proved, the only remaining probabilistic input is whether the confidence event +holds on a sample path. -* `h_conf` is the high-probability confidence event for all positive times; * `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; * `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. +The conclusion is an almost-sure implication: on almost every sample path, if +`LinUCBConfidenceEvent` holds, then the displayed regret bound holds. A future self-normalized +concentration theorem should prove that this confidence event has high probability for a concrete +textbook choice of `β`. + The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression `2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this formalization lets the deterministic algorithm play its default initial arm at time zero. -/ -lemma regret_ae_le_textbook_finite_action +lemma regret_ae_imp_le_textbook_finite_action [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) (hβ_schedule : BetaSchedule β) (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + LinUCBConfidenceEvent A R reg β x ν ω → + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by by_cases hd : d = 0 · subst d - simpa using regret_ae_le_initial_gap_of_confidence_dim_eq_zero - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) - (P := P) (d := 0) rfl h_conf + exact Filter.Eventually.of_forall fun ω h_confω ↦ by + simpa using regret_le_initial_gap_of_confidence_dim_eq_zero + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + (ω := ω) (d := 0) rfl h_confω · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with ω h_gapω intro t ht _ht0 exact h_gapω t ht - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + exact regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) - (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) h_gap_two hβ_schedule (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) @@ -3733,6 +3796,27 @@ lemma regret_ae_le_textbook_finite_action (P := P) L2 hL2) matrixDetLeTraceAveragePow) +/-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to +hold almost surely. This is stronger than the textbook high-probability route and is mainly useful +as a compatibility wrapper for earlier lemmas in this file. -/ +lemma regret_ae_le_textbook_finite_action + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2, h_conf] with ω h_regret h_confω + exact h_regret h_confω + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final log-determinant potential bound hold. From 23fa3c77c6957767c6bb02ff90e001c682831fb2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:27:08 -0400 Subject: [PATCH 72/82] feat(linUCB): new lemmas that lift sample-path implication into probability statements --- .../Online/Bandit/Algorithms/LinUCB.lean | 79 +++++++++++++++++++ 1 file changed, 79 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 947877f4..5bf068de 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3796,6 +3796,85 @@ lemma regret_ae_imp_le_textbook_finite_action (P := P) L2 hL2) matrixDetLeTraceAveragePow) +/-- The confidence event is almost surely contained in the textbook finite-action regret-bound +event. + +This is the probability bridge needed after the good-event theorem: once a concentration theorem +proves that `LinUCBConfidenceEvent` has high probability, this lemma transfers that probability +mass to the displayed regret bound. -/ +lemma probReal_confidenceEvent_le_textbook_regret_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2] with ω h_regret h_confω + exact h_regret h_confω + +/-- High-probability wrapper for the textbook finite-action LinUCB regret bound. + +If a future self-normalized concentration theorem proves that the confidence event has probability +at least `1 - δ`, then the textbook regret bound has probability at least `1 - δ` as well. -/ +lemma probReal_textbook_regret_bound_ge_of_confidenceEvent_ge + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : + 1 - δ ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by + exact h_conf_prob.trans + (probReal_confidenceEvent_le_textbook_regret_bound (A := A) (R := R) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule hreg_pos L2 hL2) + +/-- Failure-probability wrapper for the textbook finite-action LinUCB regret bound. + +If a future self-normalized concentration theorem proves that the confidence event fails with +probability at most `δ`, then the textbook regret bound fails with probability at most `δ`. -/ +lemma probReal_textbook_regret_bound_failure_le_of_confidenceEvent_failure_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_failure : + P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : + P.real {ω | + ¬ + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} ≤ δ := by + refine le_trans ?_ h_conf_failure + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω + exact h_regret_failure (h_regret h_confω) + /-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to hold almost surely. This is stronger than the textbook high-probability route and is mainly useful as a compatibility wrapper for earlier lemmas in this file. -/ From 34609628000b118ac03a5a32460842caab06d9d0 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:34:28 -0400 Subject: [PATCH 73/82] feat(linUCB): remaining deterministic/probability plumbing --- .../Online/Bandit/Algorithms/LinUCB.lean | 125 ++++++++++++++++++ 1 file changed, 125 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 5bf068de..da0dc185 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2211,6 +2211,16 @@ lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] norm_num at hgap exact hgap +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial-gap term used by this formalization is deterministically at most `2` when all arm +means lie in `[-1, 1]`. At horizon zero the initial term is exactly zero. -/ +lemma initialGapTerm_le_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] + (hμ : MeanRewardBound ν (-1) 1) : + (if n = 0 then 0 else gap ν (A 0 ω)) ≤ if n = 0 then 0 else 2 := by + by_cases hn : n = 0 + · simp [hn] + · simpa [hn] using (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) hμ (A 0 ω)) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : @@ -3796,6 +3806,121 @@ lemma regret_ae_imp_le_textbook_finite_action (P := P) L2 hL2) matrixDetLeTraceAveragePow) +/-- The deterministic textbook LinUCB bonus term +`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`. + +The final finite-action theorem keeps this as a named expression so probability statements can use +a deterministic right-hand side instead of repeating the full formula. -/ +noncomputable def textbookRegretBonus (reg : ℝ) (β : ℕ → ℝ) (L2 : ℝ) (n : ℕ) : ℝ := + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) + +/-- Good-event finite-action LinUCB regret theorem with the random initial gap replaced by the +deterministic `≤ 2` bound implied by `MeanRewardBound ν (-1) 1`. -/ +lemma regret_ae_imp_le_textbook_finite_action_deterministic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, + LinUCBConfidenceEvent A R reg β x ν ω → + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2] with ω h_regret h_confω + refine (h_regret h_confω).trans ?_ + simpa [textbookRegretBonus] using + add_le_add_right + (initialGapTerm_le_two_of_meanRewardBound_neg_one_one (A := A) (ν := ν) + (n := n) (ω := ω) h_mean_bound) + (2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))) + +/-- Almost-sure corollary of +`regret_ae_imp_le_textbook_finite_action_deterministic_bound` when the confidence event is known +to hold almost surely. -/ +lemma regret_ae_le_textbook_finite_action_deterministic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by + filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + h_mean_bound hβ_schedule hreg_pos L2 hL2, h_conf] with ω h_regret h_confω + exact h_regret h_confω + +/-- The confidence event is almost surely contained in the deterministic textbook regret-bound +event. This is the version to combine with a future high-probability confidence theorem. -/ +lemma probReal_confidenceEvent_le_textbook_regret_bound_deterministic + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_confω + exact h_regret h_confω + +/-- High-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ +lemma probReal_textbook_regret_bound_deterministic_ge_of_confidenceEvent_ge + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : + 1 - δ ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by + exact h_conf_prob.trans + (probReal_confidenceEvent_le_textbook_regret_bound_deterministic (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2) + +/-- Failure-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ +lemma probReal_textbook_regret_bound_deterministic_failure_le_of_confidenceEvent_failure_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_failure : + P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : + P.real {ω | + ¬ regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} ≤ δ := by + refine le_trans ?_ h_conf_failure + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω + exact h_regret_failure (h_regret h_confω) + /-- The confidence event is almost surely contained in the textbook finite-action regret-bound event. From f63a9cf91803eae52d026f6ca8b021735c5554de Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 25 Jun 2026 14:42:58 -0400 Subject: [PATCH 74/82] feat(LinUCB Algorithm And Process API): formalize the actual finite-action LinUCB algorithm object and its process-level API. --- .../Bandit/Algorithms/LinUCB/Basic.lean | 317 ++++++++++++++++++ 1 file changed, 317 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean new file mode 100644 index 00000000..21912a55 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean @@ -0,0 +1,317 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.Analysis.MeanInequalities +public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.Analysis.Matrix.Order +public import Mathlib.Data.Real.StarOrdered +public import Mathlib.LinearAlgebra.Matrix.PosDef +public import Mathlib.LinearAlgebra.Matrix.SchurComplement +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse +public import Mathlib.Probability.Martingale.OptionalStopping + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +/-- Feature vectors for finite-dimensional linear bandits. -/ +abbrev Feature (d : ℕ) := Fin d → ℝ + +/-- The standard coordinate direction in `Feature d`. -/ +def coordinateDirection (i : Fin d) : Feature d := + fun j ↦ if j = i then 1 else 0 + +/-- Dot product with a coordinate direction extracts that coordinate. -/ +lemma dotProduct_coordinateDirection (u : Feature d) (i : Fin d) : + dotProduct (coordinateDirection i) u = u i := by + simp only [dotProduct, coordinateDirection] + rw [Finset.sum_eq_single i] + · simp + · intro j _hj hji + simp [hji] + · intro hi + simp at hi + +/-- Dot product with the negative coordinate direction extracts the negated coordinate. -/ +lemma dotProduct_neg_coordinateDirection (u : Feature d) (i : Fin d) : + dotProduct (-coordinateDirection i) u = -u i := by + rw [neg_dotProduct, dotProduct_coordinateDirection] + +/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product +`x_aᵀ x_a`. -/ +def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := + dotProduct (x a) (x a) + +/-- The squared feature norm is nonnegative. -/ +lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : + 0 ≤ featureSqNorm x a := by + rw [featureSqNorm, dotProduct] + exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) + +/-- The squared Euclidean norm of an arbitrary feature vector is nonnegative. -/ +lemma dotProduct_self_nonneg (u : Feature d) : + 0 ≤ dotProduct u u := by + rw [dotProduct] + exact sum_nonneg fun i _ ↦ mul_self_nonneg (u i) + +/-- Euclidean Cauchy-Schwarz for the finite-dimensional `Feature d` dot product. -/ +lemma abs_dotProduct_le_sqrt_mul_sqrt (u v : Feature d) : + |dotProduct u v| ≤ √(dotProduct u u) * √(dotProduct v v) := by + have hpos : + dotProduct u v ≤ √(dotProduct u u) * √(dotProduct v v) := by + simpa [dotProduct, pow_two] using + (Real.sum_mul_le_sqrt_mul_sqrt (Finset.univ : Finset (Fin d)) u v) + have hneg : + -dotProduct u v ≤ √(dotProduct u u) * √(dotProduct v v) := by + have h := Real.sum_mul_le_sqrt_mul_sqrt (Finset.univ : Finset (Fin d)) + (fun i : Fin d ↦ -u i) v + simpa [dotProduct, pow_two, Finset.sum_neg_distrib] using h + exact abs_le.mpr ⟨by linarith, hpos⟩ + +/-- Cauchy-Schwarz with external squared-norm bounds. -/ +lemma abs_dotProduct_le_sqrt_mul_sqrt_of_sq_norm_le + (u v : Feature d) {U V : ℝ} + (hu : dotProduct u u ≤ U) (hv : dotProduct v v ≤ V) : + |dotProduct u v| ≤ √U * √V := by + refine (abs_dotProduct_le_sqrt_mul_sqrt u v).trans ?_ + exact mul_le_mul (Real.sqrt_le_sqrt hu) (Real.sqrt_le_sqrt hv) + (Real.sqrt_nonneg _) (Real.sqrt_nonneg _) + +/-- Uniform squared feature-norm bound for finite-action LinUCB. + +This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared +form as `‖x_a‖₂² ≤ L2` for every action. -/ +def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := + ∀ a, featureSqNorm x a ≤ L2 + +/-- A uniform squared feature-norm bound is nonnegative whenever the finite action set is +nonempty. -/ +lemma FeatureSqNormBound.nonneg [Nonempty (Fin K)] + {x : Fin K → Feature d} {L2 : ℝ} (hL2 : FeatureSqNormBound x L2) : + 0 ≤ L2 := by + classical + exact (featureSqNorm_nonneg x (Classical.arbitrary (Fin K))).trans + (hL2 (Classical.arbitrary (Fin K))) + +/-- A squared feature-norm bound controls every coordinate of every feature vector. -/ +lemma abs_feature_coord_le_sqrt_of_featureSqNorm_le + (x : Fin K → Feature d) {L2 : ℝ} {a : Fin K} + (hL2 : featureSqNorm x a ≤ L2) (i : Fin d) : + |x a i| ≤ √L2 := by + have hcoord_sq_le_norm : (x a i) ^ 2 ≤ featureSqNorm x a := by + rw [featureSqNorm, dotProduct] + simpa [pow_two] using + (Finset.single_le_sum + (s := Finset.univ) (a := i) + (fun j _hj ↦ mul_self_nonneg (x a j)) (Finset.mem_univ i)) + exact Real.abs_le_sqrt (hcoord_sq_le_norm.trans hL2) + +/-- A uniform squared feature-norm bound controls the coordinate projection of every feature +vector. -/ +lemma abs_dotProduct_coordinateDirection_feature_le_sqrt + (x : Fin K → Feature d) {L2 : ℝ} (hL2 : FeatureSqNormBound x L2) + (i : Fin d) (a : Fin K) : + |dotProduct (coordinateDirection i) (x a)| ≤ √L2 := by + simpa [dotProduct_coordinateDirection] using + abs_feature_coord_le_sqrt_of_featureSqNorm_le (x := x) (hL2 a) i + +/-- For a fixed direction and finite action set, all arm-feature projections are bounded. -/ +lemma exists_abs_dotProduct_feature_bound (x : Fin K → Feature d) (v : Feature d) : + ∃ Q : ℝ, 0 ≤ Q ∧ ∀ a, |dotProduct v (x a)| ≤ Q := by + refine ⟨∑ a, |dotProduct v (x a)|, ?_, ?_⟩ + · exact sum_nonneg fun a _ha ↦ abs_nonneg _ + · intro a + exact Finset.single_le_sum + (fun b _hb ↦ abs_nonneg (dotProduct v (x b))) (Finset.mem_univ a) + +/-- History-level regularized design matrix for LinUCB. -/ +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +/-- History-level response vector for LinUCB. -/ +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +/-- History-level regularized least-squares estimate. -/ +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +/-- History-level estimated reward of an arm. -/ +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +/-- History-level quadratic form underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) + +/-- History-level elliptical confidence width of an arm. -/ +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(widthQuadraticForm' reg x n h a) + +/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic +form is nonnegative. -/ +lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) + (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : + width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by + simp [width', Real.sq_sqrt h_nonneg] + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +lemma measurable_designMatrix'_apply (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (i j : Fin d) : + Measurable (fun h ↦ designMatrix' reg x n h i j) := by + unfold designMatrix' + change Measurable fun h : Iic n → Fin K × ℝ ↦ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) i j + + (∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1)) i j + refine Measurable.const_add ?_ _ + rw [show (fun h : Iic n → Fin K × ℝ ↦ + (∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1)) i j) = + fun h ↦ ∑ s : Iic n, x (h s).1 i * x (h s).1 j by + funext h + simp [Matrix.sum_apply, Matrix.vecMulVec]] + fun_prop + +@[fun_prop] +lemma measurable_responseVector'_apply (x : Fin K → Feature d) (n : ℕ) (i : Fin d) : + Measurable (fun h ↦ responseVector' x n h i) := by + unfold responseVector' + fun_prop + +lemma measurable_matrix_det_apply {α : Type*} {mα : MeasurableSpace α} + (M : α → Matrix (Fin d) (Fin d) ℝ) + (hM : ∀ i j, Measurable fun a ↦ M a i j) : + Measurable fun a ↦ (M a).det := by + simp_rw [Matrix.det_apply'] + fun_prop + +lemma measurable_matrix_adjugate_apply {α : Type*} {mα : MeasurableSpace α} + (M : α → Matrix (Fin d) (Fin d) ℝ) + (hM : ∀ i j, Measurable fun a ↦ M a i j) (i j : Fin d) : + Measurable fun a ↦ (M a).adjugate i j := by + simp_rw [Matrix.adjugate_apply] + refine measurable_matrix_det_apply (fun a ↦ (M a).updateRow j (Pi.single i 1)) ?_ + intro k l + by_cases hkj : k = j + · subst k + simp [Matrix.updateRow_self] + · simpa [Matrix.updateRow_ne hkj] using hM k l + +lemma measurable_matrix_inv_apply {α : Type*} {mα : MeasurableSpace α} + (M : α → Matrix (Fin d) (Fin d) ℝ) + (hM : ∀ i j, Measurable fun a ↦ M a i j) (i j : Fin d) : + Measurable fun a ↦ (M a)⁻¹ i j := by + simp_rw [Matrix.inv_def] + change Measurable fun a ↦ Ring.inverse (M a).det * (M a).adjugate i j + simpa [Ring.inverse_eq_inv] using + (measurable_matrix_det_apply M hM).inv.mul (measurable_matrix_adjugate_apply M hM i j) + +@[fun_prop] +lemma measurable_thetaHat'_apply (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (i : Fin d) : + Measurable (fun h ↦ thetaHat' reg x n h i) := by + unfold thetaHat' + change Measurable fun h ↦ + ∑ j, (designMatrix' reg x n h)⁻¹ i j * responseVector' x n h j + refine Finset.measurable_sum _ fun j _ ↦ ?_ + exact (measurable_matrix_inv_apply (fun h ↦ designMatrix' reg x n h) + (measurable_designMatrix'_apply reg x n) i j).mul + (measurable_responseVector'_apply x n j) + +@[fun_prop] +lemma measurable_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ estimatedReward' reg x n h a) := by + unfold estimatedReward' + change Measurable fun h ↦ ∑ i, thetaHat' reg x n h i * x a i + fun_prop + +@[fun_prop] +lemma measurable_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ widthQuadraticForm' reg x n h a) := by + unfold widthQuadraticForm' + change Measurable fun h ↦ + ∑ i, x a i * (∑ j, (designMatrix' reg x n h)⁻¹ i j * x a j) + refine Finset.measurable_sum _ fun i _ ↦ ?_ + refine Measurable.const_mul ?_ _ + refine Finset.measurable_sum _ fun j _ ↦ ?_ + exact (measurable_matrix_inv_apply (fun h ↦ designMatrix' reg x n h) + (measurable_designMatrix'_apply reg x n) i j).mul measurable_const + +@[fun_prop] +lemma measurable_width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ width' reg x n h a) := by + unfold width' + fun_prop + +@[fun_prop] +lemma measurable_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ index' reg β x n h a) := by + unfold index' + fun_prop + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (n : ℕ) : + Measurable (nextArm hK reg β x n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ measurable_index' reg β x n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +end Bandits From c62f829164e8680664620bac10c45ae575fbd6d7 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:07:02 -0400 Subject: [PATCH 75/82] feat(112): reorg the linUCB into modular files --- .../Online/Bandit/Algorithms/LinUCB.lean | 4049 +---------------- 1 file changed, 6 insertions(+), 4043 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index da0dc185..f528dfdb 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -5,4052 +5,15 @@ Authors: OpenAI, Fawad Haider -/ module -public import LeanMachineLearning.Online.Bandit.SumRewards -public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax -public import Mathlib.Analysis.MeanInequalities -public import Mathlib.Analysis.SpecialFunctions.Log.Deriv -public import Mathlib.Analysis.Matrix.Order -public import Mathlib.Data.Real.StarOrdered -public import Mathlib.LinearAlgebra.Matrix.PosDef -public import Mathlib.LinearAlgebra.Matrix.SchurComplement -public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Regret /-! # LinUCB for finite-action linear bandits -Chapter 19 of *Bandit Algorithms*: --/ - -@[expose] public section - -open MeasureTheory ProbabilityTheory Filter Real Finset Learning - -open scoped ENNReal NNReal Matrix MatrixOrder - -namespace Bandits - -variable {K d : ℕ} - -section Algorithm - -namespace LinUCB - -/-- Feature vectors for finite-dimensional linear bandits. -/ -abbrev Feature (d : ℕ) := Fin d → ℝ - -/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product -`x_aᵀ x_a`. -/ -def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := - dotProduct (x a) (x a) - -/-- The squared feature norm is nonnegative. -/ -lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : - 0 ≤ featureSqNorm x a := by - rw [featureSqNorm, dotProduct] - exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) -/-- Uniform squared feature-norm bound for finite-action LinUCB. +This module is the public entry point for the finite-action LinUCB development. -This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared -form as `‖x_a‖₂² ≤ L2` for every action. -/ -def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := - ∀ a, featureSqNorm x a ≤ L2 - -/-- History-level regularized design matrix for LinUCB. -/ -noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := - reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) - -/-- History-level response vector for LinUCB. -/ -noncomputable def responseVector' (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - ∑ s : Iic n, (h s).2 • x (h s).1 - -/-- History-level regularized least-squares estimate. -/ -noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) - -/-- History-level estimated reward of an arm. -/ -noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - dotProduct (thetaHat' reg x n h) (x a) - -/-- History-level quadratic form underlying the LinUCB confidence width. -/ -noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) - -/-- History-level elliptical confidence width of an arm. -/ -noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(widthQuadraticForm' reg x n h a) - -/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic -form is nonnegative. -/ -lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : - width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by - simp [width', Real.sq_sqrt h_nonneg] - -/-- LinUCB optimistic index of an arm. - -The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` -contains the observations through time `n`, this index is used to choose the arm -at time `n + 1`, and we evaluate the schedule at `n + 2` +The implementation is split across submodules under +`LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.*`; importing this file re-exports the full +LinUCB API, including the algorithm definition, confidence events, concentration interfaces, +deterministic regret decomposition, elliptical-potential/log-det bounds, and final regret theorems. -/ -noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a - -open Classical in -/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ -noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - measurableArgmax (fun h a ↦ index' reg β x n h a) h - -@[fun_prop] -lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → Feature d) - (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) - (n : ℕ) : - Measurable (nextArm hK reg β x n) := by - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact measurable_measurableArgmax fun a ↦ h_index n a - -end LinUCB - -/-- The finite-action LinUCB algorithm. -/ -noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → LinUCB.Feature d) - (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : - Algorithm (Fin K) ℝ := - detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ - -end Algorithm - -namespace LinUCB - -variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} - {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} - {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] - {Ω : Type*} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} - {n : ℕ} {ω : Ω} - -section AlgorithmBehavior - -/-- The process-level design matrix built from actions up to time `n` excluded. -/ -noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := - reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) - -/-- The initial design matrix before any actions are included. -/ -lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - designMatrix A reg x 0 ω = reg • 1 := by - simp [designMatrix] - -/-- The design matrix update after observing one additional action. -/ -lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designMatrix A reg x (n + 1) ω = - designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by - simp [designMatrix, sum_range_succ, add_assoc] - -/-- With nonnegative regularization, the process-level design matrix is positive semidefinite. -/ -lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : - (designMatrix A reg x n ω).PosSemidef := by - unfold designMatrix - apply Matrix.PosSemidef.add - · exact Matrix.PosSemidef.smul Matrix.PosSemidef.one hreg_nonneg - · refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - -/-- Positive regularization makes the process-level design matrix positive definite. -/ -lemma designMatrix_posDef (hreg_pos : 0 < reg) : - (designMatrix A reg x n ω).PosDef := by - unfold designMatrix - apply Matrix.PosDef.add_posSemidef - · exact Matrix.PosDef.smul Matrix.PosDef.one hreg_pos - · refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The design matrix dominates its regularization part: after subtracting `reg • I`, what remains -is the sum of observed rank-one feature matrices, hence positive semidefinite. -/ -lemma designMatrix_sub_reg_smul_one_posSemidef : - (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by - have hsum : - (∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω))).PosSemidef := by - refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - simpa [designMatrix, add_sub_cancel_left] using hsum - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix-order form of `designMatrix_sub_reg_smul_one_posSemidef`: `reg • I ≤ V_n`. -/ -lemma reg_smul_one_le_designMatrix : - reg • (1 : Matrix (Fin d) (Fin d) ℝ) ≤ designMatrix A reg x n ω := by - rw [Matrix.le_iff] - exact designMatrix_sub_reg_smul_one_posSemidef (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix order is preserved by evaluating the quadratic form against a fixed feature vector. -/ -lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} - (hMN : M ≤ N) (u : Feature d) : - dotProduct u (M *ᵥ u) ≤ dotProduct u (N *ᵥ u) := by - have h_nonneg : 0 ≤ dotProduct u ((N - M) *ᵥ u) := by - simpa using (Matrix.le_iff.mp hMN).dotProduct_mulVec_nonneg u - rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg - exact sub_nonneg.mp h_nonneg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The inverse of the regularized identity is the reciprocal-scaled identity. -/ -lemma reg_smul_one_inv (hreg : reg ≠ 0) : - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ = - reg⁻¹ • (1 : Matrix (Fin d) (Fin d) ℝ) := by - rw [Matrix.inv_eq_left_inv] - simp [smul_smul, hreg] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The quadratic form induced by `(reg • I)⁻¹` is the squared norm divided by `reg`. -/ -lemma dotProduct_reg_smul_one_inv_mulVec (hreg : reg ≠ 0) (u : Feature d) : - dotProduct u (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ u) = - dotProduct u u / reg := by - rw [reg_smul_one_inv (reg := reg) (d := d) hreg] - simp [Matrix.smul_mulVec, div_eq_inv_mul, mul_comm] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Arm-specific form of `dotProduct_reg_smul_one_inv_mulVec`. -/ -lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div - (hreg : reg ≠ 0) (a : Fin K) : - dotProduct (x a) (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) = - featureSqNorm x a / reg := by - simpa [featureSqNorm] using - dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Reusable matrix-analysis theorem needed for the LinUCB width comparison. - -It states the usual inverse anti-monotonicity of positive-definite matrices in the PSD order: -if `M` is positive definite and `M ≤ N`, then inversion reverses the order. -/ -def MatrixInvAntiMonoOnPosDef (d : ℕ) : Prop := - ∀ M N : Matrix (Fin d) (Fin d) ℝ, M.PosDef → M ≤ N → N⁻¹ ≤ M⁻¹ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret -route. - -Mathematically, this should follow from `reg • I ≤ V_t` and positive regularization: inversion -reverses the positive-definite matrix order, so `V_t⁻¹ ≤ (reg • I)⁻¹`. -/ -def DesignMatrixInvLeRegInv - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := - ∀ (n : ℕ) (ω : Ω), - (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection lemma for the named inverse-order obligation. -/ -lemma DesignMatrixInvLeRegInv.apply - (h_inv : DesignMatrixInvLeRegInv A reg x) (n : ℕ) (ω : Ω) : - (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := - h_inv n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The reusable positive-definite inverse anti-monotonicity theorem implies the LinUCB-specific -inverse-design comparison. -/ -lemma DesignMatrixInvLeRegInv.of_matrix_inv_antitone - (hreg_pos : 0 < reg) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) : - DesignMatrixInvLeRegInv A reg x := by - intro n ω - exact h_inv_antitone (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) - (designMatrix A reg x n ω) - (Matrix.PosDef.smul Matrix.PosDef.one hreg_pos) - (reg_smul_one_le_designMatrix (A := A) (reg := reg) (x := x) (n := n) (ω := ω)) - -/-- Trace of the process-level regularized design matrix. -/ -noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - Matrix.trace (designMatrix A reg x n ω) - -/-- Before any observations, the design trace is the trace of `reg • I_d`, namely `reg * d`. -/ -lemma designTrace_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - designTrace A reg x 0 ω = reg * (d : ℝ) := by - simp [designTrace, designMatrix_zero] - -/-- Updating the design matrix by `x_a x_aᵀ` increases the trace by `x_aᵀ x_a`. -/ -lemma designTrace_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designTrace A reg x (n + 1) ω = - designTrace A reg x n ω + featureSqNorm x (A n ω) := by - simp [designTrace, designMatrix_succ, featureSqNorm, Matrix.trace_vecMulVec] - -/-- Closed form for the design trace: initial regularization trace plus accumulated squared -feature norms. -/ -lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designTrace A reg x n ω = - reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by - simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] - -/-- With nonnegative regularization, the design trace is nonnegative. -/ -lemma designTrace_nonneg (hreg_nonneg : 0 ≤ reg) : - 0 ≤ designTrace A reg x n ω := by - rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] - exact add_nonneg - (mul_nonneg hreg_nonneg (Nat.cast_nonneg d)) - (sum_nonneg fun t _ ↦ featureSqNorm_nonneg x (A t ω)) - -/-- If every selected feature vector has squared norm at most `L2`, then the trace of the design -matrix is at most `reg * d + n * L2`. -/ -lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by - rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] - gcongr - calc - (∑ t ∈ range n, featureSqNorm x (A t ω)) ≤ ∑ _t ∈ range n, L2 := by - exact sum_le_sum fun t ht ↦ hL2 t ht - _ = (n : ℝ) * L2 := by - simp [nsmul_eq_mul] - -omit [IsProbabilityMeasure P] in -/-- Almost surely, bounded selected feature norms give the corresponding deterministic trace -budget `reg * d + n * L2`. -/ -lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : - ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by - filter_upwards [hL2] with ω hL2ω - exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) L2 hL2ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A uniform finite-action feature bound implies the selected-action feature bound through any -finite horizon. -/ -lemma featureSqNorm_ae_le_of_featureSqNormBound - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2 := - Filter.Eventually.of_forall fun ω t _ht ↦ hL2 (A t ω) - -/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ -noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - ∑ s ∈ range n, R s ω • x (A s ω) - -/-- The initial response vector before any rewards are included. -/ -lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (ω : Ω) : - responseVector A R x 0 ω = 0 := by - simp [responseVector] - -/-- The response-vector update after observing one additional reward. -/ -lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - responseVector A R x (n + 1) ω = - responseVector A R x n ω + R n ω • x (A n ω) := by - simp [responseVector, sum_range_succ] - -/-- The process-level regularized least-squares estimate. -/ -noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) - -/-- The initial least-squares estimate is zero because no reward-feature observations have been -included yet. -/ -lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - thetaHat A R reg x 0 ω = 0 := by - simp [thetaHat, responseVector_zero] - -/-- The process-level estimated linear reward. -/ -noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - dotProduct (thetaHat A R reg x n ω) (x a) - -/-- The initial estimated reward is zero for every arm. -/ -lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - estimatedReward A R reg x a 0 ω = 0 := by - simp [estimatedReward, thetaHat_zero] - -/-- The quadratic form `x_aᵀ V_n⁻¹ x_a` underlying the LinUCB confidence width. -/ -noncomputable def widthQuadraticForm (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) - -/-- The initial width quadratic form is induced by the inverse regularized identity. -/ -lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - widthQuadraticForm A reg x a 0 ω = - dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by - simp [widthQuadraticForm, designMatrix_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Nonnegative regularization makes every LinUCB width quadratic form nonnegative. - -The reason is purely matrix-theoretic: `V_n` is positive semidefinite, the nonsingular inverse of a -positive semidefinite matrix is positive semidefinite in mathlib, and every quadratic form induced -by a positive semidefinite matrix is nonnegative. -/ -lemma widthQuadraticForm_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) (a : Fin K) : - 0 ≤ widthQuadraticForm A reg x a n ω := by - simpa [widthQuadraticForm] using - ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg).inv.dotProduct_mulVec_nonneg (x a)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, nonnegative regularization gives nonnegative selected quadratic width forms -through any finite horizon. -/ -lemma widthQuadraticForm_ae_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - exact Filter.Eventually.of_forall fun ω t _ht ↦ - widthQuadraticForm_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := t) (ω := ω) hreg_nonneg (A t ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive-time version of `widthQuadraticForm_ae_nonneg_of_reg_nonneg`, matching the side -condition shape used by the regret/width-sum bridge lemmas. -/ -lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - filter_upwards [widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) - (x := x) (n := n) (P := P) hreg_nonneg] with ω h_nonnegω - intro t ht _ht0 - exact h_nonnegω t ht - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The matrix comparison needed to turn bounded feature vectors into the positive-time LinUCB -width cap. - -Mathematically, this says `x_aᵀ V_t⁻¹ x_a ≤ ‖x_a‖² / reg`. A later matrix-order proof should -derive it from `reg > 0` and `V_t = reg I + ∑ x_s x_sᵀ`. Keeping it as a named property makes the -remaining linear-algebra obligation precise and reusable. -/ -def WidthQuadraticFormLeFeatureSqNormDivReg - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := - ∀ (a : Fin K) (n : ℕ) (ω : Ω), - widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If the inverse design matrix is bounded by the inverse regularized identity, then the LinUCB -quadratic width is bounded by `featureSqNorm / reg` for one arm, time, and sample point. -/ -lemma widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (a : Fin K) - (h_inv : (designMatrix A reg x n ω)⁻¹ ≤ - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) - (hreg : reg ≠ 0) : - widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg := by - calc - widthQuadraticForm A reg x a n ω = - dotProduct (x a) (((designMatrix A reg x n ω)⁻¹) *ᵥ (x a)) := rfl - _ ≤ dotProduct (x a) - (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) := - dotProduct_mulVec_le_of_matrix_le h_inv (x a) - _ = featureSqNorm x a / reg := - dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div - (reg := reg) (x := x) hreg a - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A pointwise inverse-order comparison for all times and sample points gives the reusable -`WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ -lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le - (hreg : reg ≠ 0) - (h_inv : DesignMatrixInvLeRegInv A reg x) : - WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by - intro a n ω - exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a - (h_inv.apply n ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the -quadratic form is at most one. -/ -lemma widthQuadraticForm_le_one_of_featureSqNorm_le_reg - (a : Fin K) - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) - (h_feature_le : featureSqNorm x a ≤ reg) : - widthQuadraticForm A reg x a n ω ≤ 1 := by - refine (h_width a n ω).trans ?_ - rwa [div_le_one hreg_pos] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure positive-time width cap from the matrix comparison and an almost-sure -`featureSqNorm ≤ reg` bound along the selected actions. -/ -lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) - (h_feature_le : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by - filter_upwards [h_feature_le] with ω h_feature_leω - intro t ht _ht0 - exact widthQuadraticForm_le_one_of_featureSqNorm_le_reg - (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (A t ω) h_width hreg_pos - (h_feature_leω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure positive-time width cap from the matrix comparison and a selected-feature budget -`featureSqNorm ≤ L2`, when `L2 ≤ reg`. -/ -lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) {L2 : ℝ} - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by - refine widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg - (A := A) (reg := reg) (x := x) (n := n) (P := P) h_width hreg_pos ?_ - filter_upwards [hL2] with ω hL2ω - intro t ht - exact (hL2ω t ht).trans hL2_le_reg - -/-- The process-level elliptical confidence width. -/ -noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(widthQuadraticForm A reg x a n ω) - -/-- The initial width is the quadratic form induced by the inverse regularized identity. -/ -lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - width A reg x a 0 ω = - √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [width, widthQuadraticForm_zero] - -/-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that -quadratic form is nonnegative. -/ -lemma width_sq_eq_quadratic_form (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x a n ω) : - width A reg x a n ω ^ 2 = widthQuadraticForm A reg x a n ω := by - simp [width, Real.sq_sqrt h_nonneg] - -/-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ -noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 - -/-- No positive-time widths are accumulated at horizon zero. -/ -lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - widthSqSum A reg x 0 ω = 0 := by - simp [widthSqSum] - -/-- Advancing the horizon adds the next positive-time squared width term. -/ -lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + - (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by - simp [widthSqSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's squared width. -/ -lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + width A reg x (A n ω) n ω ^ 2 := by - simp [widthSqSum_succ, hn] - -/-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ -noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else widthQuadraticForm A reg x (A t ω) t ω - -/-- No positive-time quadratic width forms are accumulated at horizon zero. -/ -lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - quadraticWidthSum A reg x 0 ω = 0 := by - simp [quadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time quadratic width form. -/ -lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - if n = 0 then 0 else widthQuadraticForm A reg x (A n ω) n ω := by - simp [quadraticWidthSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ -lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by - simp [quadraticWidthSum_succ, hn] - -/-- The accumulated capped quadratic forms corresponding to the positive-time LinUCB widths. -/ -noncomputable def cappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω) - -/-- No positive-time capped quadratic width forms are accumulated at horizon zero. -/ -lemma cappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - cappedQuadraticWidthSum A reg x 0 ω = 0 := by - simp [cappedQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time capped quadratic width form. -/ -lemma cappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - cappedQuadraticWidthSum A reg x (n + 1) ω = - cappedQuadraticWidthSum A reg x n ω + - if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by - simp [cappedQuadraticWidthSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's capped quadratic width form. -/ -lemma cappedQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - cappedQuadraticWidthSum A reg x (n + 1) ω = - cappedQuadraticWidthSum A reg x n ω + min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by - simp [cappedQuadraticWidthSum_succ, hn] - -/-- If every positive-time process-level quadratic width form is at most `1`, then the uncapped -and capped process-level quadratic-width accumulators agree. -/ -lemma quadraticWidthSum_eq_cappedQuadraticWidthSum - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - quadraticWidthSum A reg x n ω = cappedQuadraticWidthSum A reg x n ω := by - rw [quadraticWidthSum, cappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact (min_eq_right (h_le_one t ht ht0)).symm - -/-- If the squared-width and quadratic-form accumulators agree through a positive time and the -next quadratic form is nonnegative, then they still agree after adding the next term. -/ -lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by - rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, - quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hn, h_eq] - rw [width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A n ω) - (n := n) (ω := ω) h_nonneg] - -/-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive -time quadratic form is nonnegative. -/ -lemma widthSqSum_eq_sum_quadratic_form - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by - rw [widthSqSum, quadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0] - rw [if_neg ht0] - exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) - (n := t) (ω := ω) (h_nonneg t ht ht0) - -/-- A quadratic-form sum bound implies the corresponding bound on `widthSqSum`. This is the shape -expected from a later elliptical-potential argument. -/ -lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - exact h_quad_le - -/-- A capped process-level quadratic-form sum bound implies the corresponding bound on -`widthSqSum`, provided the positive-time process-level quadratic forms are nonnegative and at most -`1`. -/ -lemma widthSqSum_le_of_capped_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_capped_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - rw [quadraticWidthSum_eq_cappedQuadraticWidthSum (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_le_one] - exact h_capped_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped process-level quadratic-form sum bound implies the corresponding bound -on `widthSqSum`, provided the positive-time process-level quadratic forms are almost surely -nonnegative and at most `1`. -/ -lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_nonneg, h_le_one, h_capped_le] with - ω h_nonnegω h_le_oneω h_capped_leω - exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Determinant of the process-level LinUCB design matrix. -/ -noncomputable def designDet (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - Matrix.det (designMatrix A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial design determinant is the determinant of the regularized identity. -/ -lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by - simp [designDet, designMatrix_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial design determinant is `reg ^ d`. -/ -lemma designDet_zero_eq_reg_pow (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - designDet A reg x 0 ω = reg ^ d := by - rw [designDet_zero] - simp - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A nonzero regularization parameter gives a nonzero initial design determinant. -/ -lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hreg : reg ≠ 0) : - designDet A reg x 0 ω ≠ 0 := by - rw [designDet_zero_eq_reg_pow] - exact pow_ne_zero d hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization makes every process-level design determinant nonzero. -/ -lemma designDet_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : - designDet A reg x n ω ≠ 0 := by - have hunit : IsUnit (designMatrix A reg x n ω) := - (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_pos).isUnit - have hdet_unit : IsUnit (designMatrix A reg x n ω).det := - (Matrix.isUnit_iff_isUnit_det (A := designMatrix A reg x n ω)).mp hunit - exact hdet_unit.ne_zero - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, positive regularization makes all design determinants in a finite horizon -nonzero. -/ -lemma designDet_ae_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - exact Filter.Eventually.of_forall fun ω t _ht ↦ - designDet_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) (ω := ω) - hreg_pos - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ -noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - designDet A reg x n ω / designDet A reg x 0 ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the determinant ratio is `1` when the initial design determinant is nonzero. -/ -lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - designDetRatio A reg x 0 ω = 1 := by - simp [designDetRatio, hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the determinant ratio is positive when the initial design determinant is -nonzero. -/ -lemma designDetRatio_zero_pos (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - 0 < designDetRatio A reg x 0 ω := by - rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - norm_num - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. - -This is the determinant-ratio target used by the matrix-determinant part of the elliptical -potential lemma. -/ -noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - designDet A reg x (n + 1) ω / designDet A reg x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The scalar determinant appearing in the rank-one determinant update is the quadratic form -`uᵀ M u`. -/ -lemma det_one_add_replicateRow_mul_matrix_mul_replicateCol - (M : Matrix (Fin d) (Fin d) ℝ) (u : Feature d) : - (1 + Matrix.replicateRow Unit u * M * Matrix.replicateCol Unit u).det = - 1 + dotProduct u (Matrix.mulVec M u) := by - have hsum : - (∑ j, (∑ i, u i * M i j) * u j) = - ∑ i, u i * ∑ j, M i j * u j := by - calc - (∑ j, (∑ i, u i * M i j) * u j) - = ∑ j, ∑ i, (u i * M i j) * u j := by - simp [Finset.sum_mul] - _ = ∑ i, ∑ j, (u i * M i j) * u j := by - rw [Finset.sum_comm] - _ = ∑ i, u i * ∑ j, M i j * u j := by - refine Finset.sum_congr rfl ?_ - intro i _ - rw [Finset.mul_sum] - refine Finset.sum_congr rfl ?_ - intro j _ - ring - rw [Matrix.det_unique] - simpa [Matrix.mul_apply, Matrix.replicateRow, Matrix.replicateCol, Matrix.mulVec, - dotProduct] using hsum - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Process-level matrix determinant update for the LinUCB design matrix. - -If `V_n` has nonzero determinant, then the rank-one update -`V_{n+1} = V_n + x_{A_n} x_{A_n}ᵀ` satisfies -`det(V_{n+1}) = det(V_n) * (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ -lemma designDet_succ_eq_mul_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDet A reg x (n + 1) ω = - designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - have hM : IsUnit (designMatrix A reg x n ω).det := by - simpa [designDet] using (isUnit_iff_ne_zero.mpr hdet) - calc - designDet A reg x (n + 1) ω = - (designMatrix A reg x n ω + - Matrix.vecMulVec (x (A n ω)) (x (A n ω))).det := by - simp [designDet, designMatrix_succ] - _ = (designMatrix A reg x n ω + - Matrix.replicateCol Unit (x (A n ω)) * Matrix.replicateRow Unit (x (A n ω))).det := by - rw [Matrix.vecMulVec_eq Unit] - _ = (designMatrix A reg x n ω).det * - (1 + Matrix.replicateRow Unit (x (A n ω)) * - (designMatrix A reg x n ω)⁻¹ * Matrix.replicateCol Unit (x (A n ω))).det := by - exact Matrix.det_add_replicateCol_mul_replicateRow (A := designMatrix A reg x n ω) - (ι := Unit) hM (x (A n ω)) (x (A n ω)) - _ = designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - rw [designDet] - congr 1 - exact det_one_add_replicateRow_mul_matrix_mul_replicateCol - (M := (designMatrix A reg x n ω)⁻¹) (u := x (A n ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `det(V_n)` is nonzero and the selected quadratic form is nonnegative, then -`det(V_{n+1})` is nonzero. -/ -lemma designDet_succ_ne_zero_of_widthQuadraticForm_nonneg - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - designDet A reg x (n + 1) ω ≠ 0 := by - rw [designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - exact mul_ne_zero hdet (ne_of_gt (by linarith)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms preserve -nonzero design determinants up to any fixed time. -/ -lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt - (m : ℕ) (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t < m → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - designDet A reg x m ω ≠ 0 := by - induction m with - | zero => exact hdet0 - | succ m ih => - exact designDet_succ_ne_zero_of_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := m) (ω := ω) - (ih fun t ht ↦ h_nonneg t (Nat.lt_trans ht (Nat.lt_succ_self m))) - (h_nonneg m (Nat.lt_succ_self m)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms imply that -all design determinants through horizon `n` are nonzero. -/ -lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by - intro t ht - exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := t) (ω := ω) hdet0 fun s hs ↦ - h_nonneg s (mem_range.mpr (Nat.lt_of_lt_of_le hs (Nat.le_of_lt_succ (mem_range.mp ht)))) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant and nonnegative selected quadratic forms imply -that all design determinants through horizon `n` are nonzero. -/ -lemma designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω - exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `det(V_n) ≠ 0`, then the one-step determinant ratio is -`1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n}`. -/ -lemma designDetStepRatio_eq_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDetStepRatio A reg x n ω = - 1 + widthQuadraticForm A reg x (A n ω) n ω := by - simp [designDetStepRatio, - designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet, hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The cumulative determinant ratio advances by multiplying by the one-step determinant ratio. -/ -lemma designDetRatio_succ_eq_mul_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDetRatio A reg x (n + 1) ω = - designDetRatio A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - rw [designDetRatio, designDetRatio, - designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms make the -cumulative determinant ratio positive. -/ -lemma designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - 0 < designDetRatio A reg x n ω := by - induction n with - | zero => - exact designDetRatio_zero_pos (A := A) (reg := reg) (x := x) (ω := ω) hdet0 - | succ n ih => - have hdetn : designDet A reg x n ω ≠ 0 := - designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ - h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) - rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetn] - exact mul_pos - (ih fun t ht ↦ h_nonneg t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) - (by linarith [h_nonneg n (by simp)]) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, starting from a nonzero initial determinant, nonnegative selected quadratic -forms make the cumulative determinant ratio positive. -/ -lemma designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by - filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω - exact designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter and nonnegative selected quadratic forms make -the cumulative determinant ratio positive. -/ -lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by - refine designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, the cumulative determinant ratio is the finite -product of the per-round determinant-update factors. -/ -lemma designDetRatio_eq_prod_one_add_widthQuadraticForm - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - designDetRatio A reg x n ω = - ∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω) := by - induction n with - | zero => - rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet0] - simp - | succ n ih => - have hdetn : designDet A reg x n ω ≠ 0 := - designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ - h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) - rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetn] - rw [ih fun t ht ↦ h_nonneg t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))] - simp [Finset.prod_range_succ] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If every selected quadratic form is in `[0, 1]`, the cumulative determinant ratio is at most -`2 ^ n`. -/ -lemma designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - rw [designDetRatio_eq_prod_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0 h_nonneg] - calc - (∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω)) - ≤ ∏ _t ∈ range n, (2 : ℝ) := by - exact Finset.prod_le_prod - (fun t ht ↦ by linarith [h_nonneg t ht]) - (fun t ht ↦ by linarith [h_le_one t ht]) - _ = (2 : ℝ) ^ n := by - simp - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, if every selected quadratic form is in `[0, 1]`, the cumulative determinant -ratio is at most `2 ^ n`. -/ -lemma designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - filter_upwards [hdet0, h_nonneg, h_le_one] with ω hdet0ω h_nonnegω h_le_oneω - exact designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one (A := A) - (reg := reg) (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω h_le_oneω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter and selected quadratic forms in `[0, 1]` -imply the cumulative determinant ratio is at most `2 ^ n`. -/ -lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - refine designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one - (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Converts an almost-sure trace bound into the determinant-ratio bound expected from a future -trace/determinant comparison theorem. -/ -lemma designDetRatio_ae_le_trace_budget_of_designTrace_ae_le - (T : ℝ) - (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ T → - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - filter_upwards [h_trace_le] with ω h_traceω - exact h_ratio_of_trace ω h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms give the concrete trace budget -`reg * d + n * L2`; a future trace/determinant comparison then gives the corresponding -determinant-ratio bound. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - exact designDetRatio_ae_le_trace_budget_of_designTrace_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) (T := reg * (d : ℝ) + (n : ℝ) * L2) - (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2) - h_ratio_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A determinant upper bound for `V_n` implies the corresponding determinant-ratio bound, using -`det(V_0) = reg ^ d`. -/ -lemma designDetRatio_le_trace_budget_of_designDet_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hdet_le : designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - rw [designDetRatio, designDet_zero_eq_reg_pow] - have hreg_pow_nonneg : 0 ≤ reg ^ d := (pow_pos hreg_pos d).le - have hdiv : designDet A reg x n ω / reg ^ d ≤ (T / (d : ℝ)) ^ d / reg ^ d := by - exact div_le_div_of_nonneg_right hdet_le hreg_pow_nonneg - refine hdiv.trans_eq ?_ - rw [← div_pow] - congr 1 - field_simp [hreg_pos.ne', by exact_mod_cast hd] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a determinant upper bound for `V_n` implies the corresponding determinant-ratio -bound. -/ -lemma designDetRatio_ae_le_trace_budget_of_designDet_ae_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hdet_le : ∀ᵐ ω ∂P, designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - filter_upwards [hdet_le] with ω hdetω - exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) T hreg_pos hd hdetω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Converts an almost-sure trace bound plus a future determinant/trace comparison for `det(V_n)` -into the determinant-ratio bound used by the elliptical-potential chain. -/ -lemma designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ T → designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - refine designDetRatio_ae_le_trace_budget_of_designDet_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) T hreg_pos hd ?_ - filter_upwards [h_trace_le] with ω h_traceω - exact hdet_of_trace ω h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms reduce the determinant-ratio goal to the determinant upper bound -`det(V_n) ≤ ((reg * d + n * L2) / d) ^ d`. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le - (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - exact designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd - (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2) - hdet_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix-level determinant/trace comparison needed for the finite-dimensional -elliptical-potential bound. - -For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality -`det(M) ≤ (trace(M) / d) ^ d`. -/ -def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := - ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar AM-GM in the form used for PSD matrix eigenvalues: -the product of nonnegative entries is bounded by the arithmetic mean to the `card` power. -/ -lemma prod_le_average_pow_of_nonneg {ι : Type*} [Fintype ι] [Nonempty ι] - (z : ι → ℝ) (hz : ∀ i, 0 ≤ z i) : - (∏ i, z i) ≤ ((∑ i, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by - classical - have hN_pos : 0 < (Fintype.card ι : ℝ) := by - exact_mod_cast Fintype.card_pos_iff.mpr inferInstance - have hweights_pos : 0 < ∑ i : ι, (1 : ℝ) := by - simpa using hN_pos - have h_amgm := Real.geom_mean_le_arith_mean (s := Finset.univ) - (w := fun _ : ι ↦ (1 : ℝ)) (z := z) - (by intro i hi; norm_num) hweights_pos (by intro i hi; exact hz i) - have h_amgm' : - (∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹) ≤ - (∑ i : ι, z i) / (Fintype.card ι : ℝ) := by - simpa using h_amgm - have hprod_nonneg : 0 ≤ ∏ i : ι, z i := by - exact Finset.prod_nonneg fun i _ ↦ hz i - have hraise := Real.rpow_le_rpow (Real.rpow_nonneg hprod_nonneg _) h_amgm' hN_pos.le - have hleft : - ((∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹)) ^ (Fintype.card ι : ℝ) = - ∏ i : ι, z i := by - rw [← Real.rpow_mul hprod_nonneg] - rw [inv_mul_cancel₀ hN_pos.ne'] - simp - have hright : - ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ (Fintype.card ι : ℝ) = - ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by - rw [Real.rpow_natCast] - simpa [hleft, hright] using hraise - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- PSD matrix determinant/trace comparison from AM-GM over eigenvalues: -`det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma matrixDetLeTraceAveragePow : MatrixDetLeTraceAveragePow d := by - intro M hM - by_cases hd : d = 0 - · subst d - simp - · haveI : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp (Nat.pos_of_ne_zero hd) - rw [hM.1.det_eq_prod_eigenvalues, hM.1.trace_eq_sum_eigenvalues] - simpa using prod_le_average_pow_of_nonneg - (z := fun i : Fin d ↦ hM.1.eigenvalues i) - (fun i ↦ hM.eigenvalues_nonneg i) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the -design matrix is positive semidefinite. -/ -lemma designDet_le_trace_average_pow_of_matrix_det_trace_bound - (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) : - designDet A reg x n ω ≤ (designTrace A reg x n ω / (d : ℝ)) ^ d := by - simpa [designDet, designTrace] using - hdet_trace (designMatrix A reg x n ω) - (designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Combining `det(M) ≤ (trace(M)/d)^d` with a trace budget gives the determinant upper bound -`det(V_n) ≤ (T/d)^d`. -/ -lemma designDet_le_trace_budget_of_matrix_det_trace_bound - (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) - (hd : d ≠ 0) (T : ℝ) (h_trace_le : designTrace A reg x n ω ≤ T) : - designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d := by - have hd_pos : 0 < (d : ℝ) := by - exact_mod_cast Nat.pos_of_ne_zero hd - have hbase_nonneg : 0 ≤ designTrace A reg x n ω / (d : ℝ) := - div_nonneg (designTrace_nonneg (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg) hd_pos.le - have hbase_le : designTrace A reg x n ω / (d : ℝ) ≤ T / (d : ℝ) := - (div_le_div_iff_of_pos_right hd_pos).mpr h_trace_le - exact (designDet_le_trace_average_pow_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet_trace hreg_nonneg).trans - (pow_le_pow_left₀ hbase_nonneg hbase_le d) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms and the matrix-level determinant/trace comparison give the -determinant-ratio bound used by the elliptical-potential chain. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound - (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - refine designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 ?_ - intro ω h_traceω - exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) - hdet_trace hreg_pos.le hd h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The log-determinant expression that appears in the elliptical-potential lemma. -/ -noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - 2 * Real.log (designDetRatio A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A positive determinant ratio bounded by `D` gives the corresponding log-determinant potential -bound. -/ -lemma ellipticalPotential_le_two_mul_log_of_designDetRatio_le {D : ℝ} - (h_ratio_pos : 0 < designDetRatio A reg x n ω) - (h_ratio_le : designDetRatio A reg x n ω ≤ D) : - ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by - rw [ellipticalPotential] - exact mul_le_mul_of_nonneg_left (Real.log_le_log h_ratio_pos h_ratio_le) (by norm_num) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a positive determinant ratio bounded by `D` gives the corresponding -log-determinant potential bound. -/ -lemma ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le {D : ℝ} - (h_ratio_pos : ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by - filter_upwards [h_ratio_pos, h_ratio_le] with ω h_ratio_posω h_ratio_leω - exact ellipticalPotential_le_two_mul_log_of_designDetRatio_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_ratio_posω h_ratio_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. - -The future determinant-update proof should naturally establish the capped quadratic-width term is -bounded by this quantity. A separate log/telescoping bridge then connects this one-step quantity to -`ellipticalPotentialIncrement`. -/ -noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - 2 * Real.log (designDetStepRatio A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing, the one-step log-determinant potential is -`2 * log (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ -lemma ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - ellipticalPotentialStep A reg x n ω = - 2 * Real.log (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - simp [ellipticalPotentialStep, - designDetStepRatio_eq_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar log inequality used in the elliptical-potential proof: for `0 ≤ q ≤ 1`, -`min 1 q ≤ 2 * log (1 + q)`. -/ -lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} - (hq_nonneg : 0 ≤ q) (hq_le_one : q ≤ 1) : - min 1 q ≤ 2 * Real.log (1 + q) := by - have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := - Real.le_log_one_add_of_nonneg hq_nonneg - have hq_add_two_pos : 0 < q + 2 := by linarith - have hq_le_two : q ≤ 2 := by linarith - have hq_le_log_lower : q ≤ 2 * (2 * q / (q + 2)) := by - rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] - rw [le_div_iff₀ hq_add_two_pos] - nlinarith - rw [min_eq_right hq_le_one] - exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar log inequality used in the textbook elliptical-potential proof: for `0 ≤ q`, -`min 1 q ≤ 2 * log (1 + q)`. -/ -lemma min_one_le_two_mul_log_one_add_of_nonneg {q : ℝ} - (hq_nonneg : 0 ≤ q) : - min 1 q ≤ 2 * Real.log (1 + q) := by - by_cases hq_le_one : q ≤ 1 - · exact min_one_le_two_mul_log_one_add_of_nonneg_le_one hq_nonneg hq_le_one - · have hq_one : 1 ≤ q := by linarith - have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := - Real.le_log_one_add_of_nonneg hq_nonneg - have hq_add_two_pos : 0 < q + 2 := by linarith - have hone_le_log_lower : 1 ≤ 2 * (2 * q / (q + 2)) := by - rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] - rw [le_div_iff₀ hq_add_two_pos] - nlinarith - rw [min_eq_left hq_one] - exact hone_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the -single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ -lemma cappedWidthTerm_le_ellipticalPotentialStep - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) - (h_le_one : n ≠ 0 → widthQuadraticForm A reg x (A n ω) n ω ≤ 1) : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialStep A reg x n ω := by - by_cases hn : n = 0 - · rw [if_pos hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) - · rw [if_neg hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing and nonnegativity of the selected quadratic form, the single -capped quadratic-width term is bounded by the one-step log-determinant potential. This is the -textbook form; no separate `q ≤ 1` assumption is needed because the term is already capped. -/ -lemma cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialStep A reg x n ω := by - by_cases hn : n = 0 - · rw [if_pos hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) - · rw [if_neg hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact min_one_le_two_mul_log_one_add_of_nonneg h_nonneg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply -the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ -lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω := by - filter_upwards [hdet, h_nonneg, h_le_one] with ω hdetω h_nonnegω h_le_oneω - intro t ht - exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) - (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the -per-step one-step-potential bound for the capped quadratic-width term. -/ -lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω := by - filter_upwards [hdet, h_nonneg] with ω hdetω h_nonnegω - intro t ht - exact cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg (A := A) (reg := reg) - (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the log-determinant potential is zero when the initial design determinant is -nonzero. -/ -lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - ellipticalPotential A reg x 0 ω = 0 := by - simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the log-determinant elliptical-potential inequality. At horizon zero there are no -positive-time capped quadratic width forms, and the log-determinant potential is zero when the -initial design determinant is nonzero. -/ -lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) - (hdet : designDet A reg x 0 ω ≠ 0) : - cappedQuadraticWidthSum A reg x 0 ω ≤ ellipticalPotential A reg x 0 ω := by - rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), - ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step increment of the log-determinant elliptical potential. -/ -noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The one-step determinant-ratio potential equals the increment of the cumulative -log-determinant potential, provided the relevant design determinants are nonzero. -/ -lemma ellipticalPotentialStep_eq_increment - (hdet0 : designDet A reg x 0 ω ≠ 0) - (hdetn : designDet A reg x n ω ≠ 0) - (hdet_succ : designDet A reg x (n + 1) ω ≠ 0) : - ellipticalPotentialStep A reg x n ω = ellipticalPotentialIncrement A reg x n ω := by - simp [ellipticalPotentialStep, designDetStepRatio, ellipticalPotentialIncrement, - ellipticalPotential, designDetRatio, Real.log_div hdet_succ hdetn, - Real.log_div hdet_succ hdet0, Real.log_div hdetn hdet0] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the one-step determinant-ratio potential equals the increment of the cumulative -log-determinant potential throughout the finite horizon, provided all determinants up to that -horizon are nonzero almost surely. -/ -lemma ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω = ellipticalPotentialIncrement A reg x t ω := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact ellipticalPotentialStep_eq_increment (A := A) (reg := reg) (x := x) (n := t) - (ω := ω) (hdetω 0 (by simp)) - (hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) - (hdetω (t + 1) (mem_range.mpr (Nat.succ_lt_succ (mem_range.mp ht)))) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If the next capped quadratic width term is bounded by the next log-determinant potential -increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ -lemma cappedQuadraticWidthSum_succ_le_ellipticalPotential - (h_prev : cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_step : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialIncrement A reg x n ω) : - cappedQuadraticWidthSum A reg x (n + 1) ω ≤ ellipticalPotential A reg x (n + 1) ω := by - rw [cappedQuadraticWidthSum_succ (A := A) (reg := reg) (x := x) (n := n) (ω := ω)] - calc - cappedQuadraticWidthSum A reg x n ω + - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) - ≤ ellipticalPotential A reg x n ω + ellipticalPotentialIncrement A reg x n ω := by - exact add_le_add h_prev h_step - _ = ellipticalPotential A reg x (n + 1) ω := by - simp [ellipticalPotentialIncrement] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A per-step bound by log-determinant potential increments implies the cumulative -elliptical-potential inequality. This is the induction shell for the future determinant-update -proof. -/ -lemma cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le - (hdet : designDet A reg x 0 ω ≠ 0) : - (∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) → - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - induction n with - | zero => - intro _ - exact cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) - (x := x) (ω := ω) hdet - | succ n ih => - intro h_step - refine cappedQuadraticWidthSum_succ_le_ellipticalPotential (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) ?_ ?_ - · exact ih fun t ht ↦ h_step t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - · exact h_step n (by simp) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a per-step bound by log-determinant potential increments implies the cumulative -elliptical-potential inequality. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - filter_upwards [hdet, h_step] with ω hdetω h_stepω - exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetω h_stepω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the -cumulative capped-sum/log-det inequality, provided the one-step determinant-ratio potential is -bounded by the corresponding cumulative-potential increment. - -This separates the future elliptical-potential proof into two local obligations: - -* a matrix-determinant update bounding the selected arm's capped quadratic form by - `ellipticalPotentialStep`; -* a log/telescoping bridge from `ellipticalPotentialStep` to `ellipticalPotentialIncrement`. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet ?_ - filter_upwards [h_step, h_step_le_increment] with ω h_stepω h_step_le_incrementω - intro t ht - exact (h_stepω t ht).trans (h_step_le_incrementω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the -cumulative capped-sum/log-det inequality when all design determinants up to the horizon are nonzero -almost surely. - -Compared with `cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le`, this -version discharges the log/telescoping bridge automatically from determinant nonvanishing. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - have hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - exact hdetω 0 (by simp) - refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_step ?_ - filter_upwards [ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet] with ω h_eq - intro t ht - rw [h_eq t ht] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the -capped-sum/log-determinant elliptical-potential bound. - -This is the capped form used in the textbook proof of LinUCB: the quadratic forms do not need to -be bounded by `1`, because the accumulated quantity is `min 1 q_t`. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet - (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet_range_n h_nonneg) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization discharges determinant nonvanishing and nonnegativity, yielding the -capped-sum/log-determinant elliptical-potential bound directly. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos - (hreg_pos : 0 < reg) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) - (n := n + 1) (P := P) hreg_pos) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The process-level capped quadratic-width input expected from an elliptical-potential argument. - -It packages the three facts needed to turn a capped process-level quadratic-width estimate into the -`widthSqSum` estimate used by the regret chain: - -* each positive-time process-level quadratic width form is nonnegative; -* each positive-time process-level quadratic width form is at most `1`; -* their capped process-level accumulated sum is bounded by `W`. -/ -def CappedQuadraticWidthBound (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := - (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ∧ - (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ∧ - cappedQuadraticWidthSum A reg x n ω ≤ W - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Build the packaged process-level capped quadratic-width input from its component facts. -/ -lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_sum_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : - CappedQuadraticWidthBound A reg x n ω W := by - exact ⟨h_nonneg, h_le_one, h_sum_le⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the packaged process-level capped quadratic-width input. At horizon zero, the -nonnegativity and `≤ 1` side conditions are vacuous, and the capped sum is zero. -/ -lemma cappedQuadraticWidthBound_zero {W : ℝ} (hW : 0 ≤ W) : - CappedQuadraticWidthBound A reg x 0 ω W := by - refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ - · intro t ht _ - simp at ht - · intro t ht _ - simp at ht - · simpa [cappedQuadraticWidthSum_zero] using hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the packaged process-level capped quadratic-width input when the constant bound -is supplied through the log-determinant potential. -/ -lemma cappedQuadraticWidthBound_zero_of_ellipticalPotential_le_bound {W : ℝ} - (hdet : designDet A reg x 0 ω ≠ 0) (h_potential_le : ellipticalPotential A reg x 0 ω ≤ W) : - CappedQuadraticWidthBound A reg x 0 ω W := by - refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ - · intro t ht _ - simp at ht - · intro t ht _ - simp at ht - · exact (cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) - (x := x) (ω := ω) hdet).trans h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ -lemma cappedQuadraticWidthBound_mono {W W' : ℝ} - (h_bound : CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : - CappedQuadraticWidthBound A reg x n ω W' := by - exact ⟨h_bound.1, h_bound.2.1, h_bound.2.2.trans hW⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, build the packaged process-level capped quadratic-width input from its component -facts. -/ -lemma cappedQuadraticWidthBound_ae_of_nonneg_le_one_and_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_sum_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_sum_le] with - ω h_nonnegω h_le_oneω h_sum_leω - exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_sum_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged process-level capped quadratic-width input is monotone in the -numeric bound. -/ -lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W' := by - filter_upwards [h_bound] with ω h_boundω - exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) h_boundω hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped-sum bound by the log-determinant potential, together with a constant bound on that -potential, gives the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_of_ellipticalPotential_le_bound {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_potential_le : ellipticalPotential A reg x n ω ≤ W) : - CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonneg h_le_one (h_elliptical.trans h_potential_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped-sum bound by the log-determinant potential and an almost-sure constant -bound on that potential give the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_elliptical, h_potential_le] with - ω h_nonnegω h_le_oneω h_ellipticalω h_potential_leω - exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by log-determinant potential increments and a final constant -bound on the potential give the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet h_step) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, one-step determinant-ratio potential bounds, their bridge to cumulative -potential increments, and a final constant bound on the potential give the packaged process-level -capped quadratic-width input. - -This is the packaged form of the determinant-update interface: once the true matrix determinant -lemma proves the `h_step` assumption and the log/telescoping algebra proves -`h_step_le_increment`, the existing regret chain can consume the resulting bound. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, one-step determinant-ratio potential bounds, determinant nonvanishing up to the -horizon, and a final constant bound on the potential give the packaged process-level capped -quadratic-width input. - -This is the determinant-nonvanishing version of the one-step interface: the remaining hard -elliptical-potential work is to prove the one-step matrix inequality and the final -log-determinant bound. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero - {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet h_step) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the determinant-update step, determinant nonvanishing up to the horizon, and a -final constant bound on the log-determinant potential give the packaged capped quadratic-width -input used by the regret chain. - -The assumptions now match the concrete obligations left for a full elliptical-potential proof: - -* prove all relevant design determinants are nonzero; -* prove selected quadratic forms are nonnegative and at most `1` at positive times; -* prove the final log-determinant potential is at most `W`. -/ -lemma cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound {W : ℝ} - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - have h_nonneg_positive : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - filter_upwards [h_nonneg] with ω h_nonnegω - intro t ht _ - exact h_nonnegω t ht - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) - h_nonneg_positive h_le_one hdet - (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero (A := A) (reg := reg) - (x := x) (n := n) (P := P) hdet_range_n h_nonneg h_le_one) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant, the determinant-update step, and a final constant -bound on the log-determinant potential give the packaged capped quadratic-width input used by the -regret chain. - -This removes the need to assume determinant nonvanishing at every time: it is derived inductively -from `det(V_0) ≠ 0` and nonnegative selected quadratic forms. -/ -lemma cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound {W : ℝ} - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) - h_nonneg h_le_one h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter, the determinant-update step, and a final -constant bound on the log-determinant potential give the packaged capped quadratic-width input used -by the regret chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential_le_bound {W : ℝ} - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - refine cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) ?_ h_nonneg h_le_one - h_potential_le - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization discharges the determinant-nonvanishing and quadratic-form -nonnegativity obligations in the log-determinant elliptical-potential chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound {W : ℝ} - (hreg_pos : 0 < reg) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) - (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) - (n := n + 1) (P := P) hreg_pos) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - h_le_one h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a -determinant-ratio upper bound, and the determinant-update step give the packaged capped -quadratic-width input used by the regret chain. - -This version accepts the determinant-ratio bound directly and converts it into the -`ellipticalPotential ≤ 2 * log D` bound internally. -/ -lemma cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound {D : ℝ} - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by - exact cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := 2 * Real.log D) - hdet0 h_nonneg h_le_one - (ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) - h_ratio_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter, nonnegative selected quadratic forms, a -determinant-ratio upper bound, and the determinant-update step give the packaged capped -quadratic-width input used by the regret chain. - -This is the most direct interface for the final determinant-bound part of the finite-action -elliptical-potential argument: after proving `designDetRatio ≤ D`, the theorem supplies the -`CappedQuadraticWidthBound` with bound `2 * log D`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound {D : ℝ} - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by - refine cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one h_ratio_le - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A simple explicit determinant-ratio bound for the capped quadratic-width input. - -If `reg ≠ 0` and every selected quadratic form is almost surely in `[0, 1]`, then the determinant -ratio is at most `2 ^ n`, so the existing determinant-update/elliptical-potential chain gives the -packaged capped-width bound with budget `2 * log (2 ^ n)`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω (2 * Real.log ((2 : ℝ) ^ n)) := by - refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (2 : ℝ) ^ n) - hreg h_nonneg ?_ ?_ - · filter_upwards [h_le_one] with ω h_le_oneω - exact fun t ht _ ↦ h_le_oneω t ht - · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Trace-budget interface for the determinant part of the finite-action elliptical-potential -argument. - -The future spectral/AM-GM determinant theorem should prove the hypothesis -`designDetRatio ≤ (T / (reg * d)) ^ d`, where `T` is an upper bound on `trace(V_n)`. This theorem -then feeds that determinant-ratio bound into the already-proved determinant-update and -elliptical-potential chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (T : ℝ) - (h_ratio_le : ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * Real.log ((T / (reg * (d : ℝ))) ^ d)) := by - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface for the determinant part of the finite-action -elliptical-potential argument. - -If selected feature vectors have squared norm at most `L2`, then `trace(V_n) ≤ reg * d + n * L2`. -Given a future deterministic trace/determinant comparison that turns this trace budget into the -determinant-ratio bound, this theorem supplies the packaged capped-width input with the explicit -budget `2 * log (((reg * d + n * L2) / (reg * d)) ^ d)`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d)) := by - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg h_nonneg h_le_one - (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The explicit feature-norm determinant budget can be rewritten in the common -`d * log(1 + n L² / (reg d))` form. -/ -lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add - (L2 : ℝ) (hden : reg * (d : ℝ) ≠ 0) : - 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) = - 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by - have hbase : - (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) = - 1 + (n : ℝ) * L2 / (reg * (d : ℝ)) := by - exact same_add_div hden - rw [Real.log_pow, hbase] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Textbook capped elliptical-potential budget from bounded selected feature norms and the -matrix-level determinant/trace comparison. - -Unlike `cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound`, this theorem bounds the capped -quadratic-width sum directly and does not assume the individual quadratic forms are at most `1`. -/ -lemma cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ - 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by - have hden : reg * (d : ℝ) ≠ 0 := by - exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) - rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] - have h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := - widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le - have h_potential_le : ∀ᵐ ω ∂P, - ellipticalPotential A reg x n ω ≤ - 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) := by - exact ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' h_nonneg) - (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 - hdet_trace) - filter_upwards [cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos, h_potential_le] with - ω h_capped_le h_potentialω - exact h_capped_le.trans h_potentialω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface with the log term rewritten in the standard -`2 * d * log(1 + n L² / (reg d))` shape. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (hreg : reg ≠ 0) (hd : d ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - have hden : reg * (d : ℝ) ≠ 0 := by - exact mul_ne_zero hreg (by exact_mod_cast hd) - rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 - h_ratio_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface with the determinant/trace comparison stated as a determinant -upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) - (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' - (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) - (x := x) hreg_pos h_inv_antitone)) - hreg_pos hL2 hL2_le_reg) - L2 hL2 ?_ - intro ω h_traceω - exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd - (hdet_of_trace ω h_traceω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface where the remaining matrix-analysis input is the reusable -positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - refine - cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg ?_ - intro ω h_traceω - exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd - (reg * (d : ℝ) + (n : ℝ) * L2) h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed -by the regret chain. -/ -lemma widthSqSum_le_of_capped_quadratic_width_bound {W : ℝ} - (h_bound : CappedQuadraticWidthBound A reg x n ω W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_bound.1 h_bound.2.1 h_bound.2.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged process-level capped quadratic-width input implies the `widthSqSum` -bound consumed by the regret chain. -/ -lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_bound] with ω h_boundω - exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) (W := W) h_boundω - -/-- The process-level LinUCB optimistic index. -/ -noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) - (n : ℕ) (ω : Ω) : ℝ := - estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω - -/-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is -zero. -/ -lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by - simp [index, estimatedReward_zero] - -/-- At time zero, the LinUCB index is the confidence schedule times the initial quadratic-form -width. -/ -lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - index A R reg β x a 0 ω = - √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [index_zero, width_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every least-squares reward estimate is zero. -/ -lemma estimatedReward_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - estimatedReward A R reg x a n ω = 0 := by - subst d - simp [estimatedReward, dotProduct] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB quadratic width form is zero. -/ -lemma widthQuadraticForm_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - widthQuadraticForm A reg x a n ω = 0 := by - subst d - simp [widthQuadraticForm, dotProduct] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB width is zero. -/ -lemma width_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - width A reg x a n ω = 0 := by - simp [width, widthQuadraticForm_eq_zero_of_dim_eq_zero (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hd a] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB index is zero. -/ -lemma index_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - index A R reg β x a n ω = 0 := by - simp [index, estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) hd a, - width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hd a] - -/-- The pointwise LinUCB confidence event used by the finite-action regret proof. - -For every positive process time, the best arm's true mean lies below its optimistic index, and the -selected arm's pessimistic index lies below its true mean. On this event, the max-index property of -LinUCB turns optimism into an instantaneous regret bound. -/ -def LinUCBConfidenceEvent [Nonempty (Fin K)] - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] - -omit [IsMarkovKernel ν] in -/-- Uniform bound on arm gaps, used as the finite-action analogue of the textbook bounded -instantaneous-regret assumption. -/ -def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := - ∀ a, gap ν a ≤ G - -omit [IsMarkovKernel ν] in -/-- Uniform bound on arm means. For finite-action linear bandits this is a convenient way to state -the usual bounded expected-reward assumption, for example `(ν a)[id] ∈ [-1, 1]`. -/ -def MeanRewardBound (ν : Kernel (Fin K) ℝ) (lo hi : ℝ) : Prop := - ∀ a, lo ≤ (ν a)[id] ∧ (ν a)[id] ≤ hi - -omit [IsMarkovKernel ν] in -/-- If every arm mean lies in `[lo, hi]`, then every arm gap is at most `hi - lo`. -/ -lemma gap_le_of_meanRewardBound [Nonempty (Fin K)] {lo hi : ℝ} - (hμ : MeanRewardBound ν lo hi) (a : Fin K) : - gap ν a ≤ hi - lo := by - rw [gap_eq_bestArm_sub] - have hbest_le : (ν (bestArm ν))[id] ≤ hi := (hμ (bestArm ν)).2 - have ha_ge : lo ≤ (ν a)[id] := (hμ a).1 - linarith - -omit [IsMarkovKernel ν] in -/-- Arm means in `[-1, 1]` imply the gap cap `gap ≤ 2` used by the capped regret argument. -/ -lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] - (hμ : MeanRewardBound ν (-1) 1) : - GapBound (K := K) ν 2 := by - intro a - have hgap := gap_le_of_meanRewardBound (ν := ν) (lo := -1) (hi := 1) hμ a - norm_num at hgap - exact hgap - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial-gap term used by this formalization is deterministically at most `2` when all arm -means lie in `[-1, 1]`. At horizon zero the initial term is exactly zero. -/ -lemma initialGapTerm_le_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] - (hμ : MeanRewardBound ν (-1) 1) : - (if n = 0 then 0 else gap ν (A 0 ω)) ≤ if n = 0 then 0 else 2 := by - by_cases hn : n = 0 - · simp [hn] - · simpa [hn] using (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) hμ (A 0 ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ -lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → gap ν (A t ω) ≤ G := - Filter.Eventually.of_forall fun ω t _ht ↦ hG (A t ω) - -omit [IsMarkovKernel ν] in -/-- First projection from the packaged LinUCB confidence event: optimism for the best arm. -/ -lemma LinUCBConfidenceEvent.best [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - intro t ht - exact (h_conf t ht).1 - -omit [IsMarkovKernel ν] in -/-- Second projection from the packaged LinUCB confidence event: validity of the selected arm's -lower confidence inequality. -/ -lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ t, t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - intro t ht - exact (h_conf t ht).2 - -omit [IsMarkovKernel ν] in -/-- In zero feature dimension, the confidence event forces every positive-time selected gap to be -nonpositive. The best-arm index is zero, and the selected-arm pessimistic index is also zero. -/ -lemma gap_nonpos_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) - (t : ℕ) (ht : t ≠ 0) : - gap ν (A t ω) ≤ 0 := by - have hbest := LinUCBConfidenceEvent.best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (ω := ω) h_conf t ht - have harm := LinUCBConfidenceEvent.arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (ω := ω) h_conf t ht - rw [gap_eq_bestArm_sub] - have hbest0 : (ν (bestArm ν))[id] ≤ 0 := by - simpa [index_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) (β := β) - (x := x) (n := t) (ω := ω) hd (bestArm ν)] using hbest - have harm0 : 0 ≤ (ν (A t ω))[id] := by - simpa [estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) - (x := x) (n := t) (ω := ω) hd (A t ω), - width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := t) - (ω := ω) hd (A t ω)] using harm - linarith - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ -lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - filter_upwards [h_conf] with ω h_confω - exact h_confω.best - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure projection of the packaged confidence event to the selected arm's lower confidence -inequality. -/ -lemma linUCBConfidenceEvent_ae_arm [Nonempty (Fin K)] - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - filter_upwards [h_conf] with ω h_confω - exact h_confω.arm - -lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) - (ω : Ω) (hn : n ≠ 0) : - designMatrix A reg x n ω = - designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - cases n with - | zero => exact absurd rfl hn - | succ n => - simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] - rw [Nat.range_succ_eq_Iic] - exact congrArg (fun S ↦ reg • 1 + S) <| - (Finset.sum_coe_sort (Iic n) - (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm - -lemma responseVector_eq_responseVector' (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - cases n with - | zero => exact absurd rfl hn - | succ n => - simp only [responseVector, responseVector', IsAlgEnvSeq.hist] - rw [Nat.range_succ_eq_Iic] - exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm - -lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, - responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] - -lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - estimatedReward A R reg x a n ω = - estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] - -lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthQuadraticForm A reg x a n ω = - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [widthQuadraticForm, widthQuadraticForm', - designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] - -/-- At positive process times, nonnegativity of the process-level width quadratic form is -equivalent to nonnegativity of the matching history-level width quadratic form. -/ -lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - 0 ≤ widthQuadraticForm A reg x a n ω ↔ - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] - -/-- At positive process times, the process-level quadratic width form is at most `1` iff the -matching history-level quadratic width form is at most `1`. -/ -lemma widthQuadraticForm_le_one_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthQuadraticForm A reg x a n ω ≤ 1 ↔ - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a ≤ 1 := by - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] - -/-- The all-positive-times process-level nonnegativity assumption is equivalent to the matching -history-level nonnegativity assumption. -/ -lemma widthQuadraticForm_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ - ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by - constructor - · intro h t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).1 (h t ht ht0) - · intro h t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h t ht ht0) - -/-- The all-positive-times process-level `≤ 1` assumption is equivalent to the matching -history-level `≤ 1` assumption. -/ -lemma widthQuadraticForm_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ - ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by - constructor - · intro h t ht ht0 - exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).1 (h t ht ht0) - · intro h t ht ht0 - exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h t ht ht0) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, process-level all-positive-times nonnegativity is equivalent to the matching -history-level nonnegativity assumption. -/ -lemma widthQuadraticForm_ae_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) : - (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).1 hω - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).2 hω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the process-level all-positive-times `≤ 1` assumption is equivalent to the -matching history-level `≤ 1` assumption. -/ -lemma widthQuadraticForm_ae_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) : - (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).1 hω - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).2 hω - -lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n - ω hn] - -/-- At positive process times, squaring the process-level width recovers the matching history-level -quadratic form when that history-level quadratic form is nonnegative. -/ -lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_nonneg : - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a) : - width A reg x a n ω ^ 2 = - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - rw [width_eq_width' (A := A) (R := R) reg x a n ω hn] - exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a - h_nonneg - -/-- At positive process times, advancing `widthSqSum` adds the matching history-level quadratic -form when that history-level quadratic form is nonnegative. -/ -lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_nonneg : - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] - rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] - -/-- At positive process times, advancing `quadraticWidthSum` adds the matching history-level -quadratic form. -/ -lemma quadraticWidthSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - rw [quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hn] - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn] - -/-- The history-level quadratic-form accumulator aligned with process times. - -The term at process time `t = 0` is set to zero, matching the convention used by `widthSqSum` and -`quadraticWidthSum`. At positive process time `t`, the history available to LinUCB is -`IsAlgEnvSeq.hist A R (t - 1) ω`. -/ -noncomputable def historyQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) - -/-- No positive-time history-level quadratic width forms are accumulated at horizon zero. -/ -lemma historyQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - historyQuadraticWidthSum A R reg x 0 ω = 0 := by - simp [historyQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time history-level quadratic width form. -/ -lemma historyQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - historyQuadraticWidthSum A R reg x (n + 1) ω = - historyQuadraticWidthSum A R reg x n ω + - if n = 0 then 0 else - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - simp [historyQuadraticWidthSum, sum_range_succ] - -/-- At positive process times, advancing the history-level quadratic accumulator adds the selected -arm's history-level quadratic width form. -/ -lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - historyQuadraticWidthSum A R reg x (n + 1) ω = - historyQuadraticWidthSum A R reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - simp [historyQuadraticWidthSum_succ, hn] - -/-- The capped history-level quadratic-form accumulator aligned with process times. - -This is the accumulator shape that commonly appears in elliptical-potential statements: -each positive-time quadratic width form is capped at `1`. -/ -noncomputable def historyCappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else - min 1 (widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - -/-- No positive-time capped history-level quadratic width forms are accumulated at horizon zero. -/ -lemma historyCappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - historyCappedQuadraticWidthSum A R reg x 0 ω = 0 := by - simp [historyCappedQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time capped history-level quadratic width form. -/ -lemma historyCappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - historyCappedQuadraticWidthSum A R reg x (n + 1) ω = - historyCappedQuadraticWidthSum A R reg x n ω + - if n = 0 then 0 else - min 1 - (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by - simp [historyCappedQuadraticWidthSum, sum_range_succ] - -/-- At positive process times, advancing the capped history-level quadratic accumulator adds the -selected arm's capped history-level quadratic width form. -/ -lemma historyCappedQuadraticWidthSum_succ_of_ne_zero - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - historyCappedQuadraticWidthSum A R reg x (n + 1) ω = - historyCappedQuadraticWidthSum A R reg x n ω + - min 1 - (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by - simp [historyCappedQuadraticWidthSum_succ, hn] - -/-- The process-level capped quadratic-width accumulator equals the history-level capped -accumulator aligned with the same process times. -/ -lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - cappedQuadraticWidthSum A reg x n ω = - historyCappedQuadraticWidthSum A R reg x n ω := by - rw [cappedQuadraticWidthSum, historyCappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact congrArg (fun q : ℝ ↦ min 1 q) - (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) - -/-- A process-level capped quadratic-width sum bound is equivalent to the matching history-level -capped quadratic-width sum bound. -/ -lemma cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : - cappedQuadraticWidthSum A reg x n ω ≤ W ↔ - historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by - rw [cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) - reg x n ω] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a process-level capped quadratic-width sum bound is equivalent to the matching -history-level capped quadratic-width sum bound. -/ -lemma cappedQuadraticWidthSum_ae_le_iff_historyCappedQuadraticWidthSum_ae_le - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (W : ℝ) : - (∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) ↔ - ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (A := A) (R := R) reg x n ω W).1 hω - · intro h - filter_upwards [h] with ω hω - exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (A := A) (R := R) reg x n ω W).2 hω - -/-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and -capped history-level accumulators agree. -/ -lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) : - historyQuadraticWidthSum A R reg x n ω = - historyCappedQuadraticWidthSum A R reg x n ω := by - rw [historyQuadraticWidthSum, historyCappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact (min_eq_right (h_le_one t ht ht0)).symm - -/-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with -the same process times. -/ -lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - quadraticWidthSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by - rw [quadraticWidthSum, historyQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0 - -/-- The squared-width accumulator equals the history-level quadratic-form accumulator whenever the -positive-time history-level quadratic forms are nonnegative. -/ -lemma widthSqSum_eq_historyQuadraticWidthSum - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) : - widthSqSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by - have h_process_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h_nonneg t ht ht0) - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_process_nonneg] - exact quadraticWidthSum_eq_historyQuadraticWidthSum (A := A) (R := R) reg x n ω - -/-- A bound on the history-level quadratic-form accumulator implies the corresponding bound on -`widthSqSum`, provided the positive-time history-level quadratic forms are nonnegative. -/ -lemma widthSqSum_le_of_history_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_hist_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_historyQuadraticWidthSum (A := A) (R := R) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - exact h_hist_le - -omit [IsProbabilityMeasure P] in -/-- Almost surely, a history-level quadratic-form bound gives the `widthSqSum` bound consumed by -the regret chain. -/ -lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_hist_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_nonneg, h_hist_le] with ω h_nonnegω h_hist_leω - exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The pointwise input expected from a future elliptical-potential argument. - -It packages the two facts needed to turn a history-level quadratic-width estimate into the -`widthSqSum` estimate used by the regret chain: - -* each positive-time quadratic width form is nonnegative; -* their history-level accumulated sum is bounded by `W`. -/ -def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := - (∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ - historyQuadraticWidthSum A R reg x n ω ≤ W - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Build the packaged history-level quadratic-width input from its two component facts. -/ -lemma historyQuadraticWidthBound_of_nonneg_and_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_sum_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : - HistoryQuadraticWidthBound A R reg x n ω W := by - exact ⟨h_nonneg, h_sum_le⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged history-level quadratic-width input is monotone in the numeric bound. -/ -lemma historyQuadraticWidthBound_mono {W W' : ℝ} - (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : - HistoryQuadraticWidthBound A R reg x n ω W' := by - exact ⟨h_bound.1, h_bound.2.trans hW⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, build the packaged history-level quadratic-width input from its two component -facts. -/ -lemma historyQuadraticWidthBound_ae_of_nonneg_and_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_sum_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by - filter_upwards [h_nonneg, h_sum_le] with ω h_nonnegω h_sum_leω - exact historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_sum_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged history-level quadratic-width input is monotone in the numeric -bound. -/ -lemma historyQuadraticWidthBound_ae_mono {W W' : ℝ} - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W' := by - filter_upwards [h_bound] with ω h_boundω - exact historyQuadraticWidthBound_mono (A := A) (R := R) (reg := reg) (x := x) - (n := n) (ω := ω) h_boundω hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped quadratic-width sum bound gives the packaged history-level input whenever every -positive-time quadratic width form is nonnegative and at most `1`. -/ -lemma historyQuadraticWidthBound_of_capped_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - HistoryQuadraticWidthBound A R reg x n ω W := by - refine historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg ?_ - rw [historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_le_one] - exact h_capped_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped quadratic-width sum bound gives the packaged history-level input -whenever every positive-time quadratic width form is almost surely nonnegative and at most `1`. -/ -lemma historyQuadraticWidthBound_ae_of_capped_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_capped_le] with - ω h_nonnegω h_le_oneω h_capped_leω - exact historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the -regret chain. -/ -lemma widthSqSum_le_of_history_quadratic_width_bound {W : ℝ} - (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_bound.1 h_bound.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged history-level quadratic-width input implies the `widthSqSum` bound -consumed by the regret chain. -/ -lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_bound] with ω h_boundω - exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) (W := W) h_boundω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped history-level quadratic-width sum bound implies the `widthSqSum` bound consumed by -the regret chain, provided the positive-time quadratic width forms are nonnegative and at most -`1`. -/ -lemma widthSqSum_le_of_capped_history_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) (W := W) - (historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonneg h_le_one h_capped_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped history-level quadratic-width sum bound implies the `widthSqSum` bound -consumed by the regret chain, provided the positive-time quadratic width forms are almost surely -nonnegative and at most `1`. -/ -lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - h_capped_le) - -lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - index A R reg β x a n ω = - index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - have htime : n + 1 = n - 1 + 2 := by grind - simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, - width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] - -/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ -lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (n : ℕ) : - A (n + 1) =ᵐ[P] - fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact h.action_detAlgorithm_ae_eq n - -/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ -lemma arm_ae_all_eq [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : - ∀ᵐ ω ∂P, - ∀ n, A (n + 1) ω = - nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by - simp_rw [ae_all_iff] - exact fun n ↦ arm_ae_eq_linUCBNextArm h n - -/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ -lemma index_le_index_arm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (a : Fin K) (hn : n ≠ 0) : - ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by - filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm - have hn_succ : n - 1 + 1 = n := by grind - simp only [hn_succ] at h_arm - rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, - index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] - rw [h_arm] - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) - (IsAlgEnvSeq.hist A R (n - 1) ω) a - -/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ -lemma forall_index_le_index_arm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (a : Fin K) : - ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by - simp_rw [ae_all_iff] - exact fun n hn ↦ index_le_index_arm h a hn - -end AlgorithmBehavior - -omit [IsMarkovKernel ν] in -/-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the -selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected -arm's LinUCB width. -/ -lemma mean_sub_mean_arm_le_two_mul_width (a : Fin K) - (h_best : (ν a)[id] ≤ index A R reg β x a n ω) - (h_arm : estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_le : index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω) : - (ν a)[id] - (ν (A n ω))[id] ≤ - 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [sub_le_iff_le_add'] - calc - (ν a)[id] ≤ index A R reg β x a n ω := h_best - _ ≤ index A R reg β x (A n ω) n ω := h_le - _ ≤ (ν (A n ω))[id] + - 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [index, two_mul, ← add_assoc] - gcongr - rwa [sub_le_iff_le_add] at h_arm - -omit [IsMarkovKernel ν] in -/-- The gap of the selected arm is bounded by twice its LinUCB bonus whenever the usual confidence -inequalities hold and the selected arm has maximal LinUCB index. -/ -lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] - (h_best : (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_le : index A R reg β x (bestArm ν) n ω ≤ - index A R reg β x (A n ω) n ω) : - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [gap_eq_bestArm_sub] - exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) (a := bestArm ν) h_best h_arm h_le - -/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus whenever the usual -confidence inequalities hold almost surely. -/ -lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (hn : n ≠ 0) - (h_best : ∀ᵐ ω ∂P, (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - filter_upwards [h_best, h_arm, index_le_index_arm h (bestArm ν) hn] with - ω h_bestω h_armω h_leω - exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) h_bestω h_armω h_leω - -/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus at every positive -time whenever the usual confidence inequalities hold almost surely at every positive time. -/ -lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - filter_upwards [h_best, h_arm, forall_index_le_index_arm h (bestArm ν)] with - ω h_bestω h_armω h_leω - intro n hn - exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) - -omit [IsMarkovKernel ν] in -/-- Pointwise capped LinUCB regret bound for one positive time. - -If the instantaneous gap is bounded by `2`, and the confidence/max-index argument gives the usual -`2 * sqrt(β_t) * width_t` bound, then monotonicity up to the terminal `β n` gives the textbook -capped form `2 * sqrt(β n) * sqrt(min 1 q_t)`, where `q_t` is the width quadratic form. -/ -lemma gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm - (t : ℕ) - (h_gap_two : gap ν (A t ω) ≤ 2) - (h_gap_width : gap ν (A t ω) ≤ - 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) - (hβ_le : β (t + 1) ≤ β n) - (hβn_one : 1 ≤ β n) : - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - by_cases hq_le_one : widthQuadraticForm A reg x (A t ω) t ω ≤ 1 - · have hwidth_nonneg : 0 ≤ width A reg x (A t ω) t ω := Real.sqrt_nonneg _ - have hsqrt_le : √(β (t + 1)) ≤ √(β n) := Real.sqrt_le_sqrt hβ_le - have hbonus_le : - 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) ≤ - 2 * (√(β n) * width A reg x (A t ω) t ω) := by - exact mul_le_mul_of_nonneg_left - (mul_le_mul_of_nonneg_right hsqrt_le hwidth_nonneg) (by norm_num) - have hmin : - √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)) = - width A reg x (A t ω) t ω := by - rw [min_eq_right hq_le_one, width] - simpa [hmin] using h_gap_width.trans hbonus_le - · have hq_one : 1 ≤ widthQuadraticForm A reg x (A t ω) t ω := by linarith - have hsqrt_one : 1 ≤ √(β n) := by - simpa using (Real.one_le_sqrt).2 hβn_one - have htwo_le : - 2 ≤ 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [min_eq_left hq_one, Real.sqrt_one] - nlinarith - exact h_gap_two.trans htwo_le - -omit [IsMarkovKernel ν] in -/-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded -by the corresponding sum of pointwise bounds. -/ -lemma regret_le_sum_of_gap_bound (B : ℕ → ℝ) - (hB : ∀ t, t ∈ range n → gap ν (A t ω) ≤ B t) : - regret ν A n ω ≤ ∑ t ∈ range n, B t := by - rw [regret_eq_sum_gap] - exact sum_le_sum hB - -omit [IsMarkovKernel ν] in -/-- A pathwise cumulative-regret bound obtained by summing the positive-time LinUCB width bound. - -The time-zero gap is left unchanged because the current LinUCB max-index theorem applies only at -positive times. -/ -lemma regret_le_sum_width_of_forall_gap_le - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) : - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using h_gap t ht ht0 - -omit [IsMarkovKernel ν] in -/-- A pathwise cumulative-regret bound obtained by summing the positive-time capped LinUCB width -bound. -/ -lemma regret_le_sum_sqrt_capped_width_of_forall_gap_le - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using h_gap t ht ht0 - -omit [IsMarkovKernel ν] in -/-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ -lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ≤ - 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - calc - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) - = 2 * ∑ t ∈ range n, - (if t = 0 then 0 else √(β (t + 1))) * - (if t = 0 then 0 else width A reg x (A t ω) t ω) := by - rw [mul_sum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - gcongr - exact Real.sum_mul_le_sqrt_mul_sqrt (range n) - (fun t ↦ if t = 0 then 0 else √(β (t + 1))) - (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) - -omit [IsMarkovKernel ν] in -/-- Cauchy-Schwarz bound for the positive-time capped LinUCB bonus sum. -/ -lemma sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum - (hβn_nonneg : 0 ≤ β n) - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ≤ - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - calc - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) - = 2 * ∑ t ∈ range n, - (if t = 0 then 0 else √(β n)) * - (if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [mul_sum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) * - √(∑ t ∈ range n, - (if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) ^ 2)) := by - gcongr - exact Real.sum_mul_le_sqrt_mul_sqrt (range n) - (fun t ↦ if t = 0 then 0 else √(β n)) - (fun t ↦ if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) - _ ≤ 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - gcongr - · calc - (∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) - ≤ ∑ _t ∈ range n, β n := by - refine sum_le_sum ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0, hβn_nonneg] - · simp [ht0, Real.sq_sqrt hβn_nonneg] - _ = (n : ℝ) * β n := by - simp [sum_const, nsmul_eq_mul] - · rw [cappedQuadraticWidthSum] - refine le_of_eq ?_ - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · have hmin_nonneg : - 0 ≤ min 1 (widthQuadraticForm A reg x (A t ω) t ω) := by - exact le_min zero_le_one (h_nonneg t ht ht0) - simp [ht0, Real.sq_sqrt hmin_nonneg] - -omit [IsMarkovKernel ν] in -/-- Pathwise cumulative-regret bound using the textbook capped quadratic-width sum. -/ -lemma regret_le_initial_add_sqrt_nat_mul_beta_capped_sum - (hβn_nonneg : 0 ≤ β n) - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - refine (regret_le_sum_sqrt_capped_width_of_forall_gap_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) h_gap).trans ?_ - have hsplit : - (∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) = - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - ∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * - √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [← sum_add_distrib] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - rw [hsplit] - exact add_le_add le_rfl - (sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum - (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω) - hβn_nonneg h_nonneg) - -omit [IsMarkovKernel ν] in -/-- If the capped quadratic-width sum is bounded by `W`, the pathwise capped regret bound can use -`√W` in place of the realized capped-sum square root. -/ -lemma regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω))) - (hW : cappedQuadraticWidthSum A reg x n ω ≤ W) : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √W) := by - refine h_regret.trans ?_ - gcongr - -/-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is -nonnegative. -/ -lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : - (∑ t ∈ range n, if t = 0 then 0 else √(β (t + 1)) ^ 2) = - ∑ t ∈ range n, if t = 0 then 0 else β (t + 1) := by - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0, Real.sq_sqrt (hβ t)] - -/-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the -usual confidence inequalities hold almost surely at every positive time. -/ -lemma regret_ae_le_sum_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm] with ω h_gapω - exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 - -/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound on -the positive-time LinUCB width terms. -/ -lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - filter_upwards [regret_ae_le_sum_width h h_best h_arm] with ω h_regret - refine h_regret.trans ?_ - have hsplit : - (∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) = - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - ∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - rw [← sum_add_distrib] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - rw [hsplit] - exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) - (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ - -/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound whose -beta factor has been simplified using nonnegativity of the confidence schedule. -/ -lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - filter_upwards [regret_ae_le_initial_add_cauchy (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret - simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret - -omit [IsMarkovKernel ν] in -/-- If the squared LinUCB widths are bounded by `W`, then the Cauchy-Schwarz regret bound can use -`√W` in place of the square root of the realized squared-width sum. -/ -lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) - (hW : widthSqSum A reg x n ω ≤ W) - : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by - rw [widthSqSum] at hW - refine h_regret.trans ?_ - gcongr - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √(sum beta terms) * √W` whenever the squared LinUCB widths are almost surely bounded by `W`. - -This is the interface expected from a future elliptical-potential bound. -/ -lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by - filter_upwards [regret_ae_le_initial_add_cauchy_simplified (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with - ω h_regret hWω - exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -omit [IsMarkovKernel ν] in -/-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square -root of the beta sum. -/ -lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) - (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by - refine h_regret.trans ?_ - gcongr - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √B * √W` whenever the beta sum is bounded by `B` and the squared LinUCB widths are almost -surely bounded by `W`. -/ -lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) - (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by - filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW - ] with ω h_regret - exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) - (n := n) (ω := ω) B W h_regret hB - -/-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is -bounded by the horizon times the terminal beta value. -/ -lemma beta_sum_le_nat_mul_of_monotone - (hβ_mono : Monotone β) (hβ : ∀ t, 0 ≤ β (t + 1)) : - (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ (n : ℝ) * β n := by - calc - (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) - ≤ ∑ _t ∈ range n, β n := by - refine sum_le_sum ?_ - intro t ht - by_cases ht0 : t = 0 - · rw [if_pos ht0] - have hn_pos : 0 < n := by - simpa [ht0] using mem_range.mp ht - have hn_beta : 0 ≤ β n := by - have htime : n - 1 + 1 = n := by grind - simpa [htime] using hβ (n - 1) - exact hn_beta - · rw [if_neg ht0] - exact hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) - _ = (n : ℝ) * β n := by - simp [sum_const, nsmul_eq_mul] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Minimal confidence-radius schedule assumptions used by the capped finite-action LinUCB regret -chain: the schedule starts at least at one and is monotone in time. -/ -def BetaSchedule (β : ℕ → ℝ) : Prop := - 1 ≤ β 1 ∧ Monotone β - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection from `BetaSchedule`: the confidence-radius schedule starts at least at one. -/ -lemma BetaSchedule.one (hβ : BetaSchedule β) : 1 ≤ β 1 := - hβ.1 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection from `BetaSchedule`: the confidence-radius schedule is monotone. -/ -lemma BetaSchedule.monotone (hβ : BetaSchedule β) : Monotone β := - hβ.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive -horizon. -/ -lemma beta_nonneg_of_one_le_of_monotone - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) {n : ℕ} (hn : n ≠ 0) : - 0 ≤ β n := by - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) - exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A `BetaSchedule` is nonnegative at every positive horizon. -/ -lemma BetaSchedule.nonneg_of_ne_zero (hβ : BetaSchedule β) {n : ℕ} (hn : n ≠ 0) : - 0 ≤ β n := - beta_nonneg_of_one_le_of_monotone (β := β) hβ.one hβ.monotone hn - -omit [IsMarkovKernel ν] in -/-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the -horizon is zero. -/ -lemma initial_gap_sum_eq : - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) = - if n = 0 then 0 else gap ν (A 0 ω) := by - cases n <;> simp - -omit [IsMarkovKernel ν] in -/-- In zero feature dimension, the confidence event bounds cumulative regret by the initial gap. -There is no positive-time width contribution because all widths are zero. -/ -lemma regret_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by - refine (regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ if t = 0 then gap ν (A 0 ω) else 0) ?_).trans ?_ - · intro t _ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using - gap_nonpos_of_confidence_dim_eq_zero (A := A) (R := R) (reg := reg) - (β := β) (x := x) (ν := ν) (ω := ω) hd h_conf t ht0 - · rw [initial_gap_sum_eq] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure zero-dimensional version of the finite-action LinUCB regret skeleton. -/ -lemma regret_ae_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by - filter_upwards [h_conf] with ω h_confω - exact regret_le_initial_gap_of_confidence_dim_eq_zero (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hd h_confω - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` -is nonnegative and monotone. -/ -lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W - (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) hW - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` -is nonnegative and monotone. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW - ] with ω h_regret - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a history-level quadratic-form bound supplies the future -elliptical-potential input. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (hW : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the packaged history-level quadratic-width input holds almost -surely. - -This is the theorem a future elliptical-potential lemma should feed into directly. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a capped history-level quadratic-width sum bound holds almost -surely and every positive-time quadratic width form is almost surely nonnegative and at most `1`. - -This is the direct interface for the common capped form of the elliptical-potential lemma. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (hW : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a capped process-level quadratic-width sum bound holds almost -surely and every positive-time process-level quadratic width form is almost surely nonnegative and -at most `1`. - -This is the direct interface for an elliptical-potential lemma stated using the process-level design -matrices `designMatrix A reg x t ω`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the packaged process-level capped quadratic-width input holds -almost surely. - -This is the compact theorem a process-level elliptical-potential lemma should feed into directly. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) - (x := x) (n := n) (P := P) (W := W) h_bound) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is almost surely bounded -by `W`. - -This version follows the proof structure of *Bandit Algorithms*, Theorem 19.2: optimism gives the -width bound, bounded instantaneous gaps give the cap, monotonicity of `β` moves all confidence -radii to `β n`, and Cauchy-Schwarz turns the sum into the square root of the capped quadratic-width -sum. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_schedule : BetaSchedule β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - by_cases hn : n = 0 - · subst n - exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] - have hβn_nonneg : 0 ≤ β n := - hβ_schedule.nonneg_of_ne_zero hn - filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] - with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω - have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht _ht0 - exact h_quad_nonnegω t ht - have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - intro t ht ht0 - have hβ_le : β (t + 1) ≤ β n := - hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) - have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 - have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) - exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) - (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one - have h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := - regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos - h_gap_capped - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using - regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -/-- Almost surely, on the LinUCB confidence event, cumulative regret is bounded by the simplified -initial-gap term plus `2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is -almost surely bounded by `W`. - -This is the good-event form of the deterministic regret argument. It separates the algorithmic -regret proof from the future concentration theorem: a later self-normalized concentration result -should prove that `LinUCBConfidenceEvent` holds with high probability, and this theorem converts -that event into the regret bound. -/ -lemma regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_schedule : BetaSchedule β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - by_cases hn : n = 0 - · subst n - exact Filter.Eventually.of_forall fun ω _h_confω ↦ by simp [regret] - have hβn_nonneg : 0 ≤ β n := - hβ_schedule.nonneg_of_ne_zero hn - filter_upwards [forall_index_le_index_arm h (bestArm ν), h_gap_two, h_quad_nonneg, hW] with - ω h_indexω h_gap_twoω h_quad_nonnegω hWω h_confω - have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht _ht0 - exact h_quad_nonnegω t ht - have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - intro t ht ht0 - have h_gap_width : - gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := - gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := t) (ω := ω) (h_confω.best t ht0) - (h_confω.arm t ht0) (h_indexω t ht0) - have hβ_le : β (t + 1) ≤ β n := - hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) - have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 - have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) - exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) - (h_gap_twoω t ht ht0) h_gap_width hβ_le hβn_one - have h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := - regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos - h_gap_capped - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using - regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the -feature-budget elliptical-potential term -`2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. - -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_antitone` is the -generic inverse anti-monotonicity theorem for positive-definite matrices, and `h_ratio_of_trace` -should come from a determinant/trace comparison proving that the trace budget implies the displayed -determinant-ratio bound. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) - (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' - (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) - (x := x) hreg_pos h_inv_antitone)) - hreg_pos hL2 hL2_le_reg) - L2 hL2 h_ratio_of_trace) - -/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the determinant/trace input is stated as the determinant upper bound -`det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg hdet_of_trace) - -/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the determinant/trace input is the reusable PSD matrix determinant/trace comparison -`det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) - -/-- Textbook-shaped finite-action LinUCB regret theorem on the confidence event. - -This theorem is the good-event form closest to the finite-action LinUCB proof in -*Bandit Algorithms*: after the deterministic algorithm/max-index argument and the elliptical -potential bound are proved, the only remaining probabilistic input is whether the confidence event -holds on a sample path. - -* `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; -* `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; -* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. - -The conclusion is an almost-sure implication: on almost every sample path, if -`LinUCBConfidenceEvent` holds, then the displayed regret bound holds. A future self-normalized -concentration theorem should prove that this confidence event has high probability for a concrete -textbook choice of `β`. - -The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression -`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this -formalization lets the deterministic algorithm play its default initial arm at time zero. -/ -lemma regret_ae_imp_le_textbook_finite_action - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - by_cases hd : d = 0 - · subst d - exact Filter.Eventually.of_forall fun ω h_confω ↦ by - simpa using regret_le_initial_gap_of_confidence_dim_eq_zero - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) - (ω := ω) (d := 0) rfl h_confω - · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by - filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with - ω h_gapω - intro t ht _ht0 - exact h_gapω t ht - exact regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_gap_two hβ_schedule - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 - (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) - (P := P) L2 hL2) - matrixDetLeTraceAveragePow) - -/-- The deterministic textbook LinUCB bonus term -`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`. - -The final finite-action theorem keeps this as a named expression so probability statements can use -a deterministic right-hand side instead of repeating the full formula. -/ -noncomputable def textbookRegretBonus (reg : ℝ) (β : ℕ → ℝ) (L2 : ℝ) (n : ℕ) : ℝ := - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) - -/-- Good-event finite-action LinUCB regret theorem with the random initial gap replaced by the -deterministic `≤ 2` bound implied by `MeanRewardBound ν (-1) 1`. -/ -lemma regret_ae_imp_le_textbook_finite_action_deterministic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_confω - refine (h_regret h_confω).trans ?_ - simpa [textbookRegretBonus] using - add_le_add_right - (initialGapTerm_le_two_of_meanRewardBound_neg_one_one (A := A) (ν := ν) - (n := n) (ω := ω) h_mean_bound) - (2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))) - -/-- Almost-sure corollary of -`regret_ae_imp_le_textbook_finite_action_deterministic_bound` when the confidence event is known -to hold almost surely. -/ -lemma regret_ae_le_textbook_finite_action_deterministic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2, h_conf] with ω h_regret h_confω - exact h_regret h_confω - -/-- The confidence event is almost surely contained in the deterministic textbook regret-bound -event. This is the version to combine with a future high-probability confidence theorem. -/ -lemma probReal_confidenceEvent_le_textbook_regret_bound_deterministic - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_confω - exact h_regret h_confω - -/-- High-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ -lemma probReal_textbook_regret_bound_deterministic_ge_of_confidenceEvent_ge - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : - 1 - δ ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by - exact h_conf_prob.trans - (probReal_confidenceEvent_le_textbook_regret_bound_deterministic (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2) - -/-- Failure-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ -lemma probReal_textbook_regret_bound_deterministic_failure_le_of_confidenceEvent_failure_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_failure : - P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : - P.real {ω | - ¬ regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} ≤ δ := by - refine le_trans ?_ h_conf_failure - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω - exact h_regret_failure (h_regret h_confω) - -/-- The confidence event is almost surely contained in the textbook finite-action regret-bound -event. - -This is the probability bridge needed after the good-event theorem: once a concentration theorem -proves that `LinUCBConfidenceEvent` has high probability, this lemma transfers that probability -mass to the displayed regret bound. -/ -lemma probReal_confidenceEvent_le_textbook_regret_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_confω - exact h_regret h_confω - -/-- High-probability wrapper for the textbook finite-action LinUCB regret bound. - -If a future self-normalized concentration theorem proves that the confidence event has probability -at least `1 - δ`, then the textbook regret bound has probability at least `1 - δ` as well. -/ -lemma probReal_textbook_regret_bound_ge_of_confidenceEvent_ge - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : - 1 - δ ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by - exact h_conf_prob.trans - (probReal_confidenceEvent_le_textbook_regret_bound (A := A) (R := R) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule hreg_pos L2 hL2) - -/-- Failure-probability wrapper for the textbook finite-action LinUCB regret bound. - -If a future self-normalized concentration theorem proves that the confidence event fails with -probability at most `δ`, then the textbook regret bound fails with probability at most `δ`. -/ -lemma probReal_textbook_regret_bound_failure_le_of_confidenceEvent_failure_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_failure : - P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : - P.real {ω | - ¬ - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} ≤ δ := by - refine le_trans ?_ h_conf_failure - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω - exact h_regret_failure (h_regret h_confω) - -/-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to -hold almost surely. This is stronger than the textbook high-probability route and is mainly useful -as a compatibility wrapper for earlier lemmas in this file. -/ -lemma regret_ae_le_textbook_finite_action - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2, h_conf] with ω h_regret h_confω - exact h_regret h_confω - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final -log-determinant potential bound hold. - -The capped-sum/log-determinant part of the elliptical-potential argument is proved internally: -positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while -`h_quad_le_one` lets this older theorem feed the uncapped `widthSqSum` regret route. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hreg_pos : 0 < reg) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono W - (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) hreg_pos - h_quad_le_one h_potential_le) - -end LinUCB - -end Bandits From 9257d57b1876620c85aa3bd1d8d39672d803de48 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:30:36 -0400 Subject: [PATCH 76/82] feat(112): remove the regret file reference --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f528dfdb..68192152 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -5,7 +5,7 @@ Authors: OpenAI, Fawad Haider -/ module -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Regret + /-! # LinUCB for finite-action linear bandits From 60c75ad1c6bef1a5b385fc1a01907db80ebbbe87 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 10:50:39 -0400 Subject: [PATCH 77/82] linUCB(issue-112): update the LML main import file --- LeanMachineLearning.lean | 1 + 1 file changed, 1 insertion(+) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index d48bdc7e..cd8b36ea 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -20,6 +20,7 @@ public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS public import LeanMachineLearning.Online.Bandit.Algorithms.UCB From 200901403ef8d24009a4d570014e014966acc4ec Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 11:00:00 -0400 Subject: [PATCH 78/82] linUCB(issue-112): fix author comment --- .../Online/Bandit/Algorithms/LinUCB/Basic.lean | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean index 21912a55..e492f2af 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean @@ -1,5 +1,5 @@ /- -Copyright (c) 2026. All rights reserved. +Copyright (c) 2026 Fawad Haider. All rights reserved. Released under Apache 2.0 license as described in the file LICENSE. Authors: OpenAI, Fawad Haider -/ @@ -7,11 +7,11 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.MeanInequalities public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.Analysis.Matrix.Order -public import Mathlib.Data.Real.StarOrdered +public import Mathlib.Algebra.Order.Star.Real public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse From ac1a4ecc30ff7cc4cfaffdc76937b177570d98e4 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 1 Jul 2026 11:44:27 -0400 Subject: [PATCH 79/82] libUCB(issue-112): feedback changes to use mathlibs euclideans space --- .../Bandit/Algorithms/LinUCB/Basic.lean | 101 +++++------------- 1 file changed, 27 insertions(+), 74 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean index e492f2af..e8c1b260 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean @@ -7,8 +7,8 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.MeanInequalities +public import Mathlib.Analysis.InnerProductSpace.PiL2 public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.Analysis.Matrix.Order public import Mathlib.Algebra.Order.Star.Real @@ -36,39 +36,28 @@ section Algorithm namespace LinUCB -/-- Feature vectors for finite-dimensional linear bandits. -/ -abbrev Feature (d : ℕ) := Fin d → ℝ - -/-- The standard coordinate direction in `Feature d`. -/ -def coordinateDirection (i : Fin d) : Feature d := - fun j ↦ if j = i then 1 else 0 - -/-- Dot product with a coordinate direction extracts that coordinate. -/ -lemma dotProduct_coordinateDirection (u : Feature d) (i : Fin d) : - dotProduct (coordinateDirection i) u = u i := by - simp only [dotProduct, coordinateDirection] - rw [Finset.sum_eq_single i] - · simp - · intro j _hj hji - simp [hji] - · intro hi - simp at hi - -/-- Dot product with the negative coordinate direction extracts the negated coordinate. -/ -lemma dotProduct_neg_coordinateDirection (u : Feature d) (i : Fin d) : - dotProduct (-coordinateDirection i) u = -u i := by - rw [neg_dotProduct, dotProduct_coordinateDirection] - -/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product -`x_aᵀ x_a`. -/ -def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := - dotProduct (x a) (x a) - -/-- The squared feature norm is nonnegative. -/ -lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : - 0 ≤ featureSqNorm x a := by - rw [featureSqNorm, dotProduct] - exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) +/-- Feature vectors for finite-dimensional linear bandits. + +We use mathlib's Euclidean space model so that LinUCB's feature vectors carry the standard +inner-product and norm structure of `ℝ^d`. Coordinate formulas remain available through the +coercion `Feature d → (Fin d → ℝ)`, which is useful for the matrix identities below. -/ +abbrev Feature (d : ℕ) := EuclideanSpace ℝ (Fin d) + +/-- View a coordinate matrix-vector product as a Euclidean feature vector. -/ +noncomputable def matrixMulFeature (M : Matrix (Fin d) (Fin d) ℝ) (v : Feature d) : + Feature d := + WithLp.toLp 2 (Matrix.mulVec M v) + +lemma mulVec_matrixMulFeature (M N : Matrix (Fin d) (Fin d) ℝ) (v : Feature d) : + Matrix.mulVec M (matrixMulFeature N v) = Matrix.mulVec (M * N) v := by + ext i + simp [matrixMulFeature, Matrix.mulVec_mulVec] + +/-- The coordinate dot product of a feature vector with itself is its squared Euclidean norm. -/ +lemma dotProduct_self_eq_norm_sq (u : Feature d) : + dotProduct u u = ‖u‖ ^ 2 := by + rw [← real_inner_self_eq_norm_sq] + simp [dotProduct, inner] /-- The squared Euclidean norm of an arbitrary feature vector is nonnegative. -/ lemma dotProduct_self_nonneg (u : Feature d) : @@ -104,7 +93,7 @@ lemma abs_dotProduct_le_sqrt_mul_sqrt_of_sq_norm_le This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared form as `‖x_a‖₂² ≤ L2` for every action. -/ def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := - ∀ a, featureSqNorm x a ≤ L2 + ∀ a, ‖x a‖ ^ 2 ≤ L2 /-- A uniform squared feature-norm bound is nonnegative whenever the finite action set is nonempty. -/ @@ -112,32 +101,9 @@ lemma FeatureSqNormBound.nonneg [Nonempty (Fin K)] {x : Fin K → Feature d} {L2 : ℝ} (hL2 : FeatureSqNormBound x L2) : 0 ≤ L2 := by classical - exact (featureSqNorm_nonneg x (Classical.arbitrary (Fin K))).trans + exact (sq_nonneg ‖x (Classical.arbitrary (Fin K))‖).trans (hL2 (Classical.arbitrary (Fin K))) -/-- A squared feature-norm bound controls every coordinate of every feature vector. -/ -lemma abs_feature_coord_le_sqrt_of_featureSqNorm_le - (x : Fin K → Feature d) {L2 : ℝ} {a : Fin K} - (hL2 : featureSqNorm x a ≤ L2) (i : Fin d) : - |x a i| ≤ √L2 := by - have hcoord_sq_le_norm : (x a i) ^ 2 ≤ featureSqNorm x a := by - rw [featureSqNorm, dotProduct] - simpa [pow_two] using - (Finset.single_le_sum - (s := Finset.univ) (a := i) - (fun j _hj ↦ mul_self_nonneg (x a j)) (Finset.mem_univ i)) - exact Real.abs_le_sqrt (hcoord_sq_le_norm.trans hL2) - -/-- A uniform squared feature-norm bound controls the coordinate projection of every feature -vector. -/ -lemma abs_dotProduct_coordinateDirection_feature_le_sqrt - (x : Fin K → Feature d) {L2 : ℝ} (hL2 : FeatureSqNormBound x L2) - (i : Fin d) (a : Fin K) : - |dotProduct (coordinateDirection i) (x a)| ≤ √L2 := by - simpa [dotProduct_coordinateDirection] using - abs_feature_coord_le_sqrt_of_featureSqNorm_le (x := x) (hL2 a) i - -/-- For a fixed direction and finite action set, all arm-feature projections are bounded. -/ lemma exists_abs_dotProduct_feature_bound (x : Fin K → Feature d) (v : Feature d) : ∃ Q : ℝ, 0 ≤ Q ∧ ∀ a, |dotProduct v (x a)| ≤ Q := by refine ⟨∑ a, |dotProduct v (x a)|, ?_, ?_⟩ @@ -159,7 +125,7 @@ noncomputable def responseVector' (x : Fin K → Feature d) /-- History-level regularized least-squares estimate. -/ noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + matrixMulFeature (designMatrix' reg x n h)⁻¹ (responseVector' x n h) /-- History-level estimated reward of an arm. -/ noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) @@ -176,20 +142,7 @@ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := √(widthQuadraticForm' reg x n h a) -/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic -form is nonnegative. -/ -lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : - width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by - simp [width', Real.sq_sqrt h_nonneg] - -/-- LinUCB optimistic index of an arm. - -The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` -contains the observations through time `n`, this index is used to choose the arm -at time `n + 1`, and we evaluate the schedule at `n + 2` --/ +/-- History-level LinUCB optimistic index for a candidate arm. -/ noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a From 032c01aaa3466ce4fbd8ceeac505789efc136230 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 08:30:48 -0400 Subject: [PATCH 80/82] resolve merge conflict --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index ce81269a..eb991923 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -22,6 +22,7 @@ public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.BayesRegret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 32a03c22face002ec7cbcf50b5b88d88d05851fe Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 1 Jul 2026 12:30:09 -0400 Subject: [PATCH 81/82] linUCB(issue-112): refactor related to removal of measurableArgmax --- .../Online/Bandit/Algorithms/LinUCB/Basic.lean | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean index e8c1b260..5c516c4a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean @@ -11,6 +11,7 @@ public import Mathlib.Analysis.MeanInequalities public import Mathlib.Analysis.InnerProductSpace.PiL2 public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.Analysis.Matrix.Order +public import LeanMachineLearning.ForMathlib.MeasureTheory.Order.MeasurableArg public import Mathlib.Algebra.Order.Star.Real public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement @@ -192,9 +193,13 @@ lemma measurable_matrix_inv_apply {α : Type*} {mα : MeasurableSpace α} (hM : ∀ i j, Measurable fun a ↦ M a i j) (i j : Fin d) : Measurable fun a ↦ (M a)⁻¹ i j := by simp_rw [Matrix.inv_def] - change Measurable fun a ↦ Ring.inverse (M a).det * (M a).adjugate i j - simpa [Ring.inverse_eq_inv] using - (measurable_matrix_det_apply M hM).inv.mul (measurable_matrix_adjugate_apply M hM i j) + have hdet : Measurable fun a ↦ ((M a).det)⁻¹ := + (measurable_matrix_det_apply M hM).inv + have hadj : Measurable fun a ↦ (M a).adjugate i j := + measurable_matrix_adjugate_apply M hM i j + convert hdet.mul hadj using 1 + ext a + simp [Ring.inverse_eq_inv, Matrix.smul_apply] @[fun_prop] lemma measurable_thetaHat'_apply (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (i : Fin d) : @@ -247,7 +252,7 @@ noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - measurableArgmax (fun h a ↦ index' reg β x n h a) h + argmax (fun a ↦ index' reg β x n h a) @[fun_prop] lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) @@ -255,7 +260,8 @@ lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (n : ℕ) : Measurable (nextArm hK reg β x n) := by have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact measurable_measurableArgmax fun a ↦ measurable_index' reg β x n a + unfold nextArm + fun_prop end LinUCB From 9d69692eeb0a08eba8092dbedcb1c7da2cf463e4 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 2 Jul 2026 11:44:47 -0400 Subject: [PATCH 82/82] linUCB(issue-112):barrel file update --- LeanMachineLearning.lean | 1 - 1 file changed, 1 deletion(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 08f270b3..12997800 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -24,7 +24,6 @@ public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.BayesRegret