From fbbfe2af1d595743f5ddecf414a0aabb19ffb2d3 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 29 May 2026 10:10:11 -0400 Subject: [PATCH 01/88] feat : initial algorithm foundation for linUCB --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 7d539fc3..67c48acc 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -4,6 +4,7 @@ public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.Measura public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From cce2458373a3320f1833e1453f2115a27c1e0853 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 29 May 2026 10:10:11 -0400 Subject: [PATCH 02/88] feat : initial algorithm foundation for linUCB --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 3dbb4310..7d6d4a9b 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -4,6 +4,7 @@ public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.Measura public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 7b654c310a82c41a5ce8abd95c3cf406fbb66eb9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 3 Jun 2026 15:45:44 -0400 Subject: [PATCH 03/88] feat(linUCB): proving generic one-step bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 7c45c27b..adf9767a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -225,6 +225,41 @@ lemma forall_index_le_index_arm [Nonempty (Fin K)] end AlgorithmBehavior +omit [IsMarkovKernel ν] in +/-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the +selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected +arm's LinUCB width. -/ +lemma mean_sub_mean_arm_le_two_mul_width (a : Fin K) + (h_best : (ν a)[id] ≤ index A R reg β x a n ω) + (h_arm : estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (h_le : index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω) : + (ν a)[id] - (ν (A n ω))[id] ≤ + 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + rw [sub_le_iff_le_add'] + calc + (ν a)[id] ≤ index A R reg β x a n ω := h_best + _ ≤ index A R reg β x (A n ω) n ω := h_le + _ ≤ (ν (A n ω))[id] + + 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + rw [index, two_mul, ← add_assoc] + gcongr + rwa [sub_le_iff_le_add] at h_arm + +omit [IsMarkovKernel ν] in +/-- The gap of the selected arm is bounded by twice its LinUCB bonus whenever the usual confidence +inequalities hold and the selected arm has maximal LinUCB index. -/ +lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] + (h_best : (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (h_le : index A R reg β x (bestArm ν) n ω ≤ + index A R reg β x (A n ω) n ω) : + gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + rw [gap_eq_bestArm_sub] + exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) (a := bestArm ν) h_best h_arm h_le + end LinUCB end Bandits From 394750eb24342d2e37ba7df24bb9fc1935fc9492 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 4 Jun 2026 15:35:38 -0400 Subject: [PATCH 04/88] feat(linUCB): almost-sure instantaneous regret/gap bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index adf9767a..596cfebc 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -260,6 +260,21 @@ lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (a := bestArm ν) h_best h_arm h_le +/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus whenever the usual +confidence inequalities hold almost surely. -/ +lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (hn : n ≠ 0) + (h_best : ∀ᵐ ω ∂P, (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, + gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + filter_upwards [h_best, h_arm, index_le_index_arm h (bestArm ν) hn] with + ω h_bestω h_armω h_leω + exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) h_bestω h_armω h_leω + end LinUCB end Bandits From c27ce9be436e452c02539a61b6a6fcfd8bf78e70 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 5 Jun 2026 13:18:40 -0400 Subject: [PATCH 05/88] feat(linUCB): all-positive-times instantaneous gap bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 596cfebc..4eb0e7a2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -275,6 +275,23 @@ lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h_bestω h_armω h_leω +/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus at every positive +time whenever the usual confidence inequalities hold almost surely at every positive time. -/ +lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by + filter_upwards [h_best, h_arm, forall_index_le_index_arm h (bestArm ν)] with + ω h_bestω h_armω h_leω + intro n hn + exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) + end LinUCB end Bandits From 9cde1581fb12f5bac428b3bb6c8eda195c2c07bb Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 8 Jun 2026 16:51:40 -0400 Subject: [PATCH 06/88] feat(linUCB): cumulative regret bridge --- .../Online/Bandit/Algorithms/LinUCB.lean | 48 +++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 4eb0e7a2..77e33432 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -292,6 +292,54 @@ lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) +omit [IsMarkovKernel ν] in +/-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded +by the corresponding sum of pointwise bounds. -/ +lemma regret_le_sum_of_gap_bound (B : ℕ → ℝ) + (hB : ∀ t, t ∈ range n → gap ν (A t ω) ≤ B t) : + regret ν A n ω ≤ ∑ t ∈ range n, B t := by + rw [regret_eq_sum_gap] + exact sum_le_sum hB + +omit [IsMarkovKernel ν] in +/-- A pathwise cumulative-regret bound obtained by summing the positive-time LinUCB width bound. + +The time-zero gap is left unchanged because the current LinUCB max-index theorem applies only at +positive times. -/ +lemma regret_le_sum_width_of_forall_gap_le + (h_gap : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) : + regret ν A n ω ≤ + ∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by + refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) + (B := fun t ↦ + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simpa [ht0] using h_gap t ht ht0 + +/-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the +usual confidence inequalities hold almost surely at every positive time. -/ +lemma regret_ae_le_sum_width [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + ∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by + filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm] with ω h_gapω + exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 + end LinUCB end Bandits From af7a7185596e991b161c7b9947084983b79038f8 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 9 Jun 2026 11:25:23 -0400 Subject: [PATCH 07/88] sum of widths inequality using Cauchy-Schwarz from mathlib --- .../Online/Bandit/Algorithms/LinUCB.lean | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 77e33432..0ae161df 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -322,6 +322,34 @@ lemma regret_le_sum_width_of_forall_gap_le · simp [ht0] · simpa [ht0] using h_gap t ht ht0 +omit [IsMarkovKernel ν] in +/-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ +lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ≤ + 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + calc + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) + = 2 * ∑ t ∈ range n, + (if t = 0 then 0 else √(β (t + 1))) * + (if t = 0 then 0 else width A reg x (A t ω) t ω) := by + rw [mul_sum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + gcongr + exact Real.sum_mul_le_sqrt_mul_sqrt (range n) + (fun t ↦ if t = 0 then 0 else √(β (t + 1))) + (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) + /-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the usual confidence inequalities hold almost surely at every positive time. -/ lemma regret_ae_le_sum_width [Nonempty (Fin K)] @@ -340,6 +368,40 @@ lemma regret_ae_le_sum_width [Nonempty (Fin K)] exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 +/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound on +the positive-time LinUCB width terms. -/ +lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + filter_upwards [regret_ae_le_sum_width h h_best h_arm] with ω h_regret + refine h_regret.trans ?_ + have hsplit : + (∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) = + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + ∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by + rw [← sum_add_distrib] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + rw [hsplit] + exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) + (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ + end LinUCB end Bandits From 93b64a8ac9c8403edf8aad5516784bfbdc1b5a7d Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 9 Jun 2026 13:32:14 -0400 Subject: [PATCH 08/88] feat(linUCB): simplify the Cauchy beta factor --- .../Online/Bandit/Algorithms/LinUCB.lean | 30 +++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 0ae161df..26312bda 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -350,6 +350,17 @@ lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : (fun t ↦ if t = 0 then 0 else √(β (t + 1))) (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) +/-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is +nonnegative. -/ +lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : + (∑ t ∈ range n, if t = 0 then 0 else √(β (t + 1)) ^ 2) = + ∑ t ∈ range n, if t = 0 then 0 else β (t + 1) := by + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0, Real.sq_sqrt (hβ t)] + /-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the usual confidence inequalities hold almost surely at every positive time. -/ lemma regret_ae_le_sum_width [Nonempty (Fin K)] @@ -402,6 +413,25 @@ lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ +/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound whose +beta factor has been simplified using nonnegativity of the confidence schedule. -/ +lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by + filter_upwards [regret_ae_le_initial_add_cauchy (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret + simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret + end LinUCB end Bandits From 2db39aa49ae4537e10c82be132140de438feb7f1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 12:00:40 -0400 Subject: [PATCH 09/88] squared LinUCB width sum is bounded by W, then the regret bound can use square root W --- .../Online/Bandit/Algorithms/LinUCB.lean | 43 +++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 26312bda..d9af312c 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -432,6 +432,49 @@ lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret +omit [IsMarkovKernel ν] in +/-- If the squared LinUCB widths are bounded by `W`, then the Cauchy-Schwarz regret bound can use +`√W` in place of the square root of the realized squared-width sum. -/ +lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) + (h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * + √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) + (hW : (∑ t ∈ range n, + (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (_hW_nonneg : 0 ≤ W) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by + refine h_regret.trans ?_ + gcongr + +/-- Almost surely, cumulative regret is bounded by the initial gap plus +`2 * √(sum beta terms) * √W` whenever the squared LinUCB widths are almost surely bounded by `W`. + +This is the interface expected from a future elliptical-potential bound. -/ +lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) + (hW : ∀ᵐ ω ∂P, + (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by + filter_upwards [regret_ae_le_initial_add_cauchy_simplified (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with + ω h_regret hWω + exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω hW_nonneg + end LinUCB end Bandits From f813c857cb01534ce7732bf884a6da43a7445073 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 12:14:34 -0400 Subject: [PATCH 10/88] =?UTF-8?q?feat(linUCB):=20moving=20closer=20to=20te?= =?UTF-8?q?xt=20book=20version=20regret=20=E2=89=A4=20initial=20gap=20+=20?= =?UTF-8?q?2=20*=20=E2=88=9AB=20*=20=E2=88=9AW?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index d9af312c..f86001dd 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -475,6 +475,46 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω hW_nonneg +omit [IsMarkovKernel ν] in +/-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square +root of the beta sum. -/ +lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) + (h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) + (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) + (_hB_nonneg : 0 ≤ B) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by + refine h_regret.trans ?_ + gcongr + +/-- Almost surely, cumulative regret is bounded by the initial gap plus +`2 * √B * √W` whenever the beta sum is bounded by `B` and the squared LinUCB widths are almost +surely bounded by `W`. -/ +lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) + (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) + (hB_nonneg : 0 ≤ B) + (hW : ∀ᵐ ω ∂P, + (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by + filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW + hW_nonneg] with ω h_regret + exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) + (n := n) (ω := ω) B W h_regret hB hB_nonneg + end LinUCB end Bandits From bf90410b75c453b62cd855c88cbc42e06b202655 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 15:24:37 -0400 Subject: [PATCH 11/88] =?UTF-8?q?feat(linUCB):=20moving=20closer=20to=20te?= =?UTF-8?q?xt=20book=20version=20regret=20=E2=89=A4=20initial=20gap=20+=20?= =?UTF-8?q?2=20*=20=E2=88=9A(n=20*=20=CE=B2=20n)=20*=20=E2=88=9AW?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 53 +++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f86001dd..a9f55340 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -515,6 +515,59 @@ lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) (n := n) (ω := ω) B W h_regret hB hB_nonneg +/-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is +bounded by the horizon times the terminal beta value. -/ +lemma beta_sum_le_nat_mul_of_monotone + (hβ_mono : Monotone β) (hβ : ∀ t, 0 ≤ β (t + 1)) : + (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ (n : ℝ) * β n := by + calc + (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) + ≤ ∑ _t ∈ range n, β n := by + refine sum_le_sum ?_ + intro t ht + by_cases ht0 : t = 0 + · rw [if_pos ht0] + have hn_pos : 0 < n := by + simpa [ht0] using mem_range.mp ht + have hn_beta : 0 ≤ β n := by + have htime : n - 1 + 1 = n := by grind + simpa [htime] using hβ (n - 1) + exact hn_beta + · rw [if_neg ht0] + exact hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) + _ = (n : ℝ) * β n := by + simp [sum_const, nsmul_eq_mul] + +/-- Almost surely, cumulative regret is bounded by the initial gap plus +`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` +is nonnegative and monotone. -/ +lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (hW : ∀ᵐ ω ∂P, + (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √W) := by + refine regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W + (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) ?_ + hW hW_nonneg + by_cases hn : n = 0 + · simp [hn] + · have hn_pos : 0 < n := Nat.pos_of_ne_zero hn + have hn_beta : 0 ≤ β n := by + have htime : n - 1 + 1 = n := by grind + simpa [htime] using hβ (n - 1) + positivity + end LinUCB end Bandits From d8bcffb54dd219219784b74136a154bc213c6001 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 10 Jun 2026 16:03:36 -0400 Subject: [PATCH 12/88] feat(linUCB): remaining initial gap sum + cleanup --- .../Online/Bandit/Algorithms/LinUCB.lean | 47 +++++++++++++++---- 1 file changed, 39 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a9f55340..c10aabe6 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -127,6 +127,11 @@ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) +/-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ +noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) @@ -441,12 +446,12 @@ lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) - (hW : (∑ t ∈ range n, - (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : widthSqSum A reg x n ω ≤ W) (_hW_nonneg : 0 ≤ W) : regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by + rw [widthSqSum] at hW refine h_regret.trans ?_ gcongr @@ -462,8 +467,7 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, - (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) (hW_nonneg : 0 ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -503,8 +507,7 @@ lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) (hB_nonneg : 0 ≤ B) - (hW : ∀ᵐ ω ∂P, - (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) (hW_nonneg : 0 ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -538,6 +541,14 @@ lemma beta_sum_le_nat_mul_of_monotone _ = (n : ℝ) * β n := by simp [sum_const, nsmul_eq_mul] +omit [IsMarkovKernel ν] in +/-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the +horizon is zero. -/ +lemma initial_gap_sum_eq : + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) = + if n = 0 then 0 else gap ν (A 0 ω) := by + cases n <;> simp + /-- Almost surely, cumulative regret is bounded by the initial gap plus `2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` is nonnegative and monotone. -/ @@ -549,8 +560,7 @@ lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, - (∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2) ≤ W) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) (hW_nonneg : 0 ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -568,6 +578,27 @@ lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] simpa [htime] using hβ (n - 1) positivity +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` +is nonnegative and monotone. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) + (hW_nonneg : 0 ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW + hW_nonneg] with ω h_regret + simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret + end LinUCB end Bandits From 2eaf4356ad252f232a53ff37941117bef25f18a1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 09:39:02 -0400 Subject: [PATCH 13/88] feat(linUCB): widthSqSum as quadtratic with proof --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index c10aabe6..0e24b1e2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -127,6 +127,15 @@ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) +/-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that +quadratic form is nonnegative. -/ +lemma width_sq_eq_quadratic_form (a : Fin K) + (h_nonneg : 0 ≤ + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) : + width A reg x a n ω ^ 2 = + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) := by + simp [width, Real.sq_sqrt h_nonneg] + /-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 32d15ef48925cc224e8d9238bba0caf3280df5e7 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 09:44:09 -0400 Subject: [PATCH 14/88] feat(linUCB): widthSqSum as quadtratic with proof --- .../Online/Bandit/Algorithms/LinUCB.lean | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 0e24b1e2..a95a966b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -141,6 +141,27 @@ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 +/-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive +time quadratic form is nonnegative. -/ +lemma widthSqSum_eq_sum_quadratic_form + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) : + widthSqSum A reg x n ω = + ∑ t ∈ range n, + if t = 0 then 0 else + dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) := by + rw [widthSqSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0] + rw [if_neg ht0] + exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) + (n := t) (ω := ω) (h_nonneg t ht ht0) + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) From 7bfa214b8892ff431b2bfadf7c458400bc21e183 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 10:56:51 -0400 Subject: [PATCH 15/88] feat(linUCB): bridge from a quadratic-form sum bound to a widthSqSum bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a95a966b..858473e7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -162,6 +162,22 @@ lemma widthSqSum_eq_sum_quadratic_form exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) (n := t) (ω := ω) (h_nonneg t ht ht0) +/-- A quadratic-form sum bound implies the corresponding bound on `widthSqSum`. This is the shape +expected from a later elliptical-potential argument. -/ +lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) + (h_quad_le : + (∑ t ∈ range n, + if t = 0 then 0 else + dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + exact h_quad_le + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) From ae8310916cb51d91f3808c010f44c56add8bd5ba Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 11:00:55 -0400 Subject: [PATCH 16/88] feat(linUCB): clean up repeated term with quadraticWidthSum --- .../Online/Bandit/Algorithms/LinUCB.lean | 22 +++++++++---------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 858473e7..e058569e 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -141,18 +141,22 @@ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 +/-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ +noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else + dotProduct (x (A t ω)) + (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ dotProduct (x (A t ω)) (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) : - widthSqSum A reg x n ω = - ∑ t ∈ range n, - if t = 0 then 0 else - dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) := by - rw [widthSqSum] + widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by + rw [widthSqSum, quadraticWidthSum] refine sum_congr rfl ?_ intro t ht by_cases ht0 : t = 0 @@ -168,11 +172,7 @@ lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ dotProduct (x (A t ω)) (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) - (h_quad_le : - (∑ t ∈ range n, - if t = 0 then 0 else - dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) ≤ W) : + (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : widthSqSum A reg x n ω ≤ W := by rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg] From 5c922c7618f2864858bba59b0fa91de159bcbe21 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 11:27:25 -0400 Subject: [PATCH 17/88] feat(linUCB): lint clean up --- .../Online/Bandit/Algorithms/LinUCB.lean | 58 +++++++++---------- 1 file changed, 28 insertions(+), 30 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e058569e..d4d14d1b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -29,24 +29,30 @@ section Algorithm namespace LinUCB +/-- Feature vectors for finite-dimensional linear bandits. -/ abbrev Feature (d : ℕ) := Fin d → ℝ +/-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) +/-- History-level response vector for LinUCB. -/ noncomputable def responseVector' (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := ∑ s : Iic n, (h s).2 • x (h s).1 +/-- History-level regularized least-squares estimate. -/ noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) +/-- History-level estimated reward of an arm. -/ noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := dotProduct (thetaHat' reg x n h) (x a) +/-- History-level elliptical confidence width of an arm. -/ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) @@ -65,7 +71,6 @@ open Classical in /-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK measurableArgmax (fun h a ↦ index' reg β x n h a) h @@ -75,7 +80,7 @@ lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) (n : ℕ) : - Measurable (nextArm hK reg β x h_index n) := by + Measurable (nextArm hK reg β x n) := by have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact measurable_measurableArgmax fun a ↦ h_index n a @@ -86,7 +91,7 @@ noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → LinUCB.Feature d) (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : Algorithm (Fin K) ℝ := - detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ end Algorithm @@ -107,6 +112,12 @@ noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) +/-- The design matrix update after observing one additional action. -/ +lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designMatrix A reg x (n + 1) ω = + designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by + simp [designMatrix, sum_range_succ, add_assoc] + /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := @@ -237,7 +248,7 @@ lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (n : ℕ) : A (n + 1) =ᵐ[P] - fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact h.action_detAlgorithm_ae_eq n @@ -246,7 +257,7 @@ lemma arm_ae_all_eq [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : ∀ᵐ ω ∂P, ∀ n, A (n + 1) ω = - nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by simp_rw [ae_all_iff] exact fun n ↦ arm_ae_eq_linUCBNextArm h n @@ -493,7 +504,7 @@ lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) (hW : widthSqSum A reg x n ω ≤ W) - (_hW_nonneg : 0 ≤ W) : + : regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by @@ -513,8 +524,7 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + @@ -523,7 +533,7 @@ lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with ω h_regret hWω exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω hW_nonneg + (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω omit [IsMarkovKernel ν] in /-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square @@ -534,7 +544,7 @@ lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (_hB_nonneg : 0 ≤ B) : + : regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by refine h_regret.trans ?_ @@ -552,17 +562,15 @@ lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (hB_nonneg : 0 ≤ B) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW - hW_nonneg] with ω h_regret + ] with ω h_regret exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) - (n := n) (ω := ω) B W h_regret hB hB_nonneg + (n := n) (ω := ω) B W h_regret hB /-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is bounded by the horizon times the terminal beta value. -/ @@ -606,23 +614,14 @@ lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√((n : ℝ) * β n) * √W) := by - refine regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) + exact regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W - (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) ?_ - hW hW_nonneg - by_cases hn : n = 0 - · simp [hn] - · have hn_pos : 0 < n := Nat.pos_of_ne_zero hn - have hn_beta : 0 ≤ β n := by - have htime : n - 1 + 1 = n := by grind - simpa [htime] using hβ (n - 1) - positivity + (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) hW /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` @@ -635,14 +634,13 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) - (hW_nonneg : 0 ≤ W) : + (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW - hW_nonneg] with ω h_regret + ] with ω h_regret simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret end LinUCB From d2bc9e263c059e8ec3ffdd51eff02c0e10de3a15 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 13:09:21 -0400 Subject: [PATCH 18/88] feat(linUCB): design matrix is just the regularization matrix --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index d4d14d1b..dcd7988f 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -112,6 +112,11 @@ noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) +/-- The initial design matrix before any actions are included. -/ +lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + designMatrix A reg x 0 ω = reg • 1 := by + simp [designMatrix] + /-- The design matrix update after observing one additional action. -/ lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : designMatrix A reg x (n + 1) ω = From 0b0a9045fe7ccb7068c3f11f73f16142bbc6d12a Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 14:18:46 -0400 Subject: [PATCH 19/88] feat(linUCB): proof for initial response vector --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index dcd7988f..8cb15783 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -128,6 +128,12 @@ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := ∑ s ∈ range n, R s ω • x (A s ω) +/-- The initial response vector before any rewards are included. -/ +lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (ω : Ω) : + responseVector A R x 0 ω = 0 := by + simp [responseVector] + /-- The process-level regularized least-squares estimate. -/ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := From f1da71a20b6026161946f8ca61f3512f01b39377 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 11 Jun 2026 15:18:40 -0400 Subject: [PATCH 20/88] feat(linUCB): responseVector is the LinUCB reward-feature accumulator --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 8cb15783..1f8a5872 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -134,6 +134,13 @@ lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) responseVector A R x 0 ω = 0 := by simp [responseVector] +/-- The response-vector update after observing one additional reward. -/ +lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + responseVector A R x (n + 1) ω = + responseVector A R x n ω + R n ω • x (A n ω) := by + simp [responseVector, sum_range_succ] + /-- The process-level regularized least-squares estimate. -/ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := From a96beab49491a715418f376251be0bad9ad806fc Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 10:19:15 -0400 Subject: [PATCH 21/88] feat(linUCB): estimated reward for any arm is zero --- .../Online/Bandit/Algorithms/LinUCB.lean | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 1f8a5872..fa354124 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -146,11 +146,24 @@ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) +/-- The initial least-squares estimate is zero because no reward-feature observations have been +included yet. -/ +lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + thetaHat A R reg x 0 ω = 0 := by + simp [thetaHat, responseVector_zero] + /-- The process-level estimated linear reward. -/ noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := dotProduct (thetaHat A R reg x n ω) (x a) +/-- The initial estimated reward is zero for every arm. -/ +lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + estimatedReward A R reg x a 0 ω = 0 := by + simp [estimatedReward, thetaHat_zero] + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := From 5c633f2898d723aaa4918f109d58eeff93fc7dd9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 10:33:01 -0400 Subject: [PATCH 22/88] =?UTF-8?q?feat(linUCB):=20at=20time=200,=20LinUCB?= =?UTF-8?q?=E2=80=99s=20optimistic=20index=20for=20an=20arm=20is=20just=20?= =?UTF-8?q?its=20uncertainty=20bonus?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index fa354124..373c8d05 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -226,6 +226,13 @@ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (n : ℕ) (ω : Ω) : ℝ := estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω +/-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is +zero. -/ +lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by + simp [index, estimatedReward_zero] + lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = From 523728e4b3d153076d8fd6d10e839235c1bc8c65 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 10:37:06 -0400 Subject: [PATCH 23/88] feat(linUCB): at 0 time, the LinUCB width is computed using only the initial design matrix --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 373c8d05..abd220d7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -169,6 +169,13 @@ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) +/-- The initial width is the quadratic form induced by the inverse regularized identity. -/ +lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + width A reg x a 0 ω = + √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by + simp [width, designMatrix_zero] + /-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that quadratic form is nonnegative. -/ lemma width_sq_eq_quadratic_form (a : Fin K) From c47c7e1f8362094ceff53fc71785a48b19233756 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:08:58 -0400 Subject: [PATCH 24/88] feat(linUCB): combining index_zero, width_zero --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index abd220d7..75c1153d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -240,6 +240,14 @@ lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by simp [index, estimatedReward_zero] +/-- At time zero, the LinUCB index is the confidence schedule times the initial quadratic-form +width. -/ +lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + index A R reg β x a 0 ω = + √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by + simp [index_zero, width_zero] + lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = From 84fd6fd677a71da0b5eb23a2054262c0bda50fc3 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:13:52 -0400 Subject: [PATCH 25/88] feat(linUCB): the accumulated squared-width term is zero at horizon 0 --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 75c1153d..eb1d2077 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -190,6 +190,12 @@ noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 +/-- No positive-time widths are accumulated at horizon zero. -/ +lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + widthSqSum A reg x 0 ω = 0 := by + simp [widthSqSum] + /-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 33ee65ab07d6f1cfd4700734623d84eda9243d51 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:15:03 -0400 Subject: [PATCH 26/88] feat(linUCB): proof for quadraticWidthSum --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index eb1d2077..9849935e 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -204,6 +204,12 @@ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) dotProduct (x (A t ω)) (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) +/-- No positive-time quadratic width forms are accumulated at horizon zero. -/ +lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + quadraticWidthSum A reg x 0 ω = 0 := by + simp [quadraticWidthSum] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From fe9b9f1de5b969ae9f47cbfca6b5470c201e0ce2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:20:56 -0400 Subject: [PATCH 27/88] feat(linUCB): when the horizon advances from n to n + 1, the accumulated squared-width sum equals the old sum plus the new final term at time n --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 9849935e..979b4ad5 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -196,6 +196,14 @@ lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) widthSqSum A reg x 0 ω = 0 := by simp [widthSqSum] +/-- Advancing the horizon adds the next positive-time squared width term. -/ +lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + + (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by + simp [widthSqSum, sum_range_succ] + /-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 1d1725ea6085610e1cfab7292533f56746834686 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:28:05 -0400 Subject: [PATCH 28/88] feat(linUCB): when the horizon advances from n to n + 1, the accumulated quadratic squared-width sum equals the old quadratic sum plus the new final term at time n --- .../Online/Bandit/Algorithms/LinUCB.lean | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 979b4ad5..e9d9b603 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -218,6 +218,16 @@ lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) quadraticWidthSum A reg x 0 ω = 0 := by simp [quadraticWidthSum] +/-- Advancing the horizon adds the next positive-time quadratic width form. -/ +lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + if n = 0 then 0 else + dotProduct (x (A n ω)) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + simp [quadraticWidthSum, sum_range_succ] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From c86ad3e7ae68582bbbc7129e9c155d23c499eebf Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:31:23 -0400 Subject: [PATCH 29/88] feat(linUCB): proof for widthSqSum_succ when n not equal to 0 --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e9d9b603..e5bdf266 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -204,6 +204,13 @@ lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by simp [widthSqSum, sum_range_succ] +/-- At positive times, advancing the horizon adds the selected arm's squared width. -/ +lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + width A reg x (A n ω) n ω ^ 2 := by + simp [widthSqSum_succ, hn] + /-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From 3f94ae7a2c472728de2bf4c33afaf5a242f338a3 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:34:19 -0400 Subject: [PATCH 30/88] feat(linUCB): proof for quadraticWidth_succ when n not equal to 0 --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e5bdf266..1824ab82 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -235,6 +235,15 @@ lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by simp [quadraticWidthSum, sum_range_succ] +/-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ +lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + dotProduct (x (A n ω)) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + simp [quadraticWidthSum_succ, hn] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From 94c32a3e07e4f62696c5ccb4a54de2234fafd68e Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:41:20 -0400 Subject: [PATCH 31/88] feat(linUCB): if the two accumulators agree at time n, then they also agree at time n + 1, as long as the new quadratic form is nonnegative --- .../Online/Bandit/Algorithms/LinUCB.lean | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 1824ab82..fd7a1679 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -244,6 +244,20 @@ lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by simp [quadraticWidthSum_succ, hn] +/-- If the squared-width and quadratic-form accumulators agree through a positive time and the +next quadratic form is nonnegative, then they still agree after adding the next term. -/ +lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) + (h_nonneg : 0 ≤ dotProduct (x (A n ω)) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω)))) : + widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by + rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, + quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hn, h_eq] + rw [width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A n ω) + (n := n) (ω := ω) h_nonneg] + /-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form From ee5c82bdbb63f2e8a14e92a99a933d25ec65e6ef Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 12 Jun 2026 11:52:08 -0400 Subject: [PATCH 32/88] feat(linUCB): refactor --- .../Online/Bandit/Algorithms/LinUCB.lean | 46 ++++++++++--------- 1 file changed, 24 insertions(+), 22 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index fd7a1679..97396fe0 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -164,25 +164,35 @@ lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) estimatedReward A R reg x a 0 ω = 0 := by simp [estimatedReward, thetaHat_zero] +/-- The quadratic form `x_aᵀ V_n⁻¹ x_a` underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) + +/-- The initial width quadratic form is induced by the inverse regularized identity. -/ +lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + widthQuadraticForm A reg x a 0 ω = + dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by + simp [widthQuadraticForm, designMatrix_zero] + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + √(widthQuadraticForm A reg x a n ω) /-- The initial width is the quadratic form induced by the inverse regularized identity. -/ lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : width A reg x a 0 ω = √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [width, designMatrix_zero] + simp [width, widthQuadraticForm_zero] /-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that quadratic form is nonnegative. -/ lemma width_sq_eq_quadratic_form (a : Fin K) - (h_nonneg : 0 ≤ - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) : - width A reg x a n ω ^ 2 = - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) := by + (h_nonneg : 0 ≤ widthQuadraticForm A reg x a n ω) : + width A reg x a n ω ^ 2 = widthQuadraticForm A reg x a n ω := by simp [width, Real.sq_sqrt h_nonneg] /-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ @@ -215,9 +225,7 @@ lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ∑ t ∈ range n, - if t = 0 then 0 else - dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω))) + if t = 0 then 0 else widthQuadraticForm A reg x (A t ω) t ω /-- No positive-time quadratic width forms are accumulated at horizon zero. -/ lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -230,18 +238,14 @@ lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : quadraticWidthSum A reg x (n + 1) ω = quadraticWidthSum A reg x n ω + - if n = 0 then 0 else - dotProduct (x (A n ω)) - (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + if n = 0 then 0 else widthQuadraticForm A reg x (A n ω) n ω := by simp [quadraticWidthSum, sum_range_succ] /-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - dotProduct (x (A n ω)) - (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω))) := by + quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by simp [quadraticWidthSum_succ, hn] /-- If the squared-width and quadratic-form accumulators agree through a positive time and the @@ -249,8 +253,7 @@ next quadratic form is nonnegative, then they still agree after adding the next lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) - (h_nonneg : 0 ≤ dotProduct (x (A n ω)) - (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x (A n ω)))) : + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) @@ -262,8 +265,7 @@ lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) time quadratic form is nonnegative. -/ lemma widthSqSum_eq_sum_quadratic_form (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) : + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by rw [widthSqSum, quadraticWidthSum] refine sum_congr rfl ?_ @@ -279,8 +281,7 @@ lemma widthSqSum_eq_sum_quadratic_form expected from a later elliptical-potential argument. -/ lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ dotProduct (x (A t ω)) - (Matrix.mulVec (designMatrix A reg x t ω)⁻¹ (x (A t ω)))) + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : widthSqSum A reg x n ω ≤ W := by rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) @@ -346,7 +347,8 @@ lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + simp [width, widthQuadraticForm, width', + designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : From e75d4ca5e0384dea8a60a8804960edac6bfde9c9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 11:34:38 -0400 Subject: [PATCH 33/88] feat(linUCB): history-level companion to widthQuadraticForm --- .../Online/Bandit/Algorithms/LinUCB.lean | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 97396fe0..4227a4d2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -52,10 +52,15 @@ noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := dotProduct (thetaHat' reg x n h) (x a) +/-- History-level quadratic form underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) + /-- History-level elliptical confidence width of an arm. -/ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + √(widthQuadraticForm' reg x n h a) /-- LinUCB optimistic index of an arm. @@ -344,11 +349,18 @@ lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] +lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthQuadraticForm A reg x a n ω = + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [widthQuadraticForm, widthQuadraticForm', + designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, widthQuadraticForm, width', - designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n + ω hn] lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : From 9c2a1cc5e16e6bb1256b2fa3dc9b624e4d7346ec Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:06:22 -0400 Subject: [PATCH 34/88] feat(linUCB): history-level analogue of the existing process-level square-width lemma --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 4227a4d2..9d018a2c 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -62,6 +62,14 @@ noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := √(widthQuadraticForm' reg x n h a) +/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic +form is nonnegative. -/ +lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) + (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : + width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by + simp [width', Real.sq_sqrt h_nonneg] + /-- LinUCB optimistic index of an arm. The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` From 9aea4326f1f765fabf445a241f24c933238e2d38 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:09:56 -0400 Subject: [PATCH 35/88] feat(linUCB): transports the nonnegativity condition across the process/history bridge --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 9d018a2c..04e5facc 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -364,6 +364,14 @@ lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Featu simp [widthQuadraticForm, widthQuadraticForm', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] +/-- At positive process times, nonnegativity of the process-level width quadratic form is +equivalent to nonnegativity of the matching history-level width quadratic form. -/ +lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + 0 ≤ widthQuadraticForm A reg x a n ω ↔ + 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] + lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by From a943c01a84fd7e2da8956362117a189a4fa9993b Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:18:21 -0400 Subject: [PATCH 36/88] feat(linUCB): for positive process time n, the process-level squared width can be rewritten directly as the matching history-level quadratic form --- .../Online/Bandit/Algorithms/LinUCB.lean | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 04e5facc..77249f7f 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -378,6 +378,18 @@ lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] +/-- At positive process times, squaring the process-level width recovers the matching history-level +quadratic form when that history-level quadratic form is nonnegative. -/ +lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_nonneg : + 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a) : + width A reg x a n ω ^ 2 = + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + rw [width_eq_width' (A := A) (R := R) reg x a n ω hn] + exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a + h_nonneg + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = From 49ad7c992ac0528a68d83a86d20daa60f5dbf7c0 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 12:22:26 -0400 Subject: [PATCH 37/88] feat(linUCB): widthSqSum(n + 1)=widthSqSum(n)+history-level quadratic form for the selected arm --- .../Online/Bandit/Algorithms/LinUCB.lean | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 77249f7f..38d2f204 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -390,6 +390,18 @@ lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a h_nonneg +/-- At positive process times, advancing `widthSqSum` adds the matching history-level quadratic +form when that history-level quadratic form is nonnegative. -/ +lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_nonneg : + 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] + rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = From 54db96b4a2b9d1342b95968bfec1354a28f945a1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:29:15 -0400 Subject: [PATCH 38/88] feat(linUCB): sums the history-level quadratic forms --- .../Online/Bandit/Algorithms/LinUCB.lean | 119 ++++++++++++++++++ 1 file changed, 119 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 38d2f204..3dae62f2 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -402,6 +402,103 @@ lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feat rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] +/-- At positive process times, advancing `quadraticWidthSum` adds the matching history-level +quadratic form. -/ +lemma quadraticWidthSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + rw [quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hn] + rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn] + +/-- The history-level quadratic-form accumulator aligned with process times. + +The term at process time `t = 0` is set to zero, matching the convention used by `widthSqSum` and +`quadraticWidthSum`. At positive process time `t`, the history available to LinUCB is +`IsAlgEnvSeq.hist A R (t - 1) ω`. -/ +noncomputable def historyQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) + +/-- No positive-time history-level quadratic width forms are accumulated at horizon zero. -/ +lemma historyQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + historyQuadraticWidthSum A R reg x 0 ω = 0 := by + simp [historyQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time history-level quadratic width form. -/ +lemma historyQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + historyQuadraticWidthSum A R reg x (n + 1) ω = + historyQuadraticWidthSum A R reg x n ω + + if n = 0 then 0 else + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + simp [historyQuadraticWidthSum, sum_range_succ] + +/-- At positive process times, advancing the history-level quadratic accumulator adds the selected +arm's history-level quadratic width form. -/ +lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + historyQuadraticWidthSum A R reg x (n + 1) ω = + historyQuadraticWidthSum A R reg x n ω + + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by + simp [historyQuadraticWidthSum_succ, hn] + +/-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with +the same process times. -/ +lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) : + quadraticWidthSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by + rw [quadraticWidthSum, historyQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0 + +/-- The squared-width accumulator equals the history-level quadratic-form accumulator whenever the +positive-time history-level quadratic forms are nonnegative. -/ +lemma widthSqSum_eq_historyQuadraticWidthSum + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) : + widthSqSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by + have h_process_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + intro t ht ht0 + exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).2 (h_nonneg t ht ht0) + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_process_nonneg] + exact quadraticWidthSum_eq_historyQuadraticWidthSum (A := A) (R := R) reg x n ω + +/-- A bound on the history-level quadratic-form accumulator implies the corresponding bound on +`widthSqSum`, provided the positive-time history-level quadratic forms are nonnegative. -/ +lemma widthSqSum_le_of_history_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_hist_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_historyQuadraticWidthSum (A := A) (R := R) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + exact h_hist_le + +omit [IsProbabilityMeasure P] in +/-- Almost surely, a history-level quadratic-form bound gives the `widthSqSum` bound consumed by +the regret chain. -/ +lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_hist_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_nonneg, h_hist_le] with ω h_nonnegω h_hist_leω + exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = @@ -810,6 +907,28 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin ] with ω h_regret simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever a history-level quadratic-form bound supplies the future +elliptical-potential input. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (hW : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) + end LinUCB end Bandits From 2a6a0cd63a6f6a07e977c000faf295e4759d602c Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:36:04 -0400 Subject: [PATCH 39/88] feat(linUCB): history-level quadratic-width related lemmas --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 3dae62f2..be4953bb 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -499,6 +499,39 @@ lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The pointwise input expected from a future elliptical-potential argument. + +It packages the two facts needed to turn a history-level quadratic-width estimate into the +`widthSqSum` estimate used by the regret chain: + +* each positive-time quadratic width form is nonnegative; +* their history-level accumulated sum is bounded by `W`. -/ +def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := + (∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ + historyQuadraticWidthSum A R reg x n ω ≤ W + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the +regret chain. -/ +lemma widthSqSum_le_of_history_quadratic_width_bound {W : ℝ} + (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_bound.1 h_bound.2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged history-level quadratic-width input implies the `widthSqSum` bound +consumed by the regret chain. -/ +lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} + (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_bound] with ω h_boundω + exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) (W := W) h_boundω + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = @@ -929,6 +962,29 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [No (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the packaged history-level quadratic-width input holds almost +surely. + +This is the theorem a future elliptical-potential lemma should feed into directly. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) + end LinUCB end Bandits From 38e20e1b2fcfdc4be34486ae3006f62467520d43 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:40:01 -0400 Subject: [PATCH 40/88] feat(linUCB): capped version of the quadratic-width accumulator: --- .../Online/Bandit/Algorithms/LinUCB.lean | 121 ++++++++++++++++++ 1 file changed, 121 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index be4953bb..e7b53c71 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -448,6 +448,58 @@ lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by simp [historyQuadraticWidthSum_succ, hn] +/-- The capped history-level quadratic-form accumulator aligned with process times. + +This is the accumulator shape that commonly appears in elliptical-potential statements: +each positive-time quadratic width form is capped at `1`. -/ +noncomputable def historyCappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else + min 1 (widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + +/-- No positive-time capped history-level quadratic width forms are accumulated at horizon zero. -/ +lemma historyCappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + historyCappedQuadraticWidthSum A R reg x 0 ω = 0 := by + simp [historyCappedQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time capped history-level quadratic width form. -/ +lemma historyCappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + historyCappedQuadraticWidthSum A R reg x (n + 1) ω = + historyCappedQuadraticWidthSum A R reg x n ω + + if n = 0 then 0 else + min 1 + (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by + simp [historyCappedQuadraticWidthSum, sum_range_succ] + +/-- At positive process times, advancing the capped history-level quadratic accumulator adds the +selected arm's capped history-level quadratic width form. -/ +lemma historyCappedQuadraticWidthSum_succ_of_ne_zero + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + historyCappedQuadraticWidthSum A R reg x (n + 1) ω = + historyCappedQuadraticWidthSum A R reg x n ω + + min 1 + (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by + simp [historyCappedQuadraticWidthSum_succ, hn] + +/-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and +capped history-level accumulators agree. -/ +lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) : + historyQuadraticWidthSum A R reg x n ω = + historyCappedQuadraticWidthSum A R reg x n ω := by + rw [historyQuadraticWidthSum, historyCappedQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact (min_eq_right (h_le_one t ht ht0)).symm + /-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with the same process times. -/ lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) @@ -513,6 +565,75 @@ def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ historyQuadraticWidthSum A R reg x n ω ≤ W +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Build the packaged history-level quadratic-width input from its two component facts. -/ +lemma historyQuadraticWidthBound_of_nonneg_and_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_sum_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : + HistoryQuadraticWidthBound A R reg x n ω W := by + exact ⟨h_nonneg, h_sum_le⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged history-level quadratic-width input is monotone in the numeric bound. -/ +lemma historyQuadraticWidthBound_mono {W W' : ℝ} + (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : + HistoryQuadraticWidthBound A R reg x n ω W' := by + exact ⟨h_bound.1, h_bound.2.trans hW⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, build the packaged history-level quadratic-width input from its two component +facts. -/ +lemma historyQuadraticWidthBound_ae_of_nonneg_and_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_sum_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by + filter_upwards [h_nonneg, h_sum_le] with ω h_nonnegω h_sum_leω + exact historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_sum_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged history-level quadratic-width input is monotone in the numeric +bound. -/ +lemma historyQuadraticWidthBound_ae_mono {W W' : ℝ} + (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : + ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W' := by + filter_upwards [h_bound] with ω h_boundω + exact historyQuadraticWidthBound_mono (A := A) (R := R) (reg := reg) (x := x) + (n := n) (ω := ω) h_boundω hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped quadratic-width sum bound gives the packaged history-level input whenever every +positive-time quadratic width form is nonnegative and at most `1`. -/ +lemma historyQuadraticWidthBound_of_capped_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + HistoryQuadraticWidthBound A R reg x n ω W := by + refine historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg ?_ + rw [historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) h_le_one] + exact h_capped_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped quadratic-width sum bound gives the packaged history-level input +whenever every positive-time quadratic width form is almost surely nonnegative and at most `1`. -/ +lemma historyQuadraticWidthBound_ae_of_capped_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_capped_le] with + ω h_nonnegω h_le_oneω h_capped_leω + exact historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 1bb706d1e6b68329b1d65786785739115d1edb9b Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:42:30 -0400 Subject: [PATCH 41/88] feat(linUCB): bridge from the capped elliptical-potential-style quantity to the regret chain --- .../Online/Bandit/Algorithms/LinUCB.lean | 60 +++++++++++++++++++ 1 file changed, 60 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e7b53c71..cee58c4b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -653,6 +653,39 @@ lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) (x := x) (n := n) (ω := ω) (W := W) h_boundω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped history-level quadratic-width sum bound implies the `widthSqSum` bound consumed by +the regret chain, provided the positive-time quadratic width forms are nonnegative and at most +`1`. -/ +lemma widthSqSum_le_of_capped_history_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) (W := W) + (historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonneg h_le_one h_capped_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped history-level quadratic-width sum bound implies the `widthSqSum` bound +consumed by the regret chain, provided the positive-time quadratic width forms are almost surely +nonnegative and at most `1`. -/ +lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + h_capped_le) + lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : index A R reg β x a n ω = @@ -1106,6 +1139,33 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bou (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever a capped history-level quadratic-width sum bound holds almost +surely and every positive-time quadratic width form is almost surely nonnegative and at most `1`. + +This is the direct interface for the common capped form of the elliptical-potential lemma. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) + (hW : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) + end LinUCB end Bandits From 24b5503407f99354b203f1e7728ce1cdb16bebc9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Mon, 15 Jun 2026 16:55:05 -0400 Subject: [PATCH 42/88] feat(linUCB): process-level capped accumulator --- .../Online/Bandit/Algorithms/LinUCB.lean | 117 ++++++++++++++++++ 1 file changed, 117 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index cee58c4b..e2b02f96 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -261,6 +261,47 @@ lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by simp [quadraticWidthSum_succ, hn] +/-- The accumulated capped quadratic forms corresponding to the positive-time LinUCB widths. -/ +noncomputable def cappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω) + +/-- No positive-time capped quadratic width forms are accumulated at horizon zero. -/ +lemma cappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + cappedQuadraticWidthSum A reg x 0 ω = 0 := by + simp [cappedQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time capped quadratic width form. -/ +lemma cappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + cappedQuadraticWidthSum A reg x (n + 1) ω = + cappedQuadraticWidthSum A reg x n ω + + if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by + simp [cappedQuadraticWidthSum, sum_range_succ] + +/-- At positive times, advancing the horizon adds the selected arm's capped quadratic width form. -/ +lemma cappedQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + cappedQuadraticWidthSum A reg x (n + 1) ω = + cappedQuadraticWidthSum A reg x n ω + min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by + simp [cappedQuadraticWidthSum_succ, hn] + +/-- If every positive-time process-level quadratic width form is at most `1`, then the uncapped +and capped process-level quadratic-width accumulators agree. -/ +lemma quadraticWidthSum_eq_cappedQuadraticWidthSum + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + quadraticWidthSum A reg x n ω = cappedQuadraticWidthSum A reg x n ω := by + rw [quadraticWidthSum, cappedQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact (min_eq_right (h_le_one t ht ht0)).symm + /-- If the squared-width and quadratic-form accumulators agree through a positive time and the next quadratic form is nonnegative, then they still agree after adding the next term. -/ lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -301,6 +342,38 @@ lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} (n := n) (ω := ω) h_nonneg] exact h_quad_le +/-- A capped process-level quadratic-form sum bound implies the corresponding bound on +`widthSqSum`, provided the positive-time process-level quadratic forms are nonnegative and at most +`1`. -/ +lemma widthSqSum_le_of_capped_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_capped_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + rw [quadraticWidthSum_eq_cappedQuadraticWidthSum (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_le_one] + exact h_capped_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped process-level quadratic-form sum bound implies the corresponding bound +on `widthSqSum`, provided the positive-time process-level quadratic forms are almost surely +nonnegative and at most `1`. -/ +lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_nonneg, h_le_one, h_capped_le] with + ω h_nonnegω h_le_oneω h_capped_leω + exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) @@ -485,6 +558,21 @@ lemma historyCappedQuadraticWidthSum_succ_of_ne_zero (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by simp [historyCappedQuadraticWidthSum_succ, hn] +/-- The process-level capped quadratic-width accumulator equals the history-level capped +accumulator aligned with the same process times. -/ +lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + cappedQuadraticWidthSum A reg x n ω = + historyCappedQuadraticWidthSum A R reg x n ω := by + rw [cappedQuadraticWidthSum, historyCappedQuadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact congrArg (fun q : ℝ ↦ min 1 q) + (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) + /-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and capped history-level accumulators agree. -/ lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum @@ -1166,6 +1254,35 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bo (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever a capped process-level quadratic-width sum bound holds almost +surely and every positive-time process-level quadratic width form is almost surely nonnegative and +at most `1`. + +This is the direct interface for an elliptical-potential lemma stated using the process-level design +matrices `designMatrix A reg x t ω`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) + end LinUCB end Bandits From c511de3ae55f53ceb0a3e6a636a68d800ea00c9e Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 09:29:42 -0400 Subject: [PATCH 43/88] feat(linUCB): process/history transport lemmas --- .../Online/Bandit/Algorithms/LinUCB.lean | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e2b02f96..5533b4ea 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -445,6 +445,78 @@ lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] +/-- At positive process times, the process-level quadratic width form is at most `1` iff the +matching history-level quadratic width form is at most `1`. -/ +lemma widthQuadraticForm_le_one_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthQuadraticForm A reg x a n ω ≤ 1 ↔ + widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a ≤ 1 := by + rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] + +/-- The all-positive-times process-level nonnegativity assumption is equivalent to the matching +history-level nonnegativity assumption. -/ +lemma widthQuadraticForm_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) : + (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ + ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by + constructor + · intro h t ht ht0 + exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).1 (h t ht ht0) + · intro h t ht ht0 + exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).2 (h t ht ht0) + +/-- The all-positive-times process-level `≤ 1` assumption is equivalent to the matching +history-level `≤ 1` assumption. -/ +lemma widthQuadraticForm_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) : + (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ + ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by + constructor + · intro h t ht ht0 + exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).1 (h t ht ht0) + · intro h t ht ht0 + exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x + (A t ω) t ω ht0).2 (h t ht ht0) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, process-level all-positive-times nonnegativity is equivalent to the matching +history-level nonnegativity assumption. -/ +lemma widthQuadraticForm_ae_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) : + (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by + constructor + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).1 hω + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).2 hω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the process-level all-positive-times `≤ 1` assumption is equivalent to the +matching history-level `≤ 1` assumption. -/ +lemma widthQuadraticForm_ae_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) : + (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by + constructor + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).1 hω + · intro h + filter_upwards [h] with ω hω + exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).2 hω + lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by @@ -573,6 +645,32 @@ lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) exact congrArg (fun q : ℝ ↦ min 1 q) (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) +/-- A process-level capped quadratic-width sum bound is equivalent to the matching history-level +capped quadratic-width sum bound. -/ +lemma cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : + cappedQuadraticWidthSum A reg x n ω ≤ W ↔ + historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by + rw [cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) + reg x n ω] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a process-level capped quadratic-width sum bound is equivalent to the matching +history-level capped quadratic-width sum bound. -/ +lemma cappedQuadraticWidthSum_ae_le_iff_historyCappedQuadraticWidthSum_ae_le + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (W : ℝ) : + (∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) ↔ + ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by + constructor + · intro h + filter_upwards [h] with ω hω + exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le + (A := A) (R := R) reg x n ω W).1 hω + · intro h + filter_upwards [h] with ω hω + exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le + (A := A) (R := R) reg x n ω W).2 hω + /-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and capped history-level accumulators agree. -/ lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum From 85c4425a07627455d7b5650d882a5c106f2628fd Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 09:34:21 -0400 Subject: [PATCH 44/88] feat(linUCB): compact regret theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 100 ++++++++++++++++++ 1 file changed, 100 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 5533b4ea..45657e6d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -374,6 +374,83 @@ lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The process-level capped quadratic-width input expected from an elliptical-potential argument. + +It packages the three facts needed to turn a capped process-level quadratic-width estimate into the +`widthSqSum` estimate used by the regret chain: + +* each positive-time process-level quadratic width form is nonnegative; +* each positive-time process-level quadratic width form is at most `1`; +* their capped process-level accumulated sum is bounded by `W`. -/ +def CappedQuadraticWidthBound (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := + (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ∧ + (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ∧ + cappedQuadraticWidthSum A reg x n ω ≤ W + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Build the packaged process-level capped quadratic-width input from its component facts. -/ +lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_sum_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : + CappedQuadraticWidthBound A reg x n ω W := by + exact ⟨h_nonneg, h_le_one, h_sum_le⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ +lemma cappedQuadraticWidthBound_mono {W W' : ℝ} + (h_bound : CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : + CappedQuadraticWidthBound A reg x n ω W' := by + exact ⟨h_bound.1, h_bound.2.1, h_bound.2.2.trans hW⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, build the packaged process-level capped quadratic-width input from its component +facts. -/ +lemma cappedQuadraticWidthBound_ae_of_nonneg_le_one_and_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_sum_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_sum_le] with + ω h_nonnegω h_le_oneω h_sum_leω + exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_sum_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged process-level capped quadratic-width input is monotone in the +numeric bound. -/ +lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W' := by + filter_upwards [h_bound] with ω h_boundω + exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) h_boundω hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed +by the regret chain. -/ +lemma widthSqSum_le_of_capped_quadratic_width_bound {W : ℝ} + (h_bound : CappedQuadraticWidthBound A reg x n ω W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_bound.1 h_bound.2.1 h_bound.2.2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged process-level capped quadratic-width input implies the `widthSqSum` +bound consumed by the regret chain. -/ +lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_bound] with ω h_boundω + exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) (W := W) h_boundω + /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) @@ -1381,6 +1458,29 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the packaged process-level capped quadratic-width input holds +almost surely. + +This is the compact theorem a process-level elliptical-potential lemma should feed into directly. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W + (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) + (x := x) (n := n) (P := P) (W := W) h_bound) + end LinUCB end Bandits From 2a7e5b101bf3420a472148c8d4c582891c1252cd Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:15:42 -0400 Subject: [PATCH 45/88] feat(linUCB): direct regret theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 107 ++++++++++++++++++ 1 file changed, 107 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 45657e6d..f0e087a7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -374,6 +374,46 @@ lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Determinant of the process-level LinUCB design matrix. -/ +noncomputable def designDet (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + Matrix.det (designMatrix A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial design determinant is the determinant of the regularized identity. -/ +lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by + simp [designDet, designMatrix_zero] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ +noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + designDet A reg x n ω / designDet A reg x 0 ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the determinant ratio is `1` when the initial design determinant is nonzero. -/ +lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + designDetRatio A reg x 0 ω = 1 := by + simp [designDetRatio, hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The log-determinant expression that appears in the elliptical-potential lemma. -/ +noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + 2 * Real.log (designDetRatio A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the log-determinant potential is zero when the initial design determinant is +nonzero. -/ +lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + ellipticalPotential A reg x 0 ω = 0 := by + simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -432,6 +472,38 @@ lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_boundω hW +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped-sum bound by the log-determinant potential, together with a constant bound on that +potential, gives the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_of_ellipticalPotential_le_bound {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ellipticalPotential A reg x n ω ≤ W) : + CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonneg h_le_one (h_elliptical.trans h_potential_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped-sum bound by the log-determinant potential and an almost-sure constant +bound on that potential give the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_elliptical, h_potential_le] with + ω h_nonnegω h_le_oneω h_ellipticalω h_potential_leω + exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ @@ -1481,6 +1553,41 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the +log-determinant elliptical potential and that potential is bounded by `W`. + +This is the first theorem whose assumptions match the two matrix-analysis steps of the actual +elliptical-potential argument: + +* prove `cappedQuadraticWidthSum ≤ ellipticalPotential`; +* prove `ellipticalPotential ≤ W`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono W + (cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one + h_elliptical h_potential_le) + end LinUCB end Bandits From d568d50d95be7eea25c0adf455e2555321423b11 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:22:52 -0400 Subject: [PATCH 46/88] feat(linUCB): base case for the log-determinant elliptical-potential path --- .../Online/Bandit/Algorithms/LinUCB.lean | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f0e087a7..f6b75248 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -414,6 +414,17 @@ lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) ellipticalPotential A reg x 0 ω = 0 := by simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the log-determinant elliptical-potential inequality. At horizon zero there are no +positive-time capped quadratic width forms, and the log-determinant potential is zero when the +initial design determinant is nonzero. -/ +lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) + (hdet : designDet A reg x 0 ω ≠ 0) : + cappedQuadraticWidthSum A reg x 0 ω ≤ ellipticalPotential A reg x 0 ω := by + rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), + ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -440,6 +451,34 @@ lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} CappedQuadraticWidthBound A reg x n ω W := by exact ⟨h_nonneg, h_le_one, h_sum_le⟩ +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the packaged process-level capped quadratic-width input. At horizon zero, the +nonnegativity and `≤ 1` side conditions are vacuous, and the capped sum is zero. -/ +lemma cappedQuadraticWidthBound_zero {W : ℝ} (hW : 0 ≤ W) : + CappedQuadraticWidthBound A reg x 0 ω W := by + refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ + · intro t ht _ + simp at ht + · intro t ht _ + simp at ht + · simpa [cappedQuadraticWidthSum_zero] using hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the packaged process-level capped quadratic-width input when the constant bound +is supplied through the log-determinant potential. -/ +lemma cappedQuadraticWidthBound_zero_of_ellipticalPotential_le_bound {W : ℝ} + (hdet : designDet A reg x 0 ω ≠ 0) (h_potential_le : ellipticalPotential A reg x 0 ω ≤ W) : + CappedQuadraticWidthBound A reg x 0 ω W := by + refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ + · intro t ht _ + simp at ht + · intro t ht _ + simp at ht + · exact (cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) + (x := x) (ω := ω) hdet).trans h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ lemma cappedQuadraticWidthBound_mono {W W' : ℝ} From 9766e16289de5bec74a618ed8c0dcc001b654a96 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:26:54 -0400 Subject: [PATCH 47/88] feat(linUCB): per-step potential-increment shell --- .../Online/Bandit/Algorithms/LinUCB.lean | 80 +++++++++++++++++++ 1 file changed, 80 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f6b75248..7fa0aa2a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -425,6 +425,66 @@ lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step increment of the log-determinant elliptical potential. -/ +noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the next capped quadratic width term is bounded by the next log-determinant potential +increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ +lemma cappedQuadraticWidthSum_succ_le_ellipticalPotential + (h_prev : cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_step : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialIncrement A reg x n ω) : + cappedQuadraticWidthSum A reg x (n + 1) ω ≤ ellipticalPotential A reg x (n + 1) ω := by + rw [cappedQuadraticWidthSum_succ (A := A) (reg := reg) (x := x) (n := n) (ω := ω)] + calc + cappedQuadraticWidthSum A reg x n ω + + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) + ≤ ellipticalPotential A reg x n ω + ellipticalPotentialIncrement A reg x n ω := by + exact add_le_add h_prev h_step + _ = ellipticalPotential A reg x (n + 1) ω := by + simp [ellipticalPotentialIncrement] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A per-step bound by log-determinant potential increments implies the cumulative +elliptical-potential inequality. This is the induction shell for the future determinant-update +proof. -/ +lemma cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le + (hdet : designDet A reg x 0 ω ≠ 0) : + (∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) → + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + induction n with + | zero => + intro _ + exact cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) + (x := x) (ω := ω) hdet + | succ n ih => + intro h_step + refine cappedQuadraticWidthSum_succ_le_ellipticalPotential (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) ?_ ?_ + · exact ih fun t ht ↦ h_step t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + · exact h_step n (by simp) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a per-step bound by log-determinant potential increments implies the cumulative +elliptical-potential inequality. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + filter_upwards [hdet, h_step] with ω hdetω h_stepω + exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetω h_stepω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -543,6 +603,26 @@ lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by log-determinant potential increments and a final constant +bound on the potential give the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet h_step) + h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 9d1125484a236eb30d8f7e28c4e1c0b73a47ac08 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 13:38:23 -0400 Subject: [PATCH 48/88] feat(linUCB): one-step determinant ratio related lemmas --- .../Online/Bandit/Algorithms/LinUCB.lean | 70 +++++++++++++++++++ 1 file changed, 70 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 7fa0aa2a..c9ffb5c8 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -400,12 +400,31 @@ lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) designDetRatio A reg x 0 ω = 1 := by simp [designDetRatio, hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. + +This is the determinant-ratio target used by the matrix-determinant part of the elliptical +potential lemma. -/ +noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + designDet A reg x (n + 1) ω / designDet A reg x n ω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := 2 * Real.log (designDetRatio A reg x n ω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. + +The future determinant-update proof should naturally establish the capped quadratic-width term is +bounded by this quantity. A separate log/telescoping bridge then connects this one-step quantity to +`ellipticalPotentialIncrement`. -/ +noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + 2 * Real.log (designDetStepRatio A reg x n ω) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- At horizon zero, the log-determinant potential is zero when the initial design determinant is nonzero. -/ @@ -485,6 +504,30 @@ lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdetω h_stepω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the +cumulative capped-sum/log-det inequality, provided the one-step determinant-ratio potential is +bounded by the corresponding cumulative-potential increment. + +This separates the future elliptical-potential proof into two local obligations: + +* a matrix-determinant update bounding the selected arm's capped quadratic form by + `ellipticalPotentialStep`; +* a log/telescoping bridge from `ellipticalPotentialStep` to `ellipticalPotentialIncrement`. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet ?_ + filter_upwards [h_step, h_step_le_increment] with ω h_stepω h_step_le_incrementω + intro t ht + exact (h_stepω t ht).trans (h_step_le_incrementω t ht) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -623,6 +666,33 @@ lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : (reg := reg) (x := x) (n := n) (P := P) hdet h_step) h_potential_le +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, one-step determinant-ratio potential bounds, their bridge to cumulative +potential increments, and a final constant bound on the potential give the packaged process-level +capped quadratic-width input. + +This is the packaged form of the determinant-update interface: once the true matrix determinant +lemma proves the `h_step` assumption and the log/telescoping algebra proves +`h_step_le_increment`, the existing regret chain can consume the resulting bound. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) + h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 70916c42d2fdf726b293adcb51900bf3ae6415d0 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Tue, 16 Jun 2026 14:02:08 -0400 Subject: [PATCH 49/88] feat(linUCB): elliptical-potential chain related changes --- .../Online/Bandit/Algorithms/LinUCB.lean | 484 ++++++++++++++++++ 1 file changed, 484 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index c9ffb5c8..272c7bfc 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -8,6 +8,8 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.LinearAlgebra.Matrix.SchurComplement public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse /-! @@ -387,6 +389,22 @@ lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by simp [designDet, designMatrix_zero] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial design determinant is `reg ^ d`. -/ +lemma designDet_zero_eq_reg_pow (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + designDet A reg x 0 ω = reg ^ d := by + rw [designDet_zero] + simp + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A nonzero regularization parameter gives a nonzero initial design determinant. -/ +lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hreg : reg ≠ 0) : + designDet A reg x 0 ω ≠ 0 := by + rw [designDet_zero_eq_reg_pow] + exact pow_ne_zero d hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -400,6 +418,15 @@ lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) designDetRatio A reg x 0 ω = 1 := by simp [designDetRatio, hdet] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the determinant ratio is positive when the initial design determinant is +nonzero. -/ +lemma designDetRatio_zero_pos (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + 0 < designDetRatio A reg x 0 ω := by + rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + norm_num + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. @@ -409,12 +436,210 @@ noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := designDet A reg x (n + 1) ω / designDet A reg x n ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The scalar determinant appearing in the rank-one determinant update is the quadratic form +`uᵀ M u`. -/ +lemma det_one_add_replicateRow_mul_matrix_mul_replicateCol + (M : Matrix (Fin d) (Fin d) ℝ) (u : Feature d) : + (1 + Matrix.replicateRow Unit u * M * Matrix.replicateCol Unit u).det = + 1 + dotProduct u (Matrix.mulVec M u) := by + have hsum : + (∑ j, (∑ i, u i * M i j) * u j) = + ∑ i, u i * ∑ j, M i j * u j := by + calc + (∑ j, (∑ i, u i * M i j) * u j) + = ∑ j, ∑ i, (u i * M i j) * u j := by + simp [Finset.sum_mul] + _ = ∑ i, ∑ j, (u i * M i j) * u j := by + rw [Finset.sum_comm] + _ = ∑ i, u i * ∑ j, M i j * u j := by + refine Finset.sum_congr rfl ?_ + intro i _ + rw [Finset.mul_sum] + refine Finset.sum_congr rfl ?_ + intro j _ + ring + rw [Matrix.det_unique] + simpa [Matrix.mul_apply, Matrix.replicateRow, Matrix.replicateCol, Matrix.mulVec, + dotProduct] using hsum + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Process-level matrix determinant update for the LinUCB design matrix. + +If `V_n` has nonzero determinant, then the rank-one update +`V_{n+1} = V_n + x_{A_n} x_{A_n}ᵀ` satisfies +`det(V_{n+1}) = det(V_n) * (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ +lemma designDet_succ_eq_mul_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDet A reg x (n + 1) ω = + designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + have hM : IsUnit (designMatrix A reg x n ω).det := by + simpa [designDet] using (isUnit_iff_ne_zero.mpr hdet) + calc + designDet A reg x (n + 1) ω = + (designMatrix A reg x n ω + + Matrix.vecMulVec (x (A n ω)) (x (A n ω))).det := by + simp [designDet, designMatrix_succ] + _ = (designMatrix A reg x n ω + + Matrix.replicateCol Unit (x (A n ω)) * Matrix.replicateRow Unit (x (A n ω))).det := by + rw [Matrix.vecMulVec_eq Unit] + _ = (designMatrix A reg x n ω).det * + (1 + Matrix.replicateRow Unit (x (A n ω)) * + (designMatrix A reg x n ω)⁻¹ * Matrix.replicateCol Unit (x (A n ω))).det := by + exact Matrix.det_add_replicateCol_mul_replicateRow (A := designMatrix A reg x n ω) + (ι := Unit) hM (x (A n ω)) (x (A n ω)) + _ = designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + rw [designDet] + congr 1 + exact det_one_add_replicateRow_mul_matrix_mul_replicateCol + (M := (designMatrix A reg x n ω)⁻¹) (u := x (A n ω)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `det(V_n)` is nonzero and the selected quadratic form is nonnegative, then +`det(V_{n+1})` is nonzero. -/ +lemma designDet_succ_ne_zero_of_widthQuadraticForm_nonneg + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + designDet A reg x (n + 1) ω ≠ 0 := by + rw [designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + exact mul_ne_zero hdet (ne_of_gt (by linarith)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms preserve +nonzero design determinants up to any fixed time. -/ +lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt + (m : ℕ) (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t < m → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + designDet A reg x m ω ≠ 0 := by + induction m with + | zero => exact hdet0 + | succ m ih => + exact designDet_succ_ne_zero_of_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := m) (ω := ω) + (ih fun t ht ↦ h_nonneg t (Nat.lt_trans ht (Nat.lt_succ_self m))) + (h_nonneg m (Nat.lt_succ_self m)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms imply that +all design determinants through horizon `n` are nonzero. -/ +lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by + intro t ht + exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := t) (ω := ω) hdet0 fun s hs ↦ + h_nonneg s (mem_range.mpr (Nat.lt_of_lt_of_le hs (Nat.le_of_lt_succ (mem_range.mp ht)))) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant and nonnegative selected quadratic forms imply +that all design determinants through horizon `n` are nonzero. -/ +lemma designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω + exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `det(V_n) ≠ 0`, then the one-step determinant ratio is +`1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n}`. -/ +lemma designDetStepRatio_eq_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDetStepRatio A reg x n ω = + 1 + widthQuadraticForm A reg x (A n ω) n ω := by + simp [designDetStepRatio, + designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet, hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The cumulative determinant ratio advances by multiplying by the one-step determinant ratio. -/ +lemma designDetRatio_succ_eq_mul_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDetRatio A reg x (n + 1) ω = + designDetRatio A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + rw [designDetRatio, designDetRatio, + designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms make the +cumulative determinant ratio positive. -/ +lemma designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + 0 < designDetRatio A reg x n ω := by + induction n with + | zero => + exact designDetRatio_zero_pos (A := A) (reg := reg) (x := x) (ω := ω) hdet0 + | succ n ih => + have hdetn : designDet A reg x n ω ≠ 0 := + designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ + h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) + rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetn] + exact mul_pos + (ih fun t ht ↦ h_nonneg t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) + (by linarith [h_nonneg n (by simp)]) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, starting from a nonzero initial determinant, nonnegative selected quadratic +forms make the cumulative determinant ratio positive. -/ +lemma designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by + filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω + exact designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter and nonnegative selected quadratic forms make +the cumulative determinant ratio positive. -/ +lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by + refine designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := 2 * Real.log (designDetRatio A reg x n ω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A positive determinant ratio bounded by `D` gives the corresponding log-determinant potential +bound. -/ +lemma ellipticalPotential_le_two_mul_log_of_designDetRatio_le {D : ℝ} + (h_ratio_pos : 0 < designDetRatio A reg x n ω) + (h_ratio_le : designDetRatio A reg x n ω ≤ D) : + ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by + rw [ellipticalPotential] + exact mul_le_mul_of_nonneg_left (Real.log_le_log h_ratio_pos h_ratio_le) (by norm_num) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a positive determinant ratio bounded by `D` gives the corresponding +log-determinant potential bound. -/ +lemma ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le {D : ℝ} + (h_ratio_pos : ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by + filter_upwards [h_ratio_pos, h_ratio_le] with ω h_ratio_posω h_ratio_leω + exact ellipticalPotential_le_two_mul_log_of_designDetRatio_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_ratio_posω h_ratio_leω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. @@ -425,6 +650,70 @@ noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := 2 * Real.log (designDetStepRatio A reg x n ω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing, the one-step log-determinant potential is +`2 * log (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ +lemma ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + ellipticalPotentialStep A reg x n ω = + 2 * Real.log (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + simp [ellipticalPotentialStep, + designDetStepRatio_eq_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar log inequality used in the elliptical-potential proof: for `0 ≤ q ≤ 1`, +`min 1 q ≤ 2 * log (1 + q)`. -/ +lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} + (hq_nonneg : 0 ≤ q) (hq_le_one : q ≤ 1) : + min 1 q ≤ 2 * Real.log (1 + q) := by + have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := + Real.le_log_one_add_of_nonneg hq_nonneg + have hq_add_two_pos : 0 < q + 2 := by linarith + have hq_le_two : q ≤ 2 := by linarith + have hq_le_log_lower : q ≤ 2 * (2 * q / (q + 2)) := by + rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] + rw [le_div_iff₀ hq_add_two_pos] + nlinarith + rw [min_eq_right hq_le_one] + exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the +single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ +lemma cappedWidthTerm_le_ellipticalPotentialStep + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) + (h_le_one : n ≠ 0 → widthQuadraticForm A reg x (A n ω) n ω ≤ 1) : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialStep A reg x n ω := by + by_cases hn : n = 0 + · rw [if_pos hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) + · rw [if_neg hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply +the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ +lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω := by + filter_upwards [hdet, h_nonneg, h_le_one] with ω hdetω h_nonnegω h_le_oneω + intro t ht + exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- At horizon zero, the log-determinant potential is zero when the initial design determinant is nonzero. -/ @@ -450,6 +739,34 @@ noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The one-step determinant-ratio potential equals the increment of the cumulative +log-determinant potential, provided the relevant design determinants are nonzero. -/ +lemma ellipticalPotentialStep_eq_increment + (hdet0 : designDet A reg x 0 ω ≠ 0) + (hdetn : designDet A reg x n ω ≠ 0) + (hdet_succ : designDet A reg x (n + 1) ω ≠ 0) : + ellipticalPotentialStep A reg x n ω = ellipticalPotentialIncrement A reg x n ω := by + simp [ellipticalPotentialStep, designDetStepRatio, ellipticalPotentialIncrement, + ellipticalPotential, designDetRatio, Real.log_div hdet_succ hdetn, + Real.log_div hdet_succ hdet0, Real.log_div hdetn hdet0] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the one-step determinant-ratio potential equals the increment of the cumulative +log-determinant potential throughout the finite horizon, provided all determinants up to that +horizon are nonzero almost surely. -/ +lemma ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω = ellipticalPotentialIncrement A reg x t ω := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact ellipticalPotentialStep_eq_increment (A := A) (reg := reg) (x := x) (n := t) + (ω := ω) (hdetω 0 (by simp)) + (hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) + (hdetω (t + 1) (mem_range.mpr (Nat.succ_lt_succ (mem_range.mp ht)))) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- If the next capped quadratic width term is bounded by the next log-determinant potential increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ @@ -528,6 +845,29 @@ lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le intro t ht exact (h_stepω t ht).trans (h_step_le_incrementω t ht) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the +cumulative capped-sum/log-det inequality when all design determinants up to the horizon are nonzero +almost surely. + +Compared with `cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le`, this +version discharges the log/telescoping bridge automatically from determinant nonvanishing. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + have hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + exact hdetω 0 (by simp) + refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_step ?_ + filter_upwards [ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet] with ω h_eq + intro t ht + rw [h_eq t ht] + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -693,6 +1033,150 @@ lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bo (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) h_potential_le +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, one-step determinant-ratio potential bounds, determinant nonvanishing up to the +horizon, and a final constant bound on the potential give the packaged process-level capped +quadratic-width input. + +This is the determinant-nonvanishing version of the one-step interface: the remaining hard +elliptical-potential work is to prove the one-step matrix inequality and the final +log-determinant bound. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero + {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet h_step) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the determinant-update step, determinant nonvanishing up to the horizon, and a +final constant bound on the log-determinant potential give the packaged capped quadratic-width +input used by the regret chain. + +The assumptions now match the concrete obligations left for a full elliptical-potential proof: + +* prove all relevant design determinants are nonzero; +* prove selected quadratic forms are nonnegative and at most `1` at positive times; +* prove the final log-determinant potential is at most `W`. -/ +lemma cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound {W : ℝ} + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + have h_nonneg_positive : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + filter_upwards [h_nonneg] with ω h_nonnegω + intro t ht _ + exact h_nonnegω t ht + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) + h_nonneg_positive h_le_one hdet + (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero (A := A) (reg := reg) + (x := x) (n := n) (P := P) hdet_range_n h_nonneg h_le_one) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant, the determinant-update step, and a final constant +bound on the log-determinant potential give the packaged capped quadratic-width input used by the +regret chain. + +This removes the need to assume determinant nonvanishing at every time: it is derived inductively +from `det(V_0) ≠ 0` and nonnegative selected quadratic forms. -/ +lemma cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound {W : ℝ} + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) + h_nonneg h_le_one h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter, the determinant-update step, and a final +constant bound on the log-determinant potential give the packaged capped quadratic-width input used +by the regret chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential_le_bound {W : ℝ} + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + refine cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) ?_ h_nonneg h_le_one + h_potential_le + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a +determinant-ratio upper bound, and the determinant-update step give the packaged capped +quadratic-width input used by the regret chain. + +This version accepts the determinant-ratio bound directly and converts it into the +`ellipticalPotential ≤ 2 * log D` bound internally. -/ +lemma cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound {D : ℝ} + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by + exact cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := 2 * Real.log D) + hdet0 h_nonneg h_le_one + (ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) + h_ratio_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter, nonnegative selected quadratic forms, a +determinant-ratio upper bound, and the determinant-update step give the packaged capped +quadratic-width input used by the regret chain. + +This is the most direct interface for the final determinant-bound part of the finite-action +elliptical-potential argument: after proving `designDetRatio ≤ D`, the theorem supplies the +`CappedQuadraticWidthBound` with bound `2 * log D`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound {D : ℝ} + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by + refine cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one h_ratio_le + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 88d52b661ae64c83d4ce1dfa09f45316f48db7d1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 29 May 2026 10:10:11 -0400 Subject: [PATCH 50/88] feat : initial algorithm foundation for linUCB --- LeanMachineLearning.lean | 1 + .../Online/Bandit/Algorithms/LinUCB.lean | 230 ++++++++++++++++++ 2 files changed, 231 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 3dbb4310..7d6d4a9b 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -4,6 +4,7 @@ public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.Measura public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable public import LeanMachineLearning.Online.Bandit.Algorithms.ETC +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean new file mode 100644 index 00000000..7c45c27b --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -0,0 +1,230 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +abbrev Feature (d : ℕ) := Fin d → ℝ + +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) + (n : ℕ) : + Measurable (nextArm hK reg β x h_index n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ h_index n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) + (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (ω : Ω) (hn : n ≠ 0) : + designMatrix A reg x n ω = + designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact congrArg (fun S ↦ reg • 1 + S) <| + (Finset.sum_coe_sort (Iic n) + (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm + +lemma responseVector_eq_responseVector' (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + cases n with + | zero => exact absurd rfl hn + | succ n => + simp only [responseVector, responseVector', IsAlgEnvSeq.hist] + rw [Nat.range_succ_eq_Iic] + exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm + +lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by + simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, + responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] + +lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + estimatedReward A R reg x a n ω = + estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] + +lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] + +lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + index A R reg β x a n ω = + index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by + have htime : n + 1 = n - 1 + 2 := by grind + simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, + width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] + +/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ +lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (n : ℕ) : + A (n + 1) =ᵐ[P] + fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact h.action_detAlgorithm_ae_eq n + +/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ +lemma arm_ae_all_eq [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : + ∀ᵐ ω ∂P, + ∀ n, A (n + 1) ω = + nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by + simp_rw [ae_all_iff] + exact fun n ↦ arm_ae_eq_linUCBNextArm h n + +/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ +lemma index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) (hn : n ≠ 0) : + ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm + have hn_succ : n - 1 + 1 = n := by grind + simp only [hn_succ] at h_arm + rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, + index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] + rw [h_arm] + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) + (IsAlgEnvSeq.hist A R (n - 1) ω) a + +/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ +lemma forall_index_le_index_arm [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (a : Fin K) : + ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by + simp_rw [ae_all_iff] + exact fun n hn ↦ index_le_index_arm h a hn + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 56cf08f3a23ecae5a56379395162981d12be97b2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 09:54:29 -0400 Subject: [PATCH 51/88] feat(linUCB):D = 2 ^ n intermediate bound --- .../Online/Bandit/Algorithms/LinUCB.lean | 93 +++++++++++++++++++ 1 file changed, 93 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 272c7bfc..39de230a 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -613,6 +613,77 @@ lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, the cumulative determinant ratio is the finite +product of the per-round determinant-update factors. -/ +lemma designDetRatio_eq_prod_one_add_widthQuadraticForm + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + designDetRatio A reg x n ω = + ∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω) := by + induction n with + | zero => + rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet0] + simp + | succ n ih => + have hdetn : designDet A reg x n ω ≠ 0 := + designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ + h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) + rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetn] + rw [ih fun t ht ↦ h_nonneg t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))] + simp [Finset.prod_range_succ] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If every selected quadratic form is in `[0, 1]`, the cumulative determinant ratio is at most +`2 ^ n`. -/ +lemma designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + rw [designDetRatio_eq_prod_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0 h_nonneg] + calc + (∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω)) + ≤ ∏ _t ∈ range n, (2 : ℝ) := by + exact Finset.prod_le_prod + (fun t ht ↦ by linarith [h_nonneg t ht]) + (fun t ht ↦ by linarith [h_le_one t ht]) + _ = (2 : ℝ) ^ n := by + simp + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, if every selected quadratic form is in `[0, 1]`, the cumulative determinant +ratio is at most `2 ^ n`. -/ +lemma designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + filter_upwards [hdet0, h_nonneg, h_le_one] with ω hdet0ω h_nonnegω h_le_oneω + exact designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one (A := A) + (reg := reg) (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω h_le_oneω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter and selected quadratic forms in `[0, 1]` +imply the cumulative determinant ratio is at most `2 ^ n`. -/ +lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + refine designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one + (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1177,6 +1248,28 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_b exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A simple explicit determinant-ratio bound for the capped quadratic-width input. + +If `reg ≠ 0` and every selected quadratic form is almost surely in `[0, 1]`, then the determinant +ratio is at most `2 ^ n`, so the existing determinant-update/elliptical-potential chain gives the +packaged capped-width bound with budget `2 * log (2 ^ n)`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω (2 * Real.log ((2 : ℝ) ^ n)) := by + refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (2 : ℝ) ^ n) + hreg h_nonneg ?_ ?_ + · filter_upwards [h_le_one] with ω h_le_oneω + exact fun t ht _ ↦ h_le_oneω t ht + · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From feff54daaacdc767e170e95bdbfbc47db6c083ff Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:10:56 -0400 Subject: [PATCH 52/88] feat(linUCB):trace/determinant-budget layer --- .../Online/Bandit/Algorithms/LinUCB.lean | 78 +++++++++++++++++++ 1 file changed, 78 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 39de230a..da5bdbed 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -34,6 +34,11 @@ namespace LinUCB /-- Feature vectors for finite-dimensional linear bandits. -/ abbrev Feature (d : ℕ) := Fin d → ℝ +/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product +`x_aᵀ x_a`. -/ +def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := + dotProduct (x a) (x a) + /-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := @@ -138,6 +143,55 @@ lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by simp [designMatrix, sum_range_succ, add_assoc] +/-- Trace of the process-level regularized design matrix. -/ +noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + Matrix.trace (designMatrix A reg x n ω) + +/-- Before any observations, the design trace is the trace of `reg • I_d`, namely `reg * d`. -/ +lemma designTrace_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + designTrace A reg x 0 ω = reg * (d : ℝ) := by + simp [designTrace, designMatrix_zero] + +/-- Updating the design matrix by `x_a x_aᵀ` increases the trace by `x_aᵀ x_a`. -/ +lemma designTrace_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designTrace A reg x (n + 1) ω = + designTrace A reg x n ω + featureSqNorm x (A n ω) := by + simp [designTrace, designMatrix_succ, featureSqNorm, Matrix.trace_vecMulVec] + +/-- Closed form for the design trace: initial regularization trace plus accumulated squared +feature norms. -/ +lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designTrace A reg x n ω = + reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by + simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] + +/-- If every selected feature vector has squared norm at most `L2`, then the trace of the design +matrix is at most `reg * d + n * L2`. -/ +lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by + rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] + gcongr + calc + (∑ t ∈ range n, featureSqNorm x (A t ω)) ≤ ∑ _t ∈ range n, L2 := by + exact sum_le_sum fun t ht ↦ hL2 t ht + _ = (n : ℝ) * L2 := by + simp [nsmul_eq_mul] + +omit [IsProbabilityMeasure P] in +/-- Almost surely, bounded selected feature norms give the corresponding deterministic trace +budget `reg * d + n * L2`. -/ +lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : + ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by + filter_upwards [hL2] with ω hL2ω + exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) L2 hL2ω + /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := @@ -1270,6 +1324,30 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Trace-budget interface for the determinant part of the finite-action elliptical-potential +argument. + +The future spectral/AM-GM determinant theorem should prove the hypothesis +`designDetRatio ≤ (T / (reg * d)) ^ d`, where `T` is an upper bound on `trace(V_n)`. This theorem +then feeds that determinant-ratio bound into the already-proved determinant-update and +elliptical-potential chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (T : ℝ) + (h_ratio_le : ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * Real.log ((T / (reg * (d : ℝ))) ^ d)) := by + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 0a81fe7433925aaf8404f627e9da8a7078e5dd2a Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:15:17 -0400 Subject: [PATCH 53/88] feat(linUCB):trace/determinant-budget layer --- .../Online/Bandit/Algorithms/LinUCB.lean | 72 ------------------- 1 file changed, 72 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 2d21d954..da5bdbed 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -8,11 +8,8 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax -<<<<<<< HEAD public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.LinearAlgebra.Matrix.SchurComplement -======= ->>>>>>> main public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse /-! @@ -34,7 +31,6 @@ section Algorithm namespace LinUCB -<<<<<<< HEAD /-- Feature vectors for finite-dimensional linear bandits. -/ abbrev Feature (d : ℕ) := Fin d → ℝ @@ -44,39 +40,25 @@ def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := dotProduct (x a) (x a) /-- History-level regularized design matrix for LinUCB. -/ -======= -abbrev Feature (d : ℕ) := Fin d → ℝ - ->>>>>>> main noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) -<<<<<<< HEAD /-- History-level response vector for LinUCB. -/ -======= ->>>>>>> main noncomputable def responseVector' (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := ∑ s : Iic n, (h s).2 • x (h s).1 -<<<<<<< HEAD /-- History-level regularized least-squares estimate. -/ -======= ->>>>>>> main noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) -<<<<<<< HEAD /-- History-level estimated reward of an arm. -/ -======= ->>>>>>> main noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := dotProduct (thetaHat' reg x n h) (x a) -<<<<<<< HEAD /-- History-level quadratic form underlying the LinUCB confidence width. -/ noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := @@ -94,11 +76,6 @@ lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by simp [width', Real.sq_sqrt h_nonneg] -======= -noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a))) ->>>>>>> main /-- LinUCB optimistic index of an arm. @@ -114,10 +91,6 @@ open Classical in /-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) -<<<<<<< HEAD -======= - (_h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) ->>>>>>> main (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK measurableArgmax (fun h a ↦ index' reg β x n h a) h @@ -127,11 +100,7 @@ lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) (n : ℕ) : -<<<<<<< HEAD Measurable (nextArm hK reg β x n) := by -======= - Measurable (nextArm hK reg β x h_index n) := by ->>>>>>> main have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact measurable_measurableArgmax fun a ↦ h_index n a @@ -142,11 +111,7 @@ noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → LinUCB.Feature d) (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : Algorithm (Fin K) ℝ := -<<<<<<< HEAD detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ -======= - detAlgorithm (LinUCB.nextArm hK reg β x h_index) (by fun_prop) ⟨0, hK⟩ ->>>>>>> main end Algorithm @@ -167,7 +132,6 @@ noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) -<<<<<<< HEAD /-- The initial design matrix before any actions are included. -/ lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : designMatrix A reg x 0 ω = reg • 1 := by @@ -228,14 +192,11 @@ lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) L2 hL2ω -======= ->>>>>>> main /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := ∑ s ∈ range n, R s ω • x (A s ω) -<<<<<<< HEAD /-- The initial response vector before any rewards are included. -/ lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (ω : Ω) : @@ -249,14 +210,11 @@ lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) responseVector A R x n ω + R n ω • x (A n ω) := by simp [responseVector, sum_range_succ] -======= ->>>>>>> main /-- The process-level regularized least-squares estimate. -/ noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) -<<<<<<< HEAD /-- The initial least-squares estimate is zero because no reward-feature observations have been included yet. -/ lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) @@ -264,14 +222,11 @@ lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) thetaHat A R reg x 0 ω = 0 := by simp [thetaHat, responseVector_zero] -======= ->>>>>>> main /-- The process-level estimated linear reward. -/ noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := dotProduct (thetaHat A R reg x n ω) (x a) -<<<<<<< HEAD /-- The initial estimated reward is zero for every arm. -/ lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : @@ -1411,12 +1366,6 @@ lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} filter_upwards [h_bound] with ω h_boundω exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) (W := W) h_boundω -======= -/-- The process-level elliptical confidence width. -/ -noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a))) ->>>>>>> main /-- The process-level LinUCB optimistic index. -/ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) @@ -1424,7 +1373,6 @@ noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (n : ℕ) (ω : Ω) : ℝ := estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω -<<<<<<< HEAD /-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is zero. -/ lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) @@ -1440,8 +1388,6 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by simp [index_zero, width_zero] -======= ->>>>>>> main lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = @@ -1477,7 +1423,6 @@ lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] -<<<<<<< HEAD lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : widthQuadraticForm A reg x a n ω = @@ -1919,12 +1864,6 @@ lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one h_capped_le) -======= -lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] ->>>>>>> main lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : @@ -1939,11 +1878,7 @@ lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (n : ℕ) : A (n + 1) =ᵐ[P] -<<<<<<< HEAD fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by -======= - fun ω ↦ nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by ->>>>>>> main have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK exact h.action_detAlgorithm_ae_eq n @@ -1952,11 +1887,7 @@ lemma arm_ae_all_eq [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : ∀ᵐ ω ∂P, ∀ n, A (n + 1) ω = -<<<<<<< HEAD nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by -======= - nextArm hK reg β x h_index n (IsAlgEnvSeq.hist A R n ω) := by ->>>>>>> main simp_rw [ae_all_iff] exact fun n ↦ arm_ae_eq_linUCBNextArm h n @@ -1986,7 +1917,6 @@ lemma forall_index_le_index_arm [Nonempty (Fin K)] end AlgorithmBehavior -<<<<<<< HEAD omit [IsMarkovKernel ν] in /-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected @@ -2502,8 +2432,6 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one h_elliptical h_potential_le) -======= ->>>>>>> main end LinUCB end Bandits From 51d73c5357a02db14c1e8c41f0874c8c73599c29 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:28:56 -0400 Subject: [PATCH 54/88] feat(linUCB):packaging step which allows formal slot where the hard matrix inequality will plug in --- .../Online/Bandit/Algorithms/LinUCB.lean | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index da5bdbed..65cd1781 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -738,6 +738,39 @@ lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_o exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Converts an almost-sure trace bound into the determinant-ratio bound expected from a future +trace/determinant comparison theorem. -/ +lemma designDetRatio_ae_le_trace_budget_of_designTrace_ae_le + (T : ℝ) + (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ T → + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + filter_upwards [h_trace_le] with ω h_traceω + exact h_ratio_of_trace ω h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms give the concrete trace budget +`reg * d + n * L2`; a future trace/determinant comparison then gives the corresponding +determinant-ratio bound. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + exact designDetRatio_ae_le_trace_budget_of_designTrace_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) (T := reg * (d : ℝ) + (n : ℝ) * L2) + (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2) + h_ratio_of_trace + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1348,6 +1381,35 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface for the determinant part of the finite-action +elliptical-potential argument. + +If selected feature vectors have squared norm at most `L2`, then `trace(V_n) ≤ reg * d + n * L2`. +Given a future deterministic trace/determinant comparison that turns this trace budget into the +determinant-ratio bound, this theorem supplies the packaged capped-width input with the explicit +budget `2 * log (((reg * d + n * L2) / (reg * d)) ^ d)`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d)) := by + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg h_nonneg h_le_one + (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From 13b222e59b067f1da87b415cc51ae3cdee1f454f Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:41:10 -0400 Subject: [PATCH 55/88] feat(linUCB): formal bound shaped like the standard elliptical-potential term used in LinUCB regret proofs --- .../Online/Bandit/Algorithms/LinUCB.lean | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 65cd1781..a1b5e49c 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -1410,6 +1410,45 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budge (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The explicit feature-norm determinant budget can be rewritten in the common +`d * log(1 + n L² / (reg d))` form. -/ +lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add + (L2 : ℝ) (hden : reg * (d : ℝ) ≠ 0) : + 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) = + 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by + have hbase : + (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) = + 1 + (n : ℝ) * L2 / (reg * (d : ℝ)) := by + exact same_add_div hden + rw [Real.log_pow, hbase] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface with the log term rewritten in the standard +`2 * d * log(1 + n L² / (reg d))` shape. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (hreg : reg ≠ 0) (hd : d ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg (by exact_mod_cast hd) + rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 + h_ratio_of_trace + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ From c01cb041d2a74aaec422b8dde15f1797639325aa Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:44:10 -0400 Subject: [PATCH 56/88] feat(linUCB): end-to-end regret-facing step --- .../Online/Bandit/Algorithms/LinUCB.lean | 40 +++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a1b5e49c..1cf9bcea 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2498,6 +2498,46 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the +feature-budget elliptical-potential term +`2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. + +The remaining matrix-analysis input is isolated in `h_ratio_of_trace`: a future determinant/trace +comparison theorem should prove that the trace budget implies the displayed determinant-ratio +bound. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) + (hreg : reg ≠ 0) (hd : d ≠ 0) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg hd h_quad_nonneg + h_quad_le_one L2 hL2 h_ratio_of_trace) + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the log-determinant elliptical potential and that potential is bounded by `W`. From c6e66069778944b25f515b96488627bb61a8c6d9 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 10:48:39 -0400 Subject: [PATCH 57/88] feat(linUCB): proof chain no longer needs the future matrix theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 123 ++++++++++++++++++ 1 file changed, 123 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 1cf9bcea..82dd5743 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -771,6 +771,67 @@ lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (x := x) (n := n) (P := P) L2 hL2) h_ratio_of_trace +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A determinant upper bound for `V_n` implies the corresponding determinant-ratio bound, using +`det(V_0) = reg ^ d`. -/ +lemma designDetRatio_le_trace_budget_of_designDet_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hdet_le : designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + rw [designDetRatio, designDet_zero_eq_reg_pow] + have hreg_pow_nonneg : 0 ≤ reg ^ d := (pow_pos hreg_pos d).le + have hdiv : designDet A reg x n ω / reg ^ d ≤ (T / (d : ℝ)) ^ d / reg ^ d := by + exact div_le_div_of_nonneg_right hdet_le hreg_pow_nonneg + refine hdiv.trans_eq ?_ + rw [← div_pow] + congr 1 + field_simp [hreg_pos.ne', by exact_mod_cast hd] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a determinant upper bound for `V_n` implies the corresponding determinant-ratio +bound. -/ +lemma designDetRatio_ae_le_trace_budget_of_designDet_ae_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hdet_le : ∀ᵐ ω ∂P, designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + filter_upwards [hdet_le] with ω hdetω + exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) T hreg_pos hd hdetω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Converts an almost-sure trace bound plus a future determinant/trace comparison for `det(V_n)` +into the determinant-ratio bound used by the elliptical-potential chain. -/ +lemma designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ T → designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + refine designDetRatio_ae_le_trace_budget_of_designDet_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) T hreg_pos hd ?_ + filter_upwards [h_trace_le] with ω h_traceω + exact hdet_of_trace ω h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms reduce the determinant-ratio goal to the determinant upper bound +`det(V_n) ≤ ((reg * d + n * L2) / d) ^ d`. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + exact designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd + (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2) + hdet_of_trace + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1449,6 +1510,32 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budge (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 h_ratio_of_trace +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface with the determinant/trace comparison stated as a determinant +upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd h_nonneg + h_le_one L2 hL2 ?_ + intro ω h_traceω + exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd + (hdet_of_trace ω h_traceω) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ @@ -2538,6 +2625,42 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg hd h_quad_nonneg h_quad_le_one L2 hL2 h_ratio_of_trace) +/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term +when the remaining matrix-analysis input is stated as the determinant upper bound +`det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_nonneg + h_quad_le_one L2 hL2 hdet_of_trace) + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the log-determinant elliptical potential and that potential is bounded by `W`. From 5c3eb5c7194fe2b8ba39fb34ee30ce2363580b51 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 11:13:11 -0400 Subject: [PATCH 58/88] feat(linUCB): small clean up --- .../Online/Bandit/Algorithms/LinUCB.lean | 199 ++++++++++++++++-- 1 file changed, 184 insertions(+), 15 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 82dd5743..a3b08e01 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -9,6 +9,8 @@ public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.Data.Real.StarOrdered +public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse @@ -39,6 +41,12 @@ abbrev Feature (d : ℕ) := Fin d → ℝ def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := dotProduct (x a) (x a) +/-- The squared feature norm is nonnegative. -/ +lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : + 0 ≤ featureSqNorm x a := by + rw [featureSqNorm, dotProduct] + exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) + /-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := @@ -143,6 +151,16 @@ lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by simp [designMatrix, sum_range_succ, add_assoc] +/-- With nonnegative regularization, the process-level design matrix is positive semidefinite. -/ +lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : + (designMatrix A reg x n ω).PosSemidef := by + unfold designMatrix + apply Matrix.PosSemidef.add + · exact Matrix.PosSemidef.smul Matrix.PosSemidef.one hreg_nonneg + · refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -167,6 +185,14 @@ lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] +/-- With nonnegative regularization, the design trace is nonnegative. -/ +lemma designTrace_nonneg (hreg_nonneg : 0 ≤ reg) : + 0 ≤ designTrace A reg x n ω := by + rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] + exact add_nonneg + (mul_nonneg hreg_nonneg (Nat.cast_nonneg d)) + (sum_nonneg fun t _ ↦ featureSqNorm_nonneg x (A t ω)) + /-- If every selected feature vector has squared norm at most `L2`, then the trace of the design matrix is at most `reg * d + n * L2`. -/ lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound @@ -245,6 +271,42 @@ lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by simp [widthQuadraticForm, designMatrix_zero] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Nonnegative regularization makes every LinUCB width quadratic form nonnegative. + +The reason is purely matrix-theoretic: `V_n` is positive semidefinite, the nonsingular inverse of a +positive semidefinite matrix is positive semidefinite in mathlib, and every quadratic form induced +by a positive semidefinite matrix is nonnegative. -/ +lemma widthQuadraticForm_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) (a : Fin K) : + 0 ≤ widthQuadraticForm A reg x a n ω := by + simpa [widthQuadraticForm] using + ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg).inv.dotProduct_mulVec_nonneg (x a)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, nonnegative regularization gives nonnegative selected quadratic width forms +through any finite horizon. -/ +lemma widthQuadraticForm_ae_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + exact Filter.Eventually.of_forall fun ω t _ht ↦ + widthQuadraticForm_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) hreg_nonneg (A t ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive-time version of `widthQuadraticForm_ae_nonneg_of_reg_nonneg`, matching the side +condition shape used by the regret/width-sum bridge lemmas. -/ +lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + filter_upwards [widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) + (x := x) (n := n) (P := P) hreg_nonneg] with ω h_nonnegω + intro t ht _ht0 + exact h_nonnegω t ht + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := @@ -832,6 +894,61 @@ lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le (x := x) (n := n) (P := P) L2 hL2) hdet_of_trace +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix-level determinant/trace comparison needed for the finite-dimensional +elliptical-potential bound. + +For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality +`det(M) ≤ (trace(M) / d) ^ d`. -/ +def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := + ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the +design matrix is positive semidefinite. -/ +lemma designDet_le_trace_average_pow_of_matrix_det_trace_bound + (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) : + designDet A reg x n ω ≤ (designTrace A reg x n ω / (d : ℝ)) ^ d := by + simpa [designDet, designTrace] using + hdet_trace (designMatrix A reg x n ω) + (designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Combining `det(M) ≤ (trace(M)/d)^d` with a trace budget gives the determinant upper bound +`det(V_n) ≤ (T/d)^d`. -/ +lemma designDet_le_trace_budget_of_matrix_det_trace_bound + (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) + (hd : d ≠ 0) (T : ℝ) (h_trace_le : designTrace A reg x n ω ≤ T) : + designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d := by + have hd_pos : 0 < (d : ℝ) := by + exact_mod_cast Nat.pos_of_ne_zero hd + have hbase_nonneg : 0 ≤ designTrace A reg x n ω / (d : ℝ) := + div_nonneg (designTrace_nonneg (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg) hd_pos.le + have hbase_le : designTrace A reg x n ω / (d : ℝ) ≤ T / (d : ℝ) := + (div_le_div_iff_of_pos_right hd_pos).mpr h_trace_le + exact (designDet_le_trace_average_pow_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet_trace hreg_nonneg).trans + (pow_le_pow_left₀ hbase_nonneg hbase_le d) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms and the matrix-level determinant/trace comparison give the +determinant-ratio bound used by the elliptical-potential chain. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + refine designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 ?_ + intro ω h_traceω + exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) + hdet_trace hreg_pos.le hd h_traceω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The log-determinant expression that appears in the elliptical-potential lemma. -/ noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1515,8 +1632,6 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (L2 : ℝ) @@ -1529,13 +1644,37 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo CappedQuadraticWidthBound A reg x n ω (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd h_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) h_le_one L2 hL2 ?_ intro ω h_traceω exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd (hdet_of_trace ω h_traceω) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface where the remaining matrix-analysis input is the reusable +positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + refine + cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_le_one + L2 hL2 ?_ + intro ω h_traceω + exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd + (reg * (d : ℝ) + (n : ℝ) * L2) h_traceω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed by the regret chain. -/ @@ -2601,9 +2740,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg : reg ≠ 0) (hd : d ≠ 0) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hreg_pos : 0 < reg) (hd : d ≠ 0) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (L2 : ℝ) @@ -2622,7 +2759,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg hd h_quad_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) h_quad_le_one L2 hL2 h_ratio_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term @@ -2638,8 +2777,6 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (L2 : ℝ) @@ -2658,8 +2795,39 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_nonneg - h_quad_le_one L2 hL2 hdet_of_trace) + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one + L2 hL2 hdet_of_trace) + +/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term +when the remaining hard input is the reusable PSD matrix determinant/trace comparison +`det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best + h_arm hβ hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one + L2 hL2 hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the @@ -2679,8 +2847,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hreg_nonneg : 0 ≤ reg) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) (h_elliptical : ∀ᵐ ω ∂P, @@ -2693,8 +2860,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W (cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one - h_elliptical h_potential_le) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg (A := A) (reg := reg) + (x := x) (n := n) (P := P) hreg_nonneg) + h_quad_le_one h_elliptical h_potential_le) end LinUCB From c45b44629f4e0a25fb25bbbfbd8760cb77e97271 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 11:19:29 -0400 Subject: [PATCH 59/88] feat(linUCB): positive regularization now proves the design matrix is positive definite --- .../Online/Bandit/Algorithms/LinUCB.lean | 71 +++++++++++++++---- 1 file changed, 56 insertions(+), 15 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index a3b08e01..cbf2b6b3 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -161,6 +161,16 @@ lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : intro t _ simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) +/-- Positive regularization makes the process-level design matrix positive definite. -/ +lemma designMatrix_posDef (hreg_pos : 0 < reg) : + (designMatrix A reg x n ω).PosDef := by + unfold designMatrix + apply Matrix.PosDef.add_posSemidef + · exact Matrix.PosDef.smul Matrix.PosDef.one hreg_pos + · refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -521,6 +531,26 @@ lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : rw [designDet_zero_eq_reg_pow] exact pow_ne_zero d hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization makes every process-level design determinant nonzero. -/ +lemma designDet_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : + designDet A reg x n ω ≠ 0 := by + have hunit : IsUnit (designMatrix A reg x n ω) := + (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).isUnit + have hdet_unit : IsUnit (designMatrix A reg x n ω).det := + (Matrix.isUnit_iff_isUnit_det (A := designMatrix A reg x n ω)).mp hunit + exact hdet_unit.ne_zero + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, positive regularization makes all design determinants in a finite horizon +nonzero. -/ +lemma designDet_ae_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + exact Filter.Eventually.of_forall fun ω t _ht ↦ + designDet_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) (ω := ω) + hreg_pos + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) @@ -1468,6 +1498,23 @@ lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential exact Filter.Eventually.of_forall fun ω ↦ designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization discharges the determinant-nonvanishing and quadratic-form +nonnegativity obligations in the log-determinant elliptical-potential chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound {W : ℝ} + (hreg_pos : 0 < reg) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) + (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) + (n := n + 1) (P := P) hreg_pos) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + h_le_one h_potential_le + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a determinant-ratio upper bound, and the determinant-update step give the packaged capped @@ -2830,14 +2877,12 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound L2 hL2 hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the capped quadratic-width sum is bounded by the -log-determinant elliptical potential and that potential is bounded by `W`. - -This is the first theorem whose assumptions match the two matrix-analysis steps of the actual -elliptical-potential argument: +`2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final +log-determinant potential bound hold. -* prove `cappedQuadraticWidthSum ≤ ellipticalPotential`; -* prove `ellipticalPotential ≤ W`. -/ +The capped-sum/log-determinant part of the elliptical-potential argument is now proved internally: +positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while +`h_quad_le_one` supplies the cap needed for `min 1 q ≤ 2 * log (1 + q)`. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) @@ -2847,11 +2892,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hreg_nonneg : 0 ≤ reg) + (hreg_pos : 0 < reg) (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -2859,11 +2902,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_boun exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg (A := A) (reg := reg) - (x := x) (n := n) (P := P) hreg_nonneg) - h_quad_le_one h_elliptical h_potential_le) + (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) hreg_pos + h_quad_le_one h_potential_le) end LinUCB From c79afb07592fb5c4e9f3d52d12c1d949c7f1601f Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 13:15:17 -0400 Subject: [PATCH 60/88] feat(linUCB): linear-algebra bridge for the elliptical-potential regret theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 106 ++++++++++++++---- 1 file changed, 83 insertions(+), 23 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index cbf2b6b3..90e6693d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -317,6 +317,61 @@ lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg intro t ht _ht0 exact h_nonnegω t ht +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The matrix comparison needed to turn bounded feature vectors into the positive-time LinUCB +width cap. + +Mathematically, this says `x_aᵀ V_t⁻¹ x_a ≤ ‖x_a‖² / reg`. A later matrix-order proof should +derive it from `reg > 0` and `V_t = reg I + ∑ x_s x_sᵀ`. Keeping it as a named property makes the +remaining linear-algebra obligation precise and reusable. -/ +def WidthQuadraticFormLeFeatureSqNormDivReg + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (a : Fin K) (n : ℕ) (ω : Ω), + widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the +quadratic form is at most one. -/ +lemma widthQuadraticForm_le_one_of_featureSqNorm_le_reg + (a : Fin K) + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) + (h_feature_le : featureSqNorm x a ≤ reg) : + widthQuadraticForm A reg x a n ω ≤ 1 := by + refine (h_width a n ω).trans ?_ + rwa [div_le_one hreg_pos] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure positive-time width cap from the matrix comparison and an almost-sure +`featureSqNorm ≤ reg` bound along the selected actions. -/ +lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) + (h_feature_le : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by + filter_upwards [h_feature_le] with ω h_feature_leω + intro t ht _ht0 + exact widthQuadraticForm_le_one_of_featureSqNorm_le_reg + (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (A t ω) h_width hreg_pos + (h_feature_leω t ht) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure positive-time width cap from the matrix comparison and a selected-feature budget +`featureSqNorm ≤ L2`, when `L2 ≤ reg`. -/ +lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) {L2 : ℝ} + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by + refine widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg + (A := A) (reg := reg) (x := x) (n := n) (P := P) h_width hreg_pos ?_ + filter_upwards [hL2] with ω hL2ω + intro t ht + exact (hL2ω t ht).trans hL2_le_reg + /-- The process-level elliptical confidence width. -/ noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := @@ -1679,10 +1734,10 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_of_trace : ∀ ω, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → designDet A reg x n ω ≤ @@ -1694,7 +1749,9 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) - h_le_one L2 hL2 ?_ + (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + L2 hL2 ?_ intro ω h_traceω exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd @@ -1705,18 +1762,18 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_trace : MatrixDetLeTraceAveragePow d) : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by refine cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_le_one - L2 hL2 ?_ + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + h_width_le_feature L2 hL2 hL2_le_reg ?_ intro ω h_traceω exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd @@ -2775,9 +2832,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. -The remaining matrix-analysis input is isolated in `h_ratio_of_trace`: a future determinant/trace -comparison theorem should prove that the trace budget implies the displayed determinant-ratio -bound. -/ +The remaining matrix-analysis inputs are isolated as named hypotheses: `h_width_le_feature` should +come from the inverse-design comparison `xᵀV⁻¹x ≤ ‖x‖² / reg`, and `h_ratio_of_trace` should come +from a determinant/trace comparison proving that the trace budget implies the displayed +determinant-ratio bound. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) @@ -2788,10 +2846,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (h_ratio_of_trace : ∀ ω, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → designDetRatio A reg x n ω ≤ @@ -2809,10 +2867,12 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) - h_quad_le_one L2 hL2 h_ratio_of_trace) + (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + L2 hL2 h_ratio_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the remaining matrix-analysis input is stated as the determinant upper bound +when the determinant/trace input is stated as the determinant upper bound `det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le [Nonempty (Fin K)] @@ -2824,10 +2884,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_of_trace : ∀ ω, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → designDet A reg x n ω ≤ @@ -2842,11 +2902,11 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one - L2 hL2 hdet_of_trace) + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + h_width_le_feature L2 hL2 hL2_le_reg hdet_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the remaining hard input is the reusable PSD matrix determinant/trace comparison +when the determinant/trace input is the reusable PSD matrix determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound [Nonempty (Fin K)] @@ -2858,10 +2918,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) (hdet_trace : MatrixDetLeTraceAveragePow d) : ∀ᵐ ω ∂P, regret ν A n ω ≤ @@ -2873,8 +2933,8 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound h_arm hβ hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_quad_le_one - L2 hL2 hdet_trace) + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + h_width_le_feature L2 hL2 hL2_le_reg hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 0fb3ae1b5e9f6e46a52a9cd0af22f1e76e2d87ca Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 13:21:17 -0400 Subject: [PATCH 61/88] feat(linUCB): convert matrix inequalities into scalar quadratic-form inequalities --- .../Online/Bandit/Algorithms/LinUCB.lean | 33 ++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 90e6693d..b8adacc6 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -9,6 +9,7 @@ public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.Analysis.Matrix.Order public import Mathlib.Data.Real.StarOrdered public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement @@ -23,7 +24,7 @@ Chapter 19 of *Bandit Algorithms*: open MeasureTheory ProbabilityTheory Filter Real Finset Learning -open scoped ENNReal NNReal Matrix +open scoped ENNReal NNReal Matrix MatrixOrder namespace Bandits @@ -171,6 +172,36 @@ lemma designMatrix_posDef (hreg_pos : 0 < reg) : intro t _ simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design matrix dominates its regularization part: after subtracting `reg • I`, what remains +is the sum of observed rank-one feature matrices, hence positive semidefinite. -/ +lemma designMatrix_sub_reg_smul_one_posSemidef : + (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by + have hsum : + (∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω))).PosSemidef := by + refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + simpa [designMatrix, add_sub_cancel_left] using hsum + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix-order form of `designMatrix_sub_reg_smul_one_posSemidef`: `reg • I ≤ V_n`. -/ +lemma reg_smul_one_le_designMatrix : + reg • (1 : Matrix (Fin d) (Fin d) ℝ) ≤ designMatrix A reg x n ω := by + rw [Matrix.le_iff] + exact designMatrix_sub_reg_smul_one_posSemidef (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix order is preserved by evaluating the quadratic form against a fixed feature vector. -/ +lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} + (hMN : M ≤ N) (u : Feature d) : + dotProduct u (M *ᵥ u) ≤ dotProduct u (N *ᵥ u) := by + have h_nonneg : 0 ≤ dotProduct u ((N - M) *ᵥ u) := by + simpa using (Matrix.le_iff.mp hMN).dotProduct_mulVec_nonneg u + rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg + exact sub_nonneg.mp h_nonneg + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := From e4ac7701ac93c1256753eaa17f5b5436eb744201 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 13:31:06 -0400 Subject: [PATCH 62/88] feat(linUCB): width comparison needed by the regret theorem path --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index b8adacc6..99a2929d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -202,6 +202,31 @@ lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg exact sub_nonneg.mp h_nonneg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The inverse of the regularized identity is the reciprocal-scaled identity. -/ +lemma reg_smul_one_inv (hreg : reg ≠ 0) : + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ = + reg⁻¹ • (1 : Matrix (Fin d) (Fin d) ℝ) := by + rw [Matrix.inv_eq_left_inv] + simp [smul_smul, hreg] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The quadratic form induced by `(reg • I)⁻¹` is the squared norm divided by `reg`. -/ +lemma dotProduct_reg_smul_one_inv_mulVec (hreg : reg ≠ 0) (u : Feature d) : + dotProduct u (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ u) = + dotProduct u u / reg := by + rw [reg_smul_one_inv (reg := reg) (d := d) hreg] + simp [Matrix.smul_mulVec, div_eq_inv_mul, mul_comm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Arm-specific form of `dotProduct_reg_smul_one_inv_mulVec`. -/ +lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div + (hreg : reg ≠ 0) (a : Fin K) : + dotProduct (x a) (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) = + featureSqNorm x a / reg := by + simpa [featureSqNorm] using + dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -360,6 +385,37 @@ def WidthQuadraticFormLeFeatureSqNormDivReg ∀ (a : Fin K) (n : ℕ) (ω : Ω), widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the inverse design matrix is bounded by the inverse regularized identity, then the LinUCB +quadratic width is bounded by `featureSqNorm / reg` for one arm, time, and sample point. -/ +lemma widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le + (a : Fin K) + (h_inv : (designMatrix A reg x n ω)⁻¹ ≤ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) + (hreg : reg ≠ 0) : + widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg := by + calc + widthQuadraticForm A reg x a n ω = + dotProduct (x a) (((designMatrix A reg x n ω)⁻¹) *ᵥ (x a)) := rfl + _ ≤ dotProduct (x a) + (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) := + dotProduct_mulVec_le_of_matrix_le h_inv (x a) + _ = featureSqNorm x a / reg := + dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div + (reg := reg) (x := x) hreg a + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A pointwise inverse-order comparison for all times and sample points gives the reusable +`WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ +lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le + (hreg : reg ≠ 0) + (h_inv : ∀ n ω, (designMatrix A reg x n ω)⁻¹ ≤ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) : + WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by + intro a n ω + exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le + (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a (h_inv n ω) hreg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the quadratic form is at most one. -/ From 6c16639090f29498ed55819b731b70af0999e539 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 14:17:59 -0400 Subject: [PATCH 63/88] feat(linUCB): regret theorem now using updated matrix theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++------ 1 file changed, 40 insertions(+), 16 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 99a2929d..99214ae9 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -227,6 +227,24 @@ lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div simpa [featureSqNorm] using dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret +route. + +Mathematically, this should follow from `reg • I ≤ V_t` and positive regularization: inversion +reverses the positive-definite matrix order, so `V_t⁻¹ ≤ (reg • I)⁻¹`. -/ +def DesignMatrixInvLeRegInv + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (n : ℕ) (ω : Ω), + (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection lemma for the named inverse-order obligation. -/ +lemma DesignMatrixInvLeRegInv.apply + (h_inv : DesignMatrixInvLeRegInv A reg x) (n : ℕ) (ω : Ω) : + (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := + h_inv n ω + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -409,12 +427,12 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in `WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (hreg : reg ≠ 0) - (h_inv : ∀ n ω, (designMatrix A reg x n ω)⁻¹ ≤ - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) : + (h_inv : DesignMatrixInvLeRegInv A reg x) : WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by intro a n ω exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a (h_inv n ω) hreg + (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a + (h_inv.apply n ω) hreg omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the @@ -1821,7 +1839,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1837,7 +1855,10 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + (x := x) (n := n) (P := P) + (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) + (x := x) hreg_pos.ne' h_inv_le_reg) + hreg_pos hL2 hL2_le_reg) L2 hL2 ?_ intro ω h_traceω exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) @@ -1849,7 +1870,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1860,7 +1881,7 @@ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound refine cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_width_le_feature L2 hL2 hL2_le_reg ?_ + h_inv_le_reg L2 hL2 hL2_le_reg ?_ intro ω h_traceω exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd @@ -2919,9 +2940,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_width_le_feature` should -come from the inverse-design comparison `xᵀV⁻¹x ≤ ‖x‖² / reg`, and `h_ratio_of_trace` should come -from a determinant/trace comparison proving that the trace budget implies the displayed +The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_le_reg` is the +inverse-design comparison `V_t⁻¹ ≤ (reg I)⁻¹`, and `h_ratio_of_trace` should come from a +determinant/trace comparison proving that the trace budget implies the displayed determinant-ratio bound. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound [Nonempty (Fin K)] @@ -2933,7 +2954,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -2955,7 +2976,10 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) h_width_le_feature hreg_pos hL2 hL2_le_reg) + (x := x) (n := n) (P := P) + (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) + (x := x) hreg_pos.ne' h_inv_le_reg) + hreg_pos hL2 hL2_le_reg) L2 hL2 h_ratio_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term @@ -2971,7 +2995,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -2990,7 +3014,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_width_le_feature L2 hL2 hL2_le_reg hdet_of_trace) + h_inv_le_reg L2 hL2 hL2_le_reg hdet_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term when the determinant/trace input is the reusable PSD matrix determinant/trace comparison @@ -3005,7 +3029,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_width_le_feature : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -3021,7 +3045,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_width_le_feature L2 hL2 hL2_le_reg hdet_trace) + h_inv_le_reg L2 hL2 hL2_le_reg hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 92c3df161b2b82431d68c7c368ced811ec3da352 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Wed, 17 Jun 2026 14:21:53 -0400 Subject: [PATCH 64/88] feat(linUCB): regret theorem dependent on reusable linear-algebra theorem --- .../Online/Bandit/Algorithms/LinUCB.lean | 51 ++++++++++++++----- 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 99214ae9..5955c6a1 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -227,6 +227,14 @@ lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div simpa [featureSqNorm] using dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Reusable matrix-analysis theorem needed for the LinUCB width comparison. + +It states the usual inverse anti-monotonicity of positive-definite matrices in the PSD order: +if `M` is positive definite and `M ≤ N`, then inversion reverses the order. -/ +def MatrixInvAntiMonoOnPosDef (d : ℕ) : Prop := + ∀ M N : Matrix (Fin d) (Fin d) ℝ, M.PosDef → M ≤ N → N⁻¹ ≤ M⁻¹ + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret route. @@ -245,6 +253,19 @@ lemma DesignMatrixInvLeRegInv.apply (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := h_inv n ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The reusable positive-definite inverse anti-monotonicity theorem implies the LinUCB-specific +inverse-design comparison. -/ +lemma DesignMatrixInvLeRegInv.of_matrix_inv_antitone + (hreg_pos : 0 < reg) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) : + DesignMatrixInvLeRegInv A reg x := by + intro n ω + exact h_inv_antitone (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) + (designMatrix A reg x n ω) + (Matrix.PosDef.smul Matrix.PosDef.one hreg_pos) + (reg_smul_one_le_designMatrix (A := A) (reg := reg) (x := x) (n := n) (ω := ω)) + /-- Trace of the process-level regularized design matrix. -/ noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := @@ -1839,7 +1860,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1857,7 +1878,9 @@ lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bo (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) (x := x) (n := n) (P := P) (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' h_inv_le_reg) + (x := x) hreg_pos.ne' + (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) + (x := x) hreg_pos h_inv_antitone)) hreg_pos hL2 hL2_le_reg) L2 hL2 ?_ intro ω h_traceω @@ -1870,7 +1893,7 @@ omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -1881,7 +1904,7 @@ lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound refine cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_le_reg L2 hL2 hL2_le_reg ?_ + h_inv_antitone L2 hL2 hL2_le_reg ?_ intro ω h_traceω exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd @@ -2940,9 +2963,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_le_reg` is the -inverse-design comparison `V_t⁻¹ ≤ (reg I)⁻¹`, and `h_ratio_of_trace` should come from a -determinant/trace comparison proving that the trace budget implies the displayed +The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_antitone` is the +generic inverse anti-monotonicity theorem for positive-definite matrices, and `h_ratio_of_trace` +should come from a determinant/trace comparison proving that the trace budget implies the displayed determinant-ratio bound. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound [Nonempty (Fin K)] @@ -2954,7 +2977,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -2978,7 +3001,9 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) (x := x) (n := n) (P := P) (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' h_inv_le_reg) + (x := x) hreg_pos.ne' + (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) + (x := x) hreg_pos h_inv_antitone)) hreg_pos hL2 hL2_le_reg) L2 hL2 h_ratio_of_trace) @@ -2995,7 +3020,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -3014,7 +3039,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_ (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_le_reg L2 hL2 hL2_le_reg hdet_of_trace) + h_inv_antitone L2 hL2 hL2_le_reg hdet_of_trace) /-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term when the determinant/trace input is the reusable PSD matrix determinant/trace comparison @@ -3029,7 +3054,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_le_reg : DesignMatrixInvLeRegInv A reg x) + (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) (L2 : ℝ) (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) (hL2_le_reg : L2 ≤ reg) @@ -3045,7 +3070,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_le_reg L2 hL2 hL2_le_reg hdet_trace) + h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 5fb147e6b855e514c4d4fa295a44c7317f428d97 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 09:03:25 -0400 Subject: [PATCH 65/88] feat(linUCB): remaining deterministic/regret-shell steps --- .../Online/Bandit/Algorithms/LinUCB.lean | 474 +++++++++++++++++- 1 file changed, 472 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 5955c6a1..71e93bf7 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -48,6 +48,13 @@ lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : rw [featureSqNorm, dotProduct] exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) +/-- Uniform squared feature-norm bound for finite-action LinUCB. + +This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared +form as `‖x_a‖₂² ≤ L2` for every action. -/ +def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := + ∀ a, featureSqNorm x a ≤ L2 + /-- History-level regularized design matrix for LinUCB. -/ noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := @@ -323,6 +330,14 @@ lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) (x := x) (n := n) (ω := ω) L2 hL2ω +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A uniform finite-action feature bound implies the selected-action feature bound through any +finite horizon. -/ +lemma featureSqNorm_ae_le_of_featureSqNormBound + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2 := + Filter.Eventually.of_forall fun ω t _ht ↦ hL2 (A t ω) + /-- The process-level reward-feature vector built from history up to time `n` excluded. -/ noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := @@ -1225,6 +1240,25 @@ lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} rw [min_eq_right hq_le_one] exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar log inequality used in the textbook elliptical-potential proof: for `0 ≤ q`, +`min 1 q ≤ 2 * log (1 + q)`. -/ +lemma min_one_le_two_mul_log_one_add_of_nonneg {q : ℝ} + (hq_nonneg : 0 ≤ q) : + min 1 q ≤ 2 * Real.log (1 + q) := by + by_cases hq_le_one : q ≤ 1 + · exact min_one_le_two_mul_log_one_add_of_nonneg_le_one hq_nonneg hq_le_one + · have hq_one : 1 ≤ q := by linarith + have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := + Real.le_log_one_add_of_nonneg hq_nonneg + have hq_add_two_pos : 0 < q + 2 := by linarith + have hone_le_log_lower : 1 ≤ 2 * (2 * q / (q + 2)) := by + rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] + rw [le_div_iff₀ hq_add_two_pos] + nlinarith + rw [min_eq_left hq_one] + exact hone_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ @@ -1244,6 +1278,25 @@ lemma cappedWidthTerm_le_ellipticalPotentialStep (x := x) (n := n) (ω := ω) hdet] exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing and nonnegativity of the selected quadratic form, the single +capped quadratic-width term is bounded by the one-step log-determinant potential. This is the +textbook form; no separate `q ≤ 1` assumption is needed because the term is already capped. -/ +lemma cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialStep A reg x n ω := by + by_cases hn : n = 0 + · rw [if_pos hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) + · rw [if_neg hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact min_one_le_two_mul_log_one_add_of_nonneg h_nonneg + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ @@ -1261,6 +1314,21 @@ lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the +per-step one-step-potential bound for the capped quadratic-width term. -/ +lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω := by + filter_upwards [hdet, h_nonneg] with ω hdetω h_nonnegω + intro t ht + exact cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg (A := A) (reg := reg) + (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- At horizon zero, the log-determinant potential is zero when the initial design determinant is nonzero. -/ @@ -1415,6 +1483,39 @@ lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_o intro t ht rw [h_eq t ht] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the +capped-sum/log-determinant elliptical-potential bound. + +This is the capped form used in the textbook proof of LinUCB: the quadratic forms do not need to +be bounded by `1`, because the accumulated quantity is `min 1 q_t`. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet + (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet_range_n h_nonneg) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization discharges determinant nonvanishing and nonnegativity, yielding the +capped-sum/log-determinant elliptical-potential bound directly. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos + (hreg_pos : 0 < reg) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) + (n := n + 1) (P := P) hreg_pos) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- The process-level capped quadratic-width input expected from an elliptical-potential argument. @@ -1830,6 +1931,42 @@ lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add rw [Real.log_pow, hbase] ring +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Textbook capped elliptical-potential budget from bounded selected feature norms and the +matrix-level determinant/trace comparison. + +Unlike `cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound`, this theorem bounds the capped +quadratic-width sum directly and does not assume the individual quadratic forms are at most `1`. -/ +lemma cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ + 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) + rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] + have h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := + widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le + have h_potential_le : ∀ᵐ ω ∂P, + ellipticalPotential A reg x n ω ≤ + 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) := by + exact ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' h_nonneg) + (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 + hdet_trace) + filter_upwards [cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos, h_potential_le] with + ω h_capped_le h_potentialω + exact h_capped_le.trans h_potentialω + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Feature-norm-budget interface with the log term rewritten in the standard `2 * d * log(1 + n L² / (reg d))` shape. -/ @@ -1950,6 +2087,72 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by simp [index_zero, width_zero] +/-- The pointwise LinUCB confidence event used by the finite-action regret proof. + +For every positive process time, the best arm's true mean lies below its optimistic index, and the +selected arm's pessimistic index lies below its true mean. On this event, the max-index property of +LinUCB turns optimism into an instantaneous regret bound. -/ +def LinUCBConfidenceEvent [Nonempty (Fin K)] + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] + +omit [IsMarkovKernel ν] in +/-- Uniform bound on arm gaps, used as the finite-action analogue of the textbook bounded +instantaneous-regret assumption. -/ +def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := + ∀ a, gap ν a ≤ G + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ +lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → gap ν (A t ω) ≤ G := + Filter.Eventually.of_forall fun ω t _ht ↦ hG (A t ω) + +omit [IsMarkovKernel ν] in +/-- First projection from the packaged LinUCB confidence event: optimism for the best arm. -/ +lemma LinUCBConfidenceEvent.best [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by + intro t ht + exact (h_conf t ht).1 + +omit [IsMarkovKernel ν] in +/-- Second projection from the packaged LinUCB confidence event: validity of the selected arm's +lower confidence inequality. -/ +lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ t, t ≠ 0 → + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by + intro t ht + exact (h_conf t ht).2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ +lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by + filter_upwards [h_conf] with ω h_confω + exact h_confω.best + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure projection of the packaged confidence event to the selected arm's lower confidence +inequality. -/ +lemma linUCBConfidenceEvent_ae_arm [Nonempty (Fin K)] + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by + filter_upwards [h_conf] with ω h_confω + exact h_confω.arm + lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : designMatrix A reg x n ω = @@ -2546,6 +2749,43 @@ lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) +omit [IsMarkovKernel ν] in +/-- Pointwise capped LinUCB regret bound for one positive time. + +If the instantaneous gap is bounded by `2`, and the confidence/max-index argument gives the usual +`2 * sqrt(β_t) * width_t` bound, then monotonicity up to the terminal `β n` gives the textbook +capped form `2 * sqrt(β n) * sqrt(min 1 q_t)`, where `q_t` is the width quadratic form. -/ +lemma gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm + (t : ℕ) + (h_gap_two : gap ν (A t ω) ≤ 2) + (h_gap_width : gap ν (A t ω) ≤ + 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) + (hβ_le : β (t + 1) ≤ β n) + (hβn_one : 1 ≤ β n) : + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + by_cases hq_le_one : widthQuadraticForm A reg x (A t ω) t ω ≤ 1 + · have hwidth_nonneg : 0 ≤ width A reg x (A t ω) t ω := Real.sqrt_nonneg _ + have hsqrt_le : √(β (t + 1)) ≤ √(β n) := Real.sqrt_le_sqrt hβ_le + have hbonus_le : + 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) ≤ + 2 * (√(β n) * width A reg x (A t ω) t ω) := by + exact mul_le_mul_of_nonneg_left + (mul_le_mul_of_nonneg_right hsqrt_le hwidth_nonneg) (by norm_num) + have hmin : + √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)) = + width A reg x (A t ω) t ω := by + rw [min_eq_right hq_le_one, width] + simpa [hmin] using h_gap_width.trans hbonus_le + · have hq_one : 1 ≤ widthQuadraticForm A reg x (A t ω) t ω := by linarith + have hsqrt_one : 1 ≤ √(β n) := by + simpa using (Real.one_le_sqrt).2 hβn_one + have htwo_le : + 2 ≤ 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + rw [min_eq_left hq_one, Real.sqrt_one] + nlinarith + exact h_gap_two.trans htwo_le + omit [IsMarkovKernel ν] in /-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded by the corresponding sum of pointwise bounds. -/ @@ -2576,6 +2816,26 @@ lemma regret_le_sum_width_of_forall_gap_le · simp [ht0] · simpa [ht0] using h_gap t ht ht0 +omit [IsMarkovKernel ν] in +/-- A pathwise cumulative-regret bound obtained by summing the positive-time capped LinUCB width +bound. -/ +lemma regret_le_sum_sqrt_capped_width_of_forall_gap_le + (h_gap : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : + regret ν A n ω ≤ + ∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) + (B := fun t ↦ + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simpa [ht0] using h_gap t ht ht0 + omit [IsMarkovKernel ν] in /-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : @@ -2604,6 +2864,112 @@ lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : (fun t ↦ if t = 0 then 0 else √(β (t + 1))) (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) +omit [IsMarkovKernel ν] in +/-- Cauchy-Schwarz bound for the positive-time capped LinUCB bonus sum. -/ +lemma sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum + (hβn_nonneg : 0 ≤ β n) + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ≤ + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by + calc + (∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) + = 2 * ∑ t ∈ range n, + (if t = 0 then 0 else √(β n)) * + (if t = 0 then 0 + else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + rw [mul_sum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) * + √(∑ t ∈ range n, + (if t = 0 then 0 + else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) ^ 2)) := by + gcongr + exact Real.sum_mul_le_sqrt_mul_sqrt (range n) + (fun t ↦ if t = 0 then 0 else √(β n)) + (fun t ↦ if t = 0 then 0 + else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) + _ ≤ 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by + gcongr + · calc + (∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) + ≤ ∑ _t ∈ range n, β n := by + refine sum_le_sum ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0, hβn_nonneg] + · simp [ht0, Real.sq_sqrt hβn_nonneg] + _ = (n : ℝ) * β n := by + simp [sum_const, nsmul_eq_mul] + · rw [cappedQuadraticWidthSum] + refine le_of_eq ?_ + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · have hmin_nonneg : + 0 ≤ min 1 (widthQuadraticForm A reg x (A t ω) t ω) := by + exact le_min zero_le_one (h_nonneg t ht ht0) + simp [ht0, Real.sq_sqrt hmin_nonneg] + +omit [IsMarkovKernel ν] in +/-- Pathwise cumulative-regret bound using the textbook capped quadratic-width sum. -/ +lemma regret_le_initial_add_sqrt_nat_mul_beta_capped_sum + (hβn_nonneg : 0 ≤ β n) + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_gap : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by + refine (regret_le_sum_sqrt_capped_width_of_forall_gap_le (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) h_gap).trans ?_ + have hsplit : + (∑ t ∈ range n, + if t = 0 then gap ν (A 0 ω) + else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) = + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + ∑ t ∈ range n, + if t = 0 then 0 + else 2 * (√(β n) * + √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + rw [← sum_add_distrib] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp [ht0] + rw [hsplit] + exact add_le_add le_rfl + (sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum + (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω) + hβn_nonneg h_nonneg) + +omit [IsMarkovKernel ν] in +/-- If the capped quadratic-width sum is bounded by `W`, the pathwise capped regret bound can use +`√W` in place of the realized capped-sum square root. -/ +lemma regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (W : ℝ) + (h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω))) + (hW : cappedQuadraticWidthSum A reg x n ω ≤ W) : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √W) := by + refine h_regret.trans ?_ + gcongr + /-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is nonnegative. -/ lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : @@ -2959,6 +3325,61 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_boun (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) +/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus +`2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is almost surely bounded +by `W`. + +This version follows the proof structure of *Bandit Algorithms*, Theorem 19.2: optimism gives the +width bound, bounded instantaneous gaps give the cap, monotonicity of `β` moves all confidence +radii to `β n`, and Cauchy-Schwarz turns the sum into the square root of the capped quadratic-width +sum. -/ +lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) + (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → + estimatedReward A R reg x (A n ω) n ω - + √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) + (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) + (hβ_nonneg : ∀ t, 0 ≤ β t) + (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] + with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω + have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + intro t ht _ht0 + exact h_quad_nonnegω t ht + have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + intro t ht ht0 + have hβ_le : β (t + 1) ≤ β n := + hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) + have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 + have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) + have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos + have hβn_one : 1 ≤ β n := hβ_one.trans (hβ_mono hn_one) + exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) + (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one + have h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := + regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (hβ_nonneg n) h_quad_pos + h_gap_capped + simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using + regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. @@ -3072,13 +3493,62 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) +/-- Textbook-shaped finite-action LinUCB regret theorem. + +This theorem is the same deterministic regret skeleton as the theorem above, but with assumptions +packaged in the way the finite-action linear-bandit proof is usually read: + +* `h_conf` is the high-probability confidence event for all positive times; +* `h_gap_bound` is the bounded instantaneous-regret/gap assumption; +* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`; +* `hdet_trace` is the reusable determinant/trace matrix-analysis obligation. + +The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression +`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this +formalization lets the deterministic algorithm play its default initial arm at time zero. -/ +lemma regret_ae_le_textbook_finite_action + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) + (h_gap_bound : GapBound (K := K) ν 2) + (hβ_nonneg : ∀ t, 0 ≤ β t) + (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by + filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) + 2 h_gap_bound] with ω h_gapω + intro t ht _ht0 + exact h_gapω t ht + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + h_gap_two hβ_nonneg hβ_one hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 + (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) + (P := P) L2 hL2) + hdet_trace) + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final log-determinant potential bound hold. -The capped-sum/log-determinant part of the elliptical-potential argument is now proved internally: +The capped-sum/log-determinant part of the elliptical-potential argument is proved internally: positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while -`h_quad_le_one` supplies the cap needed for `min 1 q ≤ 2 * log (1 + q)`. -/ +`h_quad_le_one` lets this older theorem feed the uncapped `widthSqSum` regret route. -/ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) From 2057e7326e2844c3fca0ab737c6d9b375cd13ee2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:00:07 -0400 Subject: [PATCH 66/88] feat(linUCB): the matrix determinant/trace assumption --- .../Online/Bandit/Algorithms/LinUCB.lean | 56 +++++++++++++++++-- 1 file changed, 51 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 71e93bf7..0ce917b4 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -8,6 +8,7 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.Analysis.MeanInequalities public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.Analysis.Matrix.Order public import Mathlib.Data.Real.StarOrdered @@ -1129,6 +1130,53 @@ For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar AM-GM in the form used for PSD matrix eigenvalues: +the product of nonnegative entries is bounded by the arithmetic mean to the `card` power. -/ +lemma prod_le_average_pow_of_nonneg {ι : Type*} [Fintype ι] [Nonempty ι] + (z : ι → ℝ) (hz : ∀ i, 0 ≤ z i) : + (∏ i, z i) ≤ ((∑ i, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by + classical + have hN_pos : 0 < (Fintype.card ι : ℝ) := by + exact_mod_cast Fintype.card_pos_iff.mpr inferInstance + have hweights_pos : 0 < ∑ i : ι, (1 : ℝ) := by + simpa using hN_pos + have h_amgm := Real.geom_mean_le_arith_mean (s := Finset.univ) + (w := fun _ : ι ↦ (1 : ℝ)) (z := z) + (by intro i hi; norm_num) hweights_pos (by intro i hi; exact hz i) + have h_amgm' : + (∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹) ≤ + (∑ i : ι, z i) / (Fintype.card ι : ℝ) := by + simpa using h_amgm + have hprod_nonneg : 0 ≤ ∏ i : ι, z i := by + exact Finset.prod_nonneg fun i _ ↦ hz i + have hraise := Real.rpow_le_rpow (Real.rpow_nonneg hprod_nonneg _) h_amgm' hN_pos.le + have hleft : + ((∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹)) ^ (Fintype.card ι : ℝ) = + ∏ i : ι, z i := by + rw [← Real.rpow_mul hprod_nonneg] + rw [inv_mul_cancel₀ hN_pos.ne'] + simp + have hright : + ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ (Fintype.card ι : ℝ) = + ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by + rw [Real.rpow_natCast] + simpa [hleft, hright] using hraise + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- PSD matrix determinant/trace comparison from AM-GM over eigenvalues: +`det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma matrixDetLeTraceAveragePow : MatrixDetLeTraceAveragePow d := by + intro M hM + by_cases hd : d = 0 + · subst d + simp + · haveI : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp (Nat.pos_of_ne_zero hd) + rw [hM.1.det_eq_prod_eigenvalues, hM.1.trace_eq_sum_eigenvalues] + simpa using prod_le_average_pow_of_nonneg + (z := fun i : Fin d ↦ hM.1.eigenvalues i) + (fun i ↦ hM.eigenvalues_nonneg i) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the design matrix is positive semidefinite. -/ @@ -3500,8 +3548,7 @@ packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; * `h_gap_bound` is the bounded instantaneous-regret/gap assumption; -* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`; -* `hdet_trace` is the reusable determinant/trace matrix-analysis obligation. +* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression `2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this @@ -3514,8 +3561,7 @@ lemma regret_ae_le_textbook_finite_action (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + @@ -3540,7 +3586,7 @@ lemma regret_ae_le_textbook_finite_action (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) (P := P) L2 hL2) - hdet_trace) + matrixDetLeTraceAveragePow) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From fdf56f92161bff5136ffeefdb38a10b108e97dcc Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:06:29 -0400 Subject: [PATCH 67/88] =?UTF-8?q?feat(linUCB):=20regret=20theorem=20no=20l?= =?UTF-8?q?onger=20requires=20the=20assumption=20hd=20:=20d=20=E2=89=A0=20?= =?UTF-8?q?0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 125 +++++++++++++++--- 1 file changed, 104 insertions(+), 21 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 0ce917b4..ca2af36b 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2135,6 +2135,36 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by simp [index_zero, width_zero] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every least-squares reward estimate is zero. -/ +lemma estimatedReward_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + estimatedReward A R reg x a n ω = 0 := by + subst d + simp [estimatedReward, dotProduct] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB quadratic width form is zero. -/ +lemma widthQuadraticForm_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + widthQuadraticForm A reg x a n ω = 0 := by + subst d + simp [widthQuadraticForm, dotProduct] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB width is zero. -/ +lemma width_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + width A reg x a n ω = 0 := by + simp [width, widthQuadraticForm_eq_zero_of_dim_eq_zero (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hd a] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB index is zero. -/ +lemma index_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + index A R reg β x a n ω = 0 := by + simp [index, estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) hd a, + width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hd a] + /-- The pointwise LinUCB confidence event used by the finite-action regret proof. For every positive process time, the best arm's true mean lies below its optimistic index, and the @@ -2181,6 +2211,28 @@ lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] intro t ht exact (h_conf t ht).2 +omit [IsMarkovKernel ν] in +/-- In zero feature dimension, the confidence event forces every positive-time selected gap to be +nonpositive. The best-arm index is zero, and the selected-arm pessimistic index is also zero. -/ +lemma gap_nonpos_of_confidence_dim_eq_zero [Nonempty (Fin K)] + (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) + (t : ℕ) (ht : t ≠ 0) : + gap ν (A t ω) ≤ 0 := by + have hbest := LinUCBConfidenceEvent.best (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (ω := ω) h_conf t ht + have harm := LinUCBConfidenceEvent.arm (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (ω := ω) h_conf t ht + rw [gap_eq_bestArm_sub] + have hbest0 : (ν (bestArm ν))[id] ≤ 0 := by + simpa [index_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) (β := β) + (x := x) (n := t) (ω := ω) hd (bestArm ν)] using hbest + have harm0 : 0 ≤ (ν (A t ω))[id] := by + simpa [estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) + (x := x) (n := t) (ω := ω) hd (A t ω), + width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := t) + (ω := ω) hd (A t ω)] using harm + linarith + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] @@ -3209,6 +3261,32 @@ lemma initial_gap_sum_eq : if n = 0 then 0 else gap ν (A 0 ω) := by cases n <;> simp +omit [IsMarkovKernel ν] in +/-- In zero feature dimension, the confidence event bounds cumulative regret by the initial gap. +There is no positive-time width contribution because all widths are zero. -/ +lemma regret_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] + (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by + refine (regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) + (B := fun t ↦ if t = 0 then gap ν (A 0 ω) else 0) ?_).trans ?_ + · intro t _ht + by_cases ht0 : t = 0 + · simp [ht0] + · simpa [ht0] using + gap_nonpos_of_confidence_dim_eq_zero (A := A) (R := R) (reg := reg) + (β := β) (x := x) (ν := ν) (ω := ω) hd h_conf t ht0 + · rw [initial_gap_sum_eq] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure zero-dimensional version of the finite-action LinUCB regret skeleton. -/ +lemma regret_ae_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] + (hd : d = 0) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : + ∀ᵐ ω ∂P, regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by + filter_upwards [h_conf] with ω h_confω + exact regret_le_initial_gap_of_confidence_dim_eq_zero (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hd h_confω + /-- Almost surely, cumulative regret is bounded by the initial gap plus `2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` is nonnegative and monotone. -/ @@ -3560,33 +3638,38 @@ lemma regret_ae_le_textbook_finite_action (h_gap_bound : GapBound (K := K) ν 2) (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by - filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 h_gap_bound] with ω h_gapω - intro t ht _ht0 - exact h_gapω t ht - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) - (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) - h_gap_two hβ_nonneg hβ_one hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 - (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) - (P := P) L2 hL2) - matrixDetLeTraceAveragePow) + by_cases hd : d = 0 + · subst d + simpa using regret_ae_le_initial_gap_of_confidence_dim_eq_zero + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + (P := P) (d := 0) rfl h_conf + · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by + filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) + 2 h_gap_bound] with ω h_gapω + intro t ht _ht0 + exact h_gapω t ht + exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (P := P) h_conf) + h_gap_two hβ_nonneg hβ_one hβ_mono + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 + (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) + (P := P) L2 hL2) + matrixDetLeTraceAveragePow) /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final From 2ca2a1b173d006db48e0f0506fcabd91f99612d5 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:10:33 -0400 Subject: [PATCH 68/88] =?UTF-8?q?feat(linUCB):=20final=20theorem=20no=20lo?= =?UTF-8?q?nger=20exposes=20the=20raw=20GapBound=20=CE=BD=202=20assumption?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 33 +++++++++++++++++-- 1 file changed, 30 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index ca2af36b..82c63fef 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2185,6 +2185,32 @@ instantaneous-regret assumption. -/ def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := ∀ a, gap ν a ≤ G +omit [IsMarkovKernel ν] in +/-- Uniform bound on arm means. For finite-action linear bandits this is a convenient way to state +the usual bounded expected-reward assumption, for example `(ν a)[id] ∈ [-1, 1]`. -/ +def MeanRewardBound (ν : Kernel (Fin K) ℝ) (lo hi : ℝ) : Prop := + ∀ a, lo ≤ (ν a)[id] ∧ (ν a)[id] ≤ hi + +omit [IsMarkovKernel ν] in +/-- If every arm mean lies in `[lo, hi]`, then every arm gap is at most `hi - lo`. -/ +lemma gap_le_of_meanRewardBound [Nonempty (Fin K)] {lo hi : ℝ} + (hμ : MeanRewardBound ν lo hi) (a : Fin K) : + gap ν a ≤ hi - lo := by + rw [gap_eq_bestArm_sub] + have hbest_le : (ν (bestArm ν))[id] ≤ hi := (hμ (bestArm ν)).2 + have ha_ge : lo ≤ (ν a)[id] := (hμ a).1 + linarith + +omit [IsMarkovKernel ν] in +/-- Arm means in `[-1, 1]` imply the gap cap `gap ≤ 2` used by the capped regret argument. -/ +lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] + (hμ : MeanRewardBound ν (-1) 1) : + GapBound (K := K) ν 2 := by + intro a + have hgap := gap_le_of_meanRewardBound (ν := ν) (lo := -1) (hi := 1) hμ a + norm_num at hgap + exact hgap + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : @@ -3625,7 +3651,7 @@ This theorem is the same deterministic regret skeleton as the theorem above, but packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; -* `h_gap_bound` is the bounded instantaneous-regret/gap assumption; +* `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression @@ -3635,7 +3661,7 @@ lemma regret_ae_le_textbook_finite_action [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_gap_bound : GapBound (K := K) ν 2) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) @@ -3652,7 +3678,8 @@ lemma regret_ae_le_textbook_finite_action (P := P) (d := 0) rfl h_conf · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 h_gap_bound] with ω h_gapω + 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with + ω h_gapω intro t ht _ht0 exact h_gapω t ht exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound From 36fa51a421e776cec1b91577fa0ef42075fa10c1 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:03:11 -0400 Subject: [PATCH 69/88] =?UTF-8?q?feat(linUCB):=20remove=20h=CE=B2=5Fnonneg?= =?UTF-8?q?=20assumption?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Online/Bandit/Algorithms/LinUCB.lean | 22 +++++++++++++++---- 1 file changed, 18 insertions(+), 4 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 82c63fef..74746837 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3279,6 +3279,15 @@ lemma beta_sum_le_nat_mul_of_monotone _ = (n : ℝ) * β n := by simp [sum_const, nsmul_eq_mul] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive +horizon. -/ +lemma beta_nonneg_of_one_le_of_monotone + (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) {n : ℕ} (hn : n ≠ 0) : + 0 ≤ β n := by + have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) + exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) + omit [IsMarkovKernel ν] in /-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the horizon is zero. -/ @@ -3494,7 +3503,6 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (W : ℝ) (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) @@ -3502,6 +3510,11 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound ∀ᵐ ω ∂P, regret ν A n ω ≤ (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + by_cases hn : n = 0 + · subst n + exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] + have hβn_nonneg : 0 ≤ β n := + beta_nonneg_of_one_le_of_monotone (β := β) hβ_one hβ_mono hn filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → @@ -3526,7 +3539,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (hβ_nonneg n) h_quad_pos + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos h_gap_capped simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) @@ -3652,6 +3665,8 @@ packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; * `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; +* `hβ_one` and `hβ_mono` state that the confidence-radius schedule starts at least at one and is + monotone; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression @@ -3662,7 +3677,6 @@ lemma regret_ae_le_textbook_finite_action (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_nonneg : ∀ t, 0 ≤ β t) (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : @@ -3688,7 +3702,7 @@ lemma regret_ae_le_textbook_finite_action (x := x) (ν := ν) (P := P) h_conf) (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (P := P) h_conf) - h_gap_two hβ_nonneg hβ_one hβ_mono + h_gap_two hβ_one hβ_mono (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) From 3ad7f30c4c6a66408d3662ae6ba527c597298d74 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:09:10 -0400 Subject: [PATCH 70/88] feat(linUCB): theorem statement closer to textbook --- .../Online/Bandit/Algorithms/LinUCB.lean | 37 +++++++++++++++---- 1 file changed, 29 insertions(+), 8 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 74746837..e1b9a9ab 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3279,6 +3279,22 @@ lemma beta_sum_le_nat_mul_of_monotone _ = (n : ℝ) * β n := by simp [sum_const, nsmul_eq_mul] +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Minimal confidence-radius schedule assumptions used by the capped finite-action LinUCB regret +chain: the schedule starts at least at one and is monotone in time. -/ +def BetaSchedule (β : ℕ → ℝ) : Prop := + 1 ≤ β 1 ∧ Monotone β + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection from `BetaSchedule`: the confidence-radius schedule starts at least at one. -/ +lemma BetaSchedule.one (hβ : BetaSchedule β) : 1 ≤ β 1 := + hβ.1 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection from `BetaSchedule`: the confidence-radius schedule is monotone. -/ +lemma BetaSchedule.monotone (hβ : BetaSchedule β) : Monotone β := + hβ.2 + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive horizon. -/ @@ -3288,6 +3304,12 @@ lemma beta_nonneg_of_one_le_of_monotone have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A `BetaSchedule` is nonnegative at every positive horizon. -/ +lemma BetaSchedule.nonneg_of_ne_zero (hβ : BetaSchedule β) {n : ℕ} (hn : n ≠ 0) : + 0 ≤ β n := + beta_nonneg_of_one_le_of_monotone (β := β) hβ.one hβ.monotone hn + omit [IsMarkovKernel ν] in /-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the horizon is zero. -/ @@ -3503,7 +3525,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound estimatedReward A R reg x (A n ω) n ω - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) (W : ℝ) + (hβ_schedule : BetaSchedule β) (W : ℝ) (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : @@ -3514,7 +3536,7 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound · subst n exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] have hβn_nonneg : 0 ≤ β n := - beta_nonneg_of_one_le_of_monotone (β := β) hβ_one hβ_mono hn + hβ_schedule.nonneg_of_ne_zero hn filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → @@ -3526,11 +3548,11 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by intro t ht ht0 have hβ_le : β (t + 1) ≤ β n := - hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) + hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_one.trans (hβ_mono hn_one) + have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one @@ -3665,8 +3687,7 @@ packaged in the way the finite-action linear-bandit proof is usually read: * `h_conf` is the high-probability confidence event for all positive times; * `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; -* `hβ_one` and `hβ_mono` state that the confidence-radius schedule starts at least at one and is - monotone; +* `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression @@ -3677,7 +3698,7 @@ lemma regret_ae_le_textbook_finite_action (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) + (hβ_schedule : BetaSchedule β) (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, @@ -3702,7 +3723,7 @@ lemma regret_ae_le_textbook_finite_action (x := x) (ν := ν) (P := P) h_conf) (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (P := P) h_conf) - h_gap_two hβ_one hβ_mono + h_gap_two hβ_schedule (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.le) From be7acde83f358b6d0aeb794db6dbe9d6ff012837 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:20:24 -0400 Subject: [PATCH 71/88] feat(linUCB): theorem statement closer to textbook --- .../Online/Bandit/Algorithms/LinUCB.lean | 120 +++++++++++++++--- 1 file changed, 102 insertions(+), 18 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index e1b9a9ab..947877f4 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3567,6 +3567,66 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω +/-- Almost surely, on the LinUCB confidence event, cumulative regret is bounded by the simplified +initial-gap term plus `2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is +almost surely bounded by `W`. + +This is the good-event form of the deterministic regret argument. It separates the algorithmic +regret proof from the future concentration theorem: a later self-normalized concentration result +should prove that `LinUCBConfidenceEvent` holds with high probability, and this theorem converts +that event into the regret bound. -/ +lemma regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) + (hβ_schedule : BetaSchedule β) (W : ℝ) + (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, + LinUCBConfidenceEvent A R reg β x ν ω → + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by + by_cases hn : n = 0 + · subst n + exact Filter.Eventually.of_forall fun ω _h_confω ↦ by simp [regret] + have hβn_nonneg : 0 ≤ β n := + hβ_schedule.nonneg_of_ne_zero hn + filter_upwards [forall_index_le_index_arm h (bestArm ν), h_gap_two, h_quad_nonneg, hW] with + ω h_indexω h_gap_twoω h_quad_nonnegω hWω h_confω + have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + intro t ht _ht0 + exact h_quad_nonnegω t ht + have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → + gap ν (A t ω) ≤ + 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by + intro t ht ht0 + have h_gap_width : + gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := + gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) + (x := x) (ν := ν) (n := t) (ω := ω) (h_confω.best t ht0) + (h_confω.arm t ht0) (h_indexω t ht0) + have hβ_le : β (t + 1) ≤ β n := + hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) + have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 + have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) + have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos + have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) + exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) + (h_gap_twoω t ht ht0) h_gap_width hβ_le hβn_one + have h_regret : + regret ν A n ω ≤ + (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + + 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := + regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos + h_gap_capped + simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using + regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the feature-budget elliptical-potential term `2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. @@ -3680,49 +3740,52 @@ lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) -/-- Textbook-shaped finite-action LinUCB regret theorem. +/-- Textbook-shaped finite-action LinUCB regret theorem on the confidence event. -This theorem is the same deterministic regret skeleton as the theorem above, but with assumptions -packaged in the way the finite-action linear-bandit proof is usually read: +This theorem is the good-event form closest to the finite-action LinUCB proof in +*Bandit Algorithms*: after the deterministic algorithm/max-index argument and the elliptical +potential bound are proved, the only remaining probabilistic input is whether the confidence event +holds on a sample path. -* `h_conf` is the high-probability confidence event for all positive times; * `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; * `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; * `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. +The conclusion is an almost-sure implication: on almost every sample path, if +`LinUCBConfidenceEvent` holds, then the displayed regret bound holds. A future self-normalized +concentration theorem should prove that this confidence event has high probability for a concrete +textbook choice of `β`. + The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression `2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this formalization lets the deterministic algorithm play its default initial arm at time zero. -/ -lemma regret_ae_le_textbook_finite_action +lemma regret_ae_imp_le_textbook_finite_action [Nonempty (Fin K)] (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) (hβ_schedule : BetaSchedule β) (hreg_pos : 0 < reg) (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + LinUCBConfidenceEvent A R reg β x ν ω → + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by by_cases hd : d = 0 · subst d - simpa using regret_ae_le_initial_gap_of_confidence_dim_eq_zero - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) - (P := P) (d := 0) rfl h_conf + exact Filter.Eventually.of_forall fun ω h_confω ↦ by + simpa using regret_le_initial_gap_of_confidence_dim_eq_zero + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + (ω := ω) (d := 0) rfl h_confω · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with ω h_gapω intro t ht _ht0 exact h_gapω t ht - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound + exact regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - (linUCBConfidenceEvent_ae_best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) - (linUCBConfidenceEvent_ae_arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (P := P) h_conf) h_gap_two hβ_schedule (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) @@ -3733,6 +3796,27 @@ lemma regret_ae_le_textbook_finite_action (P := P) L2 hL2) matrixDetLeTraceAveragePow) +/-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to +hold almost surely. This is stronger than the textbook high-probability route and is mainly useful +as a compatibility wrapper for earlier lemmas in this file. -/ +lemma regret_ae_le_textbook_finite_action + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2, h_conf] with ω h_regret h_confω + exact h_regret h_confω + /-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus `2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final log-determinant potential bound hold. From 23fa3c77c6957767c6bb02ff90e001c682831fb2 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:27:08 -0400 Subject: [PATCH 72/88] feat(linUCB): new lemmas that lift sample-path implication into probability statements --- .../Online/Bandit/Algorithms/LinUCB.lean | 79 +++++++++++++++++++ 1 file changed, 79 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 947877f4..5bf068de 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -3796,6 +3796,85 @@ lemma regret_ae_imp_le_textbook_finite_action (P := P) L2 hL2) matrixDetLeTraceAveragePow) +/-- The confidence event is almost surely contained in the textbook finite-action regret-bound +event. + +This is the probability bridge needed after the good-event theorem: once a concentration theorem +proves that `LinUCBConfidenceEvent` has high probability, this lemma transfers that probability +mass to the displayed regret bound. -/ +lemma probReal_confidenceEvent_le_textbook_regret_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2] with ω h_regret h_confω + exact h_regret h_confω + +/-- High-probability wrapper for the textbook finite-action LinUCB regret bound. + +If a future self-normalized concentration theorem proves that the confidence event has probability +at least `1 - δ`, then the textbook regret bound has probability at least `1 - δ` as well. -/ +lemma probReal_textbook_regret_bound_ge_of_confidenceEvent_ge + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : + 1 - δ ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by + exact h_conf_prob.trans + (probReal_confidenceEvent_le_textbook_regret_bound (A := A) (R := R) (reg := reg) + (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule hreg_pos L2 hL2) + +/-- Failure-probability wrapper for the textbook finite-action LinUCB regret bound. + +If a future self-normalized concentration theorem proves that the confidence event fails with +probability at most `δ`, then the textbook regret bound fails with probability at most `δ`. -/ +lemma probReal_textbook_regret_bound_failure_le_of_confidenceEvent_failure_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_failure : + P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : + P.real {ω | + ¬ + regret ν A n ω ≤ + (if n = 0 then 0 else gap ν (A 0 ω)) + + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} ≤ δ := by + refine le_trans ?_ h_conf_failure + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω + exact h_regret_failure (h_regret h_confω) + /-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to hold almost surely. This is stronger than the textbook high-probability route and is mainly useful as a compatibility wrapper for earlier lemmas in this file. -/ From 34609628000b118ac03a5a32460842caab06d9d0 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 18 Jun 2026 12:34:28 -0400 Subject: [PATCH 73/88] feat(linUCB): remaining deterministic/probability plumbing --- .../Online/Bandit/Algorithms/LinUCB.lean | 125 ++++++++++++++++++ 1 file changed, 125 insertions(+) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index 5bf068de..da0dc185 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -2211,6 +2211,16 @@ lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] norm_num at hgap exact hgap +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial-gap term used by this formalization is deterministically at most `2` when all arm +means lie in `[-1, 1]`. At horizon zero the initial term is exactly zero. -/ +lemma initialGapTerm_le_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] + (hμ : MeanRewardBound ν (-1) 1) : + (if n = 0 then 0 else gap ν (A 0 ω)) ≤ if n = 0 then 0 else 2 := by + by_cases hn : n = 0 + · simp [hn] + · simpa [hn] using (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) hμ (A 0 ω)) + omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in /-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : @@ -3796,6 +3806,121 @@ lemma regret_ae_imp_le_textbook_finite_action (P := P) L2 hL2) matrixDetLeTraceAveragePow) +/-- The deterministic textbook LinUCB bonus term +`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`. + +The final finite-action theorem keeps this as a named expression so probability statements can use +a deterministic right-hand side instead of repeating the full formula. -/ +noncomputable def textbookRegretBonus (reg : ℝ) (β : ℕ → ℝ) (L2 : ℝ) (n : ℕ) : ℝ := + 2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) + +/-- Good-event finite-action LinUCB regret theorem with the random initial gap replaced by the +deterministic `≤ 2` bound implied by `MeanRewardBound ν (-1) 1`. -/ +lemma regret_ae_imp_le_textbook_finite_action_deterministic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, + LinUCBConfidenceEvent A R reg β x ν ω → + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by + filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2] with ω h_regret h_confω + refine (h_regret h_confω).trans ?_ + simpa [textbookRegretBonus] using + add_le_add_right + (initialGapTerm_le_two_of_meanRewardBound_neg_one_one (A := A) (ν := ν) + (n := n) (ω := ω) h_mean_bound) + (2 * (√((n : ℝ) * β n) * + √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))) + +/-- Almost-sure corollary of +`regret_ae_imp_le_textbook_finite_action_deterministic_bound` when the confidence event is known +to hold almost surely. -/ +lemma regret_ae_le_textbook_finite_action_deterministic_bound + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by + filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + h_mean_bound hβ_schedule hreg_pos L2 hL2, h_conf] with ω h_regret h_confω + exact h_regret h_confω + +/-- The confidence event is almost surely contained in the deterministic textbook regret-bound +event. This is the version to combine with a future high-probability confidence theorem. -/ +lemma probReal_confidenceEvent_le_textbook_regret_bound_deterministic + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_confω + exact h_regret h_confω + +/-- High-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ +lemma probReal_textbook_regret_bound_deterministic_ge_of_confidenceEvent_ge + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : + 1 - δ ≤ + P.real {ω | + regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by + exact h_conf_prob.trans + (probReal_confidenceEvent_le_textbook_regret_bound_deterministic (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule + hreg_pos L2 hL2) + +/-- Failure-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ +lemma probReal_textbook_regret_bound_deterministic_failure_le_of_confidenceEvent_failure_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) + (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) + (hβ_schedule : BetaSchedule β) + (hreg_pos : 0 < reg) + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} + (h_conf_failure : + P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : + P.real {ω | + ¬ regret ν A n ω ≤ + (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} ≤ δ := by + refine le_trans ?_ h_conf_failure + simp_rw [measureReal_def] + gcongr 1 + · simp + refine measure_mono_ae ?_ + filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h + h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω + exact h_regret_failure (h_regret h_confω) + /-- The confidence event is almost surely contained in the textbook finite-action regret-bound event. From f63a9cf91803eae52d026f6ca8b021735c5554de Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 25 Jun 2026 14:42:58 -0400 Subject: [PATCH 74/88] feat(LinUCB Algorithm And Process API): formalize the actual finite-action LinUCB algorithm object and its process-level API. --- .../Bandit/Algorithms/LinUCB/Basic.lean | 317 ++++++++++++++++++ 1 file changed, 317 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean new file mode 100644 index 00000000..21912a55 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean @@ -0,0 +1,317 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.SequentialLearning.Deterministic +public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import Mathlib.Analysis.MeanInequalities +public import Mathlib.Analysis.SpecialFunctions.Log.Deriv +public import Mathlib.Analysis.Matrix.Order +public import Mathlib.Data.Real.StarOrdered +public import Mathlib.LinearAlgebra.Matrix.PosDef +public import Mathlib.LinearAlgebra.Matrix.SchurComplement +public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse +public import Mathlib.Probability.Martingale.OptionalStopping + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +section Algorithm + +namespace LinUCB + +/-- Feature vectors for finite-dimensional linear bandits. -/ +abbrev Feature (d : ℕ) := Fin d → ℝ + +/-- The standard coordinate direction in `Feature d`. -/ +def coordinateDirection (i : Fin d) : Feature d := + fun j ↦ if j = i then 1 else 0 + +/-- Dot product with a coordinate direction extracts that coordinate. -/ +lemma dotProduct_coordinateDirection (u : Feature d) (i : Fin d) : + dotProduct (coordinateDirection i) u = u i := by + simp only [dotProduct, coordinateDirection] + rw [Finset.sum_eq_single i] + · simp + · intro j _hj hji + simp [hji] + · intro hi + simp at hi + +/-- Dot product with the negative coordinate direction extracts the negated coordinate. -/ +lemma dotProduct_neg_coordinateDirection (u : Feature d) (i : Fin d) : + dotProduct (-coordinateDirection i) u = -u i := by + rw [neg_dotProduct, dotProduct_coordinateDirection] + +/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product +`x_aᵀ x_a`. -/ +def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := + dotProduct (x a) (x a) + +/-- The squared feature norm is nonnegative. -/ +lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : + 0 ≤ featureSqNorm x a := by + rw [featureSqNorm, dotProduct] + exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) + +/-- The squared Euclidean norm of an arbitrary feature vector is nonnegative. -/ +lemma dotProduct_self_nonneg (u : Feature d) : + 0 ≤ dotProduct u u := by + rw [dotProduct] + exact sum_nonneg fun i _ ↦ mul_self_nonneg (u i) + +/-- Euclidean Cauchy-Schwarz for the finite-dimensional `Feature d` dot product. -/ +lemma abs_dotProduct_le_sqrt_mul_sqrt (u v : Feature d) : + |dotProduct u v| ≤ √(dotProduct u u) * √(dotProduct v v) := by + have hpos : + dotProduct u v ≤ √(dotProduct u u) * √(dotProduct v v) := by + simpa [dotProduct, pow_two] using + (Real.sum_mul_le_sqrt_mul_sqrt (Finset.univ : Finset (Fin d)) u v) + have hneg : + -dotProduct u v ≤ √(dotProduct u u) * √(dotProduct v v) := by + have h := Real.sum_mul_le_sqrt_mul_sqrt (Finset.univ : Finset (Fin d)) + (fun i : Fin d ↦ -u i) v + simpa [dotProduct, pow_two, Finset.sum_neg_distrib] using h + exact abs_le.mpr ⟨by linarith, hpos⟩ + +/-- Cauchy-Schwarz with external squared-norm bounds. -/ +lemma abs_dotProduct_le_sqrt_mul_sqrt_of_sq_norm_le + (u v : Feature d) {U V : ℝ} + (hu : dotProduct u u ≤ U) (hv : dotProduct v v ≤ V) : + |dotProduct u v| ≤ √U * √V := by + refine (abs_dotProduct_le_sqrt_mul_sqrt u v).trans ?_ + exact mul_le_mul (Real.sqrt_le_sqrt hu) (Real.sqrt_le_sqrt hv) + (Real.sqrt_nonneg _) (Real.sqrt_nonneg _) + +/-- Uniform squared feature-norm bound for finite-action LinUCB. + +This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared +form as `‖x_a‖₂² ≤ L2` for every action. -/ +def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := + ∀ a, featureSqNorm x a ≤ L2 + +/-- A uniform squared feature-norm bound is nonnegative whenever the finite action set is +nonempty. -/ +lemma FeatureSqNormBound.nonneg [Nonempty (Fin K)] + {x : Fin K → Feature d} {L2 : ℝ} (hL2 : FeatureSqNormBound x L2) : + 0 ≤ L2 := by + classical + exact (featureSqNorm_nonneg x (Classical.arbitrary (Fin K))).trans + (hL2 (Classical.arbitrary (Fin K))) + +/-- A squared feature-norm bound controls every coordinate of every feature vector. -/ +lemma abs_feature_coord_le_sqrt_of_featureSqNorm_le + (x : Fin K → Feature d) {L2 : ℝ} {a : Fin K} + (hL2 : featureSqNorm x a ≤ L2) (i : Fin d) : + |x a i| ≤ √L2 := by + have hcoord_sq_le_norm : (x a i) ^ 2 ≤ featureSqNorm x a := by + rw [featureSqNorm, dotProduct] + simpa [pow_two] using + (Finset.single_le_sum + (s := Finset.univ) (a := i) + (fun j _hj ↦ mul_self_nonneg (x a j)) (Finset.mem_univ i)) + exact Real.abs_le_sqrt (hcoord_sq_le_norm.trans hL2) + +/-- A uniform squared feature-norm bound controls the coordinate projection of every feature +vector. -/ +lemma abs_dotProduct_coordinateDirection_feature_le_sqrt + (x : Fin K → Feature d) {L2 : ℝ} (hL2 : FeatureSqNormBound x L2) + (i : Fin d) (a : Fin K) : + |dotProduct (coordinateDirection i) (x a)| ≤ √L2 := by + simpa [dotProduct_coordinateDirection] using + abs_feature_coord_le_sqrt_of_featureSqNorm_le (x := x) (hL2 a) i + +/-- For a fixed direction and finite action set, all arm-feature projections are bounded. -/ +lemma exists_abs_dotProduct_feature_bound (x : Fin K → Feature d) (v : Feature d) : + ∃ Q : ℝ, 0 ≤ Q ∧ ∀ a, |dotProduct v (x a)| ≤ Q := by + refine ⟨∑ a, |dotProduct v (x a)|, ?_, ?_⟩ + · exact sum_nonneg fun a _ha ↦ abs_nonneg _ + · intro a + exact Finset.single_le_sum + (fun b _hb ↦ abs_nonneg (dotProduct v (x b))) (Finset.mem_univ a) + +/-- History-level regularized design matrix for LinUCB. -/ +noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) + +/-- History-level response vector for LinUCB. -/ +noncomputable def responseVector' (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + ∑ s : Iic n, (h s).2 • x (h s).1 + +/-- History-level regularized least-squares estimate. -/ +noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := + Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) + +/-- History-level estimated reward of an arm. -/ +noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (thetaHat' reg x n h) (x a) + +/-- History-level quadratic form underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) + +/-- History-level elliptical confidence width of an arm. -/ +noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + √(widthQuadraticForm' reg x n h a) + +/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic +form is nonnegative. -/ +lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) + (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : + width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by + simp [width', Real.sq_sqrt h_nonneg] + +/-- LinUCB optimistic index of an arm. + +The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` +contains the observations through time `n`, this index is used to choose the arm +at time `n + 1`, and we evaluate the schedule at `n + 2` +-/ +noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := + estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a + +lemma measurable_designMatrix'_apply (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) + (i j : Fin d) : + Measurable (fun h ↦ designMatrix' reg x n h i j) := by + unfold designMatrix' + change Measurable fun h : Iic n → Fin K × ℝ ↦ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) i j + + (∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1)) i j + refine Measurable.const_add ?_ _ + rw [show (fun h : Iic n → Fin K × ℝ ↦ + (∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1)) i j) = + fun h ↦ ∑ s : Iic n, x (h s).1 i * x (h s).1 j by + funext h + simp [Matrix.sum_apply, Matrix.vecMulVec]] + fun_prop + +@[fun_prop] +lemma measurable_responseVector'_apply (x : Fin K → Feature d) (n : ℕ) (i : Fin d) : + Measurable (fun h ↦ responseVector' x n h i) := by + unfold responseVector' + fun_prop + +lemma measurable_matrix_det_apply {α : Type*} {mα : MeasurableSpace α} + (M : α → Matrix (Fin d) (Fin d) ℝ) + (hM : ∀ i j, Measurable fun a ↦ M a i j) : + Measurable fun a ↦ (M a).det := by + simp_rw [Matrix.det_apply'] + fun_prop + +lemma measurable_matrix_adjugate_apply {α : Type*} {mα : MeasurableSpace α} + (M : α → Matrix (Fin d) (Fin d) ℝ) + (hM : ∀ i j, Measurable fun a ↦ M a i j) (i j : Fin d) : + Measurable fun a ↦ (M a).adjugate i j := by + simp_rw [Matrix.adjugate_apply] + refine measurable_matrix_det_apply (fun a ↦ (M a).updateRow j (Pi.single i 1)) ?_ + intro k l + by_cases hkj : k = j + · subst k + simp [Matrix.updateRow_self] + · simpa [Matrix.updateRow_ne hkj] using hM k l + +lemma measurable_matrix_inv_apply {α : Type*} {mα : MeasurableSpace α} + (M : α → Matrix (Fin d) (Fin d) ℝ) + (hM : ∀ i j, Measurable fun a ↦ M a i j) (i j : Fin d) : + Measurable fun a ↦ (M a)⁻¹ i j := by + simp_rw [Matrix.inv_def] + change Measurable fun a ↦ Ring.inverse (M a).det * (M a).adjugate i j + simpa [Ring.inverse_eq_inv] using + (measurable_matrix_det_apply M hM).inv.mul (measurable_matrix_adjugate_apply M hM i j) + +@[fun_prop] +lemma measurable_thetaHat'_apply (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (i : Fin d) : + Measurable (fun h ↦ thetaHat' reg x n h i) := by + unfold thetaHat' + change Measurable fun h ↦ + ∑ j, (designMatrix' reg x n h)⁻¹ i j * responseVector' x n h j + refine Finset.measurable_sum _ fun j _ ↦ ?_ + exact (measurable_matrix_inv_apply (fun h ↦ designMatrix' reg x n h) + (measurable_designMatrix'_apply reg x n) i j).mul + (measurable_responseVector'_apply x n j) + +@[fun_prop] +lemma measurable_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ estimatedReward' reg x n h a) := by + unfold estimatedReward' + change Measurable fun h ↦ ∑ i, thetaHat' reg x n h i * x a i + fun_prop + +@[fun_prop] +lemma measurable_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) + (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ widthQuadraticForm' reg x n h a) := by + unfold widthQuadraticForm' + change Measurable fun h ↦ + ∑ i, x a i * (∑ j, (designMatrix' reg x n h)⁻¹ i j * x a j) + refine Finset.measurable_sum _ fun i _ ↦ ?_ + refine Measurable.const_mul ?_ _ + refine Finset.measurable_sum _ fun j _ ↦ ?_ + exact (measurable_matrix_inv_apply (fun h ↦ designMatrix' reg x n h) + (measurable_designMatrix'_apply reg x n) i j).mul measurable_const + +@[fun_prop] +lemma measurable_width' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ width' reg x n h a) := by + unfold width' + fun_prop + +@[fun_prop] +lemma measurable_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (n : ℕ) (a : Fin K) : + Measurable (fun h ↦ index' reg β x n h a) := by + unfold index' + fun_prop + +open Classical in +/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ +noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + measurableArgmax (fun h a ↦ index' reg β x n h a) h + +@[fun_prop] +lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → Feature d) + (n : ℕ) : + Measurable (nextArm hK reg β x n) := by + have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK + exact measurable_measurableArgmax fun a ↦ measurable_index' reg β x n a + +end LinUCB + +/-- The finite-action LinUCB algorithm. -/ +noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) + (x : Fin K → LinUCB.Feature d) : + Algorithm (Fin K) ℝ := + detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ + +end Algorithm + +end Bandits From c68a78502ffce8c772e4a48a5347e3da89360519 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 25 Jun 2026 14:50:05 -0400 Subject: [PATCH 75/88] feat(LinUCB Matrix And Least-Squares Algebra): prove the linear algebra needed for ridge regression, widths, determinant ratios, and design-matrix updates. --- .../Bandit/Algorithms/LinUCB/Matrix.lean | 2443 +++++++++++++++++ 1 file changed, 2443 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean new file mode 100644 index 00000000..096136b9 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean @@ -0,0 +1,2443 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic +public import Mathlib.MeasureTheory.Measure.Lebesgue.EqHaar + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The process-level design matrix built from actions up to time `n` excluded. -/ +noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := + reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) + +/-- The initial design matrix before any actions are included. -/ +lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + designMatrix A reg x 0 ω = reg • 1 := by + simp [designMatrix] + +/-- The design matrix update after observing one additional action. -/ +lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designMatrix A reg x (n + 1) ω = + designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by + simp [designMatrix, sum_range_succ, add_assoc] + +/-- With nonnegative regularization, the process-level design matrix is positive semidefinite. -/ +lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : + (designMatrix A reg x n ω).PosSemidef := by + unfold designMatrix + apply Matrix.PosSemidef.add + · exact Matrix.PosSemidef.smul Matrix.PosSemidef.one hreg_nonneg + · refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + +/-- Positive regularization makes the process-level design matrix positive definite. -/ +lemma designMatrix_posDef (hreg_pos : 0 < reg) : + (designMatrix A reg x n ω).PosDef := by + unfold designMatrix + apply Matrix.PosDef.add_posSemidef + · exact Matrix.PosDef.smul Matrix.PosDef.one hreg_pos + · refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design matrix dominates its regularization part: after subtracting `reg • I`, what remains +is the sum of observed rank-one feature matrices, hence positive semidefinite. -/ +lemma designMatrix_sub_reg_smul_one_posSemidef : + (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by + have hsum : + (∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω))).PosSemidef := by + refine Matrix.posSemidef_sum (s := range n) ?_ + intro t _ + simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) + simpa [designMatrix, add_sub_cancel_left] using hsum + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix-order form of `designMatrix_sub_reg_smul_one_posSemidef`: `reg • I ≤ V_n`. -/ +lemma reg_smul_one_le_designMatrix : + reg • (1 : Matrix (Fin d) (Fin d) ℝ) ≤ designMatrix A reg x n ω := by + rw [Matrix.le_iff] + exact designMatrix_sub_reg_smul_one_posSemidef (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix order is preserved by evaluating the quadratic form against a fixed feature vector. -/ +lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} + (hMN : M ≤ N) (u : Feature d) : + dotProduct u (M *ᵥ u) ≤ dotProduct u (N *ᵥ u) := by + have h_nonneg : 0 ≤ dotProduct u ((N - M) *ᵥ u) := by + simpa using (Matrix.le_iff.mp hMN).dotProduct_mulVec_nonneg u + rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg + exact sub_nonneg.mp h_nonneg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The inverse of the regularized identity is the reciprocal-scaled identity. -/ +lemma reg_smul_one_inv (hreg : reg ≠ 0) : + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ = + reg⁻¹ • (1 : Matrix (Fin d) (Fin d) ℝ) := by + rw [Matrix.inv_eq_left_inv] + simp [smul_smul, hreg] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The quadratic form induced by `(reg • I)⁻¹` is the squared norm divided by `reg`. -/ +lemma dotProduct_reg_smul_one_inv_mulVec (hreg : reg ≠ 0) (u : Feature d) : + dotProduct u (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ u) = + dotProduct u u / reg := by + rw [reg_smul_one_inv (reg := reg) (d := d) hreg] + simp [Matrix.smul_mulVec, div_eq_inv_mul, mul_comm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Arm-specific form of `dotProduct_reg_smul_one_inv_mulVec`. -/ +lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div + (hreg : reg ≠ 0) (a : Fin K) : + dotProduct (x a) (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) = + featureSqNorm x a / reg := by + simpa [featureSqNorm] using + dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Reusable matrix-analysis theorem needed for the LinUCB width comparison. + +It states the usual inverse anti-monotonicity of positive-definite matrices in the PSD order: +if `M` is positive definite and `M ≤ N`, then inversion reverses the order. -/ +def MatrixInvAntiMonoOnPosDef (d : ℕ) : Prop := + ∀ M N : Matrix (Fin d) (Fin d) ℝ, M.PosDef → M ≤ N → N⁻¹ ≤ M⁻¹ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Inversion reverses the PSD order on positive-definite real matrices. + +The proof uses the standard Schur-complement argument. From `M ≤ N`, the difference `N - M` is +positive semidefinite, hence `N` is positive definite. The block matrix +`[N, I; I, M⁻¹]` is positive semidefinite because its lower-right Schur complement is `N - M`. +Taking the upper-left Schur complement of the same block matrix gives `M⁻¹ - N⁻¹ ≥ 0`, which is +exactly `N⁻¹ ≤ M⁻¹`. -/ +lemma MatrixInvAntiMonoOnPosDef.of_posDef : MatrixInvAntiMonoOnPosDef d := by + intro M N hM hMN + have hDiff : (N - M).PosSemidef := by + simpa using (Matrix.le_iff.mp hMN) + have hN : N.PosDef := by + have hadd := hM.add_posSemidef hDiff + simpa [sub_eq_add_neg, add_assoc, add_comm, add_left_comm] using hadd + have hMdet : IsUnit M.det := by + exact Matrix.isUnit_iff_isUnit_det (A := M) |>.mp hM.isUnit + have hNdet : IsUnit N.det := by + exact Matrix.isUnit_iff_isUnit_det (A := N) |>.mp hN.isUnit + letI : Invertible M := Matrix.invertibleOfIsUnitDet M hMdet + letI : Invertible N := Matrix.invertibleOfIsUnitDet N hNdet + have h_block : + (Matrix.fromBlocks N (1 : Matrix (Fin d) (Fin d) ℝ) + ((1 : Matrix (Fin d) (Fin d) ℝ)ᴴ) M⁻¹).PosSemidef := by + exact (Matrix.PosDef.fromBlocks₂₂ (A := N) + (B := (1 : Matrix (Fin d) (Fin d) ℝ)) (D := M⁻¹) hM.inv).mpr + (by simpa [Matrix.inv_inv_of_invertible, Matrix.mul_assoc] using hDiff) + have h_schur : + (M⁻¹ - (1 : Matrix (Fin d) (Fin d) ℝ)ᴴ * N⁻¹ * + (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by + exact (Matrix.PosDef.fromBlocks₁₁ (A := N) (B := (1 : Matrix (Fin d) (Fin d) ℝ)) + (D := M⁻¹) hN).mp h_block + rw [Matrix.le_iff] + simpa [Matrix.mul_assoc] using h_schur + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The inverse-design comparison used in the finite-action LinUCB regret route. + +Mathematically, this should follow from `reg • I ≤ V_t` and positive regularization: inversion +reverses the positive-definite matrix order, so `V_t⁻¹ ≤ (reg • I)⁻¹`. -/ +def DesignMatrixInvLeRegInv + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (n : ℕ) (ω : Ω), + (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projection lemma for the named inverse-order obligation. -/ +lemma DesignMatrixInvLeRegInv.apply + (h_inv : DesignMatrixInvLeRegInv A reg x) (n : ℕ) (ω : Ω) : + (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := + h_inv n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization gives the LinUCB-specific inverse-design comparison. -/ +lemma DesignMatrixInvLeRegInv.of_reg_pos + (hreg_pos : 0 < reg) : + DesignMatrixInvLeRegInv A reg x := by + intro n ω + exact MatrixInvAntiMonoOnPosDef.of_posDef (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) + (designMatrix A reg x n ω) + (Matrix.PosDef.smul Matrix.PosDef.one hreg_pos) + (reg_smul_one_le_designMatrix (A := A) (reg := reg) (x := x) (n := n) (ω := ω)) + +/-- Trace of the process-level regularized design matrix. -/ +noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + Matrix.trace (designMatrix A reg x n ω) + +/-- Before any observations, the design trace is the trace of `reg • I_d`, namely `reg * d`. -/ +lemma designTrace_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + designTrace A reg x 0 ω = reg * (d : ℝ) := by + simp [designTrace, designMatrix_zero] + +/-- Updating the design matrix by `x_a x_aᵀ` increases the trace by `x_aᵀ x_a`. -/ +lemma designTrace_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designTrace A reg x (n + 1) ω = + designTrace A reg x n ω + featureSqNorm x (A n ω) := by + simp [designTrace, designMatrix_succ, featureSqNorm, Matrix.trace_vecMulVec] + +/-- Closed form for the design trace: initial regularization trace plus accumulated squared +feature norms. -/ +lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + designTrace A reg x n ω = + reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by + simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] + +/-- With nonnegative regularization, the design trace is nonnegative. -/ +lemma designTrace_nonneg (hreg_nonneg : 0 ≤ reg) : + 0 ≤ designTrace A reg x n ω := by + rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] + exact add_nonneg + (mul_nonneg hreg_nonneg (Nat.cast_nonneg d)) + (sum_nonneg fun t _ ↦ featureSqNorm_nonneg x (A t ω)) + +/-- If every selected feature vector has squared norm at most `L2`, then the trace of the design +matrix is at most `reg * d + n * L2`. -/ +lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by + rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] + gcongr + calc + (∑ t ∈ range n, featureSqNorm x (A t ω)) ≤ ∑ _t ∈ range n, L2 := by + exact sum_le_sum fun t ht ↦ hL2 t ht + _ = (n : ℝ) * L2 := by + simp [nsmul_eq_mul] + +omit [IsProbabilityMeasure P] in +/-- Almost surely, bounded selected feature norms give the corresponding deterministic trace +budget `reg * d + n * L2`. -/ +lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : + ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by + filter_upwards [hL2] with ω hL2ω + exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) L2 hL2ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A uniform finite-action feature bound implies the selected-action feature bound through any +finite horizon. -/ +lemma featureSqNorm_ae_le_of_featureSqNormBound + (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2 := + Filter.Eventually.of_forall fun ω t _ht ↦ hL2 (A t ω) + +/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ +noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, R s ω • x (A s ω) + +/-- The initial response vector before any rewards are included. -/ +lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (ω : Ω) : + responseVector A R x 0 ω = 0 := by + simp [responseVector] + +/-- The response-vector update after observing one additional reward. -/ +lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + responseVector A R x (n + 1) ω = + responseVector A R x n ω + R n ω • x (A n ω) := by + simp [responseVector, sum_range_succ] + +/-- The process-level regularized least-squares estimate. -/ +noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) + +/-- The initial least-squares estimate is zero because no reward-feature observations have been +included yet. -/ +lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : + thetaHat A R reg x 0 ω = 0 := by + simp [thetaHat, responseVector_zero] + +/-- The process-level estimated linear reward. -/ +noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω) (x a) + +/-- The initial estimated reward is zero for every arm. -/ +lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + estimatedReward A R reg x a 0 ω = 0 := by + simp [estimatedReward, thetaHat_zero] + +/-- The quadratic form `x_aᵀ V_n⁻¹ x_a` underlying the LinUCB confidence width. -/ +noncomputable def widthQuadraticForm (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) + +/-- The initial width quadratic form is induced by the inverse regularized identity. -/ +lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + widthQuadraticForm A reg x a 0 ω = + dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by + simp [widthQuadraticForm, designMatrix_zero] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Nonnegative regularization makes every LinUCB width quadratic form nonnegative. + +The reason is purely matrix-theoretic: `V_n` is positive semidefinite, the nonsingular inverse of a +positive semidefinite matrix is positive semidefinite in mathlib, and every quadratic form induced +by a positive semidefinite matrix is nonnegative. -/ +lemma widthQuadraticForm_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) (a : Fin K) : + 0 ≤ widthQuadraticForm A reg x a n ω := by + simpa [widthQuadraticForm] using + ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg).inv.dotProduct_mulVec_nonneg (x a)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, nonnegative regularization gives nonnegative selected quadratic width forms +through any finite horizon. -/ +lemma widthQuadraticForm_ae_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + exact Filter.Eventually.of_forall fun ω t _ht ↦ + widthQuadraticForm_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) hreg_nonneg (A t ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive-time version of `widthQuadraticForm_ae_nonneg_of_reg_nonneg`, matching the side +condition shape used by the regret/width-sum bridge lemmas. -/ +lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + filter_upwards [widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) + (x := x) (n := n) (P := P) hreg_nonneg] with ω h_nonnegω + intro t ht _ht0 + exact h_nonnegω t ht + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The matrix comparison needed to turn bounded feature vectors into the positive-time LinUCB +width cap. + +Mathematically, this says `x_aᵀ V_t⁻¹ x_a ≤ ‖x_a‖² / reg`. Positive regularization proves it +because `reg I ≤ V_t` and positive-definite matrix inversion reverses the PSD order. -/ +def WidthQuadraticFormLeFeatureSqNormDivReg + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (a : Fin K) (n : ℕ) (ω : Ω), + widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the inverse design matrix is bounded by the inverse regularized identity, then the LinUCB +quadratic width is bounded by `featureSqNorm / reg` for one arm, time, and sample point. -/ +lemma widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le + (a : Fin K) + (h_inv : (designMatrix A reg x n ω)⁻¹ ≤ + (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) + (hreg : reg ≠ 0) : + widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg := by + calc + widthQuadraticForm A reg x a n ω = + dotProduct (x a) (((designMatrix A reg x n ω)⁻¹) *ᵥ (x a)) := rfl + _ ≤ dotProduct (x a) + (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) := + dotProduct_mulVec_le_of_matrix_le h_inv (x a) + _ = featureSqNorm x a / reg := + dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div + (reg := reg) (x := x) hreg a + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A pointwise inverse-order comparison for all times and sample points gives the reusable +`WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ +lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le + (hreg : reg ≠ 0) + (h_inv : DesignMatrixInvLeRegInv A reg x) : + WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by + intro a n ω + exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le + (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a + (h_inv.apply n ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization bounds LinUCB quadratic widths by the squared feature norm divided by +the regularization. -/ +lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_reg_pos + (hreg_pos : 0 < reg) : + WidthQuadraticFormLeFeatureSqNormDivReg A reg x := + WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) (x := x) + hreg_pos.ne' (DesignMatrixInvLeRegInv.of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the +quadratic form is at most one. -/ +lemma widthQuadraticForm_le_one_of_featureSqNorm_le_reg + (a : Fin K) + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) + (h_feature_le : featureSqNorm x a ≤ reg) : + widthQuadraticForm A reg x a n ω ≤ 1 := by + refine (h_width a n ω).trans ?_ + rwa [div_le_one hreg_pos] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure positive-time width cap from the matrix comparison and an almost-sure +`featureSqNorm ≤ reg` bound along the selected actions. -/ +lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) + (h_feature_le : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by + filter_upwards [h_feature_le] with ω h_feature_leω + intro t ht _ht0 + exact widthQuadraticForm_le_one_of_featureSqNorm_le_reg + (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (A t ω) h_width hreg_pos + (h_feature_leω t ht) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost-sure positive-time width cap from the matrix comparison and a selected-feature budget +`featureSqNorm ≤ L2`, when `L2 ≤ reg`. -/ +lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le + (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) + (hreg_pos : 0 < reg) {L2 : ℝ} + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by + refine widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg + (A := A) (reg := reg) (x := x) (n := n) (P := P) h_width hreg_pos ?_ + filter_upwards [hL2] with ω hL2ω + intro t ht + exact (hL2ω t ht).trans hL2_le_reg + +/-- The process-level elliptical confidence width. -/ +noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := + √(widthQuadraticForm A reg x a n ω) + +/-- The initial width is the quadratic form induced by the inverse regularized identity. -/ +lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + width A reg x a 0 ω = + √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by + simp [width, widthQuadraticForm_zero] + +/-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that +quadratic form is nonnegative. -/ +lemma width_sq_eq_quadratic_form (a : Fin K) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x a n ω) : + width A reg x a n ω ^ 2 = widthQuadraticForm A reg x a n ω := by + simp [width, Real.sq_sqrt h_nonneg] + +/-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ +noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 + +/-- No positive-time widths are accumulated at horizon zero. -/ +lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + widthSqSum A reg x 0 ω = 0 := by + simp [widthSqSum] + +/-- Advancing the horizon adds the next positive-time squared width term. -/ +lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + + (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by + simp [widthSqSum, sum_range_succ] + +/-- At positive times, advancing the horizon adds the selected arm's squared width. -/ +lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + widthSqSum A reg x (n + 1) ω = + widthSqSum A reg x n ω + width A reg x (A n ω) n ω ^ 2 := by + simp [widthSqSum_succ, hn] + +/-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ +noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else widthQuadraticForm A reg x (A t ω) t ω + +/-- No positive-time quadratic width forms are accumulated at horizon zero. -/ +lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + quadraticWidthSum A reg x 0 ω = 0 := by + simp [quadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time quadratic width form. -/ +lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + + if n = 0 then 0 else widthQuadraticForm A reg x (A n ω) n ω := by + simp [quadraticWidthSum, sum_range_succ] + +/-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ +lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + quadraticWidthSum A reg x (n + 1) ω = + quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by + simp [quadraticWidthSum_succ, hn] + +/-- The accumulated capped quadratic forms corresponding to the positive-time LinUCB widths. -/ +noncomputable def cappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, + if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω) + +/-- No positive-time capped quadratic width forms are accumulated at horizon zero. -/ +lemma cappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + cappedQuadraticWidthSum A reg x 0 ω = 0 := by + simp [cappedQuadraticWidthSum] + +/-- Advancing the horizon adds the next positive-time capped quadratic width form. -/ +lemma cappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : + cappedQuadraticWidthSum A reg x (n + 1) ω = + cappedQuadraticWidthSum A reg x n ω + + if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by + simp [cappedQuadraticWidthSum, sum_range_succ] + +/-- At positive times, advancing the horizon adds the selected arm's capped quadratic width form. -/ +lemma cappedQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : + cappedQuadraticWidthSum A reg x (n + 1) ω = + cappedQuadraticWidthSum A reg x n ω + min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by + simp [cappedQuadraticWidthSum_succ, hn] + +/-- If every positive-time process-level quadratic width form is at most `1`, then the uncapped +and capped process-level quadratic-width accumulators agree. -/ +lemma quadraticWidthSum_eq_cappedQuadraticWidthSum + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + quadraticWidthSum A reg x n ω = cappedQuadraticWidthSum A reg x n ω := by + rw [quadraticWidthSum, cappedQuadraticWidthSum] + refine Finset.sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0, if_neg ht0] + exact (min_eq_right (h_le_one t ht ht0)).symm + +/-- If the squared-width and quadratic-form accumulators agree through a positive time and the +next quadratic form is nonnegative, then they still agree after adding the next term. -/ +lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) + (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by + rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, + quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hn, h_eq] + rw [width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A n ω) + (n := n) (ω := ω) h_nonneg] + +/-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive +time quadratic form is nonnegative. -/ +lemma widthSqSum_eq_sum_quadratic_form + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by + rw [widthSqSum, quadraticWidthSum] + refine sum_congr rfl ?_ + intro t ht + by_cases ht0 : t = 0 + · simp [ht0] + · rw [if_neg ht0] + rw [if_neg ht0] + exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) + (n := t) (ω := ω) (h_nonneg t ht ht0) + +/-- A quadratic-form sum bound implies the corresponding bound on `widthSqSum`. This is the shape +expected from a later elliptical-potential argument. -/ +lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + exact h_quad_le + +/-- A capped process-level quadratic-form sum bound implies the corresponding bound on +`widthSqSum`, provided the positive-time process-level quadratic forms are nonnegative and at most +`1`. -/ +lemma widthSqSum_le_of_capped_quadratic_width_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_capped_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : + widthSqSum A reg x n ω ≤ W := by + rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonneg] + rw [quadraticWidthSum_eq_cappedQuadraticWidthSum (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_le_one] + exact h_capped_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped process-level quadratic-form sum bound implies the corresponding bound +on `widthSqSum`, provided the positive-time process-level quadratic forms are almost surely +nonnegative and at most `1`. -/ +lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_capped_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_nonneg, h_le_one, h_capped_le] with + ω h_nonnegω h_le_oneω h_capped_leω + exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Determinant of the process-level LinUCB design matrix. -/ +noncomputable def designDet (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + Matrix.det (designMatrix A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial design determinant is the determinant of the regularized identity. -/ +lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by + simp [designDet, designMatrix_zero] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The initial design determinant is `reg ^ d`. -/ +lemma designDet_zero_eq_reg_pow (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) : + designDet A reg x 0 ω = reg ^ d := by + rw [designDet_zero] + simp + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A nonzero regularization parameter gives a nonzero initial design determinant. -/ +lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hreg : reg ≠ 0) : + designDet A reg x 0 ω ≠ 0 := by + rw [designDet_zero_eq_reg_pow] + exact pow_ne_zero d hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization makes every process-level design determinant positive. -/ +lemma designDet_pos_of_reg_pos (hreg_pos : 0 < reg) : + 0 < designDet A reg x n ω := by + unfold designDet + exact (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).det_pos + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization makes every process-level design determinant nonzero. -/ +lemma designDet_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : + designDet A reg x n ω ≠ 0 := by + exact (designDet_pos_of_reg_pos (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).ne' + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, positive regularization makes all design determinants in a finite horizon +nonzero. -/ +lemma designDet_ae_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + exact Filter.Eventually.of_forall fun ω t _ht ↦ + designDet_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) (ω := ω) + hreg_pos + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ +noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + designDet A reg x n ω / designDet A reg x 0 ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the determinant ratio is `1` when the initial design determinant is nonzero. -/ +lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + designDetRatio A reg x 0 ω = 1 := by + simp [designDetRatio, hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the determinant ratio is positive when the initial design determinant is +nonzero. -/ +lemma designDetRatio_zero_pos (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + 0 < designDetRatio A reg x 0 ω := by + rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + norm_num + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The process-level determinant ratio is `det(Vₙ) / reg ^ d`. -/ +lemma designDetRatio_eq_div_reg_pow : + designDetRatio A reg x n ω = designDet A reg x n ω / reg ^ d := by + rw [designDetRatio, designDet_zero_eq_reg_pow] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the process-level design matrix is diagonal, its determinant is the product of its diagonal +entries. -/ +lemma designDet_eq_prod_of_designMatrix_eq_diagonal {diag : Fin d → ℝ} + (hdiag : designMatrix A reg x n ω = Matrix.diagonal diag) : + designDet A reg x n ω = ∏ i, diag i := by + unfold designDet + rw [hdiag, Matrix.det_diagonal] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If positive regularization makes the design matrix positive definite and that matrix is +diagonal, then every diagonal entry is positive. -/ +lemma diagonal_pos_of_designMatrix_eq_diagonal {diag : Fin d → ℝ} + (hreg_pos : 0 < reg) + (hdiag : designMatrix A reg x n ω = Matrix.diagonal diag) : + ∀ i, 0 < diag i := by + have hpos : (Matrix.diagonal diag).PosDef := by + simpa [hdiag] using + designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hreg_pos + exact Matrix.posDef_diagonal_iff.mp hpos + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Diagonal-design form of the determinant ratio. -/ +lemma designDetRatio_eq_prod_div_reg_pow_of_designMatrix_eq_diagonal {diag : Fin d → ℝ} + (hdiag : designMatrix A reg x n ω = Matrix.diagonal diag) : + designDetRatio A reg x n ω = (∏ i, diag i) / reg ^ d := by + rw [designDetRatio_eq_div_reg_pow, + designDet_eq_prod_of_designMatrix_eq_diagonal (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdiag] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Eigenvalues of the positive-definite process-level LinUCB design matrix. + +The positive regularization proof is part of the definition because the matrix spectral API exposes +eigenvalues through a Hermitian proof. -/ +noncomputable def designEigenvalues (hreg_pos : 0 < reg) : Fin d → ℝ := + (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hreg_pos).1.eigenvalues + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization makes every design-matrix eigenvalue positive. -/ +lemma designEigenvalues_pos (hreg_pos : 0 < reg) (i : Fin d) : + 0 < designEigenvalues (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos i := by + unfold designEigenvalues + exact (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).eigenvalues_pos i + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design determinant is the product of the positive design-matrix eigenvalues. -/ +lemma designDet_eq_prod_designEigenvalues (hreg_pos : 0 < reg) : + designDet A reg x n ω = + ∏ i, designEigenvalues (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos i := by + unfold designDet designEigenvalues + exact (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).1.det_eq_prod_eigenvalues + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design determinant ratio is the product of the positive design-matrix eigenvalues divided +by the initial determinant `reg ^ d`. -/ +lemma designDetRatio_eq_prod_designEigenvalues_div_reg_pow (hreg_pos : 0 < reg) : + designDetRatio A reg x n ω = + (∏ i, designEigenvalues (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos i) / reg ^ d := by + rw [designDetRatio_eq_div_reg_pow, + designDet_eq_prod_designEigenvalues (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hreg_pos] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The eigenvector unitary that diagonalizes the positive-definite process-level design matrix. -/ +noncomputable def designEigenvectorUnitary (hreg_pos : 0 < reg) : + Matrix.unitaryGroup (Fin d) ℝ := + (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos).1.eigenvectorUnitary + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design eigenvector unitary diagonalizes the process-level design matrix, with diagonal +entries equal to `designEigenvalues`. -/ +lemma designEigenvectorUnitary_diagonalizes (hreg_pos : 0 < reg) : + star (designEigenvectorUnitary (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos : Matrix (Fin d) (Fin d) ℝ) * designMatrix A reg x n ω * + (designEigenvectorUnitary (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos : Matrix (Fin d) (Fin d) ℝ) = + Matrix.diagonal (designEigenvalues (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos) := by + let hM := designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos + change star (hM.1.eigenvectorUnitary : Matrix (Fin d) (Fin d) ℝ) * + designMatrix A reg x n ω * (hM.1.eigenvectorUnitary : Matrix (Fin d) (Fin d) ℝ) = + Matrix.diagonal hM.1.eigenvalues + simpa using hM.1.conjStarAlgAut_star_eigenvectorUnitary + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Orthogonal/unitary change-of-coordinates preserves the corresponding quadratic form. + +This is the finite-dimensional linear-algebra part of diagonalizing the Gaussian kernel: the +quadratic form for `M` in the original coordinates equals the quadratic form for +`star U * M * U` in the `star U` coordinates. -/ +lemma unitary_conj_quadraticForm_eq + (M : Matrix (Fin d) (Fin d) ℝ) (U : Matrix.unitaryGroup (Fin d) ℝ) + (lambda : Feature d) : + dotProduct ((star (U : Matrix (Fin d) (Fin d) ℝ)) *ᵥ lambda) + (Matrix.mulVec ((star (U : Matrix (Fin d) (Fin d) ℝ)) * M * + (U : Matrix (Fin d) (Fin d) ℝ)) + ((star (U : Matrix (Fin d) (Fin d) ℝ)) *ᵥ lambda)) = + dotProduct lambda (Matrix.mulVec M lambda) := by + let Umat : Matrix (Fin d) (Fin d) ℝ := U + have hstar_left : Umat * star Umat = 1 := by + change (U : Matrix (Fin d) (Fin d) ℝ) * + star (U : Matrix (Fin d) (Fin d) ℝ) = 1 + exact Unitary.coe_mul_star_self U + have hy : star Umat *ᵥ lambda = Matrix.vecMul lambda Umat := by + simpa [Umat] using (Matrix.mulVec_transpose (U : Matrix (Fin d) (Fin d) ℝ) lambda) + have hcancel_left : Umat * (star Umat * M * Umat) = M * Umat := by + rw [Matrix.mul_assoc (star Umat) M Umat] + rw [← Matrix.mul_assoc Umat (star Umat) (M * Umat)] + rw [hstar_left, Matrix.one_mul] + have hcancel_right : (M * Umat) * star Umat = M := by + rw [Matrix.mul_assoc M Umat (star Umat)] + rw [hstar_left, Matrix.mul_one] + change dotProduct (star Umat *ᵥ lambda) + (Matrix.mulVec (star Umat * M * Umat) (star Umat *ᵥ lambda)) = + dotProduct lambda (Matrix.mulVec M lambda) + calc + dotProduct (star Umat *ᵥ lambda) + (Matrix.mulVec (star Umat * M * Umat) (star Umat *ᵥ lambda)) + = dotProduct (Matrix.vecMul lambda Umat) + (Matrix.mulVec (star Umat * M * Umat) (star Umat *ᵥ lambda)) := by + rw [hy] + _ = Matrix.vecMul (Matrix.vecMul lambda Umat) (star Umat * M * Umat) ⬝ᵥ + (star Umat *ᵥ lambda) := by + rw [Matrix.dotProduct_mulVec] + _ = Matrix.vecMul lambda (Umat * (star Umat * M * Umat)) ⬝ᵥ + (star Umat *ᵥ lambda) := by + rw [Matrix.vecMul_vecMul] + _ = Matrix.vecMul lambda (M * Umat) ⬝ᵥ (star Umat *ᵥ lambda) := by + rw [hcancel_left] + _ = Matrix.vecMul (Matrix.vecMul lambda (M * Umat)) (star Umat) ⬝ᵥ lambda := by + rw [Matrix.dotProduct_mulVec] + _ = Matrix.vecMul lambda ((M * Umat) * star Umat) ⬝ᵥ lambda := by + rw [Matrix.vecMul_vecMul] + _ = Matrix.vecMul lambda M ⬝ᵥ lambda := by + rw [hcancel_right] + _ = dotProduct lambda (Matrix.mulVec M lambda) := by + rw [Matrix.dotProduct_mulVec] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The process-level design quadratic form becomes the diagonal eigenvalue quadratic form after +applying the design eigenvector coordinates. -/ +lemma designQuadraticForm_eq_eigenvectorUnitary_diagonal + (hreg_pos : 0 < reg) (lambda : Feature d) : + dotProduct lambda (Matrix.mulVec (designMatrix A reg x n ω) lambda) = + dotProduct + ((star (designEigenvectorUnitary (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hreg_pos : Matrix (Fin d) (Fin d) ℝ)) *ᵥ lambda) + (Matrix.mulVec + (Matrix.diagonal + (designEigenvalues (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos)) + ((star (designEigenvectorUnitary (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hreg_pos : Matrix (Fin d) (Fin d) ℝ)) *ᵥ lambda)) := by + rw [← designEigenvectorUnitary_diagonalizes (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hreg_pos] + exact (unitary_conj_quadraticForm_eq (d := d) (designMatrix A reg x n ω) + (designEigenvectorUnitary (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos) lambda).symm + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The adjoint of a real unitary matrix has determinant with absolute value one. -/ +lemma abs_det_star_unitary + (U : Matrix.unitaryGroup (Fin d) ℝ) : + |(star (U : Matrix (Fin d) (Fin d) ℝ)).det| = 1 := by + rw [← abs_one, abs_eq_iff_mul_self_eq] + have hunit := Matrix.det_of_mem_unitary + (A := star (U : Matrix (Fin d) (Fin d) ℝ)) + (SetLike.coe_mem (star U : Matrix.unitaryGroup (Fin d) ℝ)) + simpa [sq, mul_comm] using hunit.1 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Multiplication by the adjoint of a real unitary matrix preserves Lebesgue volume on feature +coordinates. + +This is the measure-theoretic piece needed for the spectral reduction of the anisotropic Gaussian +integral: after diagonalizing the design matrix by a unitary eigenvector matrix, the corresponding +coordinate change has determinant of absolute value `1`, so Haar/Lebesgue volume is unchanged. -/ +lemma unitary_star_mulVec_measurePreserving + (U : Matrix.unitaryGroup (Fin d) ℝ) : + MeasurePreserving + (fun lambda : Feature d => + star (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda) + volume volume := by + let L : Feature d →ₗ[ℝ] Feature d := + Matrix.toLin' (star (U : Matrix (Fin d) (Fin d) ℝ)) + have h_abs_det : + |(star (U : Matrix (Fin d) (Fin d) ℝ)).det| = 1 := + abs_det_star_unitary U + have hdet_ne : LinearMap.det L ≠ 0 := by + simpa [L, LinearMap.det_toLin'] using + (Matrix.UnitaryGroup.det_isUnit (star U)).ne_zero + refine ⟨L.continuous_of_finiteDimensional.measurable, ?_⟩ + change Measure.map L volume = volume + rw [Measure.map_linearMap_addHaar_eq_smul_addHaar (μ := volume) hdet_ne] + have hscale : ENNReal.ofReal |(LinearMap.det L)⁻¹| = 1 := by + dsimp [L] + rw [LinearMap.det_toLin', abs_inv, h_abs_det] + norm_num + rw [hscale, one_smul] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The adjoint action of a real unitary matrix as a volume-preserving measurable equivalence on +feature coordinates. -/ +noncomputable def unitaryStarMulVecMeasurableEquiv + (U : Matrix.unitaryGroup (Fin d) ℝ) : Feature d ≃ᵐ Feature d where + toFun lambda := star (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda + invFun lambda := (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda + left_inv lambda := by + change (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ + (star (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda) = lambda + have hunit : + (U : Matrix (Fin d) (Fin d) ℝ) * + star (U : Matrix (Fin d) (Fin d) ℝ) = 1 := by + exact Unitary.coe_mul_star_self U + rw [Matrix.mulVec_mulVec, hunit, Matrix.one_mulVec] + right_inv lambda := by + change star (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ + ((U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda) = lambda + have hunit : + star (U : Matrix (Fin d) (Fin d) ℝ) * + (U : Matrix (Fin d) (Fin d) ℝ) = 1 := by + exact Unitary.coe_star_mul_self U + rw [Matrix.mulVec_mulVec, hunit, Matrix.one_mulVec] + measurable_toFun := (unitary_star_mulVec_measurePreserving U).measurable + measurable_invFun := by + change Measurable fun lambda : Feature d => + (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda + simpa using (unitary_star_mulVec_measurePreserving (star U)).measurable + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The unitary adjoint measurable equivalence preserves Lebesgue volume. -/ +lemma unitaryStarMulVecMeasurableEquiv_measurePreserving + (U : Matrix.unitaryGroup (Fin d) ℝ) : + MeasurePreserving (unitaryStarMulVecMeasurableEquiv U) volume volume := by + change MeasurePreserving + (fun lambda : Feature d => star (U : Matrix (Fin d) (Fin d) ℝ) *ᵥ lambda) + volume volume + exact unitary_star_mulVec_measurePreserving U + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. + +This is the determinant-ratio target used by the matrix-determinant part of the elliptical +potential lemma. -/ +noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + designDet A reg x (n + 1) ω / designDet A reg x n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The scalar determinant appearing in the rank-one determinant update is the quadratic form +`uᵀ M u`. -/ +lemma det_one_add_replicateRow_mul_matrix_mul_replicateCol + (M : Matrix (Fin d) (Fin d) ℝ) (u : Feature d) : + (1 + Matrix.replicateRow Unit u * M * Matrix.replicateCol Unit u).det = + 1 + dotProduct u (Matrix.mulVec M u) := by + have hsum : + (∑ j, (∑ i, u i * M i j) * u j) = + ∑ i, u i * ∑ j, M i j * u j := by + calc + (∑ j, (∑ i, u i * M i j) * u j) + = ∑ j, ∑ i, (u i * M i j) * u j := by + simp [Finset.sum_mul] + _ = ∑ i, ∑ j, (u i * M i j) * u j := by + rw [Finset.sum_comm] + _ = ∑ i, u i * ∑ j, M i j * u j := by + refine Finset.sum_congr rfl ?_ + intro i _ + rw [Finset.mul_sum] + refine Finset.sum_congr rfl ?_ + intro j _ + ring + rw [Matrix.det_unique] + simpa [Matrix.mul_apply, Matrix.replicateRow, Matrix.replicateCol, Matrix.mulVec, + dotProduct] using hsum + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Process-level matrix determinant update for the LinUCB design matrix. + +If `V_n` has nonzero determinant, then the rank-one update +`V_{n+1} = V_n + x_{A_n} x_{A_n}ᵀ` satisfies +`det(V_{n+1}) = det(V_n) * (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ +lemma designDet_succ_eq_mul_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDet A reg x (n + 1) ω = + designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + have hM : IsUnit (designMatrix A reg x n ω).det := by + simpa [designDet] using (isUnit_iff_ne_zero.mpr hdet) + calc + designDet A reg x (n + 1) ω = + (designMatrix A reg x n ω + + Matrix.vecMulVec (x (A n ω)) (x (A n ω))).det := by + simp [designDet, designMatrix_succ] + _ = (designMatrix A reg x n ω + + Matrix.replicateCol Unit (x (A n ω)) * Matrix.replicateRow Unit (x (A n ω))).det := by + rw [Matrix.vecMulVec_eq Unit] + _ = (designMatrix A reg x n ω).det * + (1 + Matrix.replicateRow Unit (x (A n ω)) * + (designMatrix A reg x n ω)⁻¹ * Matrix.replicateCol Unit (x (A n ω))).det := by + exact Matrix.det_add_replicateCol_mul_replicateRow (A := designMatrix A reg x n ω) + (ι := Unit) hM (x (A n ω)) (x (A n ω)) + _ = designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + rw [designDet] + congr 1 + exact det_one_add_replicateRow_mul_matrix_mul_replicateCol + (M := (designMatrix A reg x n ω)⁻¹) (u := x (A n ω)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `det(V_n)` is nonzero and the selected quadratic form is nonnegative, then +`det(V_{n+1})` is nonzero. -/ +lemma designDet_succ_ne_zero_of_widthQuadraticForm_nonneg + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + designDet A reg x (n + 1) ω ≠ 0 := by + rw [designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + exact mul_ne_zero hdet (ne_of_gt (by linarith)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms preserve +nonzero design determinants up to any fixed time. -/ +lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt + (m : ℕ) (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t < m → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + designDet A reg x m ω ≠ 0 := by + induction m with + | zero => exact hdet0 + | succ m ih => + exact designDet_succ_ne_zero_of_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := m) (ω := ω) + (ih fun t ht ↦ h_nonneg t (Nat.lt_trans ht (Nat.lt_succ_self m))) + (h_nonneg m (Nat.lt_succ_self m)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms imply that +all design determinants through horizon `n` are nonzero. -/ +lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by + intro t ht + exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := t) (ω := ω) hdet0 fun s hs ↦ + h_nonneg s (mem_range.mpr (Nat.lt_of_lt_of_le hs (Nat.le_of_lt_succ (mem_range.mp ht)))) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant and nonnegative selected quadratic forms imply +that all design determinants through horizon `n` are nonzero. -/ +lemma designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω + exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `det(V_n) ≠ 0`, then the one-step determinant ratio is +`1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n}`. -/ +lemma designDetStepRatio_eq_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDetStepRatio A reg x n ω = + 1 + widthQuadraticForm A reg x (A n ω) n ω := by + simp [designDetStepRatio, + designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet, hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The cumulative determinant ratio advances by multiplying by the one-step determinant ratio. -/ +lemma designDetRatio_succ_eq_mul_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + designDetRatio A reg x (n + 1) ω = + designDetRatio A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + rw [designDetRatio, designDetRatio, + designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms make the +cumulative determinant ratio positive. -/ +lemma designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + 0 < designDetRatio A reg x n ω := by + induction n with + | zero => + exact designDetRatio_zero_pos (A := A) (reg := reg) (x := x) (ω := ω) hdet0 + | succ n ih => + have hdetn : designDet A reg x n ω ≠ 0 := + designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ + h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) + rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetn] + exact mul_pos + (ih fun t ht ↦ h_nonneg t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) + (by linarith [h_nonneg n (by simp)]) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, starting from a nonzero initial determinant, nonnegative selected quadratic +forms make the cumulative determinant ratio positive. -/ +lemma designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by + filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω + exact designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter and nonnegative selected quadratic forms make +the cumulative determinant ratio positive. -/ +lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by + refine designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization makes every realized determinant ratio positive. -/ +lemma designDetRatio_pos_of_reg_pos (hreg_pos : 0 < reg) : + 0 < designDetRatio A reg x n ω := by + rw [designDetRatio_eq_div_reg_pow] + exact div_pos + (designDet_pos_of_reg_pos (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_pos) + (pow_pos hreg_pos d) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Starting from a nonzero initial determinant, the cumulative determinant ratio is the finite +product of the per-round determinant-update factors. -/ +lemma designDetRatio_eq_prod_one_add_widthQuadraticForm + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + designDetRatio A reg x n ω = + ∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω) := by + induction n with + | zero => + rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet0] + simp + | succ n ih => + have hdetn : designDet A reg x n ω ≠ 0 := + designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) + (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ + h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) + rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetn] + rw [ih fun t ht ↦ h_nonneg t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))] + simp [Finset.prod_range_succ] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If every selected quadratic form is in `[0, 1]`, the cumulative determinant ratio is at most +`2 ^ n`. -/ +lemma designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one + (hdet0 : designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + rw [designDetRatio_eq_prod_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet0 h_nonneg] + calc + (∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω)) + ≤ ∏ _t ∈ range n, (2 : ℝ) := by + exact Finset.prod_le_prod + (fun t ht ↦ by linarith [h_nonneg t ht]) + (fun t ht ↦ by linarith [h_le_one t ht]) + _ = (2 : ℝ) ^ n := by + simp + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, if every selected quadratic form is in `[0, 1]`, the cumulative determinant +ratio is at most `2 ^ n`. -/ +lemma designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + filter_upwards [hdet0, h_nonneg, h_le_one] with ω hdet0ω h_nonnegω h_le_oneω + exact designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one (A := A) + (reg := reg) (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω h_le_oneω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter and selected quadratic forms in `[0, 1]` +imply the cumulative determinant ratio is at most `2 ^ n`. -/ +lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by + refine designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one + (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Converts an almost-sure trace bound into the determinant-ratio bound supplied by a +trace/determinant comparison theorem. -/ +lemma designDetRatio_ae_le_trace_budget_of_designTrace_ae_le + (T : ℝ) + (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ T → + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + filter_upwards [h_trace_le] with ω h_traceω + exact h_ratio_of_trace ω h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms give the concrete trace budget +`reg * d + n * L2`; a trace/determinant comparison then gives the corresponding +determinant-ratio bound. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + exact designDetRatio_ae_le_trace_budget_of_designTrace_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) (T := reg * (d : ℝ) + (n : ℝ) * L2) + (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2) + h_ratio_of_trace + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A determinant upper bound for `V_n` implies the corresponding determinant-ratio bound, using +`det(V_0) = reg ^ d`. -/ +lemma designDetRatio_le_trace_budget_of_designDet_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hdet_le : designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + rw [designDetRatio, designDet_zero_eq_reg_pow] + have hreg_pow_nonneg : 0 ≤ reg ^ d := (pow_pos hreg_pos d).le + have hdiv : designDet A reg x n ω / reg ^ d ≤ (T / (d : ℝ)) ^ d / reg ^ d := by + exact div_le_div_of_nonneg_right hdet_le hreg_pow_nonneg + refine hdiv.trans_eq ?_ + rw [← div_pow] + congr 1 + field_simp [hreg_pos.ne', by exact_mod_cast hd] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a determinant upper bound for `V_n` implies the corresponding determinant-ratio +bound. -/ +lemma designDetRatio_ae_le_trace_budget_of_designDet_ae_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hdet_le : ∀ᵐ ω ∂P, designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + filter_upwards [hdet_le] with ω hdetω + exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) T hreg_pos hd hdetω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Converts an almost-sure trace bound plus a determinant/trace comparison for `det(V_n)` +into the determinant-ratio bound used by the elliptical-potential chain. -/ +lemma designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le + (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ T → designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by + refine designDetRatio_ae_le_trace_budget_of_designDet_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) T hreg_pos hd ?_ + filter_upwards [h_trace_le] with ω h_traceω + exact hdet_of_trace ω h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms reduce the determinant-ratio goal to the determinant upper bound +`det(V_n) ≤ ((reg * d + n * L2) / d) ^ d`. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + exact designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd + (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2) + hdet_of_trace + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix-level determinant/trace comparison needed for the finite-dimensional +elliptical-potential bound. + +For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality +`det(M) ≤ (trace(M) / d) ^ d`. -/ +def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := + ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar AM-GM in the form used for PSD matrix eigenvalues: +the product of nonnegative entries is bounded by the arithmetic mean to the `card` power. -/ +lemma prod_le_average_pow_of_nonneg {ι : Type*} [Fintype ι] [Nonempty ι] + (z : ι → ℝ) (hz : ∀ i, 0 ≤ z i) : + (∏ i, z i) ≤ ((∑ i, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by + classical + have hN_pos : 0 < (Fintype.card ι : ℝ) := by + exact_mod_cast Fintype.card_pos_iff.mpr inferInstance + have hweights_pos : 0 < ∑ i : ι, (1 : ℝ) := by + simpa using hN_pos + have h_amgm := Real.geom_mean_le_arith_mean (s := Finset.univ) + (w := fun _ : ι ↦ (1 : ℝ)) (z := z) + (by intro i hi; norm_num) hweights_pos (by intro i hi; exact hz i) + have h_amgm' : + (∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹) ≤ + (∑ i : ι, z i) / (Fintype.card ι : ℝ) := by + simpa using h_amgm + have hprod_nonneg : 0 ≤ ∏ i : ι, z i := by + exact Finset.prod_nonneg fun i _ ↦ hz i + have hraise := Real.rpow_le_rpow (Real.rpow_nonneg hprod_nonneg _) h_amgm' hN_pos.le + have hleft : + ((∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹)) ^ (Fintype.card ι : ℝ) = + ∏ i : ι, z i := by + rw [← Real.rpow_mul hprod_nonneg] + rw [inv_mul_cancel₀ hN_pos.ne'] + simp + have hright : + ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ (Fintype.card ι : ℝ) = + ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by + rw [Real.rpow_natCast] + simpa [hleft, hright] using hraise + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- PSD matrix determinant/trace comparison from AM-GM over eigenvalues: +`det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma matrixDetLeTraceAveragePow : MatrixDetLeTraceAveragePow d := by + intro M hM + by_cases hd : d = 0 + · subst d + simp + · haveI : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp (Nat.pos_of_ne_zero hd) + rw [hM.1.det_eq_prod_eigenvalues, hM.1.trace_eq_sum_eigenvalues] + simpa using prod_le_average_pow_of_nonneg + (z := fun i : Fin d ↦ hM.1.eigenvalues i) + (fun i ↦ hM.eigenvalues_nonneg i) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the +design matrix is positive semidefinite. -/ +lemma designDet_le_trace_average_pow_of_matrix_det_trace_bound + (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) : + designDet A reg x n ω ≤ (designTrace A reg x n ω / (d : ℝ)) ^ d := by + simpa [designDet, designTrace] using + hdet_trace (designMatrix A reg x n ω) + (designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Combining `det(M) ≤ (trace(M)/d)^d` with a trace budget gives the determinant upper bound +`det(V_n) ≤ (T/d)^d`. -/ +lemma designDet_le_trace_budget_of_matrix_det_trace_bound + (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) + (hd : d ≠ 0) (T : ℝ) (h_trace_le : designTrace A reg x n ω ≤ T) : + designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d := by + have hd_pos : 0 < (d : ℝ) := by + exact_mod_cast Nat.pos_of_ne_zero hd + have hbase_nonneg : 0 ≤ designTrace A reg x n ω / (d : ℝ) := + div_nonneg (designTrace_nonneg (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg) hd_pos.le + have hbase_le : designTrace A reg x n ω / (d : ℝ) ≤ T / (d : ℝ) := + (div_le_div_iff_of_pos_right hd_pos).mpr h_trace_le + exact (designDet_le_trace_average_pow_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet_trace hreg_nonneg).trans + (pow_le_pow_left₀ hbase_nonneg hbase_le d) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded selected feature norms and the matrix-level determinant/trace comparison give the +determinant-ratio bound used by the elliptical-potential chain. -/ +lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + refine designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 ?_ + intro ω h_traceω + exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) + hdet_trace hreg_pos.le hd h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The deterministic determinant-ratio trace budget used in the textbook LinUCB bound: +`(1 + n L² / (λ d))^d`. -/ +noncomputable def textbookDesignDetRatioTraceBound + (d : ℕ) (reg L2 : ℝ) (n : ℕ) : ℝ := + (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) ^ d + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The textbook determinant-ratio trace budget is the same as the trace-average expression +produced by the determinant/trace comparison. -/ +lemma textbookDesignDetRatioTraceBound_eq_trace_budget + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) : + textbookDesignDetRatioTraceBound d reg L2 n = + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + unfold textbookDesignDetRatioTraceBound + congr 1 + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) + field_simp [hden] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The trace-budget determinant-ratio bound is monotone in the horizon when `L2 ≥ 0`. -/ +lemma trace_budget_ratio_pow_le_textbookDesignDetRatioTraceBound + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) (hL2_nonneg : 0 ≤ L2) + {t n : ℕ} (ht : t ≤ n) : + ((reg * (d : ℝ) + (t : ℝ) * L2) / (reg * (d : ℝ))) ^ d ≤ + textbookDesignDetRatioTraceBound d reg L2 n := by + have hden_pos : 0 < reg * (d : ℝ) := by + exact mul_pos hreg_pos (by exact_mod_cast Nat.pos_of_ne_zero hd) + have ht_real : (t : ℝ) ≤ (n : ℝ) := by + exact_mod_cast ht + have hnum_nonneg : 0 ≤ reg * (d : ℝ) + (t : ℝ) * L2 := + add_nonneg hden_pos.le (mul_nonneg (Nat.cast_nonneg t) hL2_nonneg) + have hbase_le : + (reg * (d : ℝ) + (t : ℝ) * L2) / (reg * (d : ℝ)) ≤ + (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) := by + refine (div_le_div_iff_of_pos_right hden_pos).mpr ?_ + nlinarith + calc + ((reg * (d : ℝ) + (t : ℝ) * L2) / (reg * (d : ℝ))) ^ d + ≤ ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := + pow_le_pow_left₀ (div_nonneg hnum_nonneg hden_pos.le) hbase_le d + _ = textbookDesignDetRatioTraceBound d reg L2 n := + (textbookDesignDetRatioTraceBound_eq_trace_budget (reg := reg) (d := d) + (n := n) L2 hreg_pos hd).symm + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bounded finite-action features and the matrix determinant/trace comparison bound every +intermediate determinant ratio by the terminal textbook trace budget. -/ +lemma designDetRatio_ae_all_le_textbookTraceBound_of_featureSqNorm_bound + [Nonempty (Fin K)] + (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : FeatureSqNormBound x L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + designDetRatio A reg x t ω ≤ textbookDesignDetRatioTraceBound d reg L2 n := by + have hL2_nonneg : 0 ≤ L2 := hL2.nonneg + have h_all : + ∀ᵐ ω ∂P, ∀ t, + designDetRatio A reg x t ω ≤ + ((reg * (d : ℝ) + (t : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by + simp_rw [ae_all_iff] + intro t + exact designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := t) (P := P) + L2 hreg_pos hd + (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := t) + (P := P) L2 hL2) + hdet_trace + filter_upwards [h_all] with ω h_allω t ht + exact (h_allω t).trans + (trace_budget_ratio_pow_le_textbookDesignDetRatioTraceBound (reg := reg) (d := d) + L2 hreg_pos hd hL2_nonneg (le_of_lt (mem_range.mp ht))) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The log-determinant expression that appears in the elliptical-potential lemma. -/ +noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + 2 * Real.log (designDetRatio A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A positive determinant ratio bounded by `D` gives the corresponding log-determinant potential +bound. -/ +lemma ellipticalPotential_le_two_mul_log_of_designDetRatio_le {D : ℝ} + (h_ratio_pos : 0 < designDetRatio A reg x n ω) + (h_ratio_le : designDetRatio A reg x n ω ≤ D) : + ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by + rw [ellipticalPotential] + exact mul_le_mul_of_nonneg_left (Real.log_le_log h_ratio_pos h_ratio_le) (by norm_num) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a positive determinant ratio bounded by `D` gives the corresponding +log-determinant potential bound. -/ +lemma ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le {D : ℝ} + (h_ratio_pos : ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by + filter_upwards [h_ratio_pos, h_ratio_le] with ω h_ratio_posω h_ratio_leω + exact ellipticalPotential_le_two_mul_log_of_designDetRatio_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_ratio_posω h_ratio_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. + +The determinant-update lemmas below establish that the capped quadratic-width term is bounded by +this quantity. A separate log/telescoping bridge then connects this one-step quantity to +`ellipticalPotentialIncrement`. -/ +noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + 2 * Real.log (designDetStepRatio A reg x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing, the one-step log-determinant potential is +`2 * log (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ +lemma ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm + (hdet : designDet A reg x n ω ≠ 0) : + ellipticalPotentialStep A reg x n ω = + 2 * Real.log (1 + widthQuadraticForm A reg x (A n ω) n ω) := by + simp [ellipticalPotentialStep, + designDetStepRatio_eq_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar log inequality used in the elliptical-potential proof: for `0 ≤ q ≤ 1`, +`min 1 q ≤ 2 * log (1 + q)`. -/ +lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} + (hq_nonneg : 0 ≤ q) (hq_le_one : q ≤ 1) : + min 1 q ≤ 2 * Real.log (1 + q) := by + have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := + Real.le_log_one_add_of_nonneg hq_nonneg + have hq_add_two_pos : 0 < q + 2 := by linarith + have hq_le_two : q ≤ 2 := by linarith + have hq_le_log_lower : q ≤ 2 * (2 * q / (q + 2)) := by + rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] + rw [le_div_iff₀ hq_add_two_pos] + nlinarith + rw [min_eq_right hq_le_one] + exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar log inequality used in the textbook elliptical-potential proof: for `0 ≤ q`, +`min 1 q ≤ 2 * log (1 + q)`. -/ +lemma min_one_le_two_mul_log_one_add_of_nonneg {q : ℝ} + (hq_nonneg : 0 ≤ q) : + min 1 q ≤ 2 * Real.log (1 + q) := by + by_cases hq_le_one : q ≤ 1 + · exact min_one_le_two_mul_log_one_add_of_nonneg_le_one hq_nonneg hq_le_one + · have hq_one : 1 ≤ q := by linarith + have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := + Real.le_log_one_add_of_nonneg hq_nonneg + have hq_add_two_pos : 0 < q + 2 := by linarith + have hone_le_log_lower : 1 ≤ 2 * (2 * q / (q + 2)) := by + rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] + rw [le_div_iff₀ hq_add_two_pos] + nlinarith + rw [min_eq_left hq_one] + exact hone_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the +single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ +lemma cappedWidthTerm_le_ellipticalPotentialStep + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) + (h_le_one : n ≠ 0 → widthQuadraticForm A reg x (A n ω) n ω ≤ 1) : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialStep A reg x n ω := by + by_cases hn : n = 0 + · rw [if_pos hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) + · rw [if_neg hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under determinant nonvanishing and nonnegativity of the selected quadratic form, the single +capped quadratic-width term is bounded by the one-step log-determinant potential. This is the +textbook form; no separate `q ≤ 1` assumption is needed because the term is already capped. -/ +lemma cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg + (hdet : designDet A reg x n ω ≠ 0) + (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialStep A reg x n ω := by + by_cases hn : n = 0 + · rw [if_pos hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) + · rw [if_neg hn, + ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet] + exact min_one_le_two_mul_log_one_add_of_nonneg h_nonneg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply +the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ +lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω := by + filter_upwards [hdet, h_nonneg, h_le_one] with ω hdetω h_nonnegω h_le_oneω + intro t ht + exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the +per-step one-step-potential bound for the capped quadratic-width term. -/ +lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω := by + filter_upwards [hdet, h_nonneg] with ω hdetω h_nonnegω + intro t ht + exact cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg (A := A) (reg := reg) + (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At horizon zero, the log-determinant potential is zero when the initial design determinant is +nonzero. -/ +lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : + ellipticalPotential A reg x 0 ω = 0 := by + simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the log-determinant elliptical-potential inequality. At horizon zero there are no +positive-time capped quadratic width forms, and the log-determinant potential is zero when the +initial design determinant is nonzero. -/ +lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) + (hdet : designDet A reg x 0 ω ≠ 0) : + cappedQuadraticWidthSum A reg x 0 ω ≤ ellipticalPotential A reg x 0 ω := by + rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), + ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- One-step increment of the log-determinant elliptical potential. -/ +noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := + ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The one-step determinant-ratio potential equals the increment of the cumulative +log-determinant potential, provided the relevant design determinants are nonzero. -/ +lemma ellipticalPotentialStep_eq_increment + (hdet0 : designDet A reg x 0 ω ≠ 0) + (hdetn : designDet A reg x n ω ≠ 0) + (hdet_succ : designDet A reg x (n + 1) ω ≠ 0) : + ellipticalPotentialStep A reg x n ω = ellipticalPotentialIncrement A reg x n ω := by + simp [ellipticalPotentialStep, designDetStepRatio, ellipticalPotentialIncrement, + ellipticalPotential, designDetRatio, Real.log_div hdet_succ hdetn, + Real.log_div hdet_succ hdet0, Real.log_div hdetn hdet0] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the one-step determinant-ratio potential equals the increment of the cumulative +log-determinant potential throughout the finite horizon, provided all determinants up to that +horizon are nonzero almost surely. -/ +lemma ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω = ellipticalPotentialIncrement A reg x t ω := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact ellipticalPotentialStep_eq_increment (A := A) (reg := reg) (x := x) (n := t) + (ω := ω) (hdetω 0 (by simp)) + (hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) + (hdetω (t + 1) (mem_range.mpr (Nat.succ_lt_succ (mem_range.mp ht)))) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If the next capped quadratic width term is bounded by the next log-determinant potential +increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ +lemma cappedQuadraticWidthSum_succ_le_ellipticalPotential + (h_prev : cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_step : + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ + ellipticalPotentialIncrement A reg x n ω) : + cappedQuadraticWidthSum A reg x (n + 1) ω ≤ ellipticalPotential A reg x (n + 1) ω := by + rw [cappedQuadraticWidthSum_succ (A := A) (reg := reg) (x := x) (n := n) (ω := ω)] + calc + cappedQuadraticWidthSum A reg x n ω + + (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) + ≤ ellipticalPotential A reg x n ω + ellipticalPotentialIncrement A reg x n ω := by + exact add_le_add h_prev h_step + _ = ellipticalPotential A reg x (n + 1) ω := by + simp [ellipticalPotentialIncrement] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A per-step bound by log-determinant potential increments implies the cumulative +elliptical-potential inequality. This is the induction shell for the determinant-update proof. -/ +lemma cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le + (hdet : designDet A reg x 0 ω ≠ 0) : + (∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) → + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + induction n with + | zero => + intro _ + exact cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) + (x := x) (ω := ω) hdet + | succ n ih => + intro h_step + refine cappedQuadraticWidthSum_succ_le_ellipticalPotential (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) ?_ ?_ + · exact ih fun t ht ↦ h_step t + (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + · exact h_step n (by simp) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a per-step bound by log-determinant potential increments implies the cumulative +elliptical-potential inequality. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + filter_upwards [hdet, h_step] with ω hdetω h_stepω + exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdetω h_stepω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the +cumulative capped-sum/log-det inequality, provided the one-step determinant-ratio potential is +bounded by the corresponding cumulative-potential increment. + +This separates the elliptical-potential proof into two local obligations: + +* a matrix-determinant update bounding the selected arm's capped quadratic form by + `ellipticalPotentialStep`; +* a log/telescoping bridge from `ellipticalPotentialStep` to `ellipticalPotentialIncrement`. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet ?_ + filter_upwards [h_step, h_step_le_increment] with ω h_stepω h_step_le_incrementω + intro t ht + exact (h_stepω t ht).trans (h_step_le_incrementω t ht) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the +cumulative capped-sum/log-det inequality when all design determinants up to the horizon are nonzero +almost surely. + +Compared with `cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le`, this +version discharges the log/telescoping bridge automatically from determinant nonvanishing. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + have hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + exact hdetω 0 (by simp) + refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_step ?_ + filter_upwards [ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet] with ω h_eq + intro t ht + rw [h_eq t ht] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the +capped-sum/log-determinant elliptical-potential bound. + +This is the capped form used in the textbook proof of LinUCB: the quadratic forms do not need to +be bounded by `1`, because the accumulated quantity is `min 1 q_t`. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet + (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet_range_n h_nonneg) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization discharges determinant nonvanishing and nonnegativity, yielding the +capped-sum/log-determinant elliptical-potential bound directly. -/ +lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos + (hreg_pos : 0 < reg) : + ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by + exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) + (n := n + 1) (P := P) hreg_pos) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The process-level capped quadratic-width input expected from an elliptical-potential argument. + +It packages the three facts needed to turn a capped process-level quadratic-width estimate into the +`widthSqSum` estimate used by the regret chain: + +* each positive-time process-level quadratic width form is nonnegative; +* each positive-time process-level quadratic width form is at most `1`; +* their capped process-level accumulated sum is bounded by `W`. -/ +def CappedQuadraticWidthBound (A : ℕ → Ω → Fin K) (reg : ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := + (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ∧ + (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ∧ + cappedQuadraticWidthSum A reg x n ω ≤ W + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Build the packaged process-level capped quadratic-width input from its component facts. -/ +lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_sum_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : + CappedQuadraticWidthBound A reg x n ω W := by + exact ⟨h_nonneg, h_le_one, h_sum_le⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the packaged process-level capped quadratic-width input. At horizon zero, the +nonnegativity and `≤ 1` side conditions are vacuous, and the capped sum is zero. -/ +lemma cappedQuadraticWidthBound_zero {W : ℝ} (hW : 0 ≤ W) : + CappedQuadraticWidthBound A reg x 0 ω W := by + refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ + · intro t ht _ + simp at ht + · intro t ht _ + simp at ht + · simpa [cappedQuadraticWidthSum_zero] using hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Base case for the packaged process-level capped quadratic-width input when the constant bound +is supplied through the log-determinant potential. -/ +lemma cappedQuadraticWidthBound_zero_of_ellipticalPotential_le_bound {W : ℝ} + (hdet : designDet A reg x 0 ω ≠ 0) (h_potential_le : ellipticalPotential A reg x 0 ω ≤ W) : + CappedQuadraticWidthBound A reg x 0 ω W := by + refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ + · intro t ht _ + simp at ht + · intro t ht _ + simp at ht + · exact (cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) + (x := x) (ω := ω) hdet).trans h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ +lemma cappedQuadraticWidthBound_mono {W W' : ℝ} + (h_bound : CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : + CappedQuadraticWidthBound A reg x n ω W' := by + exact ⟨h_bound.1, h_bound.2.1, h_bound.2.2.trans hW⟩ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, build the packaged process-level capped quadratic-width input from its component +facts. -/ +lemma cappedQuadraticWidthBound_ae_of_nonneg_le_one_and_sum_ae_le {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_sum_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_sum_le] with + ω h_nonnegω h_le_oneω h_sum_leω + exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_sum_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged process-level capped quadratic-width input is monotone in the +numeric bound. -/ +lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W' := by + filter_upwards [h_bound] with ω h_boundω + exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) h_boundω hW + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A capped-sum bound by the log-determinant potential, together with a constant bound on that +potential, gives the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_of_ellipticalPotential_le_bound {W : ℝ} + (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ellipticalPotential A reg x n ω ≤ W) : + CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonneg h_le_one (h_elliptical.trans h_potential_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a capped-sum bound by the log-determinant potential and an almost-sure constant +bound on that potential give the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_elliptical : ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + filter_upwards [h_nonneg, h_le_one, h_elliptical, h_potential_le] with + ω h_nonnegω h_le_oneω h_ellipticalω h_potential_leω + exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, per-step bounds by log-determinant potential increments and a final constant +bound on the potential give the packaged process-level capped quadratic-width input. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialIncrement A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet h_step) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, one-step determinant-ratio potential bounds, their bridge to cumulative +potential increments, and a final constant bound on the potential give the packaged process-level +capped quadratic-width input. + +This is the packaged form of the determinant-update interface: once the true matrix determinant +lemma proves the `h_step` assumption and the log/telescoping algebra proves +`h_step_le_increment`, the existing regret chain can consume the resulting bound. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, one-step determinant-ratio potential bounds, determinant nonvanishing up to the +horizon, and a final constant bound on the potential give the packaged process-level capped +quadratic-width input. + +This is the determinant-nonvanishing version of the one-step interface: it isolates the one-step +matrix inequality from the final log-determinant bound. -/ +lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero + {W : ℝ} + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ + ellipticalPotentialStep A reg x t ω) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one + (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet h_step) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the determinant-update step, determinant nonvanishing up to the horizon, and a +final constant bound on the log-determinant potential give the packaged capped quadratic-width +input used by the regret chain. + +The assumptions now match the concrete obligations left for a full elliptical-potential proof: + +* prove all relevant design determinants are nonzero; +* prove selected quadratic forms are nonnegative and at most `1` at positive times; +* prove the final log-determinant potential is at most `W`. -/ +lemma cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound {W : ℝ} + (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by + filter_upwards [hdet] with ω hdetω + intro t ht + exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) + have h_nonneg_positive : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by + filter_upwards [h_nonneg] with ω h_nonnegω + intro t ht _ + exact h_nonnegω t ht + exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) + h_nonneg_positive h_le_one hdet + (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero (A := A) (reg := reg) + (x := x) (n := n) (P := P) hdet_range_n h_nonneg h_le_one) + h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant, the determinant-update step, and a final constant +bound on the log-determinant potential give the packaged capped quadratic-width input used by the +regret chain. + +This removes the need to assume determinant nonvanishing at every time: it is derived inductively +from `det(V_0) ≠ 0` and nonnegative selected quadratic forms. -/ +lemma cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound {W : ℝ} + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound (A := A) + (reg := reg) (x := x) (n := n) (P := P) (W := W) + (designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) + h_nonneg h_le_one h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter, the determinant-update step, and a final +constant bound on the log-determinant potential give the packaged capped quadratic-width input used +by the regret chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential_le_bound {W : ℝ} + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + refine cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) ?_ h_nonneg h_le_one + h_potential_le + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization discharges the determinant-nonvanishing and quadratic-form +nonnegativity obligations in the log-determinant elliptical-potential chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound {W : ℝ} + (hreg_pos : 0 < reg) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by + exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) + (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) + (n := n + 1) (P := P) hreg_pos) + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + h_le_one h_potential_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a +determinant-ratio upper bound, and the determinant-update step give the packaged capped +quadratic-width input used by the regret chain. + +This version accepts the determinant-ratio bound directly and converts it into the +`ellipticalPotential ≤ 2 * log D` bound internally. -/ +lemma cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound {D : ℝ} + (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by + exact cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := 2 * Real.log D) + hdet0 h_nonneg h_le_one + (ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) + (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) + h_ratio_le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, a nonzero regularization parameter, nonnegative selected quadratic forms, a +determinant-ratio upper bound, and the determinant-update step give the packaged capped +quadratic-width input used by the regret chain. + +This is the most direct interface for the final determinant-bound part of the finite-action +elliptical-potential argument: after proving `designDetRatio ≤ D`, the theorem supplies the +`CappedQuadraticWidthBound` with bound `2 * log D`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound {D : ℝ} + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : + ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by + refine cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one h_ratio_le + exact Filter.Eventually.of_forall fun ω ↦ + designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A simple explicit determinant-ratio bound for the capped quadratic-width input. + +If `reg ≠ 0` and every selected quadratic form is almost surely in `[0, 1]`, then the determinant +ratio is at most `2 ^ n`, so the existing determinant-update/elliptical-potential chain gives the +packaged capped-width bound with budget `2 * log (2 ^ n)`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω (2 * Real.log ((2 : ℝ) ^ n)) := by + refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (2 : ℝ) ^ n) + hreg h_nonneg ?_ ?_ + · filter_upwards [h_le_one] with ω h_le_oneω + exact fun t ht _ ↦ h_le_oneω t ht + · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Trace-budget interface for the determinant part of the finite-action elliptical-potential +argument. + +Given a determinant-ratio bound `designDetRatio ≤ (T / (reg * d)) ^ d`, where `T` is an upper bound +on `trace(V_n)`, this theorem feeds that determinant-ratio bound into the determinant-update and +elliptical-potential chain. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (T : ℝ) + (h_ratio_le : ∀ᵐ ω ∂P, + designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * Real.log ((T / (reg * (d : ℝ))) ^ d)) := by + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface for the determinant part of the finite-action +elliptical-potential argument. + +If selected feature vectors have squared norm at most `L2`, then `trace(V_n) ≤ reg * d + n * L2`. +Given a deterministic trace/determinant comparison that turns this trace budget into the +determinant-ratio bound, this theorem supplies the packaged capped-width input with the explicit +budget `2 * log (((reg * d + n * L2) / (reg * d)) ^ d)`. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound + (hreg : reg ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d)) := by + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg h_nonneg h_le_one + (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) + (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The explicit feature-norm determinant budget can be rewritten in the common +`d * log(1 + n L² / (reg d))` form. -/ +lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add + (L2 : ℝ) (hden : reg * (d : ℝ) ≠ 0) : + 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) = + 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by + have hbase : + (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) = + 1 + (n : ℝ) * L2 / (reg * (d : ℝ)) := by + exact same_add_div hden + rw [Real.log_pow, hbase] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Textbook capped elliptical-potential budget from bounded selected feature norms and the +matrix-level determinant/trace comparison. + +Unlike `cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound`, this theorem bounds the capped +quadratic-width sum directly and does not assume the individual quadratic forms are at most `1`. -/ +lemma cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + cappedQuadraticWidthSum A reg x n ω ≤ + 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) + rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] + have h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := + widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le + have h_potential_le : ∀ᵐ ω ∂P, + ellipticalPotential A reg x n ω ≤ + 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) := by + exact ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) + (reg := reg) (x := x) (n := n) (P := P) + (designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' h_nonneg) + (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 + hdet_trace) + filter_upwards [cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos, h_potential_le] with + ω h_capped_le h_potentialω + exact h_capped_le.trans h_potentialω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface with the log term rewritten in the standard +`2 * d * log(1 + n L² / (reg d))` shape. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (hreg : reg ≠ 0) (hd : d ≠ 0) + (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → + 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) + (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + widthQuadraticForm A reg x (A t ω) t ω ≤ 1) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (h_ratio_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDetRatio A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + have hden : reg * (d : ℝ) ≠ 0 := by + exact mul_ne_zero hreg (by exact_mod_cast hd) + rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] + exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 + h_ratio_of_trace + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface with the determinant/trace comparison stated as a determinant +upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ +lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) + (hdet_of_trace : ∀ ω, + designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → + designDet A reg x n ω ≤ + ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd + (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) + (n := n) (P := P) hreg_pos.le) + (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) + (x := x) (n := n) (P := P) + (WidthQuadraticFormLeFeatureSqNormDivReg.of_reg_pos (A := A) (reg := reg) + (x := x) hreg_pos) + hreg_pos hL2 hL2_le_reg) + L2 hL2 ?_ + intro ω h_traceω + exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd + (hdet_of_trace ω h_traceω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Feature-norm-budget interface consuming the reusable positive-semidefinite determinant/trace +comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ +lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (L2 : ℝ) + (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) + (hL2_le_reg : L2 ≤ reg) + (hdet_trace : MatrixDetLeTraceAveragePow d) : + ∀ᵐ ω ∂P, + CappedQuadraticWidthBound A reg x n ω + (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by + refine + cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le + (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd + L2 hL2 hL2_le_reg ?_ + intro ω h_traceω + exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd + (reg * (d : ℝ) + (n : ℝ) * L2) h_traceω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed +by the regret chain. -/ +lemma widthSqSum_le_of_capped_quadratic_width_bound {W : ℝ} + (h_bound : CappedQuadraticWidthBound A reg x n ω W) : + widthSqSum A reg x n ω ≤ W := by + exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) h_bound.1 h_bound.2.1 h_bound.2.2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Almost surely, the packaged process-level capped quadratic-width input implies the `widthSqSum` +bound consumed by the regret chain. -/ +lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} + (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : + ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by + filter_upwards [h_bound] with ω h_boundω + exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) (W := W) h_boundω + +/-- The process-level LinUCB optimistic index. -/ +noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) + (n : ℕ) (ω : Ω) : ℝ := + estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω + +/-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is +zero. -/ +lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by + simp [index, estimatedReward_zero] + +/-- At time zero, the LinUCB index is the confidence schedule times the initial quadratic-form +width. -/ +lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : + index A R reg β x a 0 ω = + √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by + simp [index_zero, width_zero] + +/-- The finite-action LinUCB process starts from the deterministic default arm. -/ +lemma arm_zero [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) : + A 0 =ᵐ[P] fun _ ↦ ⟨0, hK⟩ := by + exact h.action_zero_detAlgorithm + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every least-squares reward estimate is zero. -/ +lemma estimatedReward_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + estimatedReward A R reg x a n ω = 0 := by + subst d + simp [estimatedReward, dotProduct] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB quadratic width form is zero. -/ +lemma widthQuadraticForm_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + widthQuadraticForm A reg x a n ω = 0 := by + subst d + simp [widthQuadraticForm, dotProduct] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB width is zero. -/ +lemma width_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + width A reg x a n ω = 0 := by + simp [width, widthQuadraticForm_eq_zero_of_dim_eq_zero (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) hd a] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- In zero feature dimension, every LinUCB index is zero. -/ +lemma index_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : + index A R reg β x a n ω = 0 := by + simp [index, estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) + (reg := reg) (x := x) (n := n) (ω := ω) hd a, + width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := n) + (ω := ω) hd a] + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 38e87a3a410fbe66765d6f410daa4b230a9bf4cb Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 25 Jun 2026 15:14:38 -0400 Subject: [PATCH 76/88] feat(LinUCB Confidence Events): define and connect the confidence events used by the regret proof --- .../Bandit/Algorithms/LinUCB/Confidence.lean | 148 ++++++++ .../Algorithms/LinUCB/ConfidenceEvents.lean | 149 ++++++++ .../Algorithms/LinUCB/TextbookConfidence.lean | 354 ++++++++++++++++++ .../LinUCB/TextbookConfidenceBridge.lean | 354 ++++++++++++++++++ 4 files changed, 1005 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean new file mode 100644 index 00000000..7441bec4 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean @@ -0,0 +1,148 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Matrix + +/-! +# LinUCB for finite-action linear bandits +Chapter 19 of *Bandit Algorithms*: +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The pointwise LinUCB confidence event used by the finite-action regret proof. + +For every positive process time, the best arm's true mean lies below its optimistic index, and the +selected arm's pessimistic index lies below its true mean. On this event, the max-index property of +LinUCB turns optimism into an instantaneous regret bound. -/ +def LinUCBConfidenceEvent [Nonempty (Fin K)] + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local LinUCB confidence event. + +The regret theorem up to horizon `n` only uses the confidence inequalities at times +`t ∈ range n`. This finite-horizon event is the natural target for a high-probability +self-normalized concentration theorem with a fixed horizon, and avoids requiring confidence at all +future times when proving an `n`-round regret bound. -/ +def LinUCBConfidenceEventUpTo [Nonempty (Fin K)] + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] + +omit [IsMarkovKernel ν] in +/-- A global LinUCB confidence event implies its finite-horizon restriction. -/ +lemma LinUCBConfidenceEvent.toUpTo [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + LinUCBConfidenceEventUpTo A R reg β x ν n ω := by + intro t _ht ht0 + exact h_conf t ht0 + +omit [IsMarkovKernel ν] in +/-- First projection from the horizon-local LinUCB confidence event: optimism for the best arm. -/ +lemma LinUCBConfidenceEventUpTo.best [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEventUpTo A R reg β x ν n ω) : + ∀ t, t ∈ range n → t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by + intro t ht ht0 + exact (h_conf t ht ht0).1 + +omit [IsMarkovKernel ν] in +/-- Second projection from the horizon-local LinUCB confidence event: validity of the selected +arm's lower confidence inequality. -/ +lemma LinUCBConfidenceEventUpTo.arm [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEventUpTo A R reg β x ν n ω) : + ∀ t, t ∈ range n → t ≠ 0 → + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by + intro t ht ht0 + exact (h_conf t ht ht0).2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Linear realizability of the arm means by a parameter `θ`. + +The actual self-normalized concentration theorem for LinUCB needs an assumption of this shape: +each arm's mean reward must be represented by the finite-dimensional linear model +`x_aᵀ θ`. Without such a realizability assumption, no concentration theorem can imply +`LinUCBConfidenceEvent`, because the least-squares predictor may be biased away from the true arm +means for structural reasons rather than random noise. -/ +def LinearMeanModel (ν : Kernel (Fin K) ℝ) (x : Fin K → Feature d) (θ : Feature d) : Prop := + ∀ a, (ν a)[id] = dotProduct θ (x a) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Squared Euclidean-norm bound on a linear-bandit parameter. This is the finite-dimensional +version of the textbook assumption `‖θ‖₂ ≤ S`, written as `θᵀθ ≤ S2`. -/ +def ParameterSqNormBound (θ : Feature d) (S2 : ℝ) : Prop := + dotProduct θ θ ≤ S2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Mean response vector predicted by a linear mean model along the realized actions. -/ +noncomputable def meanResponseVector + (A : ℕ → Ω → Fin K) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, (ν (A s ω))[id] • x (A s ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Centered response vector, i.e. the accumulated reward-feature vector minus its conditional mean +under the finite-action linear mean model. This is the deterministic noise vector that the later +self-normalized concentration theorem must control. -/ +noncomputable def centeredResponseVector + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + responseVector A R x n ω - meanResponseVector A ν x n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar reward noise at time `t`, centered at the mean of the selected arm. -/ +noncomputable def rewardNoise + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (t : ℕ) (ω : Ω) : ℝ := + R t ω - (ν (A t ω))[id] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Arm-wise centered reward subgaussianity. + +This is the LinUCB analogue of the scalar noise assumption used in `UCB.lean`: for each arm, the +reward centered at that arm's mean has a subgaussian moment-generating-function bound. A future +vector self-normalized concentration theorem should start from this assumption, together with the +linear mean model, and prove the centered-noise-plus-bias confidence event. -/ +def RewardNoiseSubgaussian (ν : Kernel (Fin K) ℝ) (σ2 : ℝ≥0) : Prop := + ∀ a, HasSubgaussianMGF (fun r ↦ r - (ν a)[id]) σ2 (ν a) + +end AlgorithmBehavior + +end LinUCB + +end Bandits diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean new file mode 100644 index 00000000..99830570 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean @@ -0,0 +1,149 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Matrix + +/-! +# LinUCB Confidence Events + +Basic confidence events and modeling assumptions consumed by the finite-action LinUCB regret proof. +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +/-- The pointwise LinUCB confidence event used by the finite-action regret proof. + +For every positive process time, the best arm's true mean lies below its optimistic index, and the +selected arm's pessimistic index lies below its true mean. On this event, the max-index property of +LinUCB turns optimism into an instantaneous regret bound. -/ +def LinUCBConfidenceEvent [Nonempty (Fin K)] + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local LinUCB confidence event. + +The regret theorem up to horizon `n` only uses the confidence inequalities at times +`t ∈ range n`. This finite-horizon event is the natural target for a high-probability +self-normalized concentration theorem with a fixed horizon, and avoids requiring confidence at all +future times when proving an `n`-round regret bound. -/ +def LinUCBConfidenceEventUpTo [Nonempty (Fin K)] + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] + +omit [IsMarkovKernel ν] in +/-- A global LinUCB confidence event implies its finite-horizon restriction. -/ +lemma LinUCBConfidenceEvent.toUpTo [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : + LinUCBConfidenceEventUpTo A R reg β x ν n ω := by + intro t _ht ht0 + exact h_conf t ht0 + +omit [IsMarkovKernel ν] in +/-- First projection from the horizon-local LinUCB confidence event: optimism for the best arm. -/ +lemma LinUCBConfidenceEventUpTo.best [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEventUpTo A R reg β x ν n ω) : + ∀ t, t ∈ range n → t ≠ 0 → + (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by + intro t ht ht0 + exact (h_conf t ht ht0).1 + +omit [IsMarkovKernel ν] in +/-- Second projection from the horizon-local LinUCB confidence event: validity of the selected +arm's lower confidence inequality. -/ +lemma LinUCBConfidenceEventUpTo.arm [Nonempty (Fin K)] + (h_conf : LinUCBConfidenceEventUpTo A R reg β x ν n ω) : + ∀ t, t ∈ range n → t ≠ 0 → + estimatedReward A R reg x (A t ω) t ω - + √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by + intro t ht ht0 + exact (h_conf t ht ht0).2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Linear realizability of the arm means by a parameter `θ`. + +The actual self-normalized concentration theorem for LinUCB needs an assumption of this shape: +each arm's mean reward must be represented by the finite-dimensional linear model +`x_aᵀ θ`. Without such a realizability assumption, no concentration theorem can imply +`LinUCBConfidenceEvent`, because the least-squares predictor may be biased away from the true arm +means for structural reasons rather than random noise. -/ +def LinearMeanModel (ν : Kernel (Fin K) ℝ) (x : Fin K → Feature d) (θ : Feature d) : Prop := + ∀ a, (ν a)[id] = dotProduct θ (x a) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Squared Euclidean-norm bound on a linear-bandit parameter. This is the finite-dimensional +version of the textbook assumption `‖θ‖₂ ≤ S`, written as `θᵀθ ≤ S2`. -/ +def ParameterSqNormBound (θ : Feature d) (S2 : ℝ) : Prop := + dotProduct θ θ ≤ S2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Mean response vector predicted by a linear mean model along the realized actions. -/ +noncomputable def meanResponseVector + (A : ℕ → Ω → Fin K) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ s ∈ range n, (ν (A s ω))[id] • x (A s ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Centered response vector, i.e. the accumulated reward-feature vector minus its conditional mean +under the finite-action linear mean model. This is the deterministic noise vector that the later +self-normalized concentration theorem must control. -/ +noncomputable def centeredResponseVector + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + responseVector A R x n ω - meanResponseVector A ν x n ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Scalar reward noise at time `t`, centered at the mean of the selected arm. -/ +noncomputable def rewardNoise + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (t : ℕ) (ω : Ω) : ℝ := + R t ω - (ν (A t ω))[id] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Arm-wise centered reward subgaussianity. + +This is the LinUCB analogue of the scalar noise assumption used in `UCB.lean`: for each arm, the +reward centered at that arm's mean has a subgaussian moment-generating-function bound. A future +vector self-normalized concentration theorem should start from this assumption, together with the +linear mean model, and prove the centered-noise-plus-bias confidence event. -/ +def RewardNoiseSubgaussian (ν : Kernel (Fin K) ℝ) (σ2 : ℝ≥0) : Prop := + ∀ a, HasSubgaussianMGF (fun r ↦ r - (ν a)[id]) σ2 (ν a) + +end AlgorithmBehavior + +end LinUCB + +end Bandits diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean new file mode 100644 index 00000000..24fc92c2 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean @@ -0,0 +1,354 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.TextbookMixture + +/-! +# LinUCB Textbook Confidence Events + +Deterministic bridges from centered-noise and textbook self-normalized events to +LinUCB prediction-confidence events. +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local coordinate-wise centered-noise event. + +This is a conservative finite-dimensional interface for reusing scalar projection concentration: +if every coordinate of the centered response vector is bounded, then the centered-noise quadratic +form is bounded through `centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg`. -/ +def LinUCBCenteredNoiseCoordinateBoundEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → ∀ i, + |centeredResponseVector A R ν x t ω i| ≤ coordBudget (t + 1) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Upper-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ +def centeredNoiseCoordinateUpperFailure + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := + {ω | coordBudget (t + 1) < centeredResponseVector A R ν x t ω i} + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Lower-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ +def centeredNoiseCoordinateLowerFailure + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := + {ω | centeredResponseVector A R ν x t ω i < -coordBudget (t + 1)} + +omit [IsMarkovKernel ν] in +/-- A global centered-noise-plus-bias confidence event implies its finite-horizon restriction. -/ +lemma LinUCBCenteredNoiseBiasConfidenceEvent.toUpTo + (θ : Feature d) + (h_noise : LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by + intro t _ht ht0 + exact h_noise t ht0 + +omit [IsMarkovKernel ν] in +/-- A coordinate-wise centered-noise event implies the centered-noise quadratic-form event under the +corresponding conservative budget. -/ +lemma LinUCBCenteredNoiseConfidenceEventUpTo.of_coordinateBound + (hreg_pos : 0 < reg) + {coordBudget noiseBudget : ℕ → ℝ} + (hcoord : + LinUCBCenteredNoiseCoordinateBoundEventUpTo A R coordBudget x ν n ω) + (hcoord_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ coordBudget (t + 1)) + (h_budget : ∀ t, t ∈ range n → t ≠ 0 → + (d : ℝ) * coordBudget (t + 1) ^ 2 / reg ≤ noiseBudget (t + 1)) : + LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω := by + intro t ht ht0 + exact + (centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) hreg_pos + (hcoord_nonneg t ht ht0) (hcoord t ht ht0)).trans + (h_budget t ht ht0) + +omit [IsMarkovKernel ν] in +/-- A horizon-local centered-noise event plus a parameter norm bound implies the +centered-noise-plus-ridge-bias event. + +This is the finite-horizon deterministic half of the textbook confidence-set proof: +the future self-normalized concentration theorem only has to control +`centeredNoiseQuadraticForm`; the ridge-bias contribution is bounded here by +`reg * S2`. -/ +lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_centeredNoise + (θ : Feature d) (S2 : ℝ) + (hreg_pos : 0 < reg) + (hθ : ParameterSqNormBound θ S2) + {noiseBudget : ℕ → ℝ} + (h_noise : + LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω) + (h_budget : ∀ t, t ∈ range n → t ≠ 0 → + (√(noiseBudget (t + 1)) + √(reg * S2)) ^ 2 ≤ β (t + 1)) : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by + intro t ht ht0 + exact + (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos + (noiseBudget (t + 1)) (reg * S2) (h_noise t ht ht0) + (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) + (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans + (h_budget t ht ht0) + +omit [IsMarkovKernel ν] in +/-- The textbook determinant-ratio self-normalized noise event plus the ridge-bias radius implies +the existing centered-noise-plus-bias confidence event. + +This is the deterministic bridge from the future Gaussian-mixture concentration theorem to the +confidence event already consumed by the LinUCB regret proof. -/ +lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_textbookSelfNormalizedNoise + (θ : Feature d) (S2 : ℝ) {σ2 : ℝ≥0} {δ : ℝ} + (hreg_pos : 0 < reg) + (hθ : ParameterSqNormBound θ S2) + (h_noise : + LinUCBTextbookSelfNormalizedNoiseEventUpTo A R reg σ2 δ x ν n ω) + (h_budget : ∀ t, t ∈ range n → t ≠ 0 → + (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2)) ^ 2 ≤ β (t + 1)) : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by + intro t ht ht0 + exact + (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos + (textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + (reg * S2) (h_noise t ht ht0) + (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) + (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans + (h_budget t ht ht0) + +omit [IsMarkovKernel ν] in +/-- The centered-noise-plus-bias confidence event implies the textbook parameter ellipsoid event +under linear realizability and positive regularization. -/ +lemma LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_noise : + LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : + LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω := by + intro t ht + rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] + exact h_noise t ht + +omit [IsMarkovKernel ν] in +/-- The horizon-local centered-noise-plus-bias event implies the horizon-local parameter ellipsoid +event under linear realizability and positive regularization. -/ +lemma LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_noise : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : + LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω := by + intro t ht ht0 + rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] + exact h_noise t ht ht0 + +omit [IsMarkovKernel ν] in +/-- Under linear realizability and positive regularization, the parameter ellipsoid event is exactly +the centered-noise-plus-bias confidence event. -/ +lemma linUCBParameterEllipsoidConfidenceEvent_iff_centeredNoiseBiasConfidenceEvent + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) : + LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω ↔ + LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω := by + constructor + · intro h_ellipsoid t ht + rw [← parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] + exact h_ellipsoid t ht + · intro h_noise + exact LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear hreg_pos h_noise + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Action-wise prediction-error confidence event around a linear parameter `θ`. + +This is the finite-action consequence of the textbook ellipsoid event after applying the +matrix Cauchy-Schwarz inequality to each arm. -/ +def LinUCBParameterPredictionConfidenceEvent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (θ : Feature d) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → ∀ a, + |dotProduct (thetaHat A R reg x t ω - θ) (x a)| ≤ + √(β (t + 1)) * width A reg x a t ω + +/-- Uniform self-normalized prediction-error event for finite-action LinUCB. + +This is the event that a future self-normalized martingale concentration theorem should prove with +high probability, for a concrete textbook choice of `β`. It says that, at every positive time and +for every finite action, the least-squares prediction error is bounded by the LinUCB confidence +radius. -/ +def LinUCBSelfNormalizedConfidenceEvent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → ∀ a, + |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ + √(β (t + 1)) * width A reg x a t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local self-normalized prediction-error event. + +This is the finite-horizon version of `LinUCBSelfNormalizedConfidenceEvent`. For regret through +time `n`, the proof only needs prediction confidence for positive `t ∈ range n`. -/ +def LinUCBSelfNormalizedConfidenceEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → ∀ a, + |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ + √(β (t + 1)) * width A reg x a t ω + +omit [IsMarkovKernel ν] in +/-- A global self-normalized prediction-error event implies its finite-horizon restriction. -/ +lemma LinUCBSelfNormalizedConfidenceEvent.toUpTo + (h_self : LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω) : + LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by + intro t _ht ht0 a + exact h_self t ht0 a + +omit [IsMarkovKernel ν] in +/-- A parameter prediction-confidence event implies the self-normalized confidence event once the +arm means are realized by that parameter. -/ +lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (h_param : LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω) : + LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := by + intro t ht a + have h_param_t := h_param t ht a + have h_error : + estimatedReward A R reg x a t ω - (ν a)[id] = + dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + calc + estimatedReward A R reg x a t ω - (ν a)[id] + = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by + rw [estimatedReward, h_linear a] + _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + rw [sub_dotProduct] + rw [h_error] + simpa [sub_dotProduct] using h_param_t + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The textbook ellipsoid event implies action-wise parameter prediction confidence under positive +regularization, by the matrix Cauchy-Schwarz step. -/ +lemma LinUCBParameterPredictionConfidenceEvent.of_ellipsoid + (θ : Feature d) + (hreg_pos : 0 < reg) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : + LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω := by + intro t ht a + have h_cauchy_t := + linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos (thetaHat A R reg x t ω - θ) a t ω + have h_radius : + √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := + Real.sqrt_le_sqrt (h_ellipsoid t ht) + calc + |dotProduct (thetaHat A R reg x t ω - θ) (x a)| + ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by + simpa [parameterErrorQuadraticForm] using h_cauchy_t + _ ≤ √(β (t + 1)) * width A reg x a t ω := by + exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) + +omit [IsMarkovKernel ν] in +/-- The textbook ellipsoid event implies the self-normalized prediction-error event under positive +regularization and linear realizability. -/ +lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterEllipsoid + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : + LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := + LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear + (LinUCBParameterPredictionConfidenceEvent.of_ellipsoid (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ω := ω) θ hreg_pos h_ellipsoid) + +omit [IsMarkovKernel ν] in +/-- The horizon-local textbook ellipsoid event implies the horizon-local self-normalized +prediction-error event under positive regularization and linear realizability. -/ +lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω) : + LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by + intro t ht ht0 a + have h_cauchy_t := + linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos (thetaHat A R reg x t ω - θ) a t ω + have h_radius : + √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := + Real.sqrt_le_sqrt (h_ellipsoid t ht ht0) + have h_error : + estimatedReward A R reg x a t ω - (ν a)[id] = + dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + calc + estimatedReward A R reg x a t ω - (ν a)[id] + = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by + rw [estimatedReward, h_linear a] + _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + rw [sub_dotProduct] + rw [h_error] + calc + |dotProduct (thetaHat A R reg x t ω - θ) (x a)| + ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by + simpa [parameterErrorQuadraticForm] using h_cauchy_t + _ ≤ √(β (t + 1)) * width A reg x a t ω := by + exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) + +omit [IsMarkovKernel ν] in +/-- The horizon-local centered-noise-plus-bias event implies the horizon-local self-normalized +prediction-error event under linear realizability and positive regularization. -/ +lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_centeredNoiseBias + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_noise : LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : + LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := + LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) θ h_linear hreg_pos + (LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) + θ h_linear hreg_pos h_noise) + +end AlgorithmBehavior + +end LinUCB + +end Bandits diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean new file mode 100644 index 00000000..5a500a62 --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean @@ -0,0 +1,354 @@ +/- +Copyright (c) 2026. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.TextbookMixture + +/-! +# LinUCB Textbook Confidence Bridge + +Deterministic bridges from centered-noise and textbook self-normalized events to +LinUCB prediction-confidence events. +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +section AlgorithmBehavior + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local coordinate-wise centered-noise event. + +This is a conservative finite-dimensional interface for reusing scalar projection concentration: +if every coordinate of the centered response vector is bounded, then the centered-noise quadratic +form is bounded through `centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg`. -/ +def LinUCBCenteredNoiseCoordinateBoundEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → ∀ i, + |centeredResponseVector A R ν x t ω i| ≤ coordBudget (t + 1) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Upper-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ +def centeredNoiseCoordinateUpperFailure + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := + {ω | coordBudget (t + 1) < centeredResponseVector A R ν x t ω i} + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Lower-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ +def centeredNoiseCoordinateLowerFailure + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := + {ω | centeredResponseVector A R ν x t ω i < -coordBudget (t + 1)} + +omit [IsMarkovKernel ν] in +/-- A global centered-noise-plus-bias confidence event implies its finite-horizon restriction. -/ +lemma LinUCBCenteredNoiseBiasConfidenceEvent.toUpTo + (θ : Feature d) + (h_noise : LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by + intro t _ht ht0 + exact h_noise t ht0 + +omit [IsMarkovKernel ν] in +/-- A coordinate-wise centered-noise event implies the centered-noise quadratic-form event under the +corresponding conservative budget. -/ +lemma LinUCBCenteredNoiseConfidenceEventUpTo.of_coordinateBound + (hreg_pos : 0 < reg) + {coordBudget noiseBudget : ℕ → ℝ} + (hcoord : + LinUCBCenteredNoiseCoordinateBoundEventUpTo A R coordBudget x ν n ω) + (hcoord_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ coordBudget (t + 1)) + (h_budget : ∀ t, t ∈ range n → t ≠ 0 → + (d : ℝ) * coordBudget (t + 1) ^ 2 / reg ≤ noiseBudget (t + 1)) : + LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω := by + intro t ht ht0 + exact + (centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) hreg_pos + (hcoord_nonneg t ht ht0) (hcoord t ht ht0)).trans + (h_budget t ht ht0) + +omit [IsMarkovKernel ν] in +/-- A horizon-local centered-noise event plus a parameter norm bound implies the +centered-noise-plus-ridge-bias event. + +This is the finite-horizon deterministic half of the textbook confidence-set proof: +the future self-normalized concentration theorem only has to control +`centeredNoiseQuadraticForm`; the ridge-bias contribution is bounded here by +`reg * S2`. -/ +lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_centeredNoise + (θ : Feature d) (S2 : ℝ) + (hreg_pos : 0 < reg) + (hθ : ParameterSqNormBound θ S2) + {noiseBudget : ℕ → ℝ} + (h_noise : + LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω) + (h_budget : ∀ t, t ∈ range n → t ≠ 0 → + (√(noiseBudget (t + 1)) + √(reg * S2)) ^ 2 ≤ β (t + 1)) : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by + intro t ht ht0 + exact + (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos + (noiseBudget (t + 1)) (reg * S2) (h_noise t ht ht0) + (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) + (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans + (h_budget t ht ht0) + +omit [IsMarkovKernel ν] in +/-- The textbook determinant-ratio self-normalized noise event plus the ridge-bias radius implies +the existing centered-noise-plus-bias confidence event. + +This is the deterministic bridge from the future Gaussian-mixture concentration theorem to the +confidence event already consumed by the LinUCB regret proof. -/ +lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_textbookSelfNormalizedNoise + (θ : Feature d) (S2 : ℝ) {σ2 : ℝ≥0} {δ : ℝ} + (hreg_pos : 0 < reg) + (hθ : ParameterSqNormBound θ S2) + (h_noise : + LinUCBTextbookSelfNormalizedNoiseEventUpTo A R reg σ2 δ x ν n ω) + (h_budget : ∀ t, t ∈ range n → t ≠ 0 → + (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2)) ^ 2 ≤ β (t + 1)) : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by + intro t ht ht0 + exact + (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos + (textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + (reg * S2) (h_noise t ht ht0) + (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) + (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans + (h_budget t ht ht0) + +omit [IsMarkovKernel ν] in +/-- The centered-noise-plus-bias confidence event implies the textbook parameter ellipsoid event +under linear realizability and positive regularization. -/ +lemma LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_noise : + LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : + LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω := by + intro t ht + rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] + exact h_noise t ht + +omit [IsMarkovKernel ν] in +/-- The horizon-local centered-noise-plus-bias event implies the horizon-local parameter ellipsoid +event under linear realizability and positive regularization. -/ +lemma LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_noise : + LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : + LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω := by + intro t ht ht0 + rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] + exact h_noise t ht ht0 + +omit [IsMarkovKernel ν] in +/-- Under linear realizability and positive regularization, the parameter ellipsoid event is exactly +the centered-noise-plus-bias confidence event. -/ +lemma linUCBParameterEllipsoidConfidenceEvent_iff_centeredNoiseBiasConfidenceEvent + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) : + LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω ↔ + LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω := by + constructor + · intro h_ellipsoid t ht + rw [← parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] + exact h_ellipsoid t ht + · intro h_noise + exact LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear hreg_pos h_noise + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Action-wise prediction-error confidence event around a linear parameter `θ`. + +This is the finite-action consequence of the textbook ellipsoid event after applying the +matrix Cauchy-Schwarz inequality to each arm. -/ +def LinUCBParameterPredictionConfidenceEvent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (θ : Feature d) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → ∀ a, + |dotProduct (thetaHat A R reg x t ω - θ) (x a)| ≤ + √(β (t + 1)) * width A reg x a t ω + +/-- Uniform self-normalized prediction-error event for finite-action LinUCB. + +This is the event that a future self-normalized martingale concentration theorem should prove with +high probability, for a concrete textbook choice of `β`. It says that, at every positive time and +for every finite action, the least-squares prediction error is bounded by the LinUCB confidence +radius. -/ +def LinUCBSelfNormalizedConfidenceEvent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → ∀ a, + |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ + √(β (t + 1)) * width A reg x a t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local self-normalized prediction-error event. + +This is the finite-horizon version of `LinUCBSelfNormalizedConfidenceEvent`. For regret through +time `n`, the proof only needs prediction confidence for positive `t ∈ range n`. -/ +def LinUCBSelfNormalizedConfidenceEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → ∀ a, + |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ + √(β (t + 1)) * width A reg x a t ω + +omit [IsMarkovKernel ν] in +/-- A global self-normalized prediction-error event implies its finite-horizon restriction. -/ +lemma LinUCBSelfNormalizedConfidenceEvent.toUpTo + (h_self : LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω) : + LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by + intro t _ht ht0 a + exact h_self t ht0 a + +omit [IsMarkovKernel ν] in +/-- A parameter prediction-confidence event implies the self-normalized confidence event once the +arm means are realized by that parameter. -/ +lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (h_param : LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω) : + LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := by + intro t ht a + have h_param_t := h_param t ht a + have h_error : + estimatedReward A R reg x a t ω - (ν a)[id] = + dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + calc + estimatedReward A R reg x a t ω - (ν a)[id] + = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by + rw [estimatedReward, h_linear a] + _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + rw [sub_dotProduct] + rw [h_error] + simpa [sub_dotProduct] using h_param_t + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The textbook ellipsoid event implies action-wise parameter prediction confidence under positive +regularization, by the matrix Cauchy-Schwarz step. -/ +lemma LinUCBParameterPredictionConfidenceEvent.of_ellipsoid + (θ : Feature d) + (hreg_pos : 0 < reg) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : + LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω := by + intro t ht a + have h_cauchy_t := + linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos (thetaHat A R reg x t ω - θ) a t ω + have h_radius : + √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := + Real.sqrt_le_sqrt (h_ellipsoid t ht) + calc + |dotProduct (thetaHat A R reg x t ω - θ) (x a)| + ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by + simpa [parameterErrorQuadraticForm] using h_cauchy_t + _ ≤ √(β (t + 1)) * width A reg x a t ω := by + exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) + +omit [IsMarkovKernel ν] in +/-- The textbook ellipsoid event implies the self-normalized prediction-error event under positive +regularization and linear realizability. -/ +lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterEllipsoid + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : + LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := + LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear + (LinUCBParameterPredictionConfidenceEvent.of_ellipsoid (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ω := ω) θ hreg_pos h_ellipsoid) + +omit [IsMarkovKernel ν] in +/-- The horizon-local textbook ellipsoid event implies the horizon-local self-normalized +prediction-error event under positive regularization and linear realizability. -/ +lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω) : + LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by + intro t ht ht0 a + have h_cauchy_t := + linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos (thetaHat A R reg x t ω - θ) a t ω + have h_radius : + √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := + Real.sqrt_le_sqrt (h_ellipsoid t ht ht0) + have h_error : + estimatedReward A R reg x a t ω - (ν a)[id] = + dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + calc + estimatedReward A R reg x a t ω - (ν a)[id] + = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by + rw [estimatedReward, h_linear a] + _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by + rw [sub_dotProduct] + rw [h_error] + calc + |dotProduct (thetaHat A R reg x t ω - θ) (x a)| + ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by + simpa [parameterErrorQuadraticForm] using h_cauchy_t + _ ≤ √(β (t + 1)) * width A reg x a t ω := by + exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) + +omit [IsMarkovKernel ν] in +/-- The horizon-local centered-noise-plus-bias event implies the horizon-local self-normalized +prediction-error event under linear realizability and positive regularization. -/ +lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_centeredNoiseBias + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) + (h_noise : LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : + LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := + LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) θ h_linear hreg_pos + (LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias (A := A) (R := R) + (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) + θ h_linear hreg_pos h_noise) + +end AlgorithmBehavior + +end LinUCB + +end Bandits From 7fd939ccf55943bd442cbf69ad76242cd648d147 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Thu, 25 Jun 2026 15:16:12 -0400 Subject: [PATCH 77/88] feat(LinUCB Confidence Events): file renames/delets --- .../Online/Bandit/Algorithms/LinUCB.lean | 4049 +---------------- .../Bandit/Algorithms/LinUCB/Confidence.lean | 148 - .../Algorithms/LinUCB/TextbookConfidence.lean | 354 -- 3 files changed, 6 insertions(+), 4545 deletions(-) delete mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean delete mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index da0dc185..f528dfdb 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -5,4052 +5,15 @@ Authors: OpenAI, Fawad Haider -/ module -public import LeanMachineLearning.Online.Bandit.SumRewards -public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax -public import Mathlib.Analysis.MeanInequalities -public import Mathlib.Analysis.SpecialFunctions.Log.Deriv -public import Mathlib.Analysis.Matrix.Order -public import Mathlib.Data.Real.StarOrdered -public import Mathlib.LinearAlgebra.Matrix.PosDef -public import Mathlib.LinearAlgebra.Matrix.SchurComplement -public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Regret /-! # LinUCB for finite-action linear bandits -Chapter 19 of *Bandit Algorithms*: --/ - -@[expose] public section - -open MeasureTheory ProbabilityTheory Filter Real Finset Learning - -open scoped ENNReal NNReal Matrix MatrixOrder - -namespace Bandits - -variable {K d : ℕ} - -section Algorithm - -namespace LinUCB - -/-- Feature vectors for finite-dimensional linear bandits. -/ -abbrev Feature (d : ℕ) := Fin d → ℝ - -/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product -`x_aᵀ x_a`. -/ -def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := - dotProduct (x a) (x a) - -/-- The squared feature norm is nonnegative. -/ -lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : - 0 ≤ featureSqNorm x a := by - rw [featureSqNorm, dotProduct] - exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) -/-- Uniform squared feature-norm bound for finite-action LinUCB. +This module is the public entry point for the finite-action LinUCB development. -This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared -form as `‖x_a‖₂² ≤ L2` for every action. -/ -def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := - ∀ a, featureSqNorm x a ≤ L2 - -/-- History-level regularized design matrix for LinUCB. -/ -noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := - reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) - -/-- History-level response vector for LinUCB. -/ -noncomputable def responseVector' (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - ∑ s : Iic n, (h s).2 • x (h s).1 - -/-- History-level regularized least-squares estimate. -/ -noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) - -/-- History-level estimated reward of an arm. -/ -noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - dotProduct (thetaHat' reg x n h) (x a) - -/-- History-level quadratic form underlying the LinUCB confidence width. -/ -noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) - -/-- History-level elliptical confidence width of an arm. -/ -noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(widthQuadraticForm' reg x n h a) - -/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic -form is nonnegative. -/ -lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : - width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by - simp [width', Real.sq_sqrt h_nonneg] - -/-- LinUCB optimistic index of an arm. - -The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` -contains the observations through time `n`, this index is used to choose the arm -at time `n + 1`, and we evaluate the schedule at `n + 2` +The implementation is split across submodules under +`LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.*`; importing this file re-exports the full +LinUCB API, including the algorithm definition, confidence events, concentration interfaces, +deterministic regret decomposition, elliptical-potential/log-det bounds, and final regret theorems. -/ -noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a - -open Classical in -/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ -noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - measurableArgmax (fun h a ↦ index' reg β x n h a) h - -@[fun_prop] -lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → Feature d) - (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) - (n : ℕ) : - Measurable (nextArm hK reg β x n) := by - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact measurable_measurableArgmax fun a ↦ h_index n a - -end LinUCB - -/-- The finite-action LinUCB algorithm. -/ -noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → LinUCB.Feature d) - (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : - Algorithm (Fin K) ℝ := - detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ - -end Algorithm - -namespace LinUCB - -variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} - {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} - {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] - {Ω : Type*} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} - {n : ℕ} {ω : Ω} - -section AlgorithmBehavior - -/-- The process-level design matrix built from actions up to time `n` excluded. -/ -noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := - reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) - -/-- The initial design matrix before any actions are included. -/ -lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - designMatrix A reg x 0 ω = reg • 1 := by - simp [designMatrix] - -/-- The design matrix update after observing one additional action. -/ -lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designMatrix A reg x (n + 1) ω = - designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by - simp [designMatrix, sum_range_succ, add_assoc] - -/-- With nonnegative regularization, the process-level design matrix is positive semidefinite. -/ -lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : - (designMatrix A reg x n ω).PosSemidef := by - unfold designMatrix - apply Matrix.PosSemidef.add - · exact Matrix.PosSemidef.smul Matrix.PosSemidef.one hreg_nonneg - · refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - -/-- Positive regularization makes the process-level design matrix positive definite. -/ -lemma designMatrix_posDef (hreg_pos : 0 < reg) : - (designMatrix A reg x n ω).PosDef := by - unfold designMatrix - apply Matrix.PosDef.add_posSemidef - · exact Matrix.PosDef.smul Matrix.PosDef.one hreg_pos - · refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The design matrix dominates its regularization part: after subtracting `reg • I`, what remains -is the sum of observed rank-one feature matrices, hence positive semidefinite. -/ -lemma designMatrix_sub_reg_smul_one_posSemidef : - (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by - have hsum : - (∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω))).PosSemidef := by - refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - simpa [designMatrix, add_sub_cancel_left] using hsum - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix-order form of `designMatrix_sub_reg_smul_one_posSemidef`: `reg • I ≤ V_n`. -/ -lemma reg_smul_one_le_designMatrix : - reg • (1 : Matrix (Fin d) (Fin d) ℝ) ≤ designMatrix A reg x n ω := by - rw [Matrix.le_iff] - exact designMatrix_sub_reg_smul_one_posSemidef (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix order is preserved by evaluating the quadratic form against a fixed feature vector. -/ -lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} - (hMN : M ≤ N) (u : Feature d) : - dotProduct u (M *ᵥ u) ≤ dotProduct u (N *ᵥ u) := by - have h_nonneg : 0 ≤ dotProduct u ((N - M) *ᵥ u) := by - simpa using (Matrix.le_iff.mp hMN).dotProduct_mulVec_nonneg u - rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg - exact sub_nonneg.mp h_nonneg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The inverse of the regularized identity is the reciprocal-scaled identity. -/ -lemma reg_smul_one_inv (hreg : reg ≠ 0) : - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ = - reg⁻¹ • (1 : Matrix (Fin d) (Fin d) ℝ) := by - rw [Matrix.inv_eq_left_inv] - simp [smul_smul, hreg] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The quadratic form induced by `(reg • I)⁻¹` is the squared norm divided by `reg`. -/ -lemma dotProduct_reg_smul_one_inv_mulVec (hreg : reg ≠ 0) (u : Feature d) : - dotProduct u (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ u) = - dotProduct u u / reg := by - rw [reg_smul_one_inv (reg := reg) (d := d) hreg] - simp [Matrix.smul_mulVec, div_eq_inv_mul, mul_comm] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Arm-specific form of `dotProduct_reg_smul_one_inv_mulVec`. -/ -lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div - (hreg : reg ≠ 0) (a : Fin K) : - dotProduct (x a) (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) = - featureSqNorm x a / reg := by - simpa [featureSqNorm] using - dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Reusable matrix-analysis theorem needed for the LinUCB width comparison. - -It states the usual inverse anti-monotonicity of positive-definite matrices in the PSD order: -if `M` is positive definite and `M ≤ N`, then inversion reverses the order. -/ -def MatrixInvAntiMonoOnPosDef (d : ℕ) : Prop := - ∀ M N : Matrix (Fin d) (Fin d) ℝ, M.PosDef → M ≤ N → N⁻¹ ≤ M⁻¹ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret -route. - -Mathematically, this should follow from `reg • I ≤ V_t` and positive regularization: inversion -reverses the positive-definite matrix order, so `V_t⁻¹ ≤ (reg • I)⁻¹`. -/ -def DesignMatrixInvLeRegInv - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := - ∀ (n : ℕ) (ω : Ω), - (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection lemma for the named inverse-order obligation. -/ -lemma DesignMatrixInvLeRegInv.apply - (h_inv : DesignMatrixInvLeRegInv A reg x) (n : ℕ) (ω : Ω) : - (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := - h_inv n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The reusable positive-definite inverse anti-monotonicity theorem implies the LinUCB-specific -inverse-design comparison. -/ -lemma DesignMatrixInvLeRegInv.of_matrix_inv_antitone - (hreg_pos : 0 < reg) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) : - DesignMatrixInvLeRegInv A reg x := by - intro n ω - exact h_inv_antitone (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) - (designMatrix A reg x n ω) - (Matrix.PosDef.smul Matrix.PosDef.one hreg_pos) - (reg_smul_one_le_designMatrix (A := A) (reg := reg) (x := x) (n := n) (ω := ω)) - -/-- Trace of the process-level regularized design matrix. -/ -noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - Matrix.trace (designMatrix A reg x n ω) - -/-- Before any observations, the design trace is the trace of `reg • I_d`, namely `reg * d`. -/ -lemma designTrace_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - designTrace A reg x 0 ω = reg * (d : ℝ) := by - simp [designTrace, designMatrix_zero] - -/-- Updating the design matrix by `x_a x_aᵀ` increases the trace by `x_aᵀ x_a`. -/ -lemma designTrace_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designTrace A reg x (n + 1) ω = - designTrace A reg x n ω + featureSqNorm x (A n ω) := by - simp [designTrace, designMatrix_succ, featureSqNorm, Matrix.trace_vecMulVec] - -/-- Closed form for the design trace: initial regularization trace plus accumulated squared -feature norms. -/ -lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designTrace A reg x n ω = - reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by - simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] - -/-- With nonnegative regularization, the design trace is nonnegative. -/ -lemma designTrace_nonneg (hreg_nonneg : 0 ≤ reg) : - 0 ≤ designTrace A reg x n ω := by - rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] - exact add_nonneg - (mul_nonneg hreg_nonneg (Nat.cast_nonneg d)) - (sum_nonneg fun t _ ↦ featureSqNorm_nonneg x (A t ω)) - -/-- If every selected feature vector has squared norm at most `L2`, then the trace of the design -matrix is at most `reg * d + n * L2`. -/ -lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by - rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] - gcongr - calc - (∑ t ∈ range n, featureSqNorm x (A t ω)) ≤ ∑ _t ∈ range n, L2 := by - exact sum_le_sum fun t ht ↦ hL2 t ht - _ = (n : ℝ) * L2 := by - simp [nsmul_eq_mul] - -omit [IsProbabilityMeasure P] in -/-- Almost surely, bounded selected feature norms give the corresponding deterministic trace -budget `reg * d + n * L2`. -/ -lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : - ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by - filter_upwards [hL2] with ω hL2ω - exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) L2 hL2ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A uniform finite-action feature bound implies the selected-action feature bound through any -finite horizon. -/ -lemma featureSqNorm_ae_le_of_featureSqNormBound - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2 := - Filter.Eventually.of_forall fun ω t _ht ↦ hL2 (A t ω) - -/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ -noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - ∑ s ∈ range n, R s ω • x (A s ω) - -/-- The initial response vector before any rewards are included. -/ -lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (ω : Ω) : - responseVector A R x 0 ω = 0 := by - simp [responseVector] - -/-- The response-vector update after observing one additional reward. -/ -lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - responseVector A R x (n + 1) ω = - responseVector A R x n ω + R n ω • x (A n ω) := by - simp [responseVector, sum_range_succ] - -/-- The process-level regularized least-squares estimate. -/ -noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) - -/-- The initial least-squares estimate is zero because no reward-feature observations have been -included yet. -/ -lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - thetaHat A R reg x 0 ω = 0 := by - simp [thetaHat, responseVector_zero] - -/-- The process-level estimated linear reward. -/ -noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - dotProduct (thetaHat A R reg x n ω) (x a) - -/-- The initial estimated reward is zero for every arm. -/ -lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - estimatedReward A R reg x a 0 ω = 0 := by - simp [estimatedReward, thetaHat_zero] - -/-- The quadratic form `x_aᵀ V_n⁻¹ x_a` underlying the LinUCB confidence width. -/ -noncomputable def widthQuadraticForm (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) - -/-- The initial width quadratic form is induced by the inverse regularized identity. -/ -lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - widthQuadraticForm A reg x a 0 ω = - dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by - simp [widthQuadraticForm, designMatrix_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Nonnegative regularization makes every LinUCB width quadratic form nonnegative. - -The reason is purely matrix-theoretic: `V_n` is positive semidefinite, the nonsingular inverse of a -positive semidefinite matrix is positive semidefinite in mathlib, and every quadratic form induced -by a positive semidefinite matrix is nonnegative. -/ -lemma widthQuadraticForm_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) (a : Fin K) : - 0 ≤ widthQuadraticForm A reg x a n ω := by - simpa [widthQuadraticForm] using - ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg).inv.dotProduct_mulVec_nonneg (x a)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, nonnegative regularization gives nonnegative selected quadratic width forms -through any finite horizon. -/ -lemma widthQuadraticForm_ae_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - exact Filter.Eventually.of_forall fun ω t _ht ↦ - widthQuadraticForm_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := t) (ω := ω) hreg_nonneg (A t ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive-time version of `widthQuadraticForm_ae_nonneg_of_reg_nonneg`, matching the side -condition shape used by the regret/width-sum bridge lemmas. -/ -lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - filter_upwards [widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) - (x := x) (n := n) (P := P) hreg_nonneg] with ω h_nonnegω - intro t ht _ht0 - exact h_nonnegω t ht - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The matrix comparison needed to turn bounded feature vectors into the positive-time LinUCB -width cap. - -Mathematically, this says `x_aᵀ V_t⁻¹ x_a ≤ ‖x_a‖² / reg`. A later matrix-order proof should -derive it from `reg > 0` and `V_t = reg I + ∑ x_s x_sᵀ`. Keeping it as a named property makes the -remaining linear-algebra obligation precise and reusable. -/ -def WidthQuadraticFormLeFeatureSqNormDivReg - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := - ∀ (a : Fin K) (n : ℕ) (ω : Ω), - widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If the inverse design matrix is bounded by the inverse regularized identity, then the LinUCB -quadratic width is bounded by `featureSqNorm / reg` for one arm, time, and sample point. -/ -lemma widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (a : Fin K) - (h_inv : (designMatrix A reg x n ω)⁻¹ ≤ - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) - (hreg : reg ≠ 0) : - widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg := by - calc - widthQuadraticForm A reg x a n ω = - dotProduct (x a) (((designMatrix A reg x n ω)⁻¹) *ᵥ (x a)) := rfl - _ ≤ dotProduct (x a) - (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) := - dotProduct_mulVec_le_of_matrix_le h_inv (x a) - _ = featureSqNorm x a / reg := - dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div - (reg := reg) (x := x) hreg a - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A pointwise inverse-order comparison for all times and sample points gives the reusable -`WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ -lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le - (hreg : reg ≠ 0) - (h_inv : DesignMatrixInvLeRegInv A reg x) : - WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by - intro a n ω - exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a - (h_inv.apply n ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the -quadratic form is at most one. -/ -lemma widthQuadraticForm_le_one_of_featureSqNorm_le_reg - (a : Fin K) - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) - (h_feature_le : featureSqNorm x a ≤ reg) : - widthQuadraticForm A reg x a n ω ≤ 1 := by - refine (h_width a n ω).trans ?_ - rwa [div_le_one hreg_pos] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure positive-time width cap from the matrix comparison and an almost-sure -`featureSqNorm ≤ reg` bound along the selected actions. -/ -lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) - (h_feature_le : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by - filter_upwards [h_feature_le] with ω h_feature_leω - intro t ht _ht0 - exact widthQuadraticForm_le_one_of_featureSqNorm_le_reg - (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (A t ω) h_width hreg_pos - (h_feature_leω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure positive-time width cap from the matrix comparison and a selected-feature budget -`featureSqNorm ≤ L2`, when `L2 ≤ reg`. -/ -lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) {L2 : ℝ} - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by - refine widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg - (A := A) (reg := reg) (x := x) (n := n) (P := P) h_width hreg_pos ?_ - filter_upwards [hL2] with ω hL2ω - intro t ht - exact (hL2ω t ht).trans hL2_le_reg - -/-- The process-level elliptical confidence width. -/ -noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(widthQuadraticForm A reg x a n ω) - -/-- The initial width is the quadratic form induced by the inverse regularized identity. -/ -lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - width A reg x a 0 ω = - √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [width, widthQuadraticForm_zero] - -/-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that -quadratic form is nonnegative. -/ -lemma width_sq_eq_quadratic_form (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x a n ω) : - width A reg x a n ω ^ 2 = widthQuadraticForm A reg x a n ω := by - simp [width, Real.sq_sqrt h_nonneg] - -/-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ -noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 - -/-- No positive-time widths are accumulated at horizon zero. -/ -lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - widthSqSum A reg x 0 ω = 0 := by - simp [widthSqSum] - -/-- Advancing the horizon adds the next positive-time squared width term. -/ -lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + - (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by - simp [widthSqSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's squared width. -/ -lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + width A reg x (A n ω) n ω ^ 2 := by - simp [widthSqSum_succ, hn] - -/-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ -noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else widthQuadraticForm A reg x (A t ω) t ω - -/-- No positive-time quadratic width forms are accumulated at horizon zero. -/ -lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - quadraticWidthSum A reg x 0 ω = 0 := by - simp [quadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time quadratic width form. -/ -lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - if n = 0 then 0 else widthQuadraticForm A reg x (A n ω) n ω := by - simp [quadraticWidthSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ -lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by - simp [quadraticWidthSum_succ, hn] - -/-- The accumulated capped quadratic forms corresponding to the positive-time LinUCB widths. -/ -noncomputable def cappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω) - -/-- No positive-time capped quadratic width forms are accumulated at horizon zero. -/ -lemma cappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - cappedQuadraticWidthSum A reg x 0 ω = 0 := by - simp [cappedQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time capped quadratic width form. -/ -lemma cappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - cappedQuadraticWidthSum A reg x (n + 1) ω = - cappedQuadraticWidthSum A reg x n ω + - if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by - simp [cappedQuadraticWidthSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's capped quadratic width form. -/ -lemma cappedQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - cappedQuadraticWidthSum A reg x (n + 1) ω = - cappedQuadraticWidthSum A reg x n ω + min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by - simp [cappedQuadraticWidthSum_succ, hn] - -/-- If every positive-time process-level quadratic width form is at most `1`, then the uncapped -and capped process-level quadratic-width accumulators agree. -/ -lemma quadraticWidthSum_eq_cappedQuadraticWidthSum - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - quadraticWidthSum A reg x n ω = cappedQuadraticWidthSum A reg x n ω := by - rw [quadraticWidthSum, cappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact (min_eq_right (h_le_one t ht ht0)).symm - -/-- If the squared-width and quadratic-form accumulators agree through a positive time and the -next quadratic form is nonnegative, then they still agree after adding the next term. -/ -lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by - rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, - quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hn, h_eq] - rw [width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A n ω) - (n := n) (ω := ω) h_nonneg] - -/-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive -time quadratic form is nonnegative. -/ -lemma widthSqSum_eq_sum_quadratic_form - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by - rw [widthSqSum, quadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0] - rw [if_neg ht0] - exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) - (n := t) (ω := ω) (h_nonneg t ht ht0) - -/-- A quadratic-form sum bound implies the corresponding bound on `widthSqSum`. This is the shape -expected from a later elliptical-potential argument. -/ -lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - exact h_quad_le - -/-- A capped process-level quadratic-form sum bound implies the corresponding bound on -`widthSqSum`, provided the positive-time process-level quadratic forms are nonnegative and at most -`1`. -/ -lemma widthSqSum_le_of_capped_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_capped_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - rw [quadraticWidthSum_eq_cappedQuadraticWidthSum (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_le_one] - exact h_capped_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped process-level quadratic-form sum bound implies the corresponding bound -on `widthSqSum`, provided the positive-time process-level quadratic forms are almost surely -nonnegative and at most `1`. -/ -lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_nonneg, h_le_one, h_capped_le] with - ω h_nonnegω h_le_oneω h_capped_leω - exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Determinant of the process-level LinUCB design matrix. -/ -noncomputable def designDet (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - Matrix.det (designMatrix A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial design determinant is the determinant of the regularized identity. -/ -lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by - simp [designDet, designMatrix_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial design determinant is `reg ^ d`. -/ -lemma designDet_zero_eq_reg_pow (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - designDet A reg x 0 ω = reg ^ d := by - rw [designDet_zero] - simp - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A nonzero regularization parameter gives a nonzero initial design determinant. -/ -lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hreg : reg ≠ 0) : - designDet A reg x 0 ω ≠ 0 := by - rw [designDet_zero_eq_reg_pow] - exact pow_ne_zero d hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization makes every process-level design determinant nonzero. -/ -lemma designDet_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : - designDet A reg x n ω ≠ 0 := by - have hunit : IsUnit (designMatrix A reg x n ω) := - (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_pos).isUnit - have hdet_unit : IsUnit (designMatrix A reg x n ω).det := - (Matrix.isUnit_iff_isUnit_det (A := designMatrix A reg x n ω)).mp hunit - exact hdet_unit.ne_zero - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, positive regularization makes all design determinants in a finite horizon -nonzero. -/ -lemma designDet_ae_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - exact Filter.Eventually.of_forall fun ω t _ht ↦ - designDet_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) (ω := ω) - hreg_pos - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ -noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - designDet A reg x n ω / designDet A reg x 0 ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the determinant ratio is `1` when the initial design determinant is nonzero. -/ -lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - designDetRatio A reg x 0 ω = 1 := by - simp [designDetRatio, hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the determinant ratio is positive when the initial design determinant is -nonzero. -/ -lemma designDetRatio_zero_pos (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - 0 < designDetRatio A reg x 0 ω := by - rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - norm_num - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. - -This is the determinant-ratio target used by the matrix-determinant part of the elliptical -potential lemma. -/ -noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - designDet A reg x (n + 1) ω / designDet A reg x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The scalar determinant appearing in the rank-one determinant update is the quadratic form -`uᵀ M u`. -/ -lemma det_one_add_replicateRow_mul_matrix_mul_replicateCol - (M : Matrix (Fin d) (Fin d) ℝ) (u : Feature d) : - (1 + Matrix.replicateRow Unit u * M * Matrix.replicateCol Unit u).det = - 1 + dotProduct u (Matrix.mulVec M u) := by - have hsum : - (∑ j, (∑ i, u i * M i j) * u j) = - ∑ i, u i * ∑ j, M i j * u j := by - calc - (∑ j, (∑ i, u i * M i j) * u j) - = ∑ j, ∑ i, (u i * M i j) * u j := by - simp [Finset.sum_mul] - _ = ∑ i, ∑ j, (u i * M i j) * u j := by - rw [Finset.sum_comm] - _ = ∑ i, u i * ∑ j, M i j * u j := by - refine Finset.sum_congr rfl ?_ - intro i _ - rw [Finset.mul_sum] - refine Finset.sum_congr rfl ?_ - intro j _ - ring - rw [Matrix.det_unique] - simpa [Matrix.mul_apply, Matrix.replicateRow, Matrix.replicateCol, Matrix.mulVec, - dotProduct] using hsum - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Process-level matrix determinant update for the LinUCB design matrix. - -If `V_n` has nonzero determinant, then the rank-one update -`V_{n+1} = V_n + x_{A_n} x_{A_n}ᵀ` satisfies -`det(V_{n+1}) = det(V_n) * (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ -lemma designDet_succ_eq_mul_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDet A reg x (n + 1) ω = - designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - have hM : IsUnit (designMatrix A reg x n ω).det := by - simpa [designDet] using (isUnit_iff_ne_zero.mpr hdet) - calc - designDet A reg x (n + 1) ω = - (designMatrix A reg x n ω + - Matrix.vecMulVec (x (A n ω)) (x (A n ω))).det := by - simp [designDet, designMatrix_succ] - _ = (designMatrix A reg x n ω + - Matrix.replicateCol Unit (x (A n ω)) * Matrix.replicateRow Unit (x (A n ω))).det := by - rw [Matrix.vecMulVec_eq Unit] - _ = (designMatrix A reg x n ω).det * - (1 + Matrix.replicateRow Unit (x (A n ω)) * - (designMatrix A reg x n ω)⁻¹ * Matrix.replicateCol Unit (x (A n ω))).det := by - exact Matrix.det_add_replicateCol_mul_replicateRow (A := designMatrix A reg x n ω) - (ι := Unit) hM (x (A n ω)) (x (A n ω)) - _ = designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - rw [designDet] - congr 1 - exact det_one_add_replicateRow_mul_matrix_mul_replicateCol - (M := (designMatrix A reg x n ω)⁻¹) (u := x (A n ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `det(V_n)` is nonzero and the selected quadratic form is nonnegative, then -`det(V_{n+1})` is nonzero. -/ -lemma designDet_succ_ne_zero_of_widthQuadraticForm_nonneg - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - designDet A reg x (n + 1) ω ≠ 0 := by - rw [designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - exact mul_ne_zero hdet (ne_of_gt (by linarith)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms preserve -nonzero design determinants up to any fixed time. -/ -lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt - (m : ℕ) (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t < m → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - designDet A reg x m ω ≠ 0 := by - induction m with - | zero => exact hdet0 - | succ m ih => - exact designDet_succ_ne_zero_of_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := m) (ω := ω) - (ih fun t ht ↦ h_nonneg t (Nat.lt_trans ht (Nat.lt_succ_self m))) - (h_nonneg m (Nat.lt_succ_self m)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms imply that -all design determinants through horizon `n` are nonzero. -/ -lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by - intro t ht - exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := t) (ω := ω) hdet0 fun s hs ↦ - h_nonneg s (mem_range.mpr (Nat.lt_of_lt_of_le hs (Nat.le_of_lt_succ (mem_range.mp ht)))) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant and nonnegative selected quadratic forms imply -that all design determinants through horizon `n` are nonzero. -/ -lemma designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω - exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `det(V_n) ≠ 0`, then the one-step determinant ratio is -`1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n}`. -/ -lemma designDetStepRatio_eq_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDetStepRatio A reg x n ω = - 1 + widthQuadraticForm A reg x (A n ω) n ω := by - simp [designDetStepRatio, - designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet, hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The cumulative determinant ratio advances by multiplying by the one-step determinant ratio. -/ -lemma designDetRatio_succ_eq_mul_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDetRatio A reg x (n + 1) ω = - designDetRatio A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - rw [designDetRatio, designDetRatio, - designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms make the -cumulative determinant ratio positive. -/ -lemma designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - 0 < designDetRatio A reg x n ω := by - induction n with - | zero => - exact designDetRatio_zero_pos (A := A) (reg := reg) (x := x) (ω := ω) hdet0 - | succ n ih => - have hdetn : designDet A reg x n ω ≠ 0 := - designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ - h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) - rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetn] - exact mul_pos - (ih fun t ht ↦ h_nonneg t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) - (by linarith [h_nonneg n (by simp)]) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, starting from a nonzero initial determinant, nonnegative selected quadratic -forms make the cumulative determinant ratio positive. -/ -lemma designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by - filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω - exact designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter and nonnegative selected quadratic forms make -the cumulative determinant ratio positive. -/ -lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by - refine designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, the cumulative determinant ratio is the finite -product of the per-round determinant-update factors. -/ -lemma designDetRatio_eq_prod_one_add_widthQuadraticForm - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - designDetRatio A reg x n ω = - ∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω) := by - induction n with - | zero => - rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet0] - simp - | succ n ih => - have hdetn : designDet A reg x n ω ≠ 0 := - designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ - h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) - rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetn] - rw [ih fun t ht ↦ h_nonneg t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))] - simp [Finset.prod_range_succ] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If every selected quadratic form is in `[0, 1]`, the cumulative determinant ratio is at most -`2 ^ n`. -/ -lemma designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - rw [designDetRatio_eq_prod_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0 h_nonneg] - calc - (∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω)) - ≤ ∏ _t ∈ range n, (2 : ℝ) := by - exact Finset.prod_le_prod - (fun t ht ↦ by linarith [h_nonneg t ht]) - (fun t ht ↦ by linarith [h_le_one t ht]) - _ = (2 : ℝ) ^ n := by - simp - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, if every selected quadratic form is in `[0, 1]`, the cumulative determinant -ratio is at most `2 ^ n`. -/ -lemma designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - filter_upwards [hdet0, h_nonneg, h_le_one] with ω hdet0ω h_nonnegω h_le_oneω - exact designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one (A := A) - (reg := reg) (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω h_le_oneω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter and selected quadratic forms in `[0, 1]` -imply the cumulative determinant ratio is at most `2 ^ n`. -/ -lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - refine designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one - (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Converts an almost-sure trace bound into the determinant-ratio bound expected from a future -trace/determinant comparison theorem. -/ -lemma designDetRatio_ae_le_trace_budget_of_designTrace_ae_le - (T : ℝ) - (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ T → - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - filter_upwards [h_trace_le] with ω h_traceω - exact h_ratio_of_trace ω h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms give the concrete trace budget -`reg * d + n * L2`; a future trace/determinant comparison then gives the corresponding -determinant-ratio bound. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - exact designDetRatio_ae_le_trace_budget_of_designTrace_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) (T := reg * (d : ℝ) + (n : ℝ) * L2) - (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2) - h_ratio_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A determinant upper bound for `V_n` implies the corresponding determinant-ratio bound, using -`det(V_0) = reg ^ d`. -/ -lemma designDetRatio_le_trace_budget_of_designDet_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hdet_le : designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - rw [designDetRatio, designDet_zero_eq_reg_pow] - have hreg_pow_nonneg : 0 ≤ reg ^ d := (pow_pos hreg_pos d).le - have hdiv : designDet A reg x n ω / reg ^ d ≤ (T / (d : ℝ)) ^ d / reg ^ d := by - exact div_le_div_of_nonneg_right hdet_le hreg_pow_nonneg - refine hdiv.trans_eq ?_ - rw [← div_pow] - congr 1 - field_simp [hreg_pos.ne', by exact_mod_cast hd] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a determinant upper bound for `V_n` implies the corresponding determinant-ratio -bound. -/ -lemma designDetRatio_ae_le_trace_budget_of_designDet_ae_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hdet_le : ∀ᵐ ω ∂P, designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - filter_upwards [hdet_le] with ω hdetω - exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) T hreg_pos hd hdetω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Converts an almost-sure trace bound plus a future determinant/trace comparison for `det(V_n)` -into the determinant-ratio bound used by the elliptical-potential chain. -/ -lemma designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ T → designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - refine designDetRatio_ae_le_trace_budget_of_designDet_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) T hreg_pos hd ?_ - filter_upwards [h_trace_le] with ω h_traceω - exact hdet_of_trace ω h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms reduce the determinant-ratio goal to the determinant upper bound -`det(V_n) ≤ ((reg * d + n * L2) / d) ^ d`. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le - (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - exact designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd - (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2) - hdet_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix-level determinant/trace comparison needed for the finite-dimensional -elliptical-potential bound. - -For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality -`det(M) ≤ (trace(M) / d) ^ d`. -/ -def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := - ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar AM-GM in the form used for PSD matrix eigenvalues: -the product of nonnegative entries is bounded by the arithmetic mean to the `card` power. -/ -lemma prod_le_average_pow_of_nonneg {ι : Type*} [Fintype ι] [Nonempty ι] - (z : ι → ℝ) (hz : ∀ i, 0 ≤ z i) : - (∏ i, z i) ≤ ((∑ i, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by - classical - have hN_pos : 0 < (Fintype.card ι : ℝ) := by - exact_mod_cast Fintype.card_pos_iff.mpr inferInstance - have hweights_pos : 0 < ∑ i : ι, (1 : ℝ) := by - simpa using hN_pos - have h_amgm := Real.geom_mean_le_arith_mean (s := Finset.univ) - (w := fun _ : ι ↦ (1 : ℝ)) (z := z) - (by intro i hi; norm_num) hweights_pos (by intro i hi; exact hz i) - have h_amgm' : - (∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹) ≤ - (∑ i : ι, z i) / (Fintype.card ι : ℝ) := by - simpa using h_amgm - have hprod_nonneg : 0 ≤ ∏ i : ι, z i := by - exact Finset.prod_nonneg fun i _ ↦ hz i - have hraise := Real.rpow_le_rpow (Real.rpow_nonneg hprod_nonneg _) h_amgm' hN_pos.le - have hleft : - ((∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹)) ^ (Fintype.card ι : ℝ) = - ∏ i : ι, z i := by - rw [← Real.rpow_mul hprod_nonneg] - rw [inv_mul_cancel₀ hN_pos.ne'] - simp - have hright : - ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ (Fintype.card ι : ℝ) = - ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by - rw [Real.rpow_natCast] - simpa [hleft, hright] using hraise - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- PSD matrix determinant/trace comparison from AM-GM over eigenvalues: -`det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma matrixDetLeTraceAveragePow : MatrixDetLeTraceAveragePow d := by - intro M hM - by_cases hd : d = 0 - · subst d - simp - · haveI : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp (Nat.pos_of_ne_zero hd) - rw [hM.1.det_eq_prod_eigenvalues, hM.1.trace_eq_sum_eigenvalues] - simpa using prod_le_average_pow_of_nonneg - (z := fun i : Fin d ↦ hM.1.eigenvalues i) - (fun i ↦ hM.eigenvalues_nonneg i) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the -design matrix is positive semidefinite. -/ -lemma designDet_le_trace_average_pow_of_matrix_det_trace_bound - (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) : - designDet A reg x n ω ≤ (designTrace A reg x n ω / (d : ℝ)) ^ d := by - simpa [designDet, designTrace] using - hdet_trace (designMatrix A reg x n ω) - (designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Combining `det(M) ≤ (trace(M)/d)^d` with a trace budget gives the determinant upper bound -`det(V_n) ≤ (T/d)^d`. -/ -lemma designDet_le_trace_budget_of_matrix_det_trace_bound - (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) - (hd : d ≠ 0) (T : ℝ) (h_trace_le : designTrace A reg x n ω ≤ T) : - designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d := by - have hd_pos : 0 < (d : ℝ) := by - exact_mod_cast Nat.pos_of_ne_zero hd - have hbase_nonneg : 0 ≤ designTrace A reg x n ω / (d : ℝ) := - div_nonneg (designTrace_nonneg (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg) hd_pos.le - have hbase_le : designTrace A reg x n ω / (d : ℝ) ≤ T / (d : ℝ) := - (div_le_div_iff_of_pos_right hd_pos).mpr h_trace_le - exact (designDet_le_trace_average_pow_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet_trace hreg_nonneg).trans - (pow_le_pow_left₀ hbase_nonneg hbase_le d) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms and the matrix-level determinant/trace comparison give the -determinant-ratio bound used by the elliptical-potential chain. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound - (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - refine designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 ?_ - intro ω h_traceω - exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) - hdet_trace hreg_pos.le hd h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The log-determinant expression that appears in the elliptical-potential lemma. -/ -noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - 2 * Real.log (designDetRatio A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A positive determinant ratio bounded by `D` gives the corresponding log-determinant potential -bound. -/ -lemma ellipticalPotential_le_two_mul_log_of_designDetRatio_le {D : ℝ} - (h_ratio_pos : 0 < designDetRatio A reg x n ω) - (h_ratio_le : designDetRatio A reg x n ω ≤ D) : - ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by - rw [ellipticalPotential] - exact mul_le_mul_of_nonneg_left (Real.log_le_log h_ratio_pos h_ratio_le) (by norm_num) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a positive determinant ratio bounded by `D` gives the corresponding -log-determinant potential bound. -/ -lemma ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le {D : ℝ} - (h_ratio_pos : ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by - filter_upwards [h_ratio_pos, h_ratio_le] with ω h_ratio_posω h_ratio_leω - exact ellipticalPotential_le_two_mul_log_of_designDetRatio_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_ratio_posω h_ratio_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. - -The future determinant-update proof should naturally establish the capped quadratic-width term is -bounded by this quantity. A separate log/telescoping bridge then connects this one-step quantity to -`ellipticalPotentialIncrement`. -/ -noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - 2 * Real.log (designDetStepRatio A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing, the one-step log-determinant potential is -`2 * log (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ -lemma ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - ellipticalPotentialStep A reg x n ω = - 2 * Real.log (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - simp [ellipticalPotentialStep, - designDetStepRatio_eq_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar log inequality used in the elliptical-potential proof: for `0 ≤ q ≤ 1`, -`min 1 q ≤ 2 * log (1 + q)`. -/ -lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} - (hq_nonneg : 0 ≤ q) (hq_le_one : q ≤ 1) : - min 1 q ≤ 2 * Real.log (1 + q) := by - have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := - Real.le_log_one_add_of_nonneg hq_nonneg - have hq_add_two_pos : 0 < q + 2 := by linarith - have hq_le_two : q ≤ 2 := by linarith - have hq_le_log_lower : q ≤ 2 * (2 * q / (q + 2)) := by - rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] - rw [le_div_iff₀ hq_add_two_pos] - nlinarith - rw [min_eq_right hq_le_one] - exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar log inequality used in the textbook elliptical-potential proof: for `0 ≤ q`, -`min 1 q ≤ 2 * log (1 + q)`. -/ -lemma min_one_le_two_mul_log_one_add_of_nonneg {q : ℝ} - (hq_nonneg : 0 ≤ q) : - min 1 q ≤ 2 * Real.log (1 + q) := by - by_cases hq_le_one : q ≤ 1 - · exact min_one_le_two_mul_log_one_add_of_nonneg_le_one hq_nonneg hq_le_one - · have hq_one : 1 ≤ q := by linarith - have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := - Real.le_log_one_add_of_nonneg hq_nonneg - have hq_add_two_pos : 0 < q + 2 := by linarith - have hone_le_log_lower : 1 ≤ 2 * (2 * q / (q + 2)) := by - rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] - rw [le_div_iff₀ hq_add_two_pos] - nlinarith - rw [min_eq_left hq_one] - exact hone_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the -single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ -lemma cappedWidthTerm_le_ellipticalPotentialStep - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) - (h_le_one : n ≠ 0 → widthQuadraticForm A reg x (A n ω) n ω ≤ 1) : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialStep A reg x n ω := by - by_cases hn : n = 0 - · rw [if_pos hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) - · rw [if_neg hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing and nonnegativity of the selected quadratic form, the single -capped quadratic-width term is bounded by the one-step log-determinant potential. This is the -textbook form; no separate `q ≤ 1` assumption is needed because the term is already capped. -/ -lemma cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialStep A reg x n ω := by - by_cases hn : n = 0 - · rw [if_pos hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) - · rw [if_neg hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact min_one_le_two_mul_log_one_add_of_nonneg h_nonneg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply -the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ -lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω := by - filter_upwards [hdet, h_nonneg, h_le_one] with ω hdetω h_nonnegω h_le_oneω - intro t ht - exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) - (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the -per-step one-step-potential bound for the capped quadratic-width term. -/ -lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω := by - filter_upwards [hdet, h_nonneg] with ω hdetω h_nonnegω - intro t ht - exact cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg (A := A) (reg := reg) - (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the log-determinant potential is zero when the initial design determinant is -nonzero. -/ -lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - ellipticalPotential A reg x 0 ω = 0 := by - simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the log-determinant elliptical-potential inequality. At horizon zero there are no -positive-time capped quadratic width forms, and the log-determinant potential is zero when the -initial design determinant is nonzero. -/ -lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) - (hdet : designDet A reg x 0 ω ≠ 0) : - cappedQuadraticWidthSum A reg x 0 ω ≤ ellipticalPotential A reg x 0 ω := by - rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), - ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step increment of the log-determinant elliptical potential. -/ -noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The one-step determinant-ratio potential equals the increment of the cumulative -log-determinant potential, provided the relevant design determinants are nonzero. -/ -lemma ellipticalPotentialStep_eq_increment - (hdet0 : designDet A reg x 0 ω ≠ 0) - (hdetn : designDet A reg x n ω ≠ 0) - (hdet_succ : designDet A reg x (n + 1) ω ≠ 0) : - ellipticalPotentialStep A reg x n ω = ellipticalPotentialIncrement A reg x n ω := by - simp [ellipticalPotentialStep, designDetStepRatio, ellipticalPotentialIncrement, - ellipticalPotential, designDetRatio, Real.log_div hdet_succ hdetn, - Real.log_div hdet_succ hdet0, Real.log_div hdetn hdet0] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the one-step determinant-ratio potential equals the increment of the cumulative -log-determinant potential throughout the finite horizon, provided all determinants up to that -horizon are nonzero almost surely. -/ -lemma ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω = ellipticalPotentialIncrement A reg x t ω := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact ellipticalPotentialStep_eq_increment (A := A) (reg := reg) (x := x) (n := t) - (ω := ω) (hdetω 0 (by simp)) - (hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) - (hdetω (t + 1) (mem_range.mpr (Nat.succ_lt_succ (mem_range.mp ht)))) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If the next capped quadratic width term is bounded by the next log-determinant potential -increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ -lemma cappedQuadraticWidthSum_succ_le_ellipticalPotential - (h_prev : cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_step : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialIncrement A reg x n ω) : - cappedQuadraticWidthSum A reg x (n + 1) ω ≤ ellipticalPotential A reg x (n + 1) ω := by - rw [cappedQuadraticWidthSum_succ (A := A) (reg := reg) (x := x) (n := n) (ω := ω)] - calc - cappedQuadraticWidthSum A reg x n ω + - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) - ≤ ellipticalPotential A reg x n ω + ellipticalPotentialIncrement A reg x n ω := by - exact add_le_add h_prev h_step - _ = ellipticalPotential A reg x (n + 1) ω := by - simp [ellipticalPotentialIncrement] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A per-step bound by log-determinant potential increments implies the cumulative -elliptical-potential inequality. This is the induction shell for the future determinant-update -proof. -/ -lemma cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le - (hdet : designDet A reg x 0 ω ≠ 0) : - (∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) → - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - induction n with - | zero => - intro _ - exact cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) - (x := x) (ω := ω) hdet - | succ n ih => - intro h_step - refine cappedQuadraticWidthSum_succ_le_ellipticalPotential (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) ?_ ?_ - · exact ih fun t ht ↦ h_step t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - · exact h_step n (by simp) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a per-step bound by log-determinant potential increments implies the cumulative -elliptical-potential inequality. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - filter_upwards [hdet, h_step] with ω hdetω h_stepω - exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetω h_stepω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the -cumulative capped-sum/log-det inequality, provided the one-step determinant-ratio potential is -bounded by the corresponding cumulative-potential increment. - -This separates the future elliptical-potential proof into two local obligations: - -* a matrix-determinant update bounding the selected arm's capped quadratic form by - `ellipticalPotentialStep`; -* a log/telescoping bridge from `ellipticalPotentialStep` to `ellipticalPotentialIncrement`. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet ?_ - filter_upwards [h_step, h_step_le_increment] with ω h_stepω h_step_le_incrementω - intro t ht - exact (h_stepω t ht).trans (h_step_le_incrementω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the -cumulative capped-sum/log-det inequality when all design determinants up to the horizon are nonzero -almost surely. - -Compared with `cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le`, this -version discharges the log/telescoping bridge automatically from determinant nonvanishing. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - have hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - exact hdetω 0 (by simp) - refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_step ?_ - filter_upwards [ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet] with ω h_eq - intro t ht - rw [h_eq t ht] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the -capped-sum/log-determinant elliptical-potential bound. - -This is the capped form used in the textbook proof of LinUCB: the quadratic forms do not need to -be bounded by `1`, because the accumulated quantity is `min 1 q_t`. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet - (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet_range_n h_nonneg) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization discharges determinant nonvanishing and nonnegativity, yielding the -capped-sum/log-determinant elliptical-potential bound directly. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos - (hreg_pos : 0 < reg) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) - (n := n + 1) (P := P) hreg_pos) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The process-level capped quadratic-width input expected from an elliptical-potential argument. - -It packages the three facts needed to turn a capped process-level quadratic-width estimate into the -`widthSqSum` estimate used by the regret chain: - -* each positive-time process-level quadratic width form is nonnegative; -* each positive-time process-level quadratic width form is at most `1`; -* their capped process-level accumulated sum is bounded by `W`. -/ -def CappedQuadraticWidthBound (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := - (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ∧ - (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ∧ - cappedQuadraticWidthSum A reg x n ω ≤ W - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Build the packaged process-level capped quadratic-width input from its component facts. -/ -lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_sum_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : - CappedQuadraticWidthBound A reg x n ω W := by - exact ⟨h_nonneg, h_le_one, h_sum_le⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the packaged process-level capped quadratic-width input. At horizon zero, the -nonnegativity and `≤ 1` side conditions are vacuous, and the capped sum is zero. -/ -lemma cappedQuadraticWidthBound_zero {W : ℝ} (hW : 0 ≤ W) : - CappedQuadraticWidthBound A reg x 0 ω W := by - refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ - · intro t ht _ - simp at ht - · intro t ht _ - simp at ht - · simpa [cappedQuadraticWidthSum_zero] using hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the packaged process-level capped quadratic-width input when the constant bound -is supplied through the log-determinant potential. -/ -lemma cappedQuadraticWidthBound_zero_of_ellipticalPotential_le_bound {W : ℝ} - (hdet : designDet A reg x 0 ω ≠ 0) (h_potential_le : ellipticalPotential A reg x 0 ω ≤ W) : - CappedQuadraticWidthBound A reg x 0 ω W := by - refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ - · intro t ht _ - simp at ht - · intro t ht _ - simp at ht - · exact (cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) - (x := x) (ω := ω) hdet).trans h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ -lemma cappedQuadraticWidthBound_mono {W W' : ℝ} - (h_bound : CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : - CappedQuadraticWidthBound A reg x n ω W' := by - exact ⟨h_bound.1, h_bound.2.1, h_bound.2.2.trans hW⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, build the packaged process-level capped quadratic-width input from its component -facts. -/ -lemma cappedQuadraticWidthBound_ae_of_nonneg_le_one_and_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_sum_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_sum_le] with - ω h_nonnegω h_le_oneω h_sum_leω - exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_sum_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged process-level capped quadratic-width input is monotone in the -numeric bound. -/ -lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W' := by - filter_upwards [h_bound] with ω h_boundω - exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) h_boundω hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped-sum bound by the log-determinant potential, together with a constant bound on that -potential, gives the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_of_ellipticalPotential_le_bound {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_potential_le : ellipticalPotential A reg x n ω ≤ W) : - CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonneg h_le_one (h_elliptical.trans h_potential_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped-sum bound by the log-determinant potential and an almost-sure constant -bound on that potential give the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_elliptical, h_potential_le] with - ω h_nonnegω h_le_oneω h_ellipticalω h_potential_leω - exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by log-determinant potential increments and a final constant -bound on the potential give the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet h_step) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, one-step determinant-ratio potential bounds, their bridge to cumulative -potential increments, and a final constant bound on the potential give the packaged process-level -capped quadratic-width input. - -This is the packaged form of the determinant-update interface: once the true matrix determinant -lemma proves the `h_step` assumption and the log/telescoping algebra proves -`h_step_le_increment`, the existing regret chain can consume the resulting bound. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, one-step determinant-ratio potential bounds, determinant nonvanishing up to the -horizon, and a final constant bound on the potential give the packaged process-level capped -quadratic-width input. - -This is the determinant-nonvanishing version of the one-step interface: the remaining hard -elliptical-potential work is to prove the one-step matrix inequality and the final -log-determinant bound. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero - {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet h_step) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the determinant-update step, determinant nonvanishing up to the horizon, and a -final constant bound on the log-determinant potential give the packaged capped quadratic-width -input used by the regret chain. - -The assumptions now match the concrete obligations left for a full elliptical-potential proof: - -* prove all relevant design determinants are nonzero; -* prove selected quadratic forms are nonnegative and at most `1` at positive times; -* prove the final log-determinant potential is at most `W`. -/ -lemma cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound {W : ℝ} - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - have h_nonneg_positive : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - filter_upwards [h_nonneg] with ω h_nonnegω - intro t ht _ - exact h_nonnegω t ht - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) - h_nonneg_positive h_le_one hdet - (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero (A := A) (reg := reg) - (x := x) (n := n) (P := P) hdet_range_n h_nonneg h_le_one) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant, the determinant-update step, and a final constant -bound on the log-determinant potential give the packaged capped quadratic-width input used by the -regret chain. - -This removes the need to assume determinant nonvanishing at every time: it is derived inductively -from `det(V_0) ≠ 0` and nonnegative selected quadratic forms. -/ -lemma cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound {W : ℝ} - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) - h_nonneg h_le_one h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter, the determinant-update step, and a final -constant bound on the log-determinant potential give the packaged capped quadratic-width input used -by the regret chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential_le_bound {W : ℝ} - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - refine cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) ?_ h_nonneg h_le_one - h_potential_le - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization discharges the determinant-nonvanishing and quadratic-form -nonnegativity obligations in the log-determinant elliptical-potential chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound {W : ℝ} - (hreg_pos : 0 < reg) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) - (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) - (n := n + 1) (P := P) hreg_pos) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - h_le_one h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a -determinant-ratio upper bound, and the determinant-update step give the packaged capped -quadratic-width input used by the regret chain. - -This version accepts the determinant-ratio bound directly and converts it into the -`ellipticalPotential ≤ 2 * log D` bound internally. -/ -lemma cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound {D : ℝ} - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by - exact cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := 2 * Real.log D) - hdet0 h_nonneg h_le_one - (ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) - h_ratio_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter, nonnegative selected quadratic forms, a -determinant-ratio upper bound, and the determinant-update step give the packaged capped -quadratic-width input used by the regret chain. - -This is the most direct interface for the final determinant-bound part of the finite-action -elliptical-potential argument: after proving `designDetRatio ≤ D`, the theorem supplies the -`CappedQuadraticWidthBound` with bound `2 * log D`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound {D : ℝ} - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by - refine cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one h_ratio_le - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A simple explicit determinant-ratio bound for the capped quadratic-width input. - -If `reg ≠ 0` and every selected quadratic form is almost surely in `[0, 1]`, then the determinant -ratio is at most `2 ^ n`, so the existing determinant-update/elliptical-potential chain gives the -packaged capped-width bound with budget `2 * log (2 ^ n)`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω (2 * Real.log ((2 : ℝ) ^ n)) := by - refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (2 : ℝ) ^ n) - hreg h_nonneg ?_ ?_ - · filter_upwards [h_le_one] with ω h_le_oneω - exact fun t ht _ ↦ h_le_oneω t ht - · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Trace-budget interface for the determinant part of the finite-action elliptical-potential -argument. - -The future spectral/AM-GM determinant theorem should prove the hypothesis -`designDetRatio ≤ (T / (reg * d)) ^ d`, where `T` is an upper bound on `trace(V_n)`. This theorem -then feeds that determinant-ratio bound into the already-proved determinant-update and -elliptical-potential chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (T : ℝ) - (h_ratio_le : ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * Real.log ((T / (reg * (d : ℝ))) ^ d)) := by - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface for the determinant part of the finite-action -elliptical-potential argument. - -If selected feature vectors have squared norm at most `L2`, then `trace(V_n) ≤ reg * d + n * L2`. -Given a future deterministic trace/determinant comparison that turns this trace budget into the -determinant-ratio bound, this theorem supplies the packaged capped-width input with the explicit -budget `2 * log (((reg * d + n * L2) / (reg * d)) ^ d)`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d)) := by - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg h_nonneg h_le_one - (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The explicit feature-norm determinant budget can be rewritten in the common -`d * log(1 + n L² / (reg d))` form. -/ -lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add - (L2 : ℝ) (hden : reg * (d : ℝ) ≠ 0) : - 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) = - 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by - have hbase : - (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) = - 1 + (n : ℝ) * L2 / (reg * (d : ℝ)) := by - exact same_add_div hden - rw [Real.log_pow, hbase] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Textbook capped elliptical-potential budget from bounded selected feature norms and the -matrix-level determinant/trace comparison. - -Unlike `cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound`, this theorem bounds the capped -quadratic-width sum directly and does not assume the individual quadratic forms are at most `1`. -/ -lemma cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ - 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by - have hden : reg * (d : ℝ) ≠ 0 := by - exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) - rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] - have h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := - widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le - have h_potential_le : ∀ᵐ ω ∂P, - ellipticalPotential A reg x n ω ≤ - 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) := by - exact ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' h_nonneg) - (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 - hdet_trace) - filter_upwards [cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos, h_potential_le] with - ω h_capped_le h_potentialω - exact h_capped_le.trans h_potentialω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface with the log term rewritten in the standard -`2 * d * log(1 + n L² / (reg d))` shape. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (hreg : reg ≠ 0) (hd : d ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - have hden : reg * (d : ℝ) ≠ 0 := by - exact mul_ne_zero hreg (by exact_mod_cast hd) - rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 - h_ratio_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface with the determinant/trace comparison stated as a determinant -upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) - (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' - (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) - (x := x) hreg_pos h_inv_antitone)) - hreg_pos hL2 hL2_le_reg) - L2 hL2 ?_ - intro ω h_traceω - exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd - (hdet_of_trace ω h_traceω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface where the remaining matrix-analysis input is the reusable -positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - refine - cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg ?_ - intro ω h_traceω - exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd - (reg * (d : ℝ) + (n : ℝ) * L2) h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed -by the regret chain. -/ -lemma widthSqSum_le_of_capped_quadratic_width_bound {W : ℝ} - (h_bound : CappedQuadraticWidthBound A reg x n ω W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_bound.1 h_bound.2.1 h_bound.2.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged process-level capped quadratic-width input implies the `widthSqSum` -bound consumed by the regret chain. -/ -lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_bound] with ω h_boundω - exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) (W := W) h_boundω - -/-- The process-level LinUCB optimistic index. -/ -noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) - (n : ℕ) (ω : Ω) : ℝ := - estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω - -/-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is -zero. -/ -lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by - simp [index, estimatedReward_zero] - -/-- At time zero, the LinUCB index is the confidence schedule times the initial quadratic-form -width. -/ -lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - index A R reg β x a 0 ω = - √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [index_zero, width_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every least-squares reward estimate is zero. -/ -lemma estimatedReward_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - estimatedReward A R reg x a n ω = 0 := by - subst d - simp [estimatedReward, dotProduct] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB quadratic width form is zero. -/ -lemma widthQuadraticForm_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - widthQuadraticForm A reg x a n ω = 0 := by - subst d - simp [widthQuadraticForm, dotProduct] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB width is zero. -/ -lemma width_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - width A reg x a n ω = 0 := by - simp [width, widthQuadraticForm_eq_zero_of_dim_eq_zero (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hd a] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB index is zero. -/ -lemma index_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - index A R reg β x a n ω = 0 := by - simp [index, estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) hd a, - width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hd a] - -/-- The pointwise LinUCB confidence event used by the finite-action regret proof. - -For every positive process time, the best arm's true mean lies below its optimistic index, and the -selected arm's pessimistic index lies below its true mean. On this event, the max-index property of -LinUCB turns optimism into an instantaneous regret bound. -/ -def LinUCBConfidenceEvent [Nonempty (Fin K)] - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] - -omit [IsMarkovKernel ν] in -/-- Uniform bound on arm gaps, used as the finite-action analogue of the textbook bounded -instantaneous-regret assumption. -/ -def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := - ∀ a, gap ν a ≤ G - -omit [IsMarkovKernel ν] in -/-- Uniform bound on arm means. For finite-action linear bandits this is a convenient way to state -the usual bounded expected-reward assumption, for example `(ν a)[id] ∈ [-1, 1]`. -/ -def MeanRewardBound (ν : Kernel (Fin K) ℝ) (lo hi : ℝ) : Prop := - ∀ a, lo ≤ (ν a)[id] ∧ (ν a)[id] ≤ hi - -omit [IsMarkovKernel ν] in -/-- If every arm mean lies in `[lo, hi]`, then every arm gap is at most `hi - lo`. -/ -lemma gap_le_of_meanRewardBound [Nonempty (Fin K)] {lo hi : ℝ} - (hμ : MeanRewardBound ν lo hi) (a : Fin K) : - gap ν a ≤ hi - lo := by - rw [gap_eq_bestArm_sub] - have hbest_le : (ν (bestArm ν))[id] ≤ hi := (hμ (bestArm ν)).2 - have ha_ge : lo ≤ (ν a)[id] := (hμ a).1 - linarith - -omit [IsMarkovKernel ν] in -/-- Arm means in `[-1, 1]` imply the gap cap `gap ≤ 2` used by the capped regret argument. -/ -lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] - (hμ : MeanRewardBound ν (-1) 1) : - GapBound (K := K) ν 2 := by - intro a - have hgap := gap_le_of_meanRewardBound (ν := ν) (lo := -1) (hi := 1) hμ a - norm_num at hgap - exact hgap - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial-gap term used by this formalization is deterministically at most `2` when all arm -means lie in `[-1, 1]`. At horizon zero the initial term is exactly zero. -/ -lemma initialGapTerm_le_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] - (hμ : MeanRewardBound ν (-1) 1) : - (if n = 0 then 0 else gap ν (A 0 ω)) ≤ if n = 0 then 0 else 2 := by - by_cases hn : n = 0 - · simp [hn] - · simpa [hn] using (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) hμ (A 0 ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ -lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → gap ν (A t ω) ≤ G := - Filter.Eventually.of_forall fun ω t _ht ↦ hG (A t ω) - -omit [IsMarkovKernel ν] in -/-- First projection from the packaged LinUCB confidence event: optimism for the best arm. -/ -lemma LinUCBConfidenceEvent.best [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - intro t ht - exact (h_conf t ht).1 - -omit [IsMarkovKernel ν] in -/-- Second projection from the packaged LinUCB confidence event: validity of the selected arm's -lower confidence inequality. -/ -lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ t, t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - intro t ht - exact (h_conf t ht).2 - -omit [IsMarkovKernel ν] in -/-- In zero feature dimension, the confidence event forces every positive-time selected gap to be -nonpositive. The best-arm index is zero, and the selected-arm pessimistic index is also zero. -/ -lemma gap_nonpos_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) - (t : ℕ) (ht : t ≠ 0) : - gap ν (A t ω) ≤ 0 := by - have hbest := LinUCBConfidenceEvent.best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (ω := ω) h_conf t ht - have harm := LinUCBConfidenceEvent.arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (ω := ω) h_conf t ht - rw [gap_eq_bestArm_sub] - have hbest0 : (ν (bestArm ν))[id] ≤ 0 := by - simpa [index_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) (β := β) - (x := x) (n := t) (ω := ω) hd (bestArm ν)] using hbest - have harm0 : 0 ≤ (ν (A t ω))[id] := by - simpa [estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) - (x := x) (n := t) (ω := ω) hd (A t ω), - width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := t) - (ω := ω) hd (A t ω)] using harm - linarith - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ -lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - filter_upwards [h_conf] with ω h_confω - exact h_confω.best - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure projection of the packaged confidence event to the selected arm's lower confidence -inequality. -/ -lemma linUCBConfidenceEvent_ae_arm [Nonempty (Fin K)] - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - filter_upwards [h_conf] with ω h_confω - exact h_confω.arm - -lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) - (ω : Ω) (hn : n ≠ 0) : - designMatrix A reg x n ω = - designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - cases n with - | zero => exact absurd rfl hn - | succ n => - simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] - rw [Nat.range_succ_eq_Iic] - exact congrArg (fun S ↦ reg • 1 + S) <| - (Finset.sum_coe_sort (Iic n) - (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm - -lemma responseVector_eq_responseVector' (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - cases n with - | zero => exact absurd rfl hn - | succ n => - simp only [responseVector, responseVector', IsAlgEnvSeq.hist] - rw [Nat.range_succ_eq_Iic] - exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm - -lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, - responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] - -lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - estimatedReward A R reg x a n ω = - estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] - -lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthQuadraticForm A reg x a n ω = - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [widthQuadraticForm, widthQuadraticForm', - designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] - -/-- At positive process times, nonnegativity of the process-level width quadratic form is -equivalent to nonnegativity of the matching history-level width quadratic form. -/ -lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - 0 ≤ widthQuadraticForm A reg x a n ω ↔ - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] - -/-- At positive process times, the process-level quadratic width form is at most `1` iff the -matching history-level quadratic width form is at most `1`. -/ -lemma widthQuadraticForm_le_one_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthQuadraticForm A reg x a n ω ≤ 1 ↔ - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a ≤ 1 := by - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] - -/-- The all-positive-times process-level nonnegativity assumption is equivalent to the matching -history-level nonnegativity assumption. -/ -lemma widthQuadraticForm_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ - ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by - constructor - · intro h t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).1 (h t ht ht0) - · intro h t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h t ht ht0) - -/-- The all-positive-times process-level `≤ 1` assumption is equivalent to the matching -history-level `≤ 1` assumption. -/ -lemma widthQuadraticForm_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ - ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by - constructor - · intro h t ht ht0 - exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).1 (h t ht ht0) - · intro h t ht ht0 - exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h t ht ht0) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, process-level all-positive-times nonnegativity is equivalent to the matching -history-level nonnegativity assumption. -/ -lemma widthQuadraticForm_ae_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) : - (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).1 hω - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).2 hω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the process-level all-positive-times `≤ 1` assumption is equivalent to the -matching history-level `≤ 1` assumption. -/ -lemma widthQuadraticForm_ae_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) : - (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).1 hω - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).2 hω - -lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n - ω hn] - -/-- At positive process times, squaring the process-level width recovers the matching history-level -quadratic form when that history-level quadratic form is nonnegative. -/ -lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_nonneg : - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a) : - width A reg x a n ω ^ 2 = - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - rw [width_eq_width' (A := A) (R := R) reg x a n ω hn] - exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a - h_nonneg - -/-- At positive process times, advancing `widthSqSum` adds the matching history-level quadratic -form when that history-level quadratic form is nonnegative. -/ -lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_nonneg : - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] - rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] - -/-- At positive process times, advancing `quadraticWidthSum` adds the matching history-level -quadratic form. -/ -lemma quadraticWidthSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - rw [quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hn] - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn] - -/-- The history-level quadratic-form accumulator aligned with process times. - -The term at process time `t = 0` is set to zero, matching the convention used by `widthSqSum` and -`quadraticWidthSum`. At positive process time `t`, the history available to LinUCB is -`IsAlgEnvSeq.hist A R (t - 1) ω`. -/ -noncomputable def historyQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) - -/-- No positive-time history-level quadratic width forms are accumulated at horizon zero. -/ -lemma historyQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - historyQuadraticWidthSum A R reg x 0 ω = 0 := by - simp [historyQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time history-level quadratic width form. -/ -lemma historyQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - historyQuadraticWidthSum A R reg x (n + 1) ω = - historyQuadraticWidthSum A R reg x n ω + - if n = 0 then 0 else - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - simp [historyQuadraticWidthSum, sum_range_succ] - -/-- At positive process times, advancing the history-level quadratic accumulator adds the selected -arm's history-level quadratic width form. -/ -lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - historyQuadraticWidthSum A R reg x (n + 1) ω = - historyQuadraticWidthSum A R reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - simp [historyQuadraticWidthSum_succ, hn] - -/-- The capped history-level quadratic-form accumulator aligned with process times. - -This is the accumulator shape that commonly appears in elliptical-potential statements: -each positive-time quadratic width form is capped at `1`. -/ -noncomputable def historyCappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else - min 1 (widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - -/-- No positive-time capped history-level quadratic width forms are accumulated at horizon zero. -/ -lemma historyCappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - historyCappedQuadraticWidthSum A R reg x 0 ω = 0 := by - simp [historyCappedQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time capped history-level quadratic width form. -/ -lemma historyCappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - historyCappedQuadraticWidthSum A R reg x (n + 1) ω = - historyCappedQuadraticWidthSum A R reg x n ω + - if n = 0 then 0 else - min 1 - (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by - simp [historyCappedQuadraticWidthSum, sum_range_succ] - -/-- At positive process times, advancing the capped history-level quadratic accumulator adds the -selected arm's capped history-level quadratic width form. -/ -lemma historyCappedQuadraticWidthSum_succ_of_ne_zero - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - historyCappedQuadraticWidthSum A R reg x (n + 1) ω = - historyCappedQuadraticWidthSum A R reg x n ω + - min 1 - (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by - simp [historyCappedQuadraticWidthSum_succ, hn] - -/-- The process-level capped quadratic-width accumulator equals the history-level capped -accumulator aligned with the same process times. -/ -lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - cappedQuadraticWidthSum A reg x n ω = - historyCappedQuadraticWidthSum A R reg x n ω := by - rw [cappedQuadraticWidthSum, historyCappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact congrArg (fun q : ℝ ↦ min 1 q) - (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) - -/-- A process-level capped quadratic-width sum bound is equivalent to the matching history-level -capped quadratic-width sum bound. -/ -lemma cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : - cappedQuadraticWidthSum A reg x n ω ≤ W ↔ - historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by - rw [cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) - reg x n ω] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a process-level capped quadratic-width sum bound is equivalent to the matching -history-level capped quadratic-width sum bound. -/ -lemma cappedQuadraticWidthSum_ae_le_iff_historyCappedQuadraticWidthSum_ae_le - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (W : ℝ) : - (∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) ↔ - ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (A := A) (R := R) reg x n ω W).1 hω - · intro h - filter_upwards [h] with ω hω - exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (A := A) (R := R) reg x n ω W).2 hω - -/-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and -capped history-level accumulators agree. -/ -lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) : - historyQuadraticWidthSum A R reg x n ω = - historyCappedQuadraticWidthSum A R reg x n ω := by - rw [historyQuadraticWidthSum, historyCappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact (min_eq_right (h_le_one t ht ht0)).symm - -/-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with -the same process times. -/ -lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - quadraticWidthSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by - rw [quadraticWidthSum, historyQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0 - -/-- The squared-width accumulator equals the history-level quadratic-form accumulator whenever the -positive-time history-level quadratic forms are nonnegative. -/ -lemma widthSqSum_eq_historyQuadraticWidthSum - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) : - widthSqSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by - have h_process_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h_nonneg t ht ht0) - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_process_nonneg] - exact quadraticWidthSum_eq_historyQuadraticWidthSum (A := A) (R := R) reg x n ω - -/-- A bound on the history-level quadratic-form accumulator implies the corresponding bound on -`widthSqSum`, provided the positive-time history-level quadratic forms are nonnegative. -/ -lemma widthSqSum_le_of_history_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_hist_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_historyQuadraticWidthSum (A := A) (R := R) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - exact h_hist_le - -omit [IsProbabilityMeasure P] in -/-- Almost surely, a history-level quadratic-form bound gives the `widthSqSum` bound consumed by -the regret chain. -/ -lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_hist_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_nonneg, h_hist_le] with ω h_nonnegω h_hist_leω - exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The pointwise input expected from a future elliptical-potential argument. - -It packages the two facts needed to turn a history-level quadratic-width estimate into the -`widthSqSum` estimate used by the regret chain: - -* each positive-time quadratic width form is nonnegative; -* their history-level accumulated sum is bounded by `W`. -/ -def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := - (∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ - historyQuadraticWidthSum A R reg x n ω ≤ W - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Build the packaged history-level quadratic-width input from its two component facts. -/ -lemma historyQuadraticWidthBound_of_nonneg_and_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_sum_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : - HistoryQuadraticWidthBound A R reg x n ω W := by - exact ⟨h_nonneg, h_sum_le⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged history-level quadratic-width input is monotone in the numeric bound. -/ -lemma historyQuadraticWidthBound_mono {W W' : ℝ} - (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : - HistoryQuadraticWidthBound A R reg x n ω W' := by - exact ⟨h_bound.1, h_bound.2.trans hW⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, build the packaged history-level quadratic-width input from its two component -facts. -/ -lemma historyQuadraticWidthBound_ae_of_nonneg_and_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_sum_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by - filter_upwards [h_nonneg, h_sum_le] with ω h_nonnegω h_sum_leω - exact historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_sum_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged history-level quadratic-width input is monotone in the numeric -bound. -/ -lemma historyQuadraticWidthBound_ae_mono {W W' : ℝ} - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W' := by - filter_upwards [h_bound] with ω h_boundω - exact historyQuadraticWidthBound_mono (A := A) (R := R) (reg := reg) (x := x) - (n := n) (ω := ω) h_boundω hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped quadratic-width sum bound gives the packaged history-level input whenever every -positive-time quadratic width form is nonnegative and at most `1`. -/ -lemma historyQuadraticWidthBound_of_capped_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - HistoryQuadraticWidthBound A R reg x n ω W := by - refine historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg ?_ - rw [historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_le_one] - exact h_capped_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped quadratic-width sum bound gives the packaged history-level input -whenever every positive-time quadratic width form is almost surely nonnegative and at most `1`. -/ -lemma historyQuadraticWidthBound_ae_of_capped_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_capped_le] with - ω h_nonnegω h_le_oneω h_capped_leω - exact historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the -regret chain. -/ -lemma widthSqSum_le_of_history_quadratic_width_bound {W : ℝ} - (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_bound.1 h_bound.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged history-level quadratic-width input implies the `widthSqSum` bound -consumed by the regret chain. -/ -lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_bound] with ω h_boundω - exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) (W := W) h_boundω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped history-level quadratic-width sum bound implies the `widthSqSum` bound consumed by -the regret chain, provided the positive-time quadratic width forms are nonnegative and at most -`1`. -/ -lemma widthSqSum_le_of_capped_history_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) (W := W) - (historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonneg h_le_one h_capped_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped history-level quadratic-width sum bound implies the `widthSqSum` bound -consumed by the regret chain, provided the positive-time quadratic width forms are almost surely -nonnegative and at most `1`. -/ -lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - h_capped_le) - -lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - index A R reg β x a n ω = - index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - have htime : n + 1 = n - 1 + 2 := by grind - simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, - width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] - -/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ -lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (n : ℕ) : - A (n + 1) =ᵐ[P] - fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact h.action_detAlgorithm_ae_eq n - -/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ -lemma arm_ae_all_eq [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : - ∀ᵐ ω ∂P, - ∀ n, A (n + 1) ω = - nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by - simp_rw [ae_all_iff] - exact fun n ↦ arm_ae_eq_linUCBNextArm h n - -/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ -lemma index_le_index_arm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (a : Fin K) (hn : n ≠ 0) : - ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by - filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm - have hn_succ : n - 1 + 1 = n := by grind - simp only [hn_succ] at h_arm - rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, - index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] - rw [h_arm] - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) - (IsAlgEnvSeq.hist A R (n - 1) ω) a - -/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ -lemma forall_index_le_index_arm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (a : Fin K) : - ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by - simp_rw [ae_all_iff] - exact fun n hn ↦ index_le_index_arm h a hn - -end AlgorithmBehavior - -omit [IsMarkovKernel ν] in -/-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the -selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected -arm's LinUCB width. -/ -lemma mean_sub_mean_arm_le_two_mul_width (a : Fin K) - (h_best : (ν a)[id] ≤ index A R reg β x a n ω) - (h_arm : estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_le : index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω) : - (ν a)[id] - (ν (A n ω))[id] ≤ - 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [sub_le_iff_le_add'] - calc - (ν a)[id] ≤ index A R reg β x a n ω := h_best - _ ≤ index A R reg β x (A n ω) n ω := h_le - _ ≤ (ν (A n ω))[id] + - 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [index, two_mul, ← add_assoc] - gcongr - rwa [sub_le_iff_le_add] at h_arm - -omit [IsMarkovKernel ν] in -/-- The gap of the selected arm is bounded by twice its LinUCB bonus whenever the usual confidence -inequalities hold and the selected arm has maximal LinUCB index. -/ -lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] - (h_best : (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_le : index A R reg β x (bestArm ν) n ω ≤ - index A R reg β x (A n ω) n ω) : - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [gap_eq_bestArm_sub] - exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) (a := bestArm ν) h_best h_arm h_le - -/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus whenever the usual -confidence inequalities hold almost surely. -/ -lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (hn : n ≠ 0) - (h_best : ∀ᵐ ω ∂P, (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - filter_upwards [h_best, h_arm, index_le_index_arm h (bestArm ν) hn] with - ω h_bestω h_armω h_leω - exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) h_bestω h_armω h_leω - -/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus at every positive -time whenever the usual confidence inequalities hold almost surely at every positive time. -/ -lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - filter_upwards [h_best, h_arm, forall_index_le_index_arm h (bestArm ν)] with - ω h_bestω h_armω h_leω - intro n hn - exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) - -omit [IsMarkovKernel ν] in -/-- Pointwise capped LinUCB regret bound for one positive time. - -If the instantaneous gap is bounded by `2`, and the confidence/max-index argument gives the usual -`2 * sqrt(β_t) * width_t` bound, then monotonicity up to the terminal `β n` gives the textbook -capped form `2 * sqrt(β n) * sqrt(min 1 q_t)`, where `q_t` is the width quadratic form. -/ -lemma gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm - (t : ℕ) - (h_gap_two : gap ν (A t ω) ≤ 2) - (h_gap_width : gap ν (A t ω) ≤ - 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) - (hβ_le : β (t + 1) ≤ β n) - (hβn_one : 1 ≤ β n) : - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - by_cases hq_le_one : widthQuadraticForm A reg x (A t ω) t ω ≤ 1 - · have hwidth_nonneg : 0 ≤ width A reg x (A t ω) t ω := Real.sqrt_nonneg _ - have hsqrt_le : √(β (t + 1)) ≤ √(β n) := Real.sqrt_le_sqrt hβ_le - have hbonus_le : - 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) ≤ - 2 * (√(β n) * width A reg x (A t ω) t ω) := by - exact mul_le_mul_of_nonneg_left - (mul_le_mul_of_nonneg_right hsqrt_le hwidth_nonneg) (by norm_num) - have hmin : - √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)) = - width A reg x (A t ω) t ω := by - rw [min_eq_right hq_le_one, width] - simpa [hmin] using h_gap_width.trans hbonus_le - · have hq_one : 1 ≤ widthQuadraticForm A reg x (A t ω) t ω := by linarith - have hsqrt_one : 1 ≤ √(β n) := by - simpa using (Real.one_le_sqrt).2 hβn_one - have htwo_le : - 2 ≤ 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [min_eq_left hq_one, Real.sqrt_one] - nlinarith - exact h_gap_two.trans htwo_le - -omit [IsMarkovKernel ν] in -/-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded -by the corresponding sum of pointwise bounds. -/ -lemma regret_le_sum_of_gap_bound (B : ℕ → ℝ) - (hB : ∀ t, t ∈ range n → gap ν (A t ω) ≤ B t) : - regret ν A n ω ≤ ∑ t ∈ range n, B t := by - rw [regret_eq_sum_gap] - exact sum_le_sum hB - -omit [IsMarkovKernel ν] in -/-- A pathwise cumulative-regret bound obtained by summing the positive-time LinUCB width bound. - -The time-zero gap is left unchanged because the current LinUCB max-index theorem applies only at -positive times. -/ -lemma regret_le_sum_width_of_forall_gap_le - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) : - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using h_gap t ht ht0 - -omit [IsMarkovKernel ν] in -/-- A pathwise cumulative-regret bound obtained by summing the positive-time capped LinUCB width -bound. -/ -lemma regret_le_sum_sqrt_capped_width_of_forall_gap_le - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using h_gap t ht ht0 - -omit [IsMarkovKernel ν] in -/-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ -lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ≤ - 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - calc - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) - = 2 * ∑ t ∈ range n, - (if t = 0 then 0 else √(β (t + 1))) * - (if t = 0 then 0 else width A reg x (A t ω) t ω) := by - rw [mul_sum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - gcongr - exact Real.sum_mul_le_sqrt_mul_sqrt (range n) - (fun t ↦ if t = 0 then 0 else √(β (t + 1))) - (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) - -omit [IsMarkovKernel ν] in -/-- Cauchy-Schwarz bound for the positive-time capped LinUCB bonus sum. -/ -lemma sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum - (hβn_nonneg : 0 ≤ β n) - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ≤ - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - calc - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) - = 2 * ∑ t ∈ range n, - (if t = 0 then 0 else √(β n)) * - (if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [mul_sum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) * - √(∑ t ∈ range n, - (if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) ^ 2)) := by - gcongr - exact Real.sum_mul_le_sqrt_mul_sqrt (range n) - (fun t ↦ if t = 0 then 0 else √(β n)) - (fun t ↦ if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) - _ ≤ 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - gcongr - · calc - (∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) - ≤ ∑ _t ∈ range n, β n := by - refine sum_le_sum ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0, hβn_nonneg] - · simp [ht0, Real.sq_sqrt hβn_nonneg] - _ = (n : ℝ) * β n := by - simp [sum_const, nsmul_eq_mul] - · rw [cappedQuadraticWidthSum] - refine le_of_eq ?_ - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · have hmin_nonneg : - 0 ≤ min 1 (widthQuadraticForm A reg x (A t ω) t ω) := by - exact le_min zero_le_one (h_nonneg t ht ht0) - simp [ht0, Real.sq_sqrt hmin_nonneg] - -omit [IsMarkovKernel ν] in -/-- Pathwise cumulative-regret bound using the textbook capped quadratic-width sum. -/ -lemma regret_le_initial_add_sqrt_nat_mul_beta_capped_sum - (hβn_nonneg : 0 ≤ β n) - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - refine (regret_le_sum_sqrt_capped_width_of_forall_gap_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) h_gap).trans ?_ - have hsplit : - (∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) = - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - ∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * - √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [← sum_add_distrib] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - rw [hsplit] - exact add_le_add le_rfl - (sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum - (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω) - hβn_nonneg h_nonneg) - -omit [IsMarkovKernel ν] in -/-- If the capped quadratic-width sum is bounded by `W`, the pathwise capped regret bound can use -`√W` in place of the realized capped-sum square root. -/ -lemma regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω))) - (hW : cappedQuadraticWidthSum A reg x n ω ≤ W) : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √W) := by - refine h_regret.trans ?_ - gcongr - -/-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is -nonnegative. -/ -lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : - (∑ t ∈ range n, if t = 0 then 0 else √(β (t + 1)) ^ 2) = - ∑ t ∈ range n, if t = 0 then 0 else β (t + 1) := by - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0, Real.sq_sqrt (hβ t)] - -/-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the -usual confidence inequalities hold almost surely at every positive time. -/ -lemma regret_ae_le_sum_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm] with ω h_gapω - exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 - -/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound on -the positive-time LinUCB width terms. -/ -lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - filter_upwards [regret_ae_le_sum_width h h_best h_arm] with ω h_regret - refine h_regret.trans ?_ - have hsplit : - (∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) = - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - ∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - rw [← sum_add_distrib] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - rw [hsplit] - exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) - (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ - -/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound whose -beta factor has been simplified using nonnegativity of the confidence schedule. -/ -lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - filter_upwards [regret_ae_le_initial_add_cauchy (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret - simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret - -omit [IsMarkovKernel ν] in -/-- If the squared LinUCB widths are bounded by `W`, then the Cauchy-Schwarz regret bound can use -`√W` in place of the square root of the realized squared-width sum. -/ -lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) - (hW : widthSqSum A reg x n ω ≤ W) - : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by - rw [widthSqSum] at hW - refine h_regret.trans ?_ - gcongr - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √(sum beta terms) * √W` whenever the squared LinUCB widths are almost surely bounded by `W`. - -This is the interface expected from a future elliptical-potential bound. -/ -lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by - filter_upwards [regret_ae_le_initial_add_cauchy_simplified (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with - ω h_regret hWω - exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -omit [IsMarkovKernel ν] in -/-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square -root of the beta sum. -/ -lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) - (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by - refine h_regret.trans ?_ - gcongr - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √B * √W` whenever the beta sum is bounded by `B` and the squared LinUCB widths are almost -surely bounded by `W`. -/ -lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) - (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by - filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW - ] with ω h_regret - exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) - (n := n) (ω := ω) B W h_regret hB - -/-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is -bounded by the horizon times the terminal beta value. -/ -lemma beta_sum_le_nat_mul_of_monotone - (hβ_mono : Monotone β) (hβ : ∀ t, 0 ≤ β (t + 1)) : - (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ (n : ℝ) * β n := by - calc - (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) - ≤ ∑ _t ∈ range n, β n := by - refine sum_le_sum ?_ - intro t ht - by_cases ht0 : t = 0 - · rw [if_pos ht0] - have hn_pos : 0 < n := by - simpa [ht0] using mem_range.mp ht - have hn_beta : 0 ≤ β n := by - have htime : n - 1 + 1 = n := by grind - simpa [htime] using hβ (n - 1) - exact hn_beta - · rw [if_neg ht0] - exact hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) - _ = (n : ℝ) * β n := by - simp [sum_const, nsmul_eq_mul] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Minimal confidence-radius schedule assumptions used by the capped finite-action LinUCB regret -chain: the schedule starts at least at one and is monotone in time. -/ -def BetaSchedule (β : ℕ → ℝ) : Prop := - 1 ≤ β 1 ∧ Monotone β - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection from `BetaSchedule`: the confidence-radius schedule starts at least at one. -/ -lemma BetaSchedule.one (hβ : BetaSchedule β) : 1 ≤ β 1 := - hβ.1 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection from `BetaSchedule`: the confidence-radius schedule is monotone. -/ -lemma BetaSchedule.monotone (hβ : BetaSchedule β) : Monotone β := - hβ.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive -horizon. -/ -lemma beta_nonneg_of_one_le_of_monotone - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) {n : ℕ} (hn : n ≠ 0) : - 0 ≤ β n := by - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) - exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A `BetaSchedule` is nonnegative at every positive horizon. -/ -lemma BetaSchedule.nonneg_of_ne_zero (hβ : BetaSchedule β) {n : ℕ} (hn : n ≠ 0) : - 0 ≤ β n := - beta_nonneg_of_one_le_of_monotone (β := β) hβ.one hβ.monotone hn - -omit [IsMarkovKernel ν] in -/-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the -horizon is zero. -/ -lemma initial_gap_sum_eq : - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) = - if n = 0 then 0 else gap ν (A 0 ω) := by - cases n <;> simp - -omit [IsMarkovKernel ν] in -/-- In zero feature dimension, the confidence event bounds cumulative regret by the initial gap. -There is no positive-time width contribution because all widths are zero. -/ -lemma regret_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by - refine (regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ if t = 0 then gap ν (A 0 ω) else 0) ?_).trans ?_ - · intro t _ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using - gap_nonpos_of_confidence_dim_eq_zero (A := A) (R := R) (reg := reg) - (β := β) (x := x) (ν := ν) (ω := ω) hd h_conf t ht0 - · rw [initial_gap_sum_eq] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure zero-dimensional version of the finite-action LinUCB regret skeleton. -/ -lemma regret_ae_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by - filter_upwards [h_conf] with ω h_confω - exact regret_le_initial_gap_of_confidence_dim_eq_zero (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hd h_confω - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` -is nonnegative and monotone. -/ -lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W - (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) hW - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` -is nonnegative and monotone. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW - ] with ω h_regret - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a history-level quadratic-form bound supplies the future -elliptical-potential input. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (hW : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the packaged history-level quadratic-width input holds almost -surely. - -This is the theorem a future elliptical-potential lemma should feed into directly. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a capped history-level quadratic-width sum bound holds almost -surely and every positive-time quadratic width form is almost surely nonnegative and at most `1`. - -This is the direct interface for the common capped form of the elliptical-potential lemma. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (hW : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a capped process-level quadratic-width sum bound holds almost -surely and every positive-time process-level quadratic width form is almost surely nonnegative and -at most `1`. - -This is the direct interface for an elliptical-potential lemma stated using the process-level design -matrices `designMatrix A reg x t ω`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the packaged process-level capped quadratic-width input holds -almost surely. - -This is the compact theorem a process-level elliptical-potential lemma should feed into directly. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) - (x := x) (n := n) (P := P) (W := W) h_bound) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is almost surely bounded -by `W`. - -This version follows the proof structure of *Bandit Algorithms*, Theorem 19.2: optimism gives the -width bound, bounded instantaneous gaps give the cap, monotonicity of `β` moves all confidence -radii to `β n`, and Cauchy-Schwarz turns the sum into the square root of the capped quadratic-width -sum. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_schedule : BetaSchedule β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - by_cases hn : n = 0 - · subst n - exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] - have hβn_nonneg : 0 ≤ β n := - hβ_schedule.nonneg_of_ne_zero hn - filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] - with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω - have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht _ht0 - exact h_quad_nonnegω t ht - have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - intro t ht ht0 - have hβ_le : β (t + 1) ≤ β n := - hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) - have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 - have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) - exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) - (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one - have h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := - regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos - h_gap_capped - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using - regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -/-- Almost surely, on the LinUCB confidence event, cumulative regret is bounded by the simplified -initial-gap term plus `2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is -almost surely bounded by `W`. - -This is the good-event form of the deterministic regret argument. It separates the algorithmic -regret proof from the future concentration theorem: a later self-normalized concentration result -should prove that `LinUCBConfidenceEvent` holds with high probability, and this theorem converts -that event into the regret bound. -/ -lemma regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_schedule : BetaSchedule β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - by_cases hn : n = 0 - · subst n - exact Filter.Eventually.of_forall fun ω _h_confω ↦ by simp [regret] - have hβn_nonneg : 0 ≤ β n := - hβ_schedule.nonneg_of_ne_zero hn - filter_upwards [forall_index_le_index_arm h (bestArm ν), h_gap_two, h_quad_nonneg, hW] with - ω h_indexω h_gap_twoω h_quad_nonnegω hWω h_confω - have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht _ht0 - exact h_quad_nonnegω t ht - have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - intro t ht ht0 - have h_gap_width : - gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := - gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := t) (ω := ω) (h_confω.best t ht0) - (h_confω.arm t ht0) (h_indexω t ht0) - have hβ_le : β (t + 1) ≤ β n := - hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) - have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 - have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) - exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) - (h_gap_twoω t ht ht0) h_gap_width hβ_le hβn_one - have h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := - regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos - h_gap_capped - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using - regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the -feature-budget elliptical-potential term -`2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. - -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_antitone` is the -generic inverse anti-monotonicity theorem for positive-definite matrices, and `h_ratio_of_trace` -should come from a determinant/trace comparison proving that the trace budget implies the displayed -determinant-ratio bound. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) - (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' - (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) - (x := x) hreg_pos h_inv_antitone)) - hreg_pos hL2 hL2_le_reg) - L2 hL2 h_ratio_of_trace) - -/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the determinant/trace input is stated as the determinant upper bound -`det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg hdet_of_trace) - -/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the determinant/trace input is the reusable PSD matrix determinant/trace comparison -`det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) - -/-- Textbook-shaped finite-action LinUCB regret theorem on the confidence event. - -This theorem is the good-event form closest to the finite-action LinUCB proof in -*Bandit Algorithms*: after the deterministic algorithm/max-index argument and the elliptical -potential bound are proved, the only remaining probabilistic input is whether the confidence event -holds on a sample path. - -* `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; -* `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; -* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. - -The conclusion is an almost-sure implication: on almost every sample path, if -`LinUCBConfidenceEvent` holds, then the displayed regret bound holds. A future self-normalized -concentration theorem should prove that this confidence event has high probability for a concrete -textbook choice of `β`. - -The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression -`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this -formalization lets the deterministic algorithm play its default initial arm at time zero. -/ -lemma regret_ae_imp_le_textbook_finite_action - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - by_cases hd : d = 0 - · subst d - exact Filter.Eventually.of_forall fun ω h_confω ↦ by - simpa using regret_le_initial_gap_of_confidence_dim_eq_zero - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) - (ω := ω) (d := 0) rfl h_confω - · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by - filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with - ω h_gapω - intro t ht _ht0 - exact h_gapω t ht - exact regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_gap_two hβ_schedule - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 - (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) - (P := P) L2 hL2) - matrixDetLeTraceAveragePow) - -/-- The deterministic textbook LinUCB bonus term -`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`. - -The final finite-action theorem keeps this as a named expression so probability statements can use -a deterministic right-hand side instead of repeating the full formula. -/ -noncomputable def textbookRegretBonus (reg : ℝ) (β : ℕ → ℝ) (L2 : ℝ) (n : ℕ) : ℝ := - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) - -/-- Good-event finite-action LinUCB regret theorem with the random initial gap replaced by the -deterministic `≤ 2` bound implied by `MeanRewardBound ν (-1) 1`. -/ -lemma regret_ae_imp_le_textbook_finite_action_deterministic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_confω - refine (h_regret h_confω).trans ?_ - simpa [textbookRegretBonus] using - add_le_add_right - (initialGapTerm_le_two_of_meanRewardBound_neg_one_one (A := A) (ν := ν) - (n := n) (ω := ω) h_mean_bound) - (2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))) - -/-- Almost-sure corollary of -`regret_ae_imp_le_textbook_finite_action_deterministic_bound` when the confidence event is known -to hold almost surely. -/ -lemma regret_ae_le_textbook_finite_action_deterministic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2, h_conf] with ω h_regret h_confω - exact h_regret h_confω - -/-- The confidence event is almost surely contained in the deterministic textbook regret-bound -event. This is the version to combine with a future high-probability confidence theorem. -/ -lemma probReal_confidenceEvent_le_textbook_regret_bound_deterministic - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_confω - exact h_regret h_confω - -/-- High-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ -lemma probReal_textbook_regret_bound_deterministic_ge_of_confidenceEvent_ge - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : - 1 - δ ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by - exact h_conf_prob.trans - (probReal_confidenceEvent_le_textbook_regret_bound_deterministic (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2) - -/-- Failure-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ -lemma probReal_textbook_regret_bound_deterministic_failure_le_of_confidenceEvent_failure_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_failure : - P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : - P.real {ω | - ¬ regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} ≤ δ := by - refine le_trans ?_ h_conf_failure - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω - exact h_regret_failure (h_regret h_confω) - -/-- The confidence event is almost surely contained in the textbook finite-action regret-bound -event. - -This is the probability bridge needed after the good-event theorem: once a concentration theorem -proves that `LinUCBConfidenceEvent` has high probability, this lemma transfers that probability -mass to the displayed regret bound. -/ -lemma probReal_confidenceEvent_le_textbook_regret_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_confω - exact h_regret h_confω - -/-- High-probability wrapper for the textbook finite-action LinUCB regret bound. - -If a future self-normalized concentration theorem proves that the confidence event has probability -at least `1 - δ`, then the textbook regret bound has probability at least `1 - δ` as well. -/ -lemma probReal_textbook_regret_bound_ge_of_confidenceEvent_ge - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : - 1 - δ ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by - exact h_conf_prob.trans - (probReal_confidenceEvent_le_textbook_regret_bound (A := A) (R := R) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule hreg_pos L2 hL2) - -/-- Failure-probability wrapper for the textbook finite-action LinUCB regret bound. - -If a future self-normalized concentration theorem proves that the confidence event fails with -probability at most `δ`, then the textbook regret bound fails with probability at most `δ`. -/ -lemma probReal_textbook_regret_bound_failure_le_of_confidenceEvent_failure_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_failure : - P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : - P.real {ω | - ¬ - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} ≤ δ := by - refine le_trans ?_ h_conf_failure - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω - exact h_regret_failure (h_regret h_confω) - -/-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to -hold almost surely. This is stronger than the textbook high-probability route and is mainly useful -as a compatibility wrapper for earlier lemmas in this file. -/ -lemma regret_ae_le_textbook_finite_action - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2, h_conf] with ω h_regret h_confω - exact h_regret h_confω - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final -log-determinant potential bound hold. - -The capped-sum/log-determinant part of the elliptical-potential argument is proved internally: -positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while -`h_quad_le_one` lets this older theorem feed the uncapped `widthSqSum` regret route. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hreg_pos : 0 < reg) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono W - (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) hreg_pos - h_quad_le_one h_potential_le) - -end LinUCB - -end Bandits diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean deleted file mode 100644 index 7441bec4..00000000 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Confidence.lean +++ /dev/null @@ -1,148 +0,0 @@ -/- -Copyright (c) 2026. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: OpenAI, Fawad Haider --/ -module - -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Matrix - -/-! -# LinUCB for finite-action linear bandits -Chapter 19 of *Bandit Algorithms*: --/ - -@[expose] public section - -open MeasureTheory ProbabilityTheory Filter Real Finset Learning - -open scoped ENNReal NNReal Matrix MatrixOrder - -namespace Bandits - -variable {K d : ℕ} - -namespace LinUCB - -variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} - {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] - {Ω : Type*} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} - {n : ℕ} {ω : Ω} - -section AlgorithmBehavior - -/-- The pointwise LinUCB confidence event used by the finite-action regret proof. - -For every positive process time, the best arm's true mean lies below its optimistic index, and the -selected arm's pessimistic index lies below its true mean. On this event, the max-index property of -LinUCB turns optimism into an instantaneous regret bound. -/ -def LinUCBConfidenceEvent [Nonempty (Fin K)] - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Horizon-local LinUCB confidence event. - -The regret theorem up to horizon `n` only uses the confidence inequalities at times -`t ∈ range n`. This finite-horizon event is the natural target for a high-probability -self-normalized concentration theorem with a fixed horizon, and avoids requiring confidence at all -future times when proving an `n`-round regret bound. -/ -def LinUCBConfidenceEventUpTo [Nonempty (Fin K)] - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := - ∀ t, t ∈ range n → t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] - -omit [IsMarkovKernel ν] in -/-- A global LinUCB confidence event implies its finite-horizon restriction. -/ -lemma LinUCBConfidenceEvent.toUpTo [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - LinUCBConfidenceEventUpTo A R reg β x ν n ω := by - intro t _ht ht0 - exact h_conf t ht0 - -omit [IsMarkovKernel ν] in -/-- First projection from the horizon-local LinUCB confidence event: optimism for the best arm. -/ -lemma LinUCBConfidenceEventUpTo.best [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEventUpTo A R reg β x ν n ω) : - ∀ t, t ∈ range n → t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - intro t ht ht0 - exact (h_conf t ht ht0).1 - -omit [IsMarkovKernel ν] in -/-- Second projection from the horizon-local LinUCB confidence event: validity of the selected -arm's lower confidence inequality. -/ -lemma LinUCBConfidenceEventUpTo.arm [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEventUpTo A R reg β x ν n ω) : - ∀ t, t ∈ range n → t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - intro t ht ht0 - exact (h_conf t ht ht0).2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Linear realizability of the arm means by a parameter `θ`. - -The actual self-normalized concentration theorem for LinUCB needs an assumption of this shape: -each arm's mean reward must be represented by the finite-dimensional linear model -`x_aᵀ θ`. Without such a realizability assumption, no concentration theorem can imply -`LinUCBConfidenceEvent`, because the least-squares predictor may be biased away from the true arm -means for structural reasons rather than random noise. -/ -def LinearMeanModel (ν : Kernel (Fin K) ℝ) (x : Fin K → Feature d) (θ : Feature d) : Prop := - ∀ a, (ν a)[id] = dotProduct θ (x a) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Squared Euclidean-norm bound on a linear-bandit parameter. This is the finite-dimensional -version of the textbook assumption `‖θ‖₂ ≤ S`, written as `θᵀθ ≤ S2`. -/ -def ParameterSqNormBound (θ : Feature d) (S2 : ℝ) : Prop := - dotProduct θ θ ≤ S2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Mean response vector predicted by a linear mean model along the realized actions. -/ -noncomputable def meanResponseVector - (A : ℕ → Ω → Fin K) (ν : Kernel (Fin K) ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - ∑ s ∈ range n, (ν (A s ω))[id] • x (A s ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Centered response vector, i.e. the accumulated reward-feature vector minus its conditional mean -under the finite-action linear mean model. This is the deterministic noise vector that the later -self-normalized concentration theorem must control. -/ -noncomputable def centeredResponseVector - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - responseVector A R x n ω - meanResponseVector A ν x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar reward noise at time `t`, centered at the mean of the selected arm. -/ -noncomputable def rewardNoise - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) - (t : ℕ) (ω : Ω) : ℝ := - R t ω - (ν (A t ω))[id] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Arm-wise centered reward subgaussianity. - -This is the LinUCB analogue of the scalar noise assumption used in `UCB.lean`: for each arm, the -reward centered at that arm's mean has a subgaussian moment-generating-function bound. A future -vector self-normalized concentration theorem should start from this assumption, together with the -linear mean model, and prove the centered-noise-plus-bias confidence event. -/ -def RewardNoiseSubgaussian (ν : Kernel (Fin K) ℝ) (σ2 : ℝ≥0) : Prop := - ∀ a, HasSubgaussianMGF (fun r ↦ r - (ν a)[id]) σ2 (ν a) - -end AlgorithmBehavior - -end LinUCB - -end Bandits diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean deleted file mode 100644 index 24fc92c2..00000000 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidence.lean +++ /dev/null @@ -1,354 +0,0 @@ -/- -Copyright (c) 2026. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: OpenAI, Fawad Haider --/ -module - -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.TextbookMixture - -/-! -# LinUCB Textbook Confidence Events - -Deterministic bridges from centered-noise and textbook self-normalized events to -LinUCB prediction-confidence events. --/ - -@[expose] public section - -open MeasureTheory ProbabilityTheory Filter Real Finset Learning - -open scoped ENNReal NNReal Matrix MatrixOrder - -namespace Bandits - -variable {K d : ℕ} - -namespace LinUCB - -variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} - {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] - {Ω : Type*} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} - {n : ℕ} {ω : Ω} - -section AlgorithmBehavior - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Horizon-local coordinate-wise centered-noise event. - -This is a conservative finite-dimensional interface for reusing scalar projection concentration: -if every coordinate of the centered response vector is bounded, then the centered-noise quadratic -form is bounded through `centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg`. -/ -def LinUCBCenteredNoiseCoordinateBoundEventUpTo - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := - ∀ t, t ∈ range n → t ≠ 0 → ∀ i, - |centeredResponseVector A R ν x t ω i| ≤ coordBudget (t + 1) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Upper-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ -def centeredNoiseCoordinateUpperFailure - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := - {ω | coordBudget (t + 1) < centeredResponseVector A R ν x t ω i} - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Lower-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ -def centeredNoiseCoordinateLowerFailure - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := - {ω | centeredResponseVector A R ν x t ω i < -coordBudget (t + 1)} - -omit [IsMarkovKernel ν] in -/-- A global centered-noise-plus-bias confidence event implies its finite-horizon restriction. -/ -lemma LinUCBCenteredNoiseBiasConfidenceEvent.toUpTo - (θ : Feature d) - (h_noise : LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by - intro t _ht ht0 - exact h_noise t ht0 - -omit [IsMarkovKernel ν] in -/-- A coordinate-wise centered-noise event implies the centered-noise quadratic-form event under the -corresponding conservative budget. -/ -lemma LinUCBCenteredNoiseConfidenceEventUpTo.of_coordinateBound - (hreg_pos : 0 < reg) - {coordBudget noiseBudget : ℕ → ℝ} - (hcoord : - LinUCBCenteredNoiseCoordinateBoundEventUpTo A R coordBudget x ν n ω) - (hcoord_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ coordBudget (t + 1)) - (h_budget : ∀ t, t ∈ range n → t ≠ 0 → - (d : ℝ) * coordBudget (t + 1) ^ 2 / reg ≤ noiseBudget (t + 1)) : - LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω := by - intro t ht ht0 - exact - (centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) hreg_pos - (hcoord_nonneg t ht ht0) (hcoord t ht ht0)).trans - (h_budget t ht ht0) - -omit [IsMarkovKernel ν] in -/-- A horizon-local centered-noise event plus a parameter norm bound implies the -centered-noise-plus-ridge-bias event. - -This is the finite-horizon deterministic half of the textbook confidence-set proof: -the future self-normalized concentration theorem only has to control -`centeredNoiseQuadraticForm`; the ridge-bias contribution is bounded here by -`reg * S2`. -/ -lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_centeredNoise - (θ : Feature d) (S2 : ℝ) - (hreg_pos : 0 < reg) - (hθ : ParameterSqNormBound θ S2) - {noiseBudget : ℕ → ℝ} - (h_noise : - LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω) - (h_budget : ∀ t, t ∈ range n → t ≠ 0 → - (√(noiseBudget (t + 1)) + √(reg * S2)) ^ 2 ≤ β (t + 1)) : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by - intro t ht ht0 - exact - (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos - (noiseBudget (t + 1)) (reg * S2) (h_noise t ht ht0) - (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) - (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans - (h_budget t ht ht0) - -omit [IsMarkovKernel ν] in -/-- The textbook determinant-ratio self-normalized noise event plus the ridge-bias radius implies -the existing centered-noise-plus-bias confidence event. - -This is the deterministic bridge from the future Gaussian-mixture concentration theorem to the -confidence event already consumed by the LinUCB regret proof. -/ -lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_textbookSelfNormalizedNoise - (θ : Feature d) (S2 : ℝ) {σ2 : ℝ≥0} {δ : ℝ} - (hreg_pos : 0 < reg) - (hθ : ParameterSqNormBound θ S2) - (h_noise : - LinUCBTextbookSelfNormalizedNoiseEventUpTo A R reg σ2 δ x ν n ω) - (h_budget : ∀ t, t ∈ range n → t ≠ 0 → - (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + - √(reg * S2)) ^ 2 ≤ β (t + 1)) : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by - intro t ht ht0 - exact - (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos - (textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) - (reg * S2) (h_noise t ht ht0) - (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) - (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans - (h_budget t ht ht0) - -omit [IsMarkovKernel ν] in -/-- The centered-noise-plus-bias confidence event implies the textbook parameter ellipsoid event -under linear realizability and positive regularization. -/ -lemma LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_noise : - LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : - LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω := by - intro t ht - rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] - exact h_noise t ht - -omit [IsMarkovKernel ν] in -/-- The horizon-local centered-noise-plus-bias event implies the horizon-local parameter ellipsoid -event under linear realizability and positive regularization. -/ -lemma LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_noise : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : - LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω := by - intro t ht ht0 - rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] - exact h_noise t ht ht0 - -omit [IsMarkovKernel ν] in -/-- Under linear realizability and positive regularization, the parameter ellipsoid event is exactly -the centered-noise-plus-bias confidence event. -/ -lemma linUCBParameterEllipsoidConfidenceEvent_iff_centeredNoiseBiasConfidenceEvent - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) : - LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω ↔ - LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω := by - constructor - · intro h_ellipsoid t ht - rw [← parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] - exact h_ellipsoid t ht - · intro h_noise - exact LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear hreg_pos h_noise - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Action-wise prediction-error confidence event around a linear parameter `θ`. - -This is the finite-action consequence of the textbook ellipsoid event after applying the -matrix Cauchy-Schwarz inequality to each arm. -/ -def LinUCBParameterPredictionConfidenceEvent - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (θ : Feature d) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → ∀ a, - |dotProduct (thetaHat A R reg x t ω - θ) (x a)| ≤ - √(β (t + 1)) * width A reg x a t ω - -/-- Uniform self-normalized prediction-error event for finite-action LinUCB. - -This is the event that a future self-normalized martingale concentration theorem should prove with -high probability, for a concrete textbook choice of `β`. It says that, at every positive time and -for every finite action, the least-squares prediction error is bounded by the LinUCB confidence -radius. -/ -def LinUCBSelfNormalizedConfidenceEvent - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → ∀ a, - |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ - √(β (t + 1)) * width A reg x a t ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Horizon-local self-normalized prediction-error event. - -This is the finite-horizon version of `LinUCBSelfNormalizedConfidenceEvent`. For regret through -time `n`, the proof only needs prediction confidence for positive `t ∈ range n`. -/ -def LinUCBSelfNormalizedConfidenceEventUpTo - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := - ∀ t, t ∈ range n → t ≠ 0 → ∀ a, - |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ - √(β (t + 1)) * width A reg x a t ω - -omit [IsMarkovKernel ν] in -/-- A global self-normalized prediction-error event implies its finite-horizon restriction. -/ -lemma LinUCBSelfNormalizedConfidenceEvent.toUpTo - (h_self : LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω) : - LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by - intro t _ht ht0 a - exact h_self t ht0 a - -omit [IsMarkovKernel ν] in -/-- A parameter prediction-confidence event implies the self-normalized confidence event once the -arm means are realized by that parameter. -/ -lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (h_param : LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω) : - LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := by - intro t ht a - have h_param_t := h_param t ht a - have h_error : - estimatedReward A R reg x a t ω - (ν a)[id] = - dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - calc - estimatedReward A R reg x a t ω - (ν a)[id] - = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by - rw [estimatedReward, h_linear a] - _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - rw [sub_dotProduct] - rw [h_error] - simpa [sub_dotProduct] using h_param_t - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The textbook ellipsoid event implies action-wise parameter prediction confidence under positive -regularization, by the matrix Cauchy-Schwarz step. -/ -lemma LinUCBParameterPredictionConfidenceEvent.of_ellipsoid - (θ : Feature d) - (hreg_pos : 0 < reg) - (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : - LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω := by - intro t ht a - have h_cauchy_t := - linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) - hreg_pos (thetaHat A R reg x t ω - θ) a t ω - have h_radius : - √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := - Real.sqrt_le_sqrt (h_ellipsoid t ht) - calc - |dotProduct (thetaHat A R reg x t ω - θ) (x a)| - ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by - simpa [parameterErrorQuadraticForm] using h_cauchy_t - _ ≤ √(β (t + 1)) * width A reg x a t ω := by - exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) - -omit [IsMarkovKernel ν] in -/-- The textbook ellipsoid event implies the self-normalized prediction-error event under positive -regularization and linear realizability. -/ -lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterEllipsoid - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : - LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := - LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear - (LinUCBParameterPredictionConfidenceEvent.of_ellipsoid (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ω := ω) θ hreg_pos h_ellipsoid) - -omit [IsMarkovKernel ν] in -/-- The horizon-local textbook ellipsoid event implies the horizon-local self-normalized -prediction-error event under positive regularization and linear realizability. -/ -lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω) : - LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by - intro t ht ht0 a - have h_cauchy_t := - linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) - hreg_pos (thetaHat A R reg x t ω - θ) a t ω - have h_radius : - √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := - Real.sqrt_le_sqrt (h_ellipsoid t ht ht0) - have h_error : - estimatedReward A R reg x a t ω - (ν a)[id] = - dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - calc - estimatedReward A R reg x a t ω - (ν a)[id] - = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by - rw [estimatedReward, h_linear a] - _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - rw [sub_dotProduct] - rw [h_error] - calc - |dotProduct (thetaHat A R reg x t ω - θ) (x a)| - ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by - simpa [parameterErrorQuadraticForm] using h_cauchy_t - _ ≤ √(β (t + 1)) * width A reg x a t ω := by - exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) - -omit [IsMarkovKernel ν] in -/-- The horizon-local centered-noise-plus-bias event implies the horizon-local self-normalized -prediction-error event under linear realizability and positive regularization. -/ -lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_centeredNoiseBias - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_noise : LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : - LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := - LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) θ h_linear hreg_pos - (LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) - θ h_linear hreg_pos h_noise) - -end AlgorithmBehavior - -end LinUCB - -end Bandits From c62f829164e8680664620bac10c45ae575fbd6d7 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:07:02 -0400 Subject: [PATCH 78/88] feat(112): reorg the linUCB into modular files --- .../Online/Bandit/Algorithms/LinUCB.lean | 4049 +---------------- 1 file changed, 6 insertions(+), 4043 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index da0dc185..f528dfdb 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -5,4052 +5,15 @@ Authors: OpenAI, Fawad Haider -/ module -public import LeanMachineLearning.Online.Bandit.SumRewards -public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax -public import Mathlib.Analysis.MeanInequalities -public import Mathlib.Analysis.SpecialFunctions.Log.Deriv -public import Mathlib.Analysis.Matrix.Order -public import Mathlib.Data.Real.StarOrdered -public import Mathlib.LinearAlgebra.Matrix.PosDef -public import Mathlib.LinearAlgebra.Matrix.SchurComplement -public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Regret /-! # LinUCB for finite-action linear bandits -Chapter 19 of *Bandit Algorithms*: --/ - -@[expose] public section - -open MeasureTheory ProbabilityTheory Filter Real Finset Learning - -open scoped ENNReal NNReal Matrix MatrixOrder - -namespace Bandits - -variable {K d : ℕ} - -section Algorithm - -namespace LinUCB - -/-- Feature vectors for finite-dimensional linear bandits. -/ -abbrev Feature (d : ℕ) := Fin d → ℝ - -/-- Squared Euclidean norm of a finite-action feature vector, written as the dot product -`x_aᵀ x_a`. -/ -def featureSqNorm (x : Fin K → Feature d) (a : Fin K) : ℝ := - dotProduct (x a) (x a) - -/-- The squared feature norm is nonnegative. -/ -lemma featureSqNorm_nonneg (x : Fin K → Feature d) (a : Fin K) : - 0 ≤ featureSqNorm x a := by - rw [featureSqNorm, dotProduct] - exact sum_nonneg fun i _ ↦ mul_self_nonneg (x a i) -/-- Uniform squared feature-norm bound for finite-action LinUCB. +This module is the public entry point for the finite-action LinUCB development. -This is the finite-action version of the textbook assumption `‖x‖₂ ≤ L`, written here in squared -form as `‖x_a‖₂² ≤ L2` for every action. -/ -def FeatureSqNormBound (x : Fin K → Feature d) (L2 : ℝ) : Prop := - ∀ a, featureSqNorm x a ≤ L2 - -/-- History-level regularized design matrix for LinUCB. -/ -noncomputable def designMatrix' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Matrix (Fin d) (Fin d) ℝ := - reg • 1 + ∑ s : Iic n, Matrix.vecMulVec (x (h s).1) (x (h s).1) - -/-- History-level response vector for LinUCB. -/ -noncomputable def responseVector' (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - ∑ s : Iic n, (h s).2 • x (h s).1 - -/-- History-level regularized least-squares estimate. -/ -noncomputable def thetaHat' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Feature d := - Matrix.mulVec (designMatrix' reg x n h)⁻¹ (responseVector' x n h) - -/-- History-level estimated reward of an arm. -/ -noncomputable def estimatedReward' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - dotProduct (thetaHat' reg x n h) (x a) - -/-- History-level quadratic form underlying the LinUCB confidence width. -/ -noncomputable def widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - dotProduct (x a) (Matrix.mulVec (designMatrix' reg x n h)⁻¹ (x a)) - -/-- History-level elliptical confidence width of an arm. -/ -noncomputable def width' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - √(widthQuadraticForm' reg x n h a) - -/-- Squaring the history-level LinUCB width recovers its quadratic form, provided that quadratic -form is nonnegative. -/ -lemma width'_sq_eq_quadratic_form (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm' reg x n h a) : - width' reg x n h a ^ 2 = widthQuadraticForm' reg x n h a := by - simp [width', Real.sq_sqrt h_nonneg] - -/-- LinUCB optimistic index of an arm. - -The parameter `β` is a confidence-radius schedule. Since `h : Iic n → Fin K × ℝ` -contains the observations through time `n`, this index is used to choose the arm -at time `n + 1`, and we evaluate the schedule at `n + 2` +The implementation is split across submodules under +`LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.*`; importing this file re-exports the full +LinUCB API, including the algorithm definition, confidence events, concentration interfaces, +deterministic regret decomposition, elliptical-potential/log-det bounds, and final regret theorems. -/ -noncomputable def index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) (a : Fin K) : ℝ := - estimatedReward' reg x n h a + √(β (n + 2)) * width' reg x n h a - -open Classical in -/-- Arm pulled by finite-action LinUCB at time `n + 1`. -/ -noncomputable def nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → Feature d) - (n : ℕ) (h : Iic n → Fin K × ℝ) : Fin K := - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - measurableArgmax (fun h a ↦ index' reg β x n h a) h - -@[fun_prop] -lemma measurable_nextArm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → Feature d) - (h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)) - (n : ℕ) : - Measurable (nextArm hK reg β x n) := by - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact measurable_measurableArgmax fun a ↦ h_index n a - -end LinUCB - -/-- The finite-action LinUCB algorithm. -/ -noncomputable def linUCBAlgorithm (hK : 0 < K) (reg : ℝ) (β : ℕ → ℝ) - (x : Fin K → LinUCB.Feature d) - (h_index : ∀ n a, Measurable (fun h ↦ LinUCB.index' reg β x n h a)) : - Algorithm (Fin K) ℝ := - detAlgorithm (LinUCB.nextArm hK reg β x) (by fun_prop) ⟨0, hK⟩ - -end Algorithm - -namespace LinUCB - -variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} - {h_index : ∀ n a, Measurable (fun h ↦ index' reg β x n h a)} - {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] - {Ω : Type*} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} - {n : ℕ} {ω : Ω} - -section AlgorithmBehavior - -/-- The process-level design matrix built from actions up to time `n` excluded. -/ -noncomputable def designMatrix (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Matrix (Fin d) (Fin d) ℝ := - reg • 1 + ∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω)) - -/-- The initial design matrix before any actions are included. -/ -lemma designMatrix_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - designMatrix A reg x 0 ω = reg • 1 := by - simp [designMatrix] - -/-- The design matrix update after observing one additional action. -/ -lemma designMatrix_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designMatrix A reg x (n + 1) ω = - designMatrix A reg x n ω + Matrix.vecMulVec (x (A n ω)) (x (A n ω)) := by - simp [designMatrix, sum_range_succ, add_assoc] - -/-- With nonnegative regularization, the process-level design matrix is positive semidefinite. -/ -lemma designMatrix_posSemidef (hreg_nonneg : 0 ≤ reg) : - (designMatrix A reg x n ω).PosSemidef := by - unfold designMatrix - apply Matrix.PosSemidef.add - · exact Matrix.PosSemidef.smul Matrix.PosSemidef.one hreg_nonneg - · refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - -/-- Positive regularization makes the process-level design matrix positive definite. -/ -lemma designMatrix_posDef (hreg_pos : 0 < reg) : - (designMatrix A reg x n ω).PosDef := by - unfold designMatrix - apply Matrix.PosDef.add_posSemidef - · exact Matrix.PosDef.smul Matrix.PosDef.one hreg_pos - · refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The design matrix dominates its regularization part: after subtracting `reg • I`, what remains -is the sum of observed rank-one feature matrices, hence positive semidefinite. -/ -lemma designMatrix_sub_reg_smul_one_posSemidef : - (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)).PosSemidef := by - have hsum : - (∑ s ∈ range n, Matrix.vecMulVec (x (A s ω)) (x (A s ω))).PosSemidef := by - refine Matrix.posSemidef_sum (s := range n) ?_ - intro t _ - simpa using Matrix.posSemidef_vecMulVec_self_star (x (A t ω)) - simpa [designMatrix, add_sub_cancel_left] using hsum - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix-order form of `designMatrix_sub_reg_smul_one_posSemidef`: `reg • I ≤ V_n`. -/ -lemma reg_smul_one_le_designMatrix : - reg • (1 : Matrix (Fin d) (Fin d) ℝ) ≤ designMatrix A reg x n ω := by - rw [Matrix.le_iff] - exact designMatrix_sub_reg_smul_one_posSemidef (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix order is preserved by evaluating the quadratic form against a fixed feature vector. -/ -lemma dotProduct_mulVec_le_of_matrix_le {M N : Matrix (Fin d) (Fin d) ℝ} - (hMN : M ≤ N) (u : Feature d) : - dotProduct u (M *ᵥ u) ≤ dotProduct u (N *ᵥ u) := by - have h_nonneg : 0 ≤ dotProduct u ((N - M) *ᵥ u) := by - simpa using (Matrix.le_iff.mp hMN).dotProduct_mulVec_nonneg u - rw [Matrix.sub_mulVec, dotProduct_sub] at h_nonneg - exact sub_nonneg.mp h_nonneg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The inverse of the regularized identity is the reciprocal-scaled identity. -/ -lemma reg_smul_one_inv (hreg : reg ≠ 0) : - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ = - reg⁻¹ • (1 : Matrix (Fin d) (Fin d) ℝ) := by - rw [Matrix.inv_eq_left_inv] - simp [smul_smul, hreg] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The quadratic form induced by `(reg • I)⁻¹` is the squared norm divided by `reg`. -/ -lemma dotProduct_reg_smul_one_inv_mulVec (hreg : reg ≠ 0) (u : Feature d) : - dotProduct u (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ u) = - dotProduct u u / reg := by - rw [reg_smul_one_inv (reg := reg) (d := d) hreg] - simp [Matrix.smul_mulVec, div_eq_inv_mul, mul_comm] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Arm-specific form of `dotProduct_reg_smul_one_inv_mulVec`. -/ -lemma dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div - (hreg : reg ≠ 0) (a : Fin K) : - dotProduct (x a) (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) = - featureSqNorm x a / reg := by - simpa [featureSqNorm] using - dotProduct_reg_smul_one_inv_mulVec (reg := reg) (d := d) hreg (x a) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Reusable matrix-analysis theorem needed for the LinUCB width comparison. - -It states the usual inverse anti-monotonicity of positive-definite matrices in the PSD order: -if `M` is positive definite and `M ≤ N`, then inversion reverses the order. -/ -def MatrixInvAntiMonoOnPosDef (d : ℕ) : Prop := - ∀ M N : Matrix (Fin d) (Fin d) ℝ, M.PosDef → M ≤ N → N⁻¹ ≤ M⁻¹ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The remaining inverse-monotonicity matrix obligation for the finite-action LinUCB regret -route. - -Mathematically, this should follow from `reg • I ≤ V_t` and positive regularization: inversion -reverses the positive-definite matrix order, so `V_t⁻¹ ≤ (reg • I)⁻¹`. -/ -def DesignMatrixInvLeRegInv - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := - ∀ (n : ℕ) (ω : Ω), - (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection lemma for the named inverse-order obligation. -/ -lemma DesignMatrixInvLeRegInv.apply - (h_inv : DesignMatrixInvLeRegInv A reg x) (n : ℕ) (ω : Ω) : - (designMatrix A reg x n ω)⁻¹ ≤ (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ := - h_inv n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The reusable positive-definite inverse anti-monotonicity theorem implies the LinUCB-specific -inverse-design comparison. -/ -lemma DesignMatrixInvLeRegInv.of_matrix_inv_antitone - (hreg_pos : 0 < reg) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) : - DesignMatrixInvLeRegInv A reg x := by - intro n ω - exact h_inv_antitone (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) - (designMatrix A reg x n ω) - (Matrix.PosDef.smul Matrix.PosDef.one hreg_pos) - (reg_smul_one_le_designMatrix (A := A) (reg := reg) (x := x) (n := n) (ω := ω)) - -/-- Trace of the process-level regularized design matrix. -/ -noncomputable def designTrace (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - Matrix.trace (designMatrix A reg x n ω) - -/-- Before any observations, the design trace is the trace of `reg • I_d`, namely `reg * d`. -/ -lemma designTrace_zero (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - designTrace A reg x 0 ω = reg * (d : ℝ) := by - simp [designTrace, designMatrix_zero] - -/-- Updating the design matrix by `x_a x_aᵀ` increases the trace by `x_aᵀ x_a`. -/ -lemma designTrace_succ (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designTrace A reg x (n + 1) ω = - designTrace A reg x n ω + featureSqNorm x (A n ω) := by - simp [designTrace, designMatrix_succ, featureSqNorm, Matrix.trace_vecMulVec] - -/-- Closed form for the design trace: initial regularization trace plus accumulated squared -feature norms. -/ -lemma designTrace_eq_reg_mul_dim_add_sum_featureSqNorm - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - designTrace A reg x n ω = - reg * (d : ℝ) + ∑ t ∈ range n, featureSqNorm x (A t ω) := by - simp [designTrace, designMatrix, featureSqNorm, Matrix.trace_vecMulVec] - -/-- With nonnegative regularization, the design trace is nonnegative. -/ -lemma designTrace_nonneg (hreg_nonneg : 0 ≤ reg) : - 0 ≤ designTrace A reg x n ω := by - rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] - exact add_nonneg - (mul_nonneg hreg_nonneg (Nat.cast_nonneg d)) - (sum_nonneg fun t _ ↦ featureSqNorm_nonneg x (A t ω)) - -/-- If every selected feature vector has squared norm at most `L2`, then the trace of the design -matrix is at most `reg * d + n * L2`. -/ -lemma designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by - rw [designTrace_eq_reg_mul_dim_add_sum_featureSqNorm] - gcongr - calc - (∑ t ∈ range n, featureSqNorm x (A t ω)) ≤ ∑ _t ∈ range n, L2 := by - exact sum_le_sum fun t ht ↦ hL2 t ht - _ = (n : ℝ) * L2 := by - simp [nsmul_eq_mul] - -omit [IsProbabilityMeasure P] in -/-- Almost surely, bounded selected feature norms give the corresponding deterministic trace -budget `reg * d + n * L2`. -/ -lemma designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) : - ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 := by - filter_upwards [hL2] with ω hL2ω - exact designTrace_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) L2 hL2ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A uniform finite-action feature bound implies the selected-action feature bound through any -finite horizon. -/ -lemma featureSqNorm_ae_le_of_featureSqNormBound - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2 := - Filter.Eventually.of_forall fun ω t _ht ↦ hL2 (A t ω) - -/-- The process-level reward-feature vector built from history up to time `n` excluded. -/ -noncomputable def responseVector (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - ∑ s ∈ range n, R s ω • x (A s ω) - -/-- The initial response vector before any rewards are included. -/ -lemma responseVector_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (ω : Ω) : - responseVector A R x 0 ω = 0 := by - simp [responseVector] - -/-- The response-vector update after observing one additional reward. -/ -lemma responseVector_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - responseVector A R x (n + 1) ω = - responseVector A R x n ω + R n ω • x (A n ω) := by - simp [responseVector, sum_range_succ] - -/-- The process-level regularized least-squares estimate. -/ -noncomputable def thetaHat (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := - Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (responseVector A R x n ω) - -/-- The initial least-squares estimate is zero because no reward-feature observations have been -included yet. -/ -lemma thetaHat_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - thetaHat A R reg x 0 ω = 0 := by - simp [thetaHat, responseVector_zero] - -/-- The process-level estimated linear reward. -/ -noncomputable def estimatedReward (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - dotProduct (thetaHat A R reg x n ω) (x a) - -/-- The initial estimated reward is zero for every arm. -/ -lemma estimatedReward_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - estimatedReward A R reg x a 0 ω = 0 := by - simp [estimatedReward, thetaHat_zero] - -/-- The quadratic form `x_aᵀ V_n⁻¹ x_a` underlying the LinUCB confidence width. -/ -noncomputable def widthQuadraticForm (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - dotProduct (x a) (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (x a)) - -/-- The initial width quadratic form is induced by the inverse regularized identity. -/ -lemma widthQuadraticForm_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - widthQuadraticForm A reg x a 0 ω = - dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a)) := by - simp [widthQuadraticForm, designMatrix_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Nonnegative regularization makes every LinUCB width quadratic form nonnegative. - -The reason is purely matrix-theoretic: `V_n` is positive semidefinite, the nonsingular inverse of a -positive semidefinite matrix is positive semidefinite in mathlib, and every quadratic form induced -by a positive semidefinite matrix is nonnegative. -/ -lemma widthQuadraticForm_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) (a : Fin K) : - 0 ≤ widthQuadraticForm A reg x a n ω := by - simpa [widthQuadraticForm] using - ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg).inv.dotProduct_mulVec_nonneg (x a)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, nonnegative regularization gives nonnegative selected quadratic width forms -through any finite horizon. -/ -lemma widthQuadraticForm_ae_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - exact Filter.Eventually.of_forall fun ω t _ht ↦ - widthQuadraticForm_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := t) (ω := ω) hreg_nonneg (A t ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive-time version of `widthQuadraticForm_ae_nonneg_of_reg_nonneg`, matching the side -condition shape used by the regret/width-sum bridge lemmas. -/ -lemma widthQuadraticForm_ae_pos_time_nonneg_of_reg_nonneg - (hreg_nonneg : 0 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - filter_upwards [widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) - (x := x) (n := n) (P := P) hreg_nonneg] with ω h_nonnegω - intro t ht _ht0 - exact h_nonnegω t ht - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The matrix comparison needed to turn bounded feature vectors into the positive-time LinUCB -width cap. - -Mathematically, this says `x_aᵀ V_t⁻¹ x_a ≤ ‖x_a‖² / reg`. A later matrix-order proof should -derive it from `reg > 0` and `V_t = reg I + ∑ x_s x_sᵀ`. Keeping it as a named property makes the -remaining linear-algebra obligation precise and reusable. -/ -def WidthQuadraticFormLeFeatureSqNormDivReg - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := - ∀ (a : Fin K) (n : ℕ) (ω : Ω), - widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If the inverse design matrix is bounded by the inverse regularized identity, then the LinUCB -quadratic width is bounded by `featureSqNorm / reg` for one arm, time, and sample point. -/ -lemma widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (a : Fin K) - (h_inv : (designMatrix A reg x n ω)⁻¹ ≤ - (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) - (hreg : reg ≠ 0) : - widthQuadraticForm A reg x a n ω ≤ featureSqNorm x a / reg := by - calc - widthQuadraticForm A reg x a n ω = - dotProduct (x a) (((designMatrix A reg x n ω)⁻¹) *ᵥ (x a)) := rfl - _ ≤ dotProduct (x a) - (((reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹) *ᵥ (x a)) := - dotProduct_mulVec_le_of_matrix_le h_inv (x a) - _ = featureSqNorm x a / reg := - dotProduct_reg_smul_one_inv_mulVec_eq_featureSqNorm_div - (reg := reg) (x := x) hreg a - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A pointwise inverse-order comparison for all times and sample points gives the reusable -`WidthQuadraticFormLeFeatureSqNormDivReg` property consumed by the regret route. -/ -lemma WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le - (hreg : reg ≠ 0) - (h_inv : DesignMatrixInvLeRegInv A reg x) : - WidthQuadraticFormLeFeatureSqNormDivReg A reg x := by - intro a n ω - exact widthQuadraticForm_le_featureSqNorm_div_reg_of_inv_le - (A := A) (reg := reg) (x := x) (n := n) (ω := ω) a - (h_inv.apply n ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `x_aᵀ V_n⁻¹ x_a ≤ ‖x_a‖² / reg` and the squared feature norm is at most `reg`, then the -quadratic form is at most one. -/ -lemma widthQuadraticForm_le_one_of_featureSqNorm_le_reg - (a : Fin K) - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) - (h_feature_le : featureSqNorm x a ≤ reg) : - widthQuadraticForm A reg x a n ω ≤ 1 := by - refine (h_width a n ω).trans ?_ - rwa [div_le_one hreg_pos] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure positive-time width cap from the matrix comparison and an almost-sure -`featureSqNorm ≤ reg` bound along the selected actions. -/ -lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) - (h_feature_le : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by - filter_upwards [h_feature_le] with ω h_feature_leω - intro t ht _ht0 - exact widthQuadraticForm_le_one_of_featureSqNorm_le_reg - (A := A) (reg := reg) (x := x) (n := t) (ω := ω) (A t ω) h_width hreg_pos - (h_feature_leω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure positive-time width cap from the matrix comparison and a selected-feature budget -`featureSqNorm ≤ L2`, when `L2 ≤ reg`. -/ -lemma widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le - (h_width : WidthQuadraticFormLeFeatureSqNormDivReg A reg x) - (hreg_pos : 0 < reg) {L2 : ℝ} - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1 := by - refine widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le_reg - (A := A) (reg := reg) (x := x) (n := n) (P := P) h_width hreg_pos ?_ - filter_upwards [hL2] with ω hL2ω - intro t ht - exact (hL2ω t ht).trans hL2_le_reg - -/-- The process-level elliptical confidence width. -/ -noncomputable def width (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (n : ℕ) (ω : Ω) : ℝ := - √(widthQuadraticForm A reg x a n ω) - -/-- The initial width is the quadratic form induced by the inverse regularized identity. -/ -lemma width_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - width A reg x a 0 ω = - √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [width, widthQuadraticForm_zero] - -/-- Squaring the LinUCB width recovers the quadratic form inside the square root, provided that -quadratic form is nonnegative. -/ -lemma width_sq_eq_quadratic_form (a : Fin K) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x a n ω) : - width A reg x a n ω ^ 2 = widthQuadraticForm A reg x a n ω := by - simp [width, Real.sq_sqrt h_nonneg] - -/-- The accumulated squared LinUCB widths over positive times before horizon `n`. -/ -noncomputable def widthSqSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2 - -/-- No positive-time widths are accumulated at horizon zero. -/ -lemma widthSqSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - widthSqSum A reg x 0 ω = 0 := by - simp [widthSqSum] - -/-- Advancing the horizon adds the next positive-time squared width term. -/ -lemma widthSqSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + - (if n = 0 then 0 else width A reg x (A n ω) n ω) ^ 2 := by - simp [widthSqSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's squared width. -/ -lemma widthSqSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + width A reg x (A n ω) n ω ^ 2 := by - simp [widthSqSum_succ, hn] - -/-- The accumulated quadratic forms corresponding to the positive-time LinUCB widths. -/ -noncomputable def quadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else widthQuadraticForm A reg x (A t ω) t ω - -/-- No positive-time quadratic width forms are accumulated at horizon zero. -/ -lemma quadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - quadraticWidthSum A reg x 0 ω = 0 := by - simp [quadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time quadratic width form. -/ -lemma quadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - if n = 0 then 0 else widthQuadraticForm A reg x (A n ω) n ω := by - simp [quadraticWidthSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's quadratic width form. -/ -lemma quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + widthQuadraticForm A reg x (A n ω) n ω := by - simp [quadraticWidthSum_succ, hn] - -/-- The accumulated capped quadratic forms corresponding to the positive-time LinUCB widths. -/ -noncomputable def cappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω) - -/-- No positive-time capped quadratic width forms are accumulated at horizon zero. -/ -lemma cappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - cappedQuadraticWidthSum A reg x 0 ω = 0 := by - simp [cappedQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time capped quadratic width form. -/ -lemma cappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - cappedQuadraticWidthSum A reg x (n + 1) ω = - cappedQuadraticWidthSum A reg x n ω + - if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by - simp [cappedQuadraticWidthSum, sum_range_succ] - -/-- At positive times, advancing the horizon adds the selected arm's capped quadratic width form. -/ -lemma cappedQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - cappedQuadraticWidthSum A reg x (n + 1) ω = - cappedQuadraticWidthSum A reg x n ω + min 1 (widthQuadraticForm A reg x (A n ω) n ω) := by - simp [cappedQuadraticWidthSum_succ, hn] - -/-- If every positive-time process-level quadratic width form is at most `1`, then the uncapped -and capped process-level quadratic-width accumulators agree. -/ -lemma quadraticWidthSum_eq_cappedQuadraticWidthSum - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - quadraticWidthSum A reg x n ω = cappedQuadraticWidthSum A reg x n ω := by - rw [quadraticWidthSum, cappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact (min_eq_right (h_le_one t ht ht0)).symm - -/-- If the squared-width and quadratic-form accumulators agree through a positive time and the -next quadratic form is nonnegative, then they still agree after adding the next term. -/ -lemma widthSqSum_eq_quadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_eq : widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - widthSqSum A reg x (n + 1) ω = quadraticWidthSum A reg x (n + 1) ω := by - rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn, - quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hn, h_eq] - rw [width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A n ω) - (n := n) (ω := ω) h_nonneg] - -/-- The accumulated squared widths equal the accumulated quadratic forms, provided each positive -time quadratic form is nonnegative. -/ -lemma widthSqSum_eq_sum_quadratic_form - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - widthSqSum A reg x n ω = quadraticWidthSum A reg x n ω := by - rw [widthSqSum, quadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0] - rw [if_neg ht0] - exact width_sq_eq_quadratic_form (A := A) (reg := reg) (x := x) (a := A t ω) - (n := t) (ω := ω) (h_nonneg t ht ht0) - -/-- A quadratic-form sum bound implies the corresponding bound on `widthSqSum`. This is the shape -expected from a later elliptical-potential argument. -/ -lemma widthSqSum_le_of_sum_quadratic_form_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_quad_le : quadraticWidthSum A reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - exact h_quad_le - -/-- A capped process-level quadratic-form sum bound implies the corresponding bound on -`widthSqSum`, provided the positive-time process-level quadratic forms are nonnegative and at most -`1`. -/ -lemma widthSqSum_le_of_capped_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_capped_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - rw [quadraticWidthSum_eq_cappedQuadraticWidthSum (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_le_one] - exact h_capped_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped process-level quadratic-form sum bound implies the corresponding bound -on `widthSqSum`, provided the positive-time process-level quadratic forms are almost surely -nonnegative and at most `1`. -/ -lemma widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_nonneg, h_le_one, h_capped_le] with - ω h_nonnegω h_le_oneω h_capped_leω - exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Determinant of the process-level LinUCB design matrix. -/ -noncomputable def designDet (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - Matrix.det (designMatrix A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial design determinant is the determinant of the regularized identity. -/ -lemma designDet_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - designDet A reg x 0 ω = Matrix.det (reg • (1 : Matrix (Fin d) (Fin d) ℝ)) := by - simp [designDet, designMatrix_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial design determinant is `reg ^ d`. -/ -lemma designDet_zero_eq_reg_pow (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) : - designDet A reg x 0 ω = reg ^ d := by - rw [designDet_zero] - simp - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A nonzero regularization parameter gives a nonzero initial design determinant. -/ -lemma designDet_zero_ne_zero_of_reg_ne_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hreg : reg ≠ 0) : - designDet A reg x 0 ω ≠ 0 := by - rw [designDet_zero_eq_reg_pow] - exact pow_ne_zero d hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization makes every process-level design determinant nonzero. -/ -lemma designDet_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : - designDet A reg x n ω ≠ 0 := by - have hunit : IsUnit (designMatrix A reg x n ω) := - (designMatrix_posDef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_pos).isUnit - have hdet_unit : IsUnit (designMatrix A reg x n ω).det := - (Matrix.isUnit_iff_isUnit_det (A := designMatrix A reg x n ω)).mp hunit - exact hdet_unit.ne_zero - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, positive regularization makes all design determinants in a finite horizon -nonzero. -/ -lemma designDet_ae_ne_zero_of_reg_pos (hreg_pos : 0 < reg) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - exact Filter.Eventually.of_forall fun ω t _ht ↦ - designDet_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) (ω := ω) - hreg_pos - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Determinant ratio `det(V_n) / det(V_0)` for the process-level design matrices. -/ -noncomputable def designDetRatio (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - designDet A reg x n ω / designDet A reg x 0 ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the determinant ratio is `1` when the initial design determinant is nonzero. -/ -lemma designDetRatio_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - designDetRatio A reg x 0 ω = 1 := by - simp [designDetRatio, hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the determinant ratio is positive when the initial design determinant is -nonzero. -/ -lemma designDetRatio_zero_pos (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - 0 < designDetRatio A reg x 0 ω := by - rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - norm_num - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step determinant ratio `det(V_{n+1}) / det(V_n)` for the process-level design matrices. - -This is the determinant-ratio target used by the matrix-determinant part of the elliptical -potential lemma. -/ -noncomputable def designDetStepRatio (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - designDet A reg x (n + 1) ω / designDet A reg x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The scalar determinant appearing in the rank-one determinant update is the quadratic form -`uᵀ M u`. -/ -lemma det_one_add_replicateRow_mul_matrix_mul_replicateCol - (M : Matrix (Fin d) (Fin d) ℝ) (u : Feature d) : - (1 + Matrix.replicateRow Unit u * M * Matrix.replicateCol Unit u).det = - 1 + dotProduct u (Matrix.mulVec M u) := by - have hsum : - (∑ j, (∑ i, u i * M i j) * u j) = - ∑ i, u i * ∑ j, M i j * u j := by - calc - (∑ j, (∑ i, u i * M i j) * u j) - = ∑ j, ∑ i, (u i * M i j) * u j := by - simp [Finset.sum_mul] - _ = ∑ i, ∑ j, (u i * M i j) * u j := by - rw [Finset.sum_comm] - _ = ∑ i, u i * ∑ j, M i j * u j := by - refine Finset.sum_congr rfl ?_ - intro i _ - rw [Finset.mul_sum] - refine Finset.sum_congr rfl ?_ - intro j _ - ring - rw [Matrix.det_unique] - simpa [Matrix.mul_apply, Matrix.replicateRow, Matrix.replicateCol, Matrix.mulVec, - dotProduct] using hsum - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Process-level matrix determinant update for the LinUCB design matrix. - -If `V_n` has nonzero determinant, then the rank-one update -`V_{n+1} = V_n + x_{A_n} x_{A_n}ᵀ` satisfies -`det(V_{n+1}) = det(V_n) * (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ -lemma designDet_succ_eq_mul_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDet A reg x (n + 1) ω = - designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - have hM : IsUnit (designMatrix A reg x n ω).det := by - simpa [designDet] using (isUnit_iff_ne_zero.mpr hdet) - calc - designDet A reg x (n + 1) ω = - (designMatrix A reg x n ω + - Matrix.vecMulVec (x (A n ω)) (x (A n ω))).det := by - simp [designDet, designMatrix_succ] - _ = (designMatrix A reg x n ω + - Matrix.replicateCol Unit (x (A n ω)) * Matrix.replicateRow Unit (x (A n ω))).det := by - rw [Matrix.vecMulVec_eq Unit] - _ = (designMatrix A reg x n ω).det * - (1 + Matrix.replicateRow Unit (x (A n ω)) * - (designMatrix A reg x n ω)⁻¹ * Matrix.replicateCol Unit (x (A n ω))).det := by - exact Matrix.det_add_replicateCol_mul_replicateRow (A := designMatrix A reg x n ω) - (ι := Unit) hM (x (A n ω)) (x (A n ω)) - _ = designDet A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - rw [designDet] - congr 1 - exact det_one_add_replicateRow_mul_matrix_mul_replicateCol - (M := (designMatrix A reg x n ω)⁻¹) (u := x (A n ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `det(V_n)` is nonzero and the selected quadratic form is nonnegative, then -`det(V_{n+1})` is nonzero. -/ -lemma designDet_succ_ne_zero_of_widthQuadraticForm_nonneg - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - designDet A reg x (n + 1) ω ≠ 0 := by - rw [designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - exact mul_ne_zero hdet (ne_of_gt (by linarith)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms preserve -nonzero design determinants up to any fixed time. -/ -lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt - (m : ℕ) (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t < m → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - designDet A reg x m ω ≠ 0 := by - induction m with - | zero => exact hdet0 - | succ m ih => - exact designDet_succ_ne_zero_of_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := m) (ω := ω) - (ih fun t ht ↦ h_nonneg t (Nat.lt_trans ht (Nat.lt_succ_self m))) - (h_nonneg m (Nat.lt_succ_self m)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms imply that -all design determinants through horizon `n` are nonzero. -/ -lemma designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by - intro t ht - exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := t) (ω := ω) hdet0 fun s hs ↦ - h_nonneg s (mem_range.mpr (Nat.lt_of_lt_of_le hs (Nat.le_of_lt_succ (mem_range.mp ht)))) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant and nonnegative selected quadratic forms imply -that all design determinants through horizon `n` are nonzero. -/ -lemma designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω - exact designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If `det(V_n) ≠ 0`, then the one-step determinant ratio is -`1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n}`. -/ -lemma designDetStepRatio_eq_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDetStepRatio A reg x n ω = - 1 + widthQuadraticForm A reg x (A n ω) n ω := by - simp [designDetStepRatio, - designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet, hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The cumulative determinant ratio advances by multiplying by the one-step determinant ratio. -/ -lemma designDetRatio_succ_eq_mul_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - designDetRatio A reg x (n + 1) ω = - designDetRatio A reg x n ω * (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - rw [designDetRatio, designDetRatio, - designDet_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, nonnegative selected quadratic forms make the -cumulative determinant ratio positive. -/ -lemma designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - 0 < designDetRatio A reg x n ω := by - induction n with - | zero => - exact designDetRatio_zero_pos (A := A) (reg := reg) (x := x) (ω := ω) hdet0 - | succ n ih => - have hdetn : designDet A reg x n ω ≠ 0 := - designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ - h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) - rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetn] - exact mul_pos - (ih fun t ht ↦ h_nonneg t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) - (by linarith [h_nonneg n (by simp)]) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, starting from a nonzero initial determinant, nonnegative selected quadratic -forms make the cumulative determinant ratio positive. -/ -lemma designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by - filter_upwards [hdet0, h_nonneg] with ω hdet0ω h_nonnegω - exact designDetRatio_pos_of_initial_and_widthQuadraticForm_nonneg (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter and nonnegative selected quadratic forms make -the cumulative determinant ratio positive. -/ -lemma designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω := by - refine designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Starting from a nonzero initial determinant, the cumulative determinant ratio is the finite -product of the per-round determinant-update factors. -/ -lemma designDetRatio_eq_prod_one_add_widthQuadraticForm - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - designDetRatio A reg x n ω = - ∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω) := by - induction n with - | zero => - rw [designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet0] - simp - | succ n ih => - have hdetn : designDet A reg x n ω ≠ 0 := - designDet_ne_zero_of_initial_and_widthQuadraticForm_nonneg_lt (A := A) (reg := reg) - (x := x) (m := n) (ω := ω) hdet0 fun t ht ↦ - h_nonneg t (mem_range.mpr (Nat.lt_trans ht (Nat.lt_succ_self n))) - rw [designDetRatio_succ_eq_mul_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetn] - rw [ih fun t ht ↦ h_nonneg t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))] - simp [Finset.prod_range_succ] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If every selected quadratic form is in `[0, 1]`, the cumulative determinant ratio is at most -`2 ^ n`. -/ -lemma designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one - (hdet0 : designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ t, t ∈ range n → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - rw [designDetRatio_eq_prod_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet0 h_nonneg] - calc - (∏ t ∈ range n, (1 + widthQuadraticForm A reg x (A t ω) t ω)) - ≤ ∏ _t ∈ range n, (2 : ℝ) := by - exact Finset.prod_le_prod - (fun t ht ↦ by linarith [h_nonneg t ht]) - (fun t ht ↦ by linarith [h_le_one t ht]) - _ = (2 : ℝ) ^ n := by - simp - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, if every selected quadratic form is in `[0, 1]`, the cumulative determinant -ratio is at most `2 ^ n`. -/ -lemma designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - filter_upwards [hdet0, h_nonneg, h_le_one] with ω hdet0ω h_nonnegω h_le_oneω - exact designDetRatio_le_two_pow_of_initial_and_widthQuadraticForm_le_one (A := A) - (reg := reg) (x := x) (n := n) (ω := ω) hdet0ω h_nonnegω h_le_oneω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter and selected quadratic forms in `[0, 1]` -imply the cumulative determinant ratio is at most `2 ^ n`. -/ -lemma designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (2 : ℝ) ^ n := by - refine designDetRatio_ae_le_two_pow_of_initial_and_widthQuadraticForm_ae_le_one - (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Converts an almost-sure trace bound into the determinant-ratio bound expected from a future -trace/determinant comparison theorem. -/ -lemma designDetRatio_ae_le_trace_budget_of_designTrace_ae_le - (T : ℝ) - (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ T → - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - filter_upwards [h_trace_le] with ω h_traceω - exact h_ratio_of_trace ω h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms give the concrete trace budget -`reg * d + n * L2`; a future trace/determinant comparison then gives the corresponding -determinant-ratio bound. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - exact designDetRatio_ae_le_trace_budget_of_designTrace_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) (T := reg * (d : ℝ) + (n : ℝ) * L2) - (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2) - h_ratio_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A determinant upper bound for `V_n` implies the corresponding determinant-ratio bound, using -`det(V_0) = reg ^ d`. -/ -lemma designDetRatio_le_trace_budget_of_designDet_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hdet_le : designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - rw [designDetRatio, designDet_zero_eq_reg_pow] - have hreg_pow_nonneg : 0 ≤ reg ^ d := (pow_pos hreg_pos d).le - have hdiv : designDet A reg x n ω / reg ^ d ≤ (T / (d : ℝ)) ^ d / reg ^ d := by - exact div_le_div_of_nonneg_right hdet_le hreg_pow_nonneg - refine hdiv.trans_eq ?_ - rw [← div_pow] - congr 1 - field_simp [hreg_pos.ne', by exact_mod_cast hd] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a determinant upper bound for `V_n` implies the corresponding determinant-ratio -bound. -/ -lemma designDetRatio_ae_le_trace_budget_of_designDet_ae_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hdet_le : ∀ᵐ ω ∂P, designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - filter_upwards [hdet_le] with ω hdetω - exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) T hreg_pos hd hdetω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Converts an almost-sure trace bound plus a future determinant/trace comparison for `det(V_n)` -into the determinant-ratio bound used by the elliptical-potential chain. -/ -lemma designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le - (T : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_trace_le : ∀ᵐ ω ∂P, designTrace A reg x n ω ≤ T) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ T → designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d := by - refine designDetRatio_ae_le_trace_budget_of_designDet_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) T hreg_pos hd ?_ - filter_upwards [h_trace_le] with ω h_traceω - exact hdet_of_trace ω h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms reduce the determinant-ratio goal to the determinant upper bound -`det(V_n) ≤ ((reg * d + n * L2) / d) ^ d`. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le - (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - exact designDetRatio_ae_le_trace_budget_of_designDet_le_of_designTrace_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd - (designTrace_ae_le_reg_mul_dim_add_nat_mul_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2) - hdet_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Matrix-level determinant/trace comparison needed for the finite-dimensional -elliptical-potential bound. - -For positive semidefinite `d × d` matrices, this is the AM-GM-style inequality -`det(M) ≤ (trace(M) / d) ^ d`. -/ -def MatrixDetLeTraceAveragePow (d : ℕ) : Prop := - ∀ M : Matrix (Fin d) (Fin d) ℝ, M.PosSemidef → M.det ≤ (M.trace / (d : ℝ)) ^ d - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar AM-GM in the form used for PSD matrix eigenvalues: -the product of nonnegative entries is bounded by the arithmetic mean to the `card` power. -/ -lemma prod_le_average_pow_of_nonneg {ι : Type*} [Fintype ι] [Nonempty ι] - (z : ι → ℝ) (hz : ∀ i, 0 ≤ z i) : - (∏ i, z i) ≤ ((∑ i, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by - classical - have hN_pos : 0 < (Fintype.card ι : ℝ) := by - exact_mod_cast Fintype.card_pos_iff.mpr inferInstance - have hweights_pos : 0 < ∑ i : ι, (1 : ℝ) := by - simpa using hN_pos - have h_amgm := Real.geom_mean_le_arith_mean (s := Finset.univ) - (w := fun _ : ι ↦ (1 : ℝ)) (z := z) - (by intro i hi; norm_num) hweights_pos (by intro i hi; exact hz i) - have h_amgm' : - (∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹) ≤ - (∑ i : ι, z i) / (Fintype.card ι : ℝ) := by - simpa using h_amgm - have hprod_nonneg : 0 ≤ ∏ i : ι, z i := by - exact Finset.prod_nonneg fun i _ ↦ hz i - have hraise := Real.rpow_le_rpow (Real.rpow_nonneg hprod_nonneg _) h_amgm' hN_pos.le - have hleft : - ((∏ i : ι, z i) ^ ((Fintype.card ι : ℝ)⁻¹)) ^ (Fintype.card ι : ℝ) = - ∏ i : ι, z i := by - rw [← Real.rpow_mul hprod_nonneg] - rw [inv_mul_cancel₀ hN_pos.ne'] - simp - have hright : - ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ (Fintype.card ι : ℝ) = - ((∑ i : ι, z i) / (Fintype.card ι : ℝ)) ^ Fintype.card ι := by - rw [Real.rpow_natCast] - simpa [hleft, hright] using hraise - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- PSD matrix determinant/trace comparison from AM-GM over eigenvalues: -`det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma matrixDetLeTraceAveragePow : MatrixDetLeTraceAveragePow d := by - intro M hM - by_cases hd : d = 0 - · subst d - simp - · haveI : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp (Nat.pos_of_ne_zero hd) - rw [hM.1.det_eq_prod_eigenvalues, hM.1.trace_eq_sum_eigenvalues] - simpa using prod_le_average_pow_of_nonneg - (z := fun i : Fin d ↦ hM.1.eigenvalues i) - (fun i ↦ hM.eigenvalues_nonneg i) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A matrix-level determinant/trace comparison applies to the LinUCB design matrix because the -design matrix is positive semidefinite. -/ -lemma designDet_le_trace_average_pow_of_matrix_det_trace_bound - (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) : - designDet A reg x n ω ≤ (designTrace A reg x n ω / (d : ℝ)) ^ d := by - simpa [designDet, designTrace] using - hdet_trace (designMatrix A reg x n ω) - (designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Combining `det(M) ≤ (trace(M)/d)^d` with a trace budget gives the determinant upper bound -`det(V_n) ≤ (T/d)^d`. -/ -lemma designDet_le_trace_budget_of_matrix_det_trace_bound - (hdet_trace : MatrixDetLeTraceAveragePow d) (hreg_nonneg : 0 ≤ reg) - (hd : d ≠ 0) (T : ℝ) (h_trace_le : designTrace A reg x n ω ≤ T) : - designDet A reg x n ω ≤ (T / (d : ℝ)) ^ d := by - have hd_pos : 0 < (d : ℝ) := by - exact_mod_cast Nat.pos_of_ne_zero hd - have hbase_nonneg : 0 ≤ designTrace A reg x n ω / (d : ℝ) := - div_nonneg (designTrace_nonneg (A := A) (reg := reg) (x := x) (n := n) (ω := ω) - hreg_nonneg) hd_pos.le - have hbase_le : designTrace A reg x n ω / (d : ℝ) ≤ T / (d : ℝ) := - (div_le_div_iff_of_pos_right hd_pos).mpr h_trace_le - exact (designDet_le_trace_average_pow_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet_trace hreg_nonneg).trans - (pow_le_pow_left₀ hbase_nonneg hbase_le d) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Bounded selected feature norms and the matrix-level determinant/trace comparison give the -determinant-ratio bound used by the elliptical-potential chain. -/ -lemma designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound - (L2 : ℝ) (hreg_pos : 0 < reg) (hd : d ≠ 0) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d := by - refine designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 ?_ - intro ω h_traceω - exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) - hdet_trace hreg_pos.le hd h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The log-determinant expression that appears in the elliptical-potential lemma. -/ -noncomputable def ellipticalPotential (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - 2 * Real.log (designDetRatio A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A positive determinant ratio bounded by `D` gives the corresponding log-determinant potential -bound. -/ -lemma ellipticalPotential_le_two_mul_log_of_designDetRatio_le {D : ℝ} - (h_ratio_pos : 0 < designDetRatio A reg x n ω) - (h_ratio_le : designDetRatio A reg x n ω ≤ D) : - ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by - rw [ellipticalPotential] - exact mul_le_mul_of_nonneg_left (Real.log_le_log h_ratio_pos h_ratio_le) (by norm_num) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a positive determinant ratio bounded by `D` gives the corresponding -log-determinant potential bound. -/ -lemma ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le {D : ℝ} - (h_ratio_pos : ∀ᵐ ω ∂P, 0 < designDetRatio A reg x n ω) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ 2 * Real.log D := by - filter_upwards [h_ratio_pos, h_ratio_le] with ω h_ratio_posω h_ratio_leω - exact ellipticalPotential_le_two_mul_log_of_designDetRatio_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_ratio_posω h_ratio_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step log-determinant potential term based on `det(V_{n+1}) / det(V_n)`. - -The future determinant-update proof should naturally establish the capped quadratic-width term is -bounded by this quantity. A separate log/telescoping bridge then connects this one-step quantity to -`ellipticalPotentialIncrement`. -/ -noncomputable def ellipticalPotentialStep (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - 2 * Real.log (designDetStepRatio A reg x n ω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing, the one-step log-determinant potential is -`2 * log (1 + x_{A_n}ᵀ V_n⁻¹ x_{A_n})`. -/ -lemma ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm - (hdet : designDet A reg x n ω ≠ 0) : - ellipticalPotentialStep A reg x n ω = - 2 * Real.log (1 + widthQuadraticForm A reg x (A n ω) n ω) := by - simp [ellipticalPotentialStep, - designDetStepRatio_eq_one_add_widthQuadraticForm (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar log inequality used in the elliptical-potential proof: for `0 ≤ q ≤ 1`, -`min 1 q ≤ 2 * log (1 + q)`. -/ -lemma min_one_le_two_mul_log_one_add_of_nonneg_le_one {q : ℝ} - (hq_nonneg : 0 ≤ q) (hq_le_one : q ≤ 1) : - min 1 q ≤ 2 * Real.log (1 + q) := by - have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := - Real.le_log_one_add_of_nonneg hq_nonneg - have hq_add_two_pos : 0 < q + 2 := by linarith - have hq_le_two : q ≤ 2 := by linarith - have hq_le_log_lower : q ≤ 2 * (2 * q / (q + 2)) := by - rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] - rw [le_div_iff₀ hq_add_two_pos] - nlinarith - rw [min_eq_right hq_le_one] - exact hq_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Scalar log inequality used in the textbook elliptical-potential proof: for `0 ≤ q`, -`min 1 q ≤ 2 * log (1 + q)`. -/ -lemma min_one_le_two_mul_log_one_add_of_nonneg {q : ℝ} - (hq_nonneg : 0 ≤ q) : - min 1 q ≤ 2 * Real.log (1 + q) := by - by_cases hq_le_one : q ≤ 1 - · exact min_one_le_two_mul_log_one_add_of_nonneg_le_one hq_nonneg hq_le_one - · have hq_one : 1 ≤ q := by linarith - have hlog : 2 * q / (q + 2) ≤ Real.log (1 + q) := - Real.le_log_one_add_of_nonneg hq_nonneg - have hq_add_two_pos : 0 < q + 2 := by linarith - have hone_le_log_lower : 1 ≤ 2 * (2 * q / (q + 2)) := by - rw [show 2 * (2 * q / (q + 2)) = 4 * q / (q + 2) by ring] - rw [le_div_iff₀ hq_add_two_pos] - nlinarith - rw [min_eq_left hq_one] - exact hone_le_log_lower.trans (mul_le_mul_of_nonneg_left hlog (by norm_num)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing and the usual `0 ≤ q ≤ 1` quadratic-form side conditions, the -single capped quadratic-width term is bounded by the one-step log-determinant potential. -/ -lemma cappedWidthTerm_le_ellipticalPotentialStep - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) - (h_le_one : n ≠ 0 → widthQuadraticForm A reg x (A n ω) n ω ≤ 1) : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialStep A reg x n ω := by - by_cases hn : n = 0 - · rw [if_pos hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) - · rw [if_neg hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact min_one_le_two_mul_log_one_add_of_nonneg_le_one h_nonneg (h_le_one hn) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Under determinant nonvanishing and nonnegativity of the selected quadratic form, the single -capped quadratic-width term is bounded by the one-step log-determinant potential. This is the -textbook form; no separate `q ≤ 1` assumption is needed because the term is already capped. -/ -lemma cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg - (hdet : designDet A reg x n ω ≠ 0) - (h_nonneg : 0 ≤ widthQuadraticForm A reg x (A n ω) n ω) : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialStep A reg x n ω := by - by_cases hn : n = 0 - · rw [if_pos hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact mul_nonneg (by norm_num) (Real.log_nonneg (by linarith)) - · rw [if_neg hn, - ellipticalPotentialStep_eq_two_mul_log_one_add_widthQuadraticForm (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet] - exact min_one_le_two_mul_log_one_add_of_nonneg h_nonneg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and the standard quadratic-form side conditions imply -the per-step one-step-potential bound required by the elliptical-potential induction shell. -/ -lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω := by - filter_upwards [hdet, h_nonneg, h_le_one] with ω hdetω h_nonnegω h_le_oneω - intro t ht - exact cappedWidthTerm_le_ellipticalPotentialStep (A := A) (reg := reg) (x := x) - (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) (h_le_oneω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the -per-step one-step-potential bound for the capped quadratic-width term. -/ -lemma cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω := by - filter_upwards [hdet, h_nonneg] with ω hdetω h_nonnegω - intro t ht - exact cappedWidthTerm_le_ellipticalPotentialStep_of_nonneg (A := A) (reg := reg) - (x := x) (n := t) (ω := ω) (hdetω t ht) (h_nonnegω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- At horizon zero, the log-determinant potential is zero when the initial design determinant is -nonzero. -/ -lemma ellipticalPotential_zero (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (ω : Ω) (hdet : designDet A reg x 0 ω ≠ 0) : - ellipticalPotential A reg x 0 ω = 0 := by - simp [ellipticalPotential, designDetRatio_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the log-determinant elliptical-potential inequality. At horizon zero there are no -positive-time capped quadratic width forms, and the log-determinant potential is zero when the -initial design determinant is nonzero. -/ -lemma cappedQuadraticWidthSum_le_ellipticalPotential_zero - (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) - (hdet : designDet A reg x 0 ω ≠ 0) : - cappedQuadraticWidthSum A reg x 0 ω ≤ ellipticalPotential A reg x 0 ω := by - rw [cappedQuadraticWidthSum_zero (A := A) (reg := reg) (x := x) (ω := ω), - ellipticalPotential_zero (A := A) (reg := reg) (x := x) (ω := ω) hdet] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- One-step increment of the log-determinant elliptical potential. -/ -noncomputable def ellipticalPotentialIncrement (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ellipticalPotential A reg x (n + 1) ω - ellipticalPotential A reg x n ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The one-step determinant-ratio potential equals the increment of the cumulative -log-determinant potential, provided the relevant design determinants are nonzero. -/ -lemma ellipticalPotentialStep_eq_increment - (hdet0 : designDet A reg x 0 ω ≠ 0) - (hdetn : designDet A reg x n ω ≠ 0) - (hdet_succ : designDet A reg x (n + 1) ω ≠ 0) : - ellipticalPotentialStep A reg x n ω = ellipticalPotentialIncrement A reg x n ω := by - simp [ellipticalPotentialStep, designDetStepRatio, ellipticalPotentialIncrement, - ellipticalPotential, designDetRatio, Real.log_div hdet_succ hdetn, - Real.log_div hdet_succ hdet0, Real.log_div hdetn hdet0] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the one-step determinant-ratio potential equals the increment of the cumulative -log-determinant potential throughout the finite horizon, provided all determinants up to that -horizon are nonzero almost surely. -/ -lemma ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω = ellipticalPotentialIncrement A reg x t ω := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact ellipticalPotentialStep_eq_increment (A := A) (reg := reg) (x := x) (n := t) - (ω := ω) (hdetω 0 (by simp)) - (hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n)))) - (hdetω (t + 1) (mem_range.mpr (Nat.succ_lt_succ (mem_range.mp ht)))) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- If the next capped quadratic width term is bounded by the next log-determinant potential -increment, then the cumulative capped-sum/log-det inequality advances by one step. -/ -lemma cappedQuadraticWidthSum_succ_le_ellipticalPotential - (h_prev : cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_step : - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) ≤ - ellipticalPotentialIncrement A reg x n ω) : - cappedQuadraticWidthSum A reg x (n + 1) ω ≤ ellipticalPotential A reg x (n + 1) ω := by - rw [cappedQuadraticWidthSum_succ (A := A) (reg := reg) (x := x) (n := n) (ω := ω)] - calc - cappedQuadraticWidthSum A reg x n ω + - (if n = 0 then 0 else min 1 (widthQuadraticForm A reg x (A n ω) n ω)) - ≤ ellipticalPotential A reg x n ω + ellipticalPotentialIncrement A reg x n ω := by - exact add_le_add h_prev h_step - _ = ellipticalPotential A reg x (n + 1) ω := by - simp [ellipticalPotentialIncrement] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A per-step bound by log-determinant potential increments implies the cumulative -elliptical-potential inequality. This is the induction shell for the future determinant-update -proof. -/ -lemma cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le - (hdet : designDet A reg x 0 ω ≠ 0) : - (∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) → - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - induction n with - | zero => - intro _ - exact cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) - (x := x) (ω := ω) hdet - | succ n ih => - intro h_step - refine cappedQuadraticWidthSum_succ_le_ellipticalPotential (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) ?_ ?_ - · exact ih fun t ht ↦ h_step t - (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - · exact h_step n (by simp) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a per-step bound by log-determinant potential increments implies the cumulative -elliptical-potential inequality. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - filter_upwards [hdet, h_step] with ω hdetω h_stepω - exact cappedQuadraticWidthSum_le_ellipticalPotential_of_step_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdetω h_stepω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the -cumulative capped-sum/log-det inequality, provided the one-step determinant-ratio potential is -bounded by the corresponding cumulative-potential increment. - -This separates the future elliptical-potential proof into two local obligations: - -* a matrix-determinant update bounding the selected arm's capped quadratic form by - `ellipticalPotentialStep`; -* a log/telescoping bridge from `ellipticalPotentialStep` to `ellipticalPotentialIncrement`. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet ?_ - filter_upwards [h_step, h_step_le_increment] with ω h_stepω h_step_le_incrementω - intro t ht - exact (h_stepω t ht).trans (h_step_le_incrementω t ht) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by the one-step determinant-ratio potential imply the -cumulative capped-sum/log-det inequality when all design determinants up to the horizon are nonzero -almost surely. - -Compared with `cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le`, this -version discharges the log/telescoping bridge automatically from determinant nonvanishing. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - have hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - exact hdetω 0 (by simp) - refine cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_step ?_ - filter_upwards [ellipticalPotentialStep_ae_eq_increment_of_det_ne_zero (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet] with ω h_eq - intro t ht - rw [h_eq t ht] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, determinant nonvanishing and nonnegative selected quadratic forms imply the -capped-sum/log-determinant elliptical-potential bound. - -This is the capped form used in the textbook proof of LinUCB: the quadratic forms do not need to -be bounded by `1`, because the accumulated quantity is `min 1 q_t`. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet - (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero_of_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet_range_n h_nonneg) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization discharges determinant nonvanishing and nonnegativity, yielding the -capped-sum/log-determinant elliptical-potential bound directly. -/ -lemma cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos - (hreg_pos : 0 < reg) : - ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω := by - exact cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_det_ne_zero_and_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) - (n := n + 1) (P := P) hreg_pos) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The process-level capped quadratic-width input expected from an elliptical-potential argument. - -It packages the three facts needed to turn a capped process-level quadratic-width estimate into the -`widthSqSum` estimate used by the regret chain: - -* each positive-time process-level quadratic width form is nonnegative; -* each positive-time process-level quadratic width form is at most `1`; -* their capped process-level accumulated sum is bounded by `W`. -/ -def CappedQuadraticWidthBound (A : ℕ → Ω → Fin K) (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := - (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ∧ - (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ∧ - cappedQuadraticWidthSum A reg x n ω ≤ W - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Build the packaged process-level capped quadratic-width input from its component facts. -/ -lemma cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_sum_le : cappedQuadraticWidthSum A reg x n ω ≤ W) : - CappedQuadraticWidthBound A reg x n ω W := by - exact ⟨h_nonneg, h_le_one, h_sum_le⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the packaged process-level capped quadratic-width input. At horizon zero, the -nonnegativity and `≤ 1` side conditions are vacuous, and the capped sum is zero. -/ -lemma cappedQuadraticWidthBound_zero {W : ℝ} (hW : 0 ≤ W) : - CappedQuadraticWidthBound A reg x 0 ω W := by - refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ - · intro t ht _ - simp at ht - · intro t ht _ - simp at ht - · simpa [cappedQuadraticWidthSum_zero] using hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Base case for the packaged process-level capped quadratic-width input when the constant bound -is supplied through the log-determinant potential. -/ -lemma cappedQuadraticWidthBound_zero_of_ellipticalPotential_le_bound {W : ℝ} - (hdet : designDet A reg x 0 ω ≠ 0) (h_potential_le : ellipticalPotential A reg x 0 ω ≤ W) : - CappedQuadraticWidthBound A reg x 0 ω W := by - refine cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := 0) (ω := ω) ?_ ?_ ?_ - · intro t ht _ - simp at ht - · intro t ht _ - simp at ht - · exact (cappedQuadraticWidthSum_le_ellipticalPotential_zero (A := A) (reg := reg) - (x := x) (ω := ω) hdet).trans h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged process-level capped quadratic-width input is monotone in the numeric bound. -/ -lemma cappedQuadraticWidthBound_mono {W W' : ℝ} - (h_bound : CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : - CappedQuadraticWidthBound A reg x n ω W' := by - exact ⟨h_bound.1, h_bound.2.1, h_bound.2.2.trans hW⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, build the packaged process-level capped quadratic-width input from its component -facts. -/ -lemma cappedQuadraticWidthBound_ae_of_nonneg_le_one_and_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_sum_le : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_sum_le] with - ω h_nonnegω h_le_oneω h_sum_leω - exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_sum_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged process-level capped quadratic-width input is monotone in the -numeric bound. -/ -lemma cappedQuadraticWidthBound_ae_mono {W W' : ℝ} - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) (hW : W ≤ W') : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W' := by - filter_upwards [h_bound] with ω h_boundω - exact cappedQuadraticWidthBound_mono (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) h_boundω hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped-sum bound by the log-determinant potential, together with a constant bound on that -potential, gives the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_of_ellipticalPotential_le_bound {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_potential_le : ellipticalPotential A reg x n ω ≤ W) : - CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_of_nonneg_le_one_and_sum_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonneg h_le_one (h_elliptical.trans h_potential_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped-sum bound by the log-determinant potential and an almost-sure constant -bound on that potential give the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_elliptical : ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ ellipticalPotential A reg x n ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_elliptical, h_potential_le] with - ω h_nonnegω h_le_oneω h_ellipticalω h_potential_leω - exact cappedQuadraticWidthBound_of_ellipticalPotential_le_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_ellipticalω h_potential_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, per-step bounds by log-determinant potential increments and a final constant -bound on the potential give the packaged process-level capped quadratic-width input. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_step_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialIncrement A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_step_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet h_step) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, one-step determinant-ratio potential bounds, their bridge to cumulative -potential increments, and a final constant bound on the potential give the packaged process-level -capped quadratic-width input. - -This is the packaged form of the determinant-update interface: once the true matrix determinant -lemma proves the `h_step` assumption and the log/telescoping algebra proves -`h_step_le_increment`, the existing regret chain can consume the resulting bound. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_step_le_increment : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - ellipticalPotentialStep A reg x t ω ≤ ellipticalPotentialIncrement A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet h_step h_step_le_increment) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, one-step determinant-ratio potential bounds, determinant nonvanishing up to the -horizon, and a final constant bound on the potential give the packaged process-level capped -quadratic-width input. - -This is the determinant-nonvanishing version of the one-step interface: the remaining hard -elliptical-potential work is to prove the one-step matrix inequality and the final -log-determinant bound. -/ -lemma cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero - {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_step : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - (if t = 0 then 0 else min 1 (widthQuadraticForm A reg x (A t ω) t ω)) ≤ - ellipticalPotentialStep A reg x t ω) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_ae_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - (cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_stepPotential_ae_le_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) hdet h_step) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the determinant-update step, determinant nonvanishing up to the horizon, and a -final constant bound on the log-determinant potential give the packaged capped quadratic-width -input used by the regret chain. - -The assumptions now match the concrete obligations left for a full elliptical-potential proof: - -* prove all relevant design determinants are nonzero; -* prove selected quadratic forms are nonnegative and at most `1` at positive times; -* prove the final log-determinant potential is at most `W`. -/ -lemma cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound {W : ℝ} - (hdet : ∀ᵐ ω ∂P, ∀ t, t ∈ range (n + 1) → designDet A reg x t ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - have hdet_range_n : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → designDet A reg x t ω ≠ 0 := by - filter_upwards [hdet] with ω hdetω - intro t ht - exact hdetω t (mem_range.mpr (Nat.lt_trans (mem_range.mp ht) (Nat.lt_succ_self n))) - have h_nonneg_positive : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - filter_upwards [h_nonneg] with ω h_nonnegω - intro t ht _ - exact h_nonnegω t ht - exact cappedQuadraticWidthBound_ae_of_ellipticalPotential_stepPotential_ae_le_bound_of_det_ne_zero - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) - h_nonneg_positive h_le_one hdet - (cappedWidthTerm_ae_le_ellipticalPotentialStep_of_det_ne_zero (A := A) (reg := reg) - (x := x) (n := n) (P := P) hdet_range_n h_nonneg h_le_one) - h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant, the determinant-update step, and a final constant -bound on the log-determinant potential give the packaged capped quadratic-width input used by the -regret chain. - -This removes the need to assume determinant nonvanishing at every time: it is derived inductively -from `det(V_0) ≠ 0` and nonnegative selected quadratic forms. -/ -lemma cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound {W : ℝ} - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound (A := A) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (designDet_ae_ne_zero_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) - h_nonneg h_le_one h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter, the determinant-update step, and a final -constant bound on the log-determinant potential give the packaged capped quadratic-width input used -by the regret chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_ellipticalPotential_le_bound {W : ℝ} - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - refine cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) ?_ h_nonneg h_le_one - h_potential_le - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Positive regularization discharges the determinant-nonvanishing and quadratic-form -nonnegativity obligations in the log-determinant elliptical-potential chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound {W : ℝ} - (hreg_pos : 0 < reg) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W := by - exact cappedQuadraticWidthBound_ae_of_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) - (designDet_ae_ne_zero_of_reg_pos (A := A) (reg := reg) (x := x) - (n := n + 1) (P := P) hreg_pos) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - h_le_one h_potential_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero initial determinant, nonnegative selected quadratic forms, a -determinant-ratio upper bound, and the determinant-update step give the packaged capped -quadratic-width input used by the regret chain. - -This version accepts the determinant-ratio bound directly and converts it into the -`ellipticalPotential ≤ 2 * log D` bound internally. -/ -lemma cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound {D : ℝ} - (hdet0 : ∀ᵐ ω ∂P, designDet A reg x 0 ω ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by - exact cappedQuadraticWidthBound_ae_of_initial_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := 2 * Real.log D) - hdet0 h_nonneg h_le_one - (ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (designDetRatio_ae_pos_of_initial_and_widthQuadraticForm_ae_nonneg (A := A) - (reg := reg) (x := x) (n := n) (P := P) hdet0 h_nonneg) - h_ratio_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a nonzero regularization parameter, nonnegative selected quadratic forms, a -determinant-ratio upper bound, and the determinant-update step give the packaged capped -quadratic-width input used by the regret chain. - -This is the most direct interface for the final determinant-bound part of the finite-action -elliptical-potential argument: after proving `designDetRatio ≤ D`, the theorem supplies the -`CappedQuadraticWidthBound` with bound `2 * log D`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound {D : ℝ} - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_ratio_le : ∀ᵐ ω ∂P, designDetRatio A reg x n ω ≤ D) : - ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω (2 * Real.log D) := by - refine cappedQuadraticWidthBound_ae_of_initial_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) ?_ h_nonneg h_le_one h_ratio_le - exact Filter.Eventually.of_forall fun ω ↦ - designDet_zero_ne_zero_of_reg_ne_zero (A := A) (reg := reg) (x := x) (ω := ω) hreg - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A simple explicit determinant-ratio bound for the capped quadratic-width input. - -If `reg ≠ 0` and every selected quadratic form is almost surely in `[0, 1]`, then the determinant -ratio is at most `2 ^ n`, so the existing determinant-update/elliptical-potential chain gives the -packaged capped-width bound with budget `2 * log (2 ^ n)`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_two_pow_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω (2 * Real.log ((2 : ℝ) ^ n)) := by - refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (D := (2 : ℝ) ^ n) - hreg h_nonneg ?_ ?_ - · filter_upwards [h_le_one] with ω h_le_oneω - exact fun t ht _ ↦ h_le_oneω t ht - · exact designDetRatio_ae_le_two_pow_of_reg_ne_zero_and_widthQuadraticForm_ae_le_one - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Trace-budget interface for the determinant part of the finite-action elliptical-potential -argument. - -The future spectral/AM-GM determinant theorem should prove the hypothesis -`designDetRatio ≤ (T / (reg * d)) ^ d`, where `T` is an upper bound on `trace(V_n)`. This theorem -then feeds that determinant-ratio bound into the already-proved determinant-update and -elliptical-potential chain. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (T : ℝ) - (h_ratio_le : ∀ᵐ ω ∂P, - designDetRatio A reg x n ω ≤ (T / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * Real.log ((T / (reg * (d : ℝ))) ^ d)) := by - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_designDetRatio_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (D := (T / (reg * (d : ℝ))) ^ d) hreg h_nonneg h_le_one h_ratio_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface for the determinant part of the finite-action -elliptical-potential argument. - -If selected feature vectors have squared norm at most `L2`, then `trace(V_n) ≤ reg * d + n * L2`. -Given a future deterministic trace/determinant comparison that turns this trace budget into the -determinant-ratio bound, this theorem supplies the packaged capped-width input with the explicit -budget `2 * log (((reg * d + n * L2) / (reg * d)) ^ d)`. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound - (hreg : reg ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d)) := by - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_trace_budget_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) - (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg h_nonneg h_le_one - (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound (A := A) (reg := reg) - (x := x) (n := n) (P := P) L2 hL2 h_ratio_of_trace) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The explicit feature-norm determinant budget can be rewritten in the common -`d * log(1 + n L² / (reg d))` form. -/ -lemma featureSqNorm_budget_log_eq_dim_mul_log_one_add - (L2 : ℝ) (hden : reg * (d : ℝ) ≠ 0) : - 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) = - 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by - have hbase : - (reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ)) = - 1 + (n : ℝ) * L2 / (reg * (d : ℝ)) := by - exact same_add_div hden - rw [Real.log_pow, hbase] - ring - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Textbook capped elliptical-potential budget from bounded selected feature norms and the -matrix-level determinant/trace comparison. - -Unlike `cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound`, this theorem bounds the capped -quadratic-width sum directly and does not assume the individual quadratic forms are at most `1`. -/ -lemma cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - cappedQuadraticWidthSum A reg x n ω ≤ - 2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))) := by - have hden : reg * (d : ℝ) ≠ 0 := by - exact mul_ne_zero hreg_pos.ne' (by exact_mod_cast hd) - rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] - have h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := - widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le - have h_potential_le : ∀ᵐ ω ∂P, - ellipticalPotential A reg x n ω ≤ - 2 * Real.log (((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) := by - exact ellipticalPotential_ae_le_two_mul_log_of_designDetRatio_ae_le (A := A) - (reg := reg) (x := x) (n := n) (P := P) - (designDetRatio_ae_pos_of_reg_ne_zero_and_widthQuadraticForm_ae_nonneg - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' h_nonneg) - (designDetRatio_ae_le_trace_budget_of_featureSqNorm_bound_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) L2 hreg_pos hd hL2 - hdet_trace) - filter_upwards [cappedQuadraticWidthSum_ae_le_ellipticalPotential_of_reg_pos - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos, h_potential_le] with - ω h_capped_le h_potentialω - exact h_capped_le.trans h_potentialω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface with the log term rewritten in the standard -`2 * d * log(1 + n L² / (reg d))` shape. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (hreg : reg ≠ 0) (hd : d ≠ 0) - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - have hden : reg * (d : ℝ) ≠ 0 := by - exact mul_ne_zero hreg (by exact_mod_cast hd) - rw [← featureSqNorm_budget_log_eq_dim_mul_log_one_add (reg := reg) (n := n) L2 hden] - exact cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg h_nonneg h_le_one L2 hL2 - h_ratio_of_trace - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface with the determinant/trace comparison stated as a determinant -upper bound for `V_n`, rather than directly as a determinant-ratio bound. -/ -lemma cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - refine cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) - (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' - (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) - (x := x) hreg_pos h_inv_antitone)) - hreg_pos hL2 hL2_le_reg) - L2 hL2 ?_ - intro ω h_traceω - exact designDetRatio_le_trace_budget_of_designDet_le (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) (T := reg * (d : ℝ) + (n : ℝ) * L2) hreg_pos hd - (hdet_of_trace ω h_traceω) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Feature-norm-budget interface where the remaining matrix-analysis input is the reusable -positive-semidefinite determinant/trace comparison `det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - CappedQuadraticWidthBound A reg x n ω - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) := by - refine - cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg ?_ - intro ω h_traceω - exact designDet_le_trace_budget_of_matrix_det_trace_bound (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hdet_trace hreg_pos.le hd - (reg * (d : ℝ) + (n : ℝ) * L2) h_traceω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged process-level capped quadratic-width input implies the `widthSqSum` bound consumed -by the regret chain. -/ -lemma widthSqSum_le_of_capped_quadratic_width_bound {W : ℝ} - (h_bound : CappedQuadraticWidthBound A reg x n ω W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_capped_quadratic_width_sum_le (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_bound.1 h_bound.2.1 h_bound.2.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged process-level capped quadratic-width input implies the `widthSqSum` -bound consumed by the regret chain. -/ -lemma widthSqSum_ae_le_of_capped_quadratic_width_bound_ae {W : ℝ} - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_bound] with ω h_boundω - exact widthSqSum_le_of_capped_quadratic_width_bound (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) (W := W) h_boundω - -/-- The process-level LinUCB optimistic index. -/ -noncomputable def index (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) - (n : ℕ) (ω : Ω) : ℝ := - estimatedReward A R reg x a n ω + √(β (n + 1)) * width A reg x a n ω - -/-- At time zero, the LinUCB index is only the confidence bonus because the estimated reward is -zero. -/ -lemma index_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - index A R reg β x a 0 ω = √(β 1) * width A reg x a 0 ω := by - simp [index, estimatedReward_zero] - -/-- At time zero, the LinUCB index is the confidence schedule times the initial quadratic-form -width. -/ -lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) (a : Fin K) (ω : Ω) : - index A R reg β x a 0 ω = - √(β 1) * √(dotProduct (x a) (Matrix.mulVec (reg • 1)⁻¹ (x a))) := by - simp [index_zero, width_zero] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every least-squares reward estimate is zero. -/ -lemma estimatedReward_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - estimatedReward A R reg x a n ω = 0 := by - subst d - simp [estimatedReward, dotProduct] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB quadratic width form is zero. -/ -lemma widthQuadraticForm_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - widthQuadraticForm A reg x a n ω = 0 := by - subst d - simp [widthQuadraticForm, dotProduct] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB width is zero. -/ -lemma width_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - width A reg x a n ω = 0 := by - simp [width, widthQuadraticForm_eq_zero_of_dim_eq_zero (A := A) (reg := reg) - (x := x) (n := n) (ω := ω) hd a] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- In zero feature dimension, every LinUCB index is zero. -/ -lemma index_eq_zero_of_dim_eq_zero (hd : d = 0) (a : Fin K) : - index A R reg β x a n ω = 0 := by - simp [index, estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) hd a, - width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hd a] - -/-- The pointwise LinUCB confidence event used by the finite-action regret proof. - -For every positive process time, the best arm's true mean lies below its optimistic index, and the -selected arm's pessimistic index lies below its true mean. On this event, the max-index property of -LinUCB turns optimism into an instantaneous regret bound. -/ -def LinUCBConfidenceEvent [Nonempty (Fin K)] - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω ∧ - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] - -omit [IsMarkovKernel ν] in -/-- Uniform bound on arm gaps, used as the finite-action analogue of the textbook bounded -instantaneous-regret assumption. -/ -def GapBound (ν : Kernel (Fin K) ℝ) (G : ℝ) : Prop := - ∀ a, gap ν a ≤ G - -omit [IsMarkovKernel ν] in -/-- Uniform bound on arm means. For finite-action linear bandits this is a convenient way to state -the usual bounded expected-reward assumption, for example `(ν a)[id] ∈ [-1, 1]`. -/ -def MeanRewardBound (ν : Kernel (Fin K) ℝ) (lo hi : ℝ) : Prop := - ∀ a, lo ≤ (ν a)[id] ∧ (ν a)[id] ≤ hi - -omit [IsMarkovKernel ν] in -/-- If every arm mean lies in `[lo, hi]`, then every arm gap is at most `hi - lo`. -/ -lemma gap_le_of_meanRewardBound [Nonempty (Fin K)] {lo hi : ℝ} - (hμ : MeanRewardBound ν lo hi) (a : Fin K) : - gap ν a ≤ hi - lo := by - rw [gap_eq_bestArm_sub] - have hbest_le : (ν (bestArm ν))[id] ≤ hi := (hμ (bestArm ν)).2 - have ha_ge : lo ≤ (ν a)[id] := (hμ a).1 - linarith - -omit [IsMarkovKernel ν] in -/-- Arm means in `[-1, 1]` imply the gap cap `gap ≤ 2` used by the capped regret argument. -/ -lemma gapBound_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] - (hμ : MeanRewardBound ν (-1) 1) : - GapBound (K := K) ν 2 := by - intro a - have hgap := gap_le_of_meanRewardBound (ν := ν) (lo := -1) (hi := 1) hμ a - norm_num at hgap - exact hgap - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The initial-gap term used by this formalization is deterministically at most `2` when all arm -means lie in `[-1, 1]`. At horizon zero the initial term is exactly zero. -/ -lemma initialGapTerm_le_two_of_meanRewardBound_neg_one_one [Nonempty (Fin K)] - (hμ : MeanRewardBound ν (-1) 1) : - (if n = 0 then 0 else gap ν (A 0 ω)) ≤ if n = 0 then 0 else 2 := by - by_cases hn : n = 0 - · simp [hn] - · simpa [hn] using (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) hμ (A 0 ω)) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A uniform gap bound implies the selected-action gap bound through any finite horizon. -/ -lemma gap_ae_le_of_GapBound (G : ℝ) (hG : GapBound (K := K) ν G) : - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → gap ν (A t ω) ≤ G := - Filter.Eventually.of_forall fun ω t _ht ↦ hG (A t ω) - -omit [IsMarkovKernel ν] in -/-- First projection from the packaged LinUCB confidence event: optimism for the best arm. -/ -lemma LinUCBConfidenceEvent.best [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - intro t ht - exact (h_conf t ht).1 - -omit [IsMarkovKernel ν] in -/-- Second projection from the packaged LinUCB confidence event: validity of the selected arm's -lower confidence inequality. -/ -lemma LinUCBConfidenceEvent.arm [Nonempty (Fin K)] - (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ t, t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - intro t ht - exact (h_conf t ht).2 - -omit [IsMarkovKernel ν] in -/-- In zero feature dimension, the confidence event forces every positive-time selected gap to be -nonpositive. The best-arm index is zero, and the selected-arm pessimistic index is also zero. -/ -lemma gap_nonpos_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) - (t : ℕ) (ht : t ≠ 0) : - gap ν (A t ω) ≤ 0 := by - have hbest := LinUCBConfidenceEvent.best (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (ω := ω) h_conf t ht - have harm := LinUCBConfidenceEvent.arm (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (ω := ω) h_conf t ht - rw [gap_eq_bestArm_sub] - have hbest0 : (ν (bestArm ν))[id] ≤ 0 := by - simpa [index_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) (β := β) - (x := x) (n := t) (ω := ω) hd (bestArm ν)] using hbest - have harm0 : 0 ≤ (ν (A t ω))[id] := by - simpa [estimatedReward_eq_zero_of_dim_eq_zero (A := A) (R := R) (reg := reg) - (x := x) (n := t) (ω := ω) hd (A t ω), - width_eq_zero_of_dim_eq_zero (A := A) (reg := reg) (x := x) (n := t) - (ω := ω) hd (A t ω)] using harm - linarith - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure projection of the packaged confidence event to optimism for the best arm. -/ -lemma linUCBConfidenceEvent_ae_best [Nonempty (Fin K)] - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) t ω := by - filter_upwards [h_conf] with ω h_confω - exact h_confω.best - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure projection of the packaged confidence event to the selected arm's lower confidence -inequality. -/ -lemma linUCBConfidenceEvent_ae_arm [Nonempty (Fin K)] - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, ∀ t, t ≠ 0 → - estimatedReward A R reg x (A t ω) t ω - - √(β (t + 1)) * width A reg x (A t ω) t ω ≤ (ν (A t ω))[id] := by - filter_upwards [h_conf] with ω h_confω - exact h_confω.arm - -lemma designMatrix_eq_designMatrix' (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) - (ω : Ω) (hn : n ≠ 0) : - designMatrix A reg x n ω = - designMatrix' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - cases n with - | zero => exact absurd rfl hn - | succ n => - simp only [designMatrix, designMatrix', IsAlgEnvSeq.hist] - rw [Nat.range_succ_eq_Iic] - exact congrArg (fun S ↦ reg • 1 + S) <| - (Finset.sum_coe_sort (Iic n) - (fun s ↦ Matrix.vecMulVec (x (A s ω)) (x (A s ω)))).symm - -lemma responseVector_eq_responseVector' (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - responseVector A R x n ω = responseVector' x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - cases n with - | zero => exact absurd rfl hn - | succ n => - simp only [responseVector, responseVector', IsAlgEnvSeq.hist] - rw [Nat.range_succ_eq_Iic] - exact (Finset.sum_coe_sort (Iic n) (fun s ↦ R s ω • x (A s ω))).symm - -lemma thetaHat_eq_thetaHat' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - thetaHat A R reg x n ω = thetaHat' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) := by - simp [thetaHat, thetaHat', designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn, - responseVector_eq_responseVector' (A := A) (R := R) x n ω hn] - -lemma estimatedReward_eq_estimatedReward' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - estimatedReward A R reg x a n ω = - estimatedReward' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [estimatedReward, estimatedReward', thetaHat_eq_thetaHat' (A := A) (R := R) reg x n ω hn] - -lemma widthQuadraticForm_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthQuadraticForm A reg x a n ω = - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [widthQuadraticForm, widthQuadraticForm', - designMatrix_eq_designMatrix' (A := A) (R := R) reg x n ω hn] - -/-- At positive process times, nonnegativity of the process-level width quadratic form is -equivalent to nonnegativity of the matching history-level width quadratic form. -/ -lemma widthQuadraticForm_nonneg_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - 0 ≤ widthQuadraticForm A reg x a n ω ↔ - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] - -/-- At positive process times, the process-level quadratic width form is at most `1` iff the -matching history-level quadratic width form is at most `1`. -/ -lemma widthQuadraticForm_le_one_iff_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - widthQuadraticForm A reg x a n ω ≤ 1 ↔ - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a ≤ 1 := by - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n ω hn] - -/-- The all-positive-times process-level nonnegativity assumption is equivalent to the matching -history-level nonnegativity assumption. -/ -lemma widthQuadraticForm_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - (∀ t, t ∈ range n → t ≠ 0 → 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ - ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by - constructor - · intro h t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).1 (h t ht ht0) - · intro h t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h t ht ht0) - -/-- The all-positive-times process-level `≤ 1` assumption is equivalent to the matching -history-level `≤ 1` assumption. -/ -lemma widthQuadraticForm_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - (∀ t, t ∈ range n → t ≠ 0 → widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ - ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by - constructor - · intro h t ht ht0 - exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).1 (h t ht ht0) - · intro h t ht ht0 - exact (widthQuadraticForm_le_one_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h t ht ht0) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, process-level all-positive-times nonnegativity is equivalent to the matching -history-level nonnegativity assumption. -/ -lemma widthQuadraticForm_ae_all_nonneg_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) : - (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) ↔ - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).1 hω - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_nonneg_iff_history (A := A) (R := R) reg x n ω).2 hω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the process-level all-positive-times `≤ 1` assumption is equivalent to the -matching history-level `≤ 1` assumption. -/ -lemma widthQuadraticForm_ae_all_le_one_iff_history (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) : - (∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) ↔ - ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1 := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).1 hω - · intro h - filter_upwards [h] with ω hω - exact (widthQuadraticForm_all_le_one_iff_history (A := A) (R := R) reg x n ω).2 hω - -lemma width_eq_width' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - width A reg x a n ω = width' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - simp [width, width', widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x a n - ω hn] - -/-- At positive process times, squaring the process-level width recovers the matching history-level -quadratic form when that history-level quadratic form is nonnegative. -/ -lemma width_sq_eq_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_nonneg : - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a) : - width A reg x a n ω ^ 2 = - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - rw [width_eq_width' (A := A) (R := R) reg x a n ω hn] - exact width'_sq_eq_quadratic_form reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a - h_nonneg - -/-- At positive process times, advancing `widthSqSum` adds the matching history-level quadratic -form when that history-level quadratic form is nonnegative. -/ -lemma widthSqSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) - (h_nonneg : - 0 ≤ widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) : - widthSqSum A reg x (n + 1) ω = - widthSqSum A reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - rw [widthSqSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) (ω := ω) hn] - rw [width_sq_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn h_nonneg] - -/-- At positive process times, advancing `quadraticWidthSum` adds the matching history-level -quadratic form. -/ -lemma quadraticWidthSum_succ_eq_add_widthQuadraticForm' (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - quadraticWidthSum A reg x (n + 1) ω = - quadraticWidthSum A reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - rw [quadraticWidthSum_succ_of_ne_zero (A := A) (reg := reg) (x := x) (n := n) - (ω := ω) hn] - rw [widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A n ω) n ω hn] - -/-- The history-level quadratic-form accumulator aligned with process times. - -The term at process time `t = 0` is set to zero, matching the convention used by `widthSqSum` and -`quadraticWidthSum`. At positive process time `t`, the history available to LinUCB is -`IsAlgEnvSeq.hist A R (t - 1) ω`. -/ -noncomputable def historyQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) - -/-- No positive-time history-level quadratic width forms are accumulated at horizon zero. -/ -lemma historyQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - historyQuadraticWidthSum A R reg x 0 ω = 0 := by - simp [historyQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time history-level quadratic width form. -/ -lemma historyQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - historyQuadraticWidthSum A R reg x (n + 1) ω = - historyQuadraticWidthSum A R reg x n ω + - if n = 0 then 0 else - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - simp [historyQuadraticWidthSum, sum_range_succ] - -/-- At positive process times, advancing the history-level quadratic accumulator adds the selected -arm's history-level quadratic width form. -/ -lemma historyQuadraticWidthSum_succ_of_ne_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - historyQuadraticWidthSum A R reg x (n + 1) ω = - historyQuadraticWidthSum A R reg x n ω + - widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω) := by - simp [historyQuadraticWidthSum_succ, hn] - -/-- The capped history-level quadratic-form accumulator aligned with process times. - -This is the accumulator shape that commonly appears in elliptical-potential statements: -each positive-time quadratic width form is capped at `1`. -/ -noncomputable def historyCappedQuadraticWidthSum (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : ℝ := - ∑ t ∈ range n, - if t = 0 then 0 else - min 1 (widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - -/-- No positive-time capped history-level quadratic width forms are accumulated at horizon zero. -/ -lemma historyCappedQuadraticWidthSum_zero (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (ω : Ω) : - historyCappedQuadraticWidthSum A R reg x 0 ω = 0 := by - simp [historyCappedQuadraticWidthSum] - -/-- Advancing the horizon adds the next positive-time capped history-level quadratic width form. -/ -lemma historyCappedQuadraticWidthSum_succ (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - historyCappedQuadraticWidthSum A R reg x (n + 1) ω = - historyCappedQuadraticWidthSum A R reg x n ω + - if n = 0 then 0 else - min 1 - (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by - simp [historyCappedQuadraticWidthSum, sum_range_succ] - -/-- At positive process times, advancing the capped history-level quadratic accumulator adds the -selected arm's capped history-level quadratic width form. -/ -lemma historyCappedQuadraticWidthSum_succ_of_ne_zero - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - historyCappedQuadraticWidthSum A R reg x (n + 1) ω = - historyCappedQuadraticWidthSum A R reg x n ω + - min 1 - (widthQuadraticForm' reg x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) (A n ω)) := by - simp [historyCappedQuadraticWidthSum_succ, hn] - -/-- The process-level capped quadratic-width accumulator equals the history-level capped -accumulator aligned with the same process times. -/ -lemma cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (reg : ℝ) - (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : - cappedQuadraticWidthSum A reg x n ω = - historyCappedQuadraticWidthSum A R reg x n ω := by - rw [cappedQuadraticWidthSum, historyCappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact congrArg (fun q : ℝ ↦ min 1 q) - (widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0) - -/-- A process-level capped quadratic-width sum bound is equivalent to the matching history-level -capped quadratic-width sum bound. -/ -lemma cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : - cappedQuadraticWidthSum A reg x n ω ≤ W ↔ - historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by - rw [cappedQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) - reg x n ω] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a process-level capped quadratic-width sum bound is equivalent to the matching -history-level capped quadratic-width sum bound. -/ -lemma cappedQuadraticWidthSum_ae_le_iff_historyCappedQuadraticWidthSum_ae_le - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (W : ℝ) : - (∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) ↔ - ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W := by - constructor - · intro h - filter_upwards [h] with ω hω - exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (A := A) (R := R) reg x n ω W).1 hω - · intro h - filter_upwards [h] with ω hω - exact (cappedQuadraticWidthSum_le_iff_historyCappedQuadraticWidthSum_le - (A := A) (R := R) reg x n ω W).2 hω - -/-- If every positive-time history-level quadratic width form is at most `1`, then the uncapped and -capped history-level accumulators agree. -/ -lemma historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) : - historyQuadraticWidthSum A R reg x n ω = - historyCappedQuadraticWidthSum A R reg x n ω := by - rw [historyQuadraticWidthSum, historyCappedQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact (min_eq_right (h_le_one t ht ht0)).symm - -/-- The process-level quadratic-width accumulator equals the history-level accumulator aligned with -the same process times. -/ -lemma quadraticWidthSum_eq_historyQuadraticWidthSum (reg : ℝ) (x : Fin K → Feature d) - (n : ℕ) (ω : Ω) : - quadraticWidthSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by - rw [quadraticWidthSum, historyQuadraticWidthSum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · rw [if_neg ht0, if_neg ht0] - exact widthQuadraticForm_eq_widthQuadraticForm' (A := A) (R := R) reg x (A t ω) t ω ht0 - -/-- The squared-width accumulator equals the history-level quadratic-form accumulator whenever the -positive-time history-level quadratic forms are nonnegative. -/ -lemma widthSqSum_eq_historyQuadraticWidthSum - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) : - widthSqSum A reg x n ω = historyQuadraticWidthSum A R reg x n ω := by - have h_process_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht ht0 - exact (widthQuadraticForm_nonneg_iff_widthQuadraticForm' (A := A) (R := R) reg x - (A t ω) t ω ht0).2 (h_nonneg t ht ht0) - rw [widthSqSum_eq_sum_quadratic_form (A := A) (reg := reg) (x := x) - (n := n) (ω := ω) h_process_nonneg] - exact quadraticWidthSum_eq_historyQuadraticWidthSum (A := A) (R := R) reg x n ω - -/-- A bound on the history-level quadratic-form accumulator implies the corresponding bound on -`widthSqSum`, provided the positive-time history-level quadratic forms are nonnegative. -/ -lemma widthSqSum_le_of_history_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_hist_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - rw [widthSqSum_eq_historyQuadraticWidthSum (A := A) (R := R) (reg := reg) (x := x) - (n := n) (ω := ω) h_nonneg] - exact h_hist_le - -omit [IsProbabilityMeasure P] in -/-- Almost surely, a history-level quadratic-form bound gives the `widthSqSum` bound consumed by -the regret chain. -/ -lemma widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_hist_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_nonneg, h_hist_le] with ω h_nonnegω h_hist_leω - exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_hist_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The pointwise input expected from a future elliptical-potential argument. - -It packages the two facts needed to turn a history-level quadratic-width estimate into the -`widthSqSum` estimate used by the regret chain: - -* each positive-time quadratic width form is nonnegative; -* their history-level accumulated sum is bounded by `W`. -/ -def HistoryQuadraticWidthBound (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) (W : ℝ) : Prop := - (∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) ∧ - historyQuadraticWidthSum A R reg x n ω ≤ W - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Build the packaged history-level quadratic-width input from its two component facts. -/ -lemma historyQuadraticWidthBound_of_nonneg_and_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_sum_le : historyQuadraticWidthSum A R reg x n ω ≤ W) : - HistoryQuadraticWidthBound A R reg x n ω W := by - exact ⟨h_nonneg, h_sum_le⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged history-level quadratic-width input is monotone in the numeric bound. -/ -lemma historyQuadraticWidthBound_mono {W W' : ℝ} - (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : - HistoryQuadraticWidthBound A R reg x n ω W' := by - exact ⟨h_bound.1, h_bound.2.trans hW⟩ - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, build the packaged history-level quadratic-width input from its two component -facts. -/ -lemma historyQuadraticWidthBound_ae_of_nonneg_and_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_sum_le : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by - filter_upwards [h_nonneg, h_sum_le] with ω h_nonnegω h_sum_leω - exact historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_nonnegω h_sum_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged history-level quadratic-width input is monotone in the numeric -bound. -/ -lemma historyQuadraticWidthBound_ae_mono {W W' : ℝ} - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) (hW : W ≤ W') : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W' := by - filter_upwards [h_bound] with ω h_boundω - exact historyQuadraticWidthBound_mono (A := A) (R := R) (reg := reg) (x := x) - (n := n) (ω := ω) h_boundω hW - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped quadratic-width sum bound gives the packaged history-level input whenever every -positive-time quadratic width form is nonnegative and at most `1`. -/ -lemma historyQuadraticWidthBound_of_capped_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - HistoryQuadraticWidthBound A R reg x n ω W := by - refine historyQuadraticWidthBound_of_nonneg_and_sum_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_nonneg ?_ - rw [historyQuadraticWidthSum_eq_historyCappedQuadraticWidthSum (A := A) (R := R) - (reg := reg) (x := x) (n := n) (ω := ω) h_le_one] - exact h_capped_le - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped quadratic-width sum bound gives the packaged history-level input -whenever every positive-time quadratic width form is almost surely nonnegative and at most `1`. -/ -lemma historyQuadraticWidthBound_ae_of_capped_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W := by - filter_upwards [h_nonneg, h_le_one, h_capped_le] with - ω h_nonnegω h_le_oneω h_capped_leω - exact historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonnegω h_le_oneω h_capped_leω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The packaged history-level quadratic-width input implies the `widthSqSum` bound consumed by the -regret chain. -/ -lemma widthSqSum_le_of_history_quadratic_width_bound {W : ℝ} - (h_bound : HistoryQuadraticWidthBound A R reg x n ω W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_history_quadratic_width_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_bound.1 h_bound.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, the packaged history-level quadratic-width input implies the `widthSqSum` bound -consumed by the regret chain. -/ -lemma widthSqSum_ae_le_of_history_quadratic_width_bound_ae {W : ℝ} - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - filter_upwards [h_bound] with ω h_boundω - exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) (W := W) h_boundω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A capped history-level quadratic-width sum bound implies the `widthSqSum` bound consumed by -the regret chain, provided the positive-time quadratic width forms are nonnegative and at most -`1`. -/ -lemma widthSqSum_le_of_capped_history_quadratic_width_sum_le {W : ℝ} - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_le_of_history_quadratic_width_bound (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) (W := W) - (historyQuadraticWidthBound_of_capped_sum_le (A := A) (R := R) (reg := reg) - (x := x) (n := n) (ω := ω) h_nonneg h_le_one h_capped_le) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost surely, a capped history-level quadratic-width sum bound implies the `widthSqSum` bound -consumed by the regret chain, provided the positive-time quadratic width forms are almost surely -nonnegative and at most `1`. -/ -lemma widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le {W : ℝ} - (h_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (h_capped_le : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W := by - exact widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) - (historyQuadraticWidthBound_ae_of_capped_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_nonneg h_le_one - h_capped_le) - -lemma index_eq_index' (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (a : Fin K) (n : ℕ) (ω : Ω) (hn : n ≠ 0) : - index A R reg β x a n ω = - index' reg β x (n - 1) (IsAlgEnvSeq.hist A R (n - 1) ω) a := by - have htime : n + 1 = n - 1 + 2 := by grind - simp [index, index', estimatedReward_eq_estimatedReward' (A := A) (R := R) reg x a n ω hn, - width_eq_width' (A := A) (R := R) reg x a n ω hn, htime] - -/-- The action at time `n + 1` is the finite-action LinUCB argmax for the observed history. -/ -lemma arm_ae_eq_linUCBNextArm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (n : ℕ) : - A (n + 1) =ᵐ[P] - fun ω ↦ nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact h.action_detAlgorithm_ae_eq n - -/-- Almost surely, every positive-time action is the finite-action LinUCB argmax. -/ -lemma arm_ae_all_eq [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) : - ∀ᵐ ω ∂P, - ∀ n, A (n + 1) ω = - nextArm hK reg β x n (IsAlgEnvSeq.hist A R n ω) := by - simp_rw [ae_all_iff] - exact fun n ↦ arm_ae_eq_linUCBNextArm h n - -/-- Finite-action LinUCB chooses an arm maximizing the LinUCB index. -/ -lemma index_le_index_arm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (a : Fin K) (hn : n ≠ 0) : - ∀ᵐ ω ∂P, index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by - filter_upwards [arm_ae_eq_linUCBNextArm h (n - 1)] with ω h_arm - have hn_succ : n - 1 + 1 = n := by grind - simp only [hn_succ] at h_arm - rw [index_eq_index' (A := A) (R := R) reg β x a n ω hn, - index_eq_index' (A := A) (R := R) reg β x (A n ω) n ω hn] - rw [h_arm] - have : Nonempty (Fin K) := Fin.pos_iff_nonempty.mp hK - exact isMaxOn_measurableArgmax (fun h a ↦ index' reg β x (n - 1) h a) - (IsAlgEnvSeq.hist A R (n - 1) ω) a - -/-- Almost surely, the selected arm maximizes the LinUCB index at every positive time. -/ -lemma forall_index_le_index_arm [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (a : Fin K) : - ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω := by - simp_rw [ae_all_iff] - exact fun n hn ↦ index_le_index_arm h a hn - -end AlgorithmBehavior - -omit [IsMarkovKernel ν] in -/-- If the LinUCB confidence inequalities hold for a comparator arm and the selected arm, and the -selected arm has maximal LinUCB index, then instantaneous regret is controlled by the selected -arm's LinUCB width. -/ -lemma mean_sub_mean_arm_le_two_mul_width (a : Fin K) - (h_best : (ν a)[id] ≤ index A R reg β x a n ω) - (h_arm : estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_le : index A R reg β x a n ω ≤ index A R reg β x (A n ω) n ω) : - (ν a)[id] - (ν (A n ω))[id] ≤ - 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [sub_le_iff_le_add'] - calc - (ν a)[id] ≤ index A R reg β x a n ω := h_best - _ ≤ index A R reg β x (A n ω) n ω := h_le - _ ≤ (ν (A n ω))[id] + - 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [index, two_mul, ← add_assoc] - gcongr - rwa [sub_le_iff_le_add] at h_arm - -omit [IsMarkovKernel ν] in -/-- The gap of the selected arm is bounded by twice its LinUCB bonus whenever the usual confidence -inequalities hold and the selected arm has maximal LinUCB index. -/ -lemma gap_arm_le_two_mul_width [Nonempty (Fin K)] - (h_best : (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_le : index A R reg β x (bestArm ν) n ω ≤ - index A R reg β x (A n ω) n ω) : - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - rw [gap_eq_bestArm_sub] - exact mean_sub_mean_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) (a := bestArm ν) h_best h_arm h_le - -/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus whenever the usual -confidence inequalities hold almost surely. -/ -lemma gap_arm_ae_le_two_mul_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (hn : n ≠ 0) - (h_best : ∀ᵐ ω ∂P, (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - filter_upwards [h_best, h_arm, index_le_index_arm h (bestArm ν) hn] with - ω h_bestω h_armω h_leω - exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) h_bestω h_armω h_leω - -/-- Almost surely, the selected arm's gap is bounded by twice its LinUCB bonus at every positive -time whenever the usual confidence inequalities hold almost surely at every positive time. -/ -lemma forall_gap_arm_le_two_mul_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - gap ν (A n ω) ≤ 2 * (√(β (n + 1)) * width A reg x (A n ω) n ω) := by - filter_upwards [h_best, h_arm, forall_index_le_index_arm h (bestArm ν)] with - ω h_bestω h_armω h_leω - intro n hn - exact gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) (x := x) - (ν := ν) (n := n) (ω := ω) (h_bestω n hn) (h_armω n hn) (h_leω n hn) - -omit [IsMarkovKernel ν] in -/-- Pointwise capped LinUCB regret bound for one positive time. - -If the instantaneous gap is bounded by `2`, and the confidence/max-index argument gives the usual -`2 * sqrt(β_t) * width_t` bound, then monotonicity up to the terminal `β n` gives the textbook -capped form `2 * sqrt(β n) * sqrt(min 1 q_t)`, where `q_t` is the width quadratic form. -/ -lemma gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm - (t : ℕ) - (h_gap_two : gap ν (A t ω) ≤ 2) - (h_gap_width : gap ν (A t ω) ≤ - 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) - (hβ_le : β (t + 1) ≤ β n) - (hβn_one : 1 ≤ β n) : - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - by_cases hq_le_one : widthQuadraticForm A reg x (A t ω) t ω ≤ 1 - · have hwidth_nonneg : 0 ≤ width A reg x (A t ω) t ω := Real.sqrt_nonneg _ - have hsqrt_le : √(β (t + 1)) ≤ √(β n) := Real.sqrt_le_sqrt hβ_le - have hbonus_le : - 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) ≤ - 2 * (√(β n) * width A reg x (A t ω) t ω) := by - exact mul_le_mul_of_nonneg_left - (mul_le_mul_of_nonneg_right hsqrt_le hwidth_nonneg) (by norm_num) - have hmin : - √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)) = - width A reg x (A t ω) t ω := by - rw [min_eq_right hq_le_one, width] - simpa [hmin] using h_gap_width.trans hbonus_le - · have hq_one : 1 ≤ widthQuadraticForm A reg x (A t ω) t ω := by linarith - have hsqrt_one : 1 ≤ √(β n) := by - simpa using (Real.one_le_sqrt).2 hβn_one - have htwo_le : - 2 ≤ 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [min_eq_left hq_one, Real.sqrt_one] - nlinarith - exact h_gap_two.trans htwo_le - -omit [IsMarkovKernel ν] in -/-- If every realized gap up to horizon `n` is bounded pointwise, then regret up to `n` is bounded -by the corresponding sum of pointwise bounds. -/ -lemma regret_le_sum_of_gap_bound (B : ℕ → ℝ) - (hB : ∀ t, t ∈ range n → gap ν (A t ω) ≤ B t) : - regret ν A n ω ≤ ∑ t ∈ range n, B t := by - rw [regret_eq_sum_gap] - exact sum_le_sum hB - -omit [IsMarkovKernel ν] in -/-- A pathwise cumulative-regret bound obtained by summing the positive-time LinUCB width bound. - -The time-zero gap is left unchanged because the current LinUCB max-index theorem applies only at -positive times. -/ -lemma regret_le_sum_width_of_forall_gap_le - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) : - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using h_gap t ht ht0 - -omit [IsMarkovKernel ν] in -/-- A pathwise cumulative-regret bound obtained by summing the positive-time capped LinUCB width -bound. -/ -lemma regret_le_sum_sqrt_capped_width_of_forall_gap_le - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - refine regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using h_gap t ht ht0 - -omit [IsMarkovKernel ν] in -/-- Cauchy-Schwarz bound for the positive-time LinUCB bonus sum. -/ -lemma sum_positive_bonus_le_two_mul_sqrt_sum_sq : - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) ≤ - 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - calc - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) - = 2 * ∑ t ∈ range n, - (if t = 0 then 0 else √(β (t + 1))) * - (if t = 0 then 0 else width A reg x (A t ω) t ω) := by - rw [mul_sum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - gcongr - exact Real.sum_mul_le_sqrt_mul_sqrt (range n) - (fun t ↦ if t = 0 then 0 else √(β (t + 1))) - (fun t ↦ if t = 0 then 0 else width A reg x (A t ω) t ω) - -omit [IsMarkovKernel ν] in -/-- Cauchy-Schwarz bound for the positive-time capped LinUCB bonus sum. -/ -lemma sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum - (hβn_nonneg : 0 ≤ β n) - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) : - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) ≤ - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - calc - (∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) - = 2 * ∑ t ∈ range n, - (if t = 0 then 0 else √(β n)) * - (if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [mul_sum] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - _ ≤ 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) * - √(∑ t ∈ range n, - (if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) ^ 2)) := by - gcongr - exact Real.sum_mul_le_sqrt_mul_sqrt (range n) - (fun t ↦ if t = 0 then 0 else √(β n)) - (fun t ↦ if t = 0 then 0 - else √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) - _ ≤ 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - gcongr - · calc - (∑ t ∈ range n, (if t = 0 then 0 else √(β n)) ^ 2) - ≤ ∑ _t ∈ range n, β n := by - refine sum_le_sum ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0, hβn_nonneg] - · simp [ht0, Real.sq_sqrt hβn_nonneg] - _ = (n : ℝ) * β n := by - simp [sum_const, nsmul_eq_mul] - · rw [cappedQuadraticWidthSum] - refine le_of_eq ?_ - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · have hmin_nonneg : - 0 ≤ min 1 (widthQuadraticForm A reg x (A t ω) t ω) := by - exact le_min zero_le_one (h_nonneg t ht ht0) - simp [ht0, Real.sq_sqrt hmin_nonneg] - -omit [IsMarkovKernel ν] in -/-- Pathwise cumulative-regret bound using the textbook capped quadratic-width sum. -/ -lemma regret_le_initial_add_sqrt_nat_mul_beta_capped_sum - (hβn_nonneg : 0 ≤ β n) - (h_nonneg : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_gap : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := by - refine (regret_le_sum_sqrt_capped_width_of_forall_gap_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) h_gap).trans ?_ - have hsplit : - (∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω)))) = - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - ∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β n) * - √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - rw [← sum_add_distrib] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - rw [hsplit] - exact add_le_add le_rfl - (sum_positive_capped_bonus_le_two_mul_sqrt_nat_mul_beta_mul_sqrt_capped_sum - (A := A) (reg := reg) (β := β) (x := x) (n := n) (ω := ω) - hβn_nonneg h_nonneg) - -omit [IsMarkovKernel ν] in -/-- If the capped quadratic-width sum is bounded by `W`, the pathwise capped regret bound can use -`√W` in place of the realized capped-sum square root. -/ -lemma regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω))) - (hW : cappedQuadraticWidthSum A reg x n ω ≤ W) : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √W) := by - refine h_regret.trans ?_ - gcongr - -/-- The squared beta factor in the Cauchy-Schwarz bound simplifies when the confidence schedule is -nonnegative. -/ -lemma sum_sqrt_beta_sq_eq (hβ : ∀ t, 0 ≤ β (t + 1)) : - (∑ t ∈ range n, if t = 0 then 0 else √(β (t + 1)) ^ 2) = - ∑ t ∈ range n, if t = 0 then 0 else β (t + 1) := by - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0, Real.sq_sqrt (hβ t)] - -/-- Almost surely, the cumulative regret is bounded by the sum of LinUCB width terms whenever the -usual confidence inequalities hold almost surely at every positive time. -/ -lemma regret_ae_le_sum_width [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - ∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm] with ω h_gapω - exact regret_le_sum_width_of_forall_gap_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) fun t ht ht0 ↦ h_gapω t ht0 - -/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound on -the positive-time LinUCB width terms. -/ -lemma regret_ae_le_initial_add_cauchy [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, (if t = 0 then 0 else √(β (t + 1))) ^ 2) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - filter_upwards [regret_ae_le_sum_width h h_best h_arm] with ω h_regret - refine h_regret.trans ?_ - have hsplit : - (∑ t ∈ range n, - if t = 0 then gap ν (A 0 ω) - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω)) = - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - ∑ t ∈ range n, - if t = 0 then 0 - else 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := by - rw [← sum_add_distrib] - refine sum_congr rfl ?_ - intro t ht - by_cases ht0 : t = 0 - · simp [ht0] - · simp [ht0] - rw [hsplit] - exact add_le_add_right (sum_positive_bonus_le_two_mul_sqrt_sum_sq (A := A) - (reg := reg) (β := β) (x := x) (n := n) (ω := ω)) _ - -/-- Almost surely, cumulative regret is bounded by the initial gap plus a Cauchy-Schwarz bound whose -beta factor has been simplified using nonnegativity of the confidence schedule. -/ -lemma regret_ae_le_initial_add_cauchy_simplified [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2)) := by - filter_upwards [regret_ae_le_initial_add_cauchy (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) h h_best h_arm] with ω h_regret - simpa [sum_sqrt_beta_sq_eq (β := β) (n := n) hβ] using h_regret - -omit [IsMarkovKernel ν] in -/-- If the squared LinUCB widths are bounded by `W`, then the Cauchy-Schwarz regret bound can use -`√W` in place of the square root of the realized squared-width sum. -/ -lemma regret_le_initial_add_cauchy_of_width_sq_le (W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * - √(∑ t ∈ range n, (if t = 0 then 0 else width A reg x (A t ω) t ω) ^ 2))) - (hW : widthSqSum A reg x n ω ≤ W) - : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by - rw [widthSqSum] at hW - refine h_regret.trans ?_ - gcongr - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √(sum beta terms) * √W` whenever the squared LinUCB widths are almost surely bounded by `W`. - -This is the interface expected from a future elliptical-potential bound. -/ -lemma regret_ae_le_initial_add_sqrt_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W) := by - filter_upwards [regret_ae_le_initial_add_cauchy_simplified (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ, hW] with - ω h_regret hWω - exact regret_le_initial_add_cauchy_of_width_sq_le (A := A) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -omit [IsMarkovKernel ν] in -/-- If the beta sum is bounded by `B`, then the regret bound can use `√B` in place of the square -root of the beta sum. -/ -lemma regret_le_initial_add_sqrt_bounds_of_beta_sum_le (B W : ℝ) - (h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√(∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) * √W)) - (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by - refine h_regret.trans ?_ - gcongr - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √B * √W` whenever the beta sum is bounded by `B` and the squared LinUCB widths are almost -surely bounded by `W`. -/ -lemma regret_ae_le_initial_add_sqrt_bounds [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (B W : ℝ) - (hB : (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ B) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + 2 * (√B * √W) := by - filter_upwards [regret_ae_le_initial_add_sqrt_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ W hW - ] with ω h_regret - exact regret_le_initial_add_sqrt_bounds_of_beta_sum_le (A := A) (β := β) (ν := ν) - (n := n) (ω := ω) B W h_regret hB - -/-- If the confidence-radius schedule is nonnegative and monotone, the positive-time beta sum is -bounded by the horizon times the terminal beta value. -/ -lemma beta_sum_le_nat_mul_of_monotone - (hβ_mono : Monotone β) (hβ : ∀ t, 0 ≤ β (t + 1)) : - (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) ≤ (n : ℝ) * β n := by - calc - (∑ t ∈ range n, if t = 0 then 0 else β (t + 1)) - ≤ ∑ _t ∈ range n, β n := by - refine sum_le_sum ?_ - intro t ht - by_cases ht0 : t = 0 - · rw [if_pos ht0] - have hn_pos : 0 < n := by - simpa [ht0] using mem_range.mp ht - have hn_beta : 0 ≤ β n := by - have htime : n - 1 + 1 = n := by grind - simpa [htime] using hβ (n - 1) - exact hn_beta - · rw [if_neg ht0] - exact hβ_mono (Nat.succ_le_iff.mpr (mem_range.mp ht)) - _ = (n : ℝ) * β n := by - simp [sum_const, nsmul_eq_mul] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Minimal confidence-radius schedule assumptions used by the capped finite-action LinUCB regret -chain: the schedule starts at least at one and is monotone in time. -/ -def BetaSchedule (β : ℕ → ℝ) : Prop := - 1 ≤ β 1 ∧ Monotone β - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection from `BetaSchedule`: the confidence-radius schedule starts at least at one. -/ -lemma BetaSchedule.one (hβ : BetaSchedule β) : 1 ≤ β 1 := - hβ.1 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Projection from `BetaSchedule`: the confidence-radius schedule is monotone. -/ -lemma BetaSchedule.monotone (hβ : BetaSchedule β) : Monotone β := - hβ.2 - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A confidence-radius schedule with `1 ≤ β 1` and monotone `β` is nonnegative at every positive -horizon. -/ -lemma beta_nonneg_of_one_le_of_monotone - (hβ_one : 1 ≤ β 1) (hβ_mono : Monotone β) {n : ℕ} (hn : n ≠ 0) : - 0 ≤ β n := by - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr (Nat.pos_of_ne_zero hn) - exact ((zero_le_one : (0 : ℝ) ≤ 1).trans hβ_one).trans (hβ_mono hn_one) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- A `BetaSchedule` is nonnegative at every positive horizon. -/ -lemma BetaSchedule.nonneg_of_ne_zero (hβ : BetaSchedule β) {n : ℕ} (hn : n ≠ 0) : - 0 ≤ β n := - beta_nonneg_of_one_le_of_monotone (β := β) hβ.one hβ.monotone hn - -omit [IsMarkovKernel ν] in -/-- The initial-gap sum is just the time-zero gap when the horizon is positive, and zero when the -horizon is zero. -/ -lemma initial_gap_sum_eq : - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) = - if n = 0 then 0 else gap ν (A 0 ω) := by - cases n <;> simp - -omit [IsMarkovKernel ν] in -/-- In zero feature dimension, the confidence event bounds cumulative regret by the initial gap. -There is no positive-time width contribution because all widths are zero. -/ -lemma regret_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) (h_conf : LinUCBConfidenceEvent A R reg β x ν ω) : - regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by - refine (regret_le_sum_of_gap_bound (A := A) (ν := ν) (n := n) (ω := ω) - (B := fun t ↦ if t = 0 then gap ν (A 0 ω) else 0) ?_).trans ?_ - · intro t _ht - by_cases ht0 : t = 0 - · simp [ht0] - · simpa [ht0] using - gap_nonpos_of_confidence_dim_eq_zero (A := A) (R := R) (reg := reg) - (β := β) (x := x) (ν := ν) (ω := ω) hd h_conf t ht0 - · rw [initial_gap_sum_eq] - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Almost-sure zero-dimensional version of the finite-action LinUCB regret skeleton. -/ -lemma regret_ae_le_initial_gap_of_confidence_dim_eq_zero [Nonempty (Fin K)] - (hd : d = 0) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) : - ∀ᵐ ω ∂P, regret ν A n ω ≤ if n = 0 then 0 else gap ν (A 0 ω) := by - filter_upwards [h_conf] with ω h_confω - exact regret_le_initial_gap_of_confidence_dim_eq_zero (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hd h_confω - -/-- Almost surely, cumulative regret is bounded by the initial gap plus -`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` -is nonnegative and monotone. -/ -lemma regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_add_sqrt_bounds (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := n) h h_best h_arm hβ ((n : ℝ) * β n) W - (beta_sum_le_nat_mul_of_monotone (β := β) (n := n) hβ_mono hβ) hW - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the squared LinUCB widths are almost surely bounded by `W` and `β` -is nonnegative and monotone. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hW : ∀ᵐ ω ∂P, widthSqSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - filter_upwards [regret_ae_le_initial_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W hW - ] with ω h_regret - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using h_regret - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a history-level quadratic-form bound supplies the future -elliptical-potential input. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_bound [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (hW : ∀ᵐ ω ∂P, historyQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_history_quadratic_width_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the packaged history-level quadratic-width input holds almost -surely. - -This is the theorem a future elliptical-potential lemma should feed into directly. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_history_quadratic_width_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_bound : ∀ᵐ ω ∂P, HistoryQuadraticWidthBound A R reg x n ω W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_history_quadratic_width_bound_ae (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_bound) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a capped history-level quadratic-width sum bound holds almost -surely and every positive-time quadratic width form is almost surely nonnegative and at most `1`. - -This is the direct interface for the common capped form of the elliptical-potential lemma. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_history_quadratic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω)) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm' reg x (t - 1) (IsAlgEnvSeq.hist A R (t - 1) ω) (A t ω) ≤ 1) - (hW : ∀ᵐ ω ∂P, historyCappedQuadraticWidthSum A R reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_history_quadratic_width_sum_ae_le (A := A) (R := R) - (reg := reg) (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever a capped process-level quadratic-width sum bound holds almost -surely and every positive-time process-level quadratic width form is almost surely nonnegative and -at most `1`. - -This is the direct interface for an elliptical-potential lemma stated using the process-level design -matrices `designMatrix A reg x t ω`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_quadratic_width_sum_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) (W := W) h_quad_nonneg h_quad_le_one hW) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the packaged process-level capped quadratic-width input holds -almost surely. - -This is the compact theorem a process-level elliptical-potential lemma should feed into directly. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (h_bound : ∀ᵐ ω ∂P, CappedQuadraticWidthBound A reg x n ω W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_width_bound (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best h_arm hβ hβ_mono W - (widthSqSum_ae_le_of_capped_quadratic_width_bound_ae (A := A) (reg := reg) - (x := x) (n := n) (P := P) (W := W) h_bound) - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is almost surely bounded -by `W`. - -This version follows the proof structure of *Bandit Algorithms*, Theorem 19.2: optimism gives the -width bound, bounded instantaneous gaps give the cap, monotonicity of `β` moves all confidence -radii to `β n`, and Cauchy-Schwarz turns the sum into the square root of the capped quadratic-width -sum. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_schedule : BetaSchedule β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - by_cases hn : n = 0 - · subst n - exact Filter.Eventually.of_forall fun ω ↦ by simp [regret] - have hβn_nonneg : 0 ≤ β n := - hβ_schedule.nonneg_of_ne_zero hn - filter_upwards [forall_gap_arm_le_two_mul_width h h_best h_arm, h_gap_two, h_quad_nonneg, hW] - with ω h_gap_widthω h_gap_twoω h_quad_nonnegω hWω - have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht _ht0 - exact h_quad_nonnegω t ht - have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - intro t ht ht0 - have hβ_le : β (t + 1) ≤ β n := - hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) - have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 - have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) - exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) - (h_gap_twoω t ht ht0) (h_gap_widthω t ht0) hβ_le hβn_one - have h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := - regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos - h_gap_capped - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using - regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -/-- Almost surely, on the LinUCB confidence event, cumulative regret is bounded by the simplified -initial-gap term plus `2 * √(n * β n) * √W` whenever the textbook capped quadratic-width sum is -almost surely bounded by `W`. - -This is the good-event form of the deterministic regret argument. It separates the algorithmic -regret proof from the future concentration theorem: a later self-normalized concentration result -should prove that `LinUCBConfidenceEvent` holds with high probability, and this theorem converts -that event into the regret bound. -/ -lemma regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2) - (hβ_schedule : BetaSchedule β) (W : ℝ) - (h_quad_nonneg : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω) - (hW : ∀ᵐ ω ∂P, cappedQuadraticWidthSum A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - by_cases hn : n = 0 - · subst n - exact Filter.Eventually.of_forall fun ω _h_confω ↦ by simp [regret] - have hβn_nonneg : 0 ≤ β n := - hβ_schedule.nonneg_of_ne_zero hn - filter_upwards [forall_index_le_index_arm h (bestArm ν), h_gap_two, h_quad_nonneg, hW] with - ω h_indexω h_gap_twoω h_quad_nonnegω hWω h_confω - have h_quad_pos : ∀ t, t ∈ range n → t ≠ 0 → - 0 ≤ widthQuadraticForm A reg x (A t ω) t ω := by - intro t ht _ht0 - exact h_quad_nonnegω t ht - have h_gap_capped : ∀ t, t ∈ range n → t ≠ 0 → - gap ν (A t ω) ≤ - 2 * (√(β n) * √(min 1 (widthQuadraticForm A reg x (A t ω) t ω))) := by - intro t ht ht0 - have h_gap_width : - gap ν (A t ω) ≤ 2 * (√(β (t + 1)) * width A reg x (A t ω) t ω) := - gap_arm_le_two_mul_width (A := A) (R := R) (reg := reg) (β := β) - (x := x) (ν := ν) (n := t) (ω := ω) (h_confω.best t ht0) - (h_confω.arm t ht0) (h_indexω t ht0) - have hβ_le : β (t + 1) ≤ β n := - hβ_schedule.monotone (Nat.succ_le_iff.mpr (mem_range.mp ht)) - have ht_pos : 0 < t := Nat.pos_of_ne_zero ht0 - have hn_pos : 0 < n := Nat.lt_trans ht_pos (mem_range.mp ht) - have hn_one : 1 ≤ n := Nat.succ_le_iff.mpr hn_pos - have hβn_one : 1 ≤ β n := hβ_schedule.one.trans (hβ_schedule.monotone hn_one) - exact gap_le_two_mul_sqrt_beta_mul_sqrt_min_widthQuadraticForm (A := A) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) (t := t) - (h_gap_twoω t ht ht0) h_gap_width hβ_le hβn_one - have h_regret : - regret ν A n ω ≤ - (∑ t ∈ range n, if t = 0 then gap ν (A 0 ω) else 0) + - 2 * (√((n : ℝ) * β n) * √(cappedQuadraticWidthSum A reg x n ω)) := - regret_le_initial_add_sqrt_nat_mul_beta_capped_sum (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) hβn_nonneg h_quad_pos - h_gap_capped - simpa [initial_gap_sum_eq (A := A) (ν := ν) (n := n) (ω := ω)] using - regret_le_initial_add_sqrt_nat_mul_beta_of_capped_sum_le (A := A) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) (ω := ω) W h_regret hWω - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus the -feature-budget elliptical-potential term -`2 * √(n * β n) * √(2 * d * log(1 + n L² / (reg d)))`. - -The remaining matrix-analysis inputs are isolated as named hypotheses: `h_inv_antitone` is the -generic inverse anti-monotonicity theorem for positive-definite matrices, and `h_ratio_of_trace` -should come from a determinant/trace comparison proving that the trace budget implies the displayed -determinant-ratio bound. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (h_ratio_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDetRatio A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (reg * (d : ℝ))) ^ d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_reg_ne_zero_det_update_featureSqNorm_budget_bound' - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos.ne' hd - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (widthQuadraticForm_ae_le_one_of_featureSqNorm_ae_le (A := A) (reg := reg) - (x := x) (n := n) (P := P) - (WidthQuadraticFormLeFeatureSqNormDivReg.of_inv_le (A := A) (reg := reg) - (x := x) hreg_pos.ne' - (DesignMatrixInvLeRegInv.of_matrix_inv_antitone (A := A) (reg := reg) - (x := x) hreg_pos h_inv_antitone)) - hreg_pos hL2 hL2_le_reg) - L2 hL2 h_ratio_of_trace) - -/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the determinant/trace input is stated as the determinant upper bound -`det(V_n) ≤ ((reg * d + n * L²) / d) ^ d`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_featureSqNorm_budget_bound_of_designDet_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_of_trace : ∀ ω, - designTrace A reg x n ω ≤ reg * (d : ℝ) + (n : ℝ) * L2 → - designDet A reg x n ω ≤ - ((reg * (d : ℝ) + (n : ℝ) * L2) / (d : ℝ)) ^ d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_featureSqNorm_budget_bound_of_designDet_le - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg hdet_of_trace) - -/-- Almost surely, cumulative regret is bounded by the feature-budget elliptical-potential term -when the determinant/trace input is the reusable PSD matrix determinant/trace comparison -`det(M) ≤ (trace(M) / d) ^ d`. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_matrix_det_trace_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) - (hreg_pos : 0 < reg) (hd : d ≠ 0) - (h_inv_antitone : MatrixInvAntiMonoOnPosDef d) - (L2 : ℝ) - (hL2 : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → featureSqNorm x (A t ω) ≤ L2) - (hL2_le_reg : L2 ≤ reg) - (hdet_trace : MatrixDetLeTraceAveragePow d) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (cappedQuadraticWidthBound_ae_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd - h_inv_antitone L2 hL2 hL2_le_reg hdet_trace) - -/-- Textbook-shaped finite-action LinUCB regret theorem on the confidence event. - -This theorem is the good-event form closest to the finite-action LinUCB proof in -*Bandit Algorithms*: after the deterministic algorithm/max-index argument and the elliptical -potential bound are proved, the only remaining probabilistic input is whether the confidence event -holds on a sample path. - -* `h_mean_bound` bounds every arm's mean reward in `[-1, 1]`; -* `hβ_schedule` states that the confidence-radius schedule starts at least at one and is monotone; -* `hL2` is the uniform finite-action feature bound `‖x_a‖₂² ≤ L2`. - -The conclusion is an almost-sure implication: on almost every sample path, if -`LinUCBConfidenceEvent` holds, then the displayed regret bound holds. A future self-normalized -concentration theorem should prove that this confidence event has high probability for a concrete -textbook choice of `β`. - -The displayed bound is the standard Cauchy-Schwarz plus elliptical-potential expression -`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`, with one extra initial gap because this -formalization lets the deterministic algorithm play its default initial arm at time zero. -/ -lemma regret_ae_imp_le_textbook_finite_action - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - by_cases hd : d = 0 - · subst d - exact Filter.Eventually.of_forall fun ω h_confω ↦ by - simpa using regret_le_initial_gap_of_confidence_dim_eq_zero - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) - (ω := ω) (d := 0) rfl h_confω - · have h_gap_two : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → gap ν (A t ω) ≤ 2 := by - filter_upwards [gap_ae_le_of_GapBound (A := A) (ν := ν) (n := n) (P := P) - 2 (gapBound_two_of_meanRewardBound_neg_one_one (ν := ν) h_mean_bound)] with - ω h_gapω - intro t ht _ht0 - exact h_gapω t ht - exact regret_ae_imp_le_initial_gap_add_sqrt_nat_mul_beta_capped_sum_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_gap_two hβ_schedule - (2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))) - (widthQuadraticForm_ae_nonneg_of_reg_nonneg (A := A) (reg := reg) (x := x) - (n := n) (P := P) hreg_pos.le) - (cappedQuadraticWidthSum_ae_le_featureSqNorm_budget_of_matrix_det_trace_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) hreg_pos hd L2 - (featureSqNorm_ae_le_of_featureSqNormBound (A := A) (x := x) (n := n) - (P := P) L2 hL2) - matrixDetLeTraceAveragePow) - -/-- The deterministic textbook LinUCB bonus term -`2 * sqrt(n * β_n) * sqrt(2 d log(1 + n L² / (reg d)))`. - -The final finite-action theorem keeps this as a named expression so probability statements can use -a deterministic right-hand side instead of repeating the full formula. -/ -noncomputable def textbookRegretBonus (reg : ℝ) (β : ℕ → ℝ) (L2 : ℝ) (n : ℕ) : ℝ := - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) - -/-- Good-event finite-action LinUCB regret theorem with the random initial gap replaced by the -deterministic `≤ 2` bound implied by `MeanRewardBound ν (-1) 1`. -/ -lemma regret_ae_imp_le_textbook_finite_action_deterministic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - LinUCBConfidenceEvent A R reg β x ν ω → - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_confω - refine (h_regret h_confω).trans ?_ - simpa [textbookRegretBonus] using - add_le_add_right - (initialGapTerm_le_two_of_meanRewardBound_neg_one_one (A := A) (ν := ν) - (n := n) (ω := ω) h_mean_bound) - (2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))) - -/-- Almost-sure corollary of -`regret_ae_imp_le_textbook_finite_action_deterministic_bound` when the confidence event is known -to hold almost surely. -/ -lemma regret_ae_le_textbook_finite_action_deterministic_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n := by - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2, h_conf] with ω h_regret h_confω - exact h_regret h_confω - -/-- The confidence event is almost surely contained in the deterministic textbook regret-bound -event. This is the version to combine with a future high-probability confidence theorem. -/ -lemma probReal_confidenceEvent_le_textbook_regret_bound_deterministic - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_confω - exact h_regret h_confω - -/-- High-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ -lemma probReal_textbook_regret_bound_deterministic_ge_of_confidenceEvent_ge - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : - 1 - δ ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} := by - exact h_conf_prob.trans - (probReal_confidenceEvent_le_textbook_regret_bound_deterministic (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2) - -/-- Failure-probability wrapper for the deterministic textbook finite-action LinUCB regret bound. -/ -lemma probReal_textbook_regret_bound_deterministic_failure_le_of_confidenceEvent_failure_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_failure : - P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : - P.real {ω | - ¬ regret ν A n ω ≤ - (if n = 0 then 0 else 2) + textbookRegretBonus (d := d) reg β L2 n} ≤ δ := by - refine le_trans ?_ h_conf_failure - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action_deterministic_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h - h_mean_bound hβ_schedule hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω - exact h_regret_failure (h_regret h_confω) - -/-- The confidence event is almost surely contained in the textbook finite-action regret-bound -event. - -This is the probability bridge needed after the good-event theorem: once a concentration theorem -proves that `LinUCBConfidenceEvent` has high probability, this lemma transfers that probability -mass to the displayed regret bound. -/ -lemma probReal_confidenceEvent_le_textbook_regret_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω} ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_confω - exact h_regret h_confω - -/-- High-probability wrapper for the textbook finite-action LinUCB regret bound. - -If a future self-normalized concentration theorem proves that the confidence event has probability -at least `1 - δ`, then the textbook regret bound has probability at least `1 - δ` as well. -/ -lemma probReal_textbook_regret_bound_ge_of_confidenceEvent_ge - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_prob : 1 - δ ≤ P.real {ω | LinUCBConfidenceEvent A R reg β x ν ω}) : - 1 - δ ≤ - P.real {ω | - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} := by - exact h_conf_prob.trans - (probReal_confidenceEvent_le_textbook_regret_bound (A := A) (R := R) (reg := reg) - (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule hreg_pos L2 hL2) - -/-- Failure-probability wrapper for the textbook finite-action LinUCB regret bound. - -If a future self-normalized concentration theorem proves that the confidence event fails with -probability at most `δ`, then the textbook regret bound fails with probability at most `δ`. -/ -lemma probReal_textbook_regret_bound_failure_le_of_confidenceEvent_failure_le - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) {δ : ℝ} - (h_conf_failure : - P.real {ω | ¬ LinUCBConfidenceEvent A R reg β x ν ω} ≤ δ) : - P.real {ω | - ¬ - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ)))))} ≤ δ := by - refine le_trans ?_ h_conf_failure - simp_rw [measureReal_def] - gcongr 1 - · simp - refine measure_mono_ae ?_ - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2] with ω h_regret h_regret_failure h_confω - exact h_regret_failure (h_regret h_confω) - -/-- Corollary of `regret_ae_imp_le_textbook_finite_action` when the confidence event is known to -hold almost surely. This is stronger than the textbook high-probability route and is mainly useful -as a compatibility wrapper for earlier lemmas in this file. -/ -lemma regret_ae_le_textbook_finite_action - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_conf : ∀ᵐ ω ∂P, LinUCBConfidenceEvent A R reg β x ν ω) - (h_mean_bound : MeanRewardBound (K := K) ν (-1) 1) - (hβ_schedule : BetaSchedule β) - (hreg_pos : 0 < reg) - (L2 : ℝ) (hL2 : FeatureSqNormBound x L2) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + - 2 * (√((n : ℝ) * β n) * - √(2 * (d : ℝ) * Real.log (1 + (n : ℝ) * L2 / (reg * (d : ℝ))))) := by - filter_upwards [regret_ae_imp_le_textbook_finite_action (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_mean_bound hβ_schedule - hreg_pos L2 hL2, h_conf] with ω h_regret h_confω - exact h_regret h_confω - -/-- Almost surely, cumulative regret is bounded by the simplified initial-gap term plus -`2 * √(n * β n) * √W` whenever positive regularization, the positive-time width cap, and the final -log-determinant potential bound hold. - -The capped-sum/log-determinant part of the elliptical-potential argument is proved internally: -positive regularization gives determinant nonvanishing and nonnegative quadratic forms, while -`h_quad_le_one` lets this older theorem feed the uncapped `widthSqSum` regret route. -/ -lemma regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_of_ellipticalPotential_bound - [Nonempty (Fin K)] - (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x h_index) (stationaryEnv ν) P) - (h_best : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - (ν (bestArm ν))[id] ≤ index A R reg β x (bestArm ν) n ω) - (h_arm : ∀ᵐ ω ∂P, ∀ n, n ≠ 0 → - estimatedReward A R reg x (A n ω) n ω - - √(β (n + 1)) * width A reg x (A n ω) n ω ≤ (ν (A n ω))[id]) - (hβ : ∀ t, 0 ≤ β (t + 1)) (hβ_mono : Monotone β) (W : ℝ) - (hreg_pos : 0 < reg) - (h_quad_le_one : ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → - widthQuadraticForm A reg x (A t ω) t ω ≤ 1) - (h_potential_le : ∀ᵐ ω ∂P, ellipticalPotential A reg x n ω ≤ W) : - ∀ᵐ ω ∂P, - regret ν A n ω ≤ - (if n = 0 then 0 else gap ν (A 0 ω)) + 2 * (√((n : ℝ) * β n) * √W) := by - exact regret_ae_le_initial_gap_add_sqrt_nat_mul_beta_capped_quadratic_width_bound - (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) h h_best - h_arm hβ hβ_mono W - (cappedQuadraticWidthBound_ae_of_reg_pos_det_update_ellipticalPotential_le_bound - (A := A) (reg := reg) (x := x) (n := n) (P := P) (W := W) hreg_pos - h_quad_le_one h_potential_le) - -end LinUCB - -end Bandits From 9257d57b1876620c85aa3bd1d8d39672d803de48 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:30:36 -0400 Subject: [PATCH 79/88] feat(112): remove the regret file reference --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f528dfdb..68192152 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -5,7 +5,7 @@ Authors: OpenAI, Fawad Haider -/ module -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Regret + /-! # LinUCB for finite-action linear bandits From 962cf11211b2a0319d7ee56a5f57bcf0e5df994d Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:39:02 -0400 Subject: [PATCH 80/88] fix merge issue --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean | 1 - 1 file changed, 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean index f528dfdb..b829ead4 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB.lean @@ -5,7 +5,6 @@ Authors: OpenAI, Fawad Haider -/ module -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Regret /-! # LinUCB for finite-action linear bandits From 60c75ad1c6bef1a5b385fc1a01907db80ebbbe87 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 10:50:39 -0400 Subject: [PATCH 81/88] linUCB(issue-112): update the LML main import file --- LeanMachineLearning.lean | 1 + 1 file changed, 1 insertion(+) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index d48bdc7e..cd8b36ea 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -20,6 +20,7 @@ public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS public import LeanMachineLearning.Online.Bandit.Algorithms.UCB From 200901403ef8d24009a4d570014e014966acc4ec Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 11:00:00 -0400 Subject: [PATCH 82/88] linUCB(issue-112): fix author comment --- .../Online/Bandit/Algorithms/LinUCB/Basic.lean | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean index 21912a55..e492f2af 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Basic.lean @@ -1,5 +1,5 @@ /- -Copyright (c) 2026. All rights reserved. +Copyright (c) 2026 Fawad Haider. All rights reserved. Released under Apache 2.0 license as described in the file LICENSE. Authors: OpenAI, Fawad Haider -/ @@ -7,11 +7,11 @@ module public import LeanMachineLearning.Online.Bandit.SumRewards public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +public import LeanMachineLearning.ForMathlib.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import Mathlib.Analysis.MeanInequalities public import Mathlib.Analysis.SpecialFunctions.Log.Deriv public import Mathlib.Analysis.Matrix.Order -public import Mathlib.Data.Real.StarOrdered +public import Mathlib.Algebra.Order.Star.Real public import Mathlib.LinearAlgebra.Matrix.PosDef public import Mathlib.LinearAlgebra.Matrix.SchurComplement public import Mathlib.LinearAlgebra.Matrix.NonsingularInverse From 6c9a36c1a264ef3e9e7d03069192ae559bc9368d Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 11:02:16 -0400 Subject: [PATCH 83/88] linUCB(issue-157): update leanmachinelearning.lean imports --- LeanMachineLearning.lean | 1 + 1 file changed, 1 insertion(+) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index cd8b36ea..1e69f14a 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -21,6 +21,7 @@ public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Matrix public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS public import LeanMachineLearning.Online.Bandit.Algorithms.UCB From 5fe1e27f03e5c80c165ed3e0f17b983fad075ae6 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 11:09:36 -0400 Subject: [PATCH 84/88] linUCB(issue-157): proof update for Matrix.mulVec_transpose --- .../Online/Bandit/Algorithms/LinUCB/Matrix.lean | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean index 096136b9..23e19aac 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean @@ -1,5 +1,5 @@ /- -Copyright (c) 2026. All rights reserved. +Copyright (c) 2026 Fawad Haider. All rights reserved. Released under Apache 2.0 license as described in the file LICENSE. Authors: OpenAI, Fawad Haider -/ @@ -823,6 +823,7 @@ lemma unitary_conj_quadraticForm_eq star (U : Matrix (Fin d) (Fin d) ℝ) = 1 exact Unitary.coe_mul_star_self U have hy : star Umat *ᵥ lambda = Matrix.vecMul lambda Umat := by + rw [Matrix.star_eq_conjTranspose, Matrix.conjTranspose_eq_transpose_of_trivial] simpa [Umat] using (Matrix.mulVec_transpose (U : Matrix (Fin d) (Fin d) ℝ) lambda) have hcancel_left : Umat * (star Umat * M * Umat) = M * Umat := by rw [Matrix.mul_assoc (star Umat) M Umat] From cb9f4c94d09dec89a957f12e33ef28fb291a27da Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 11:28:18 -0400 Subject: [PATCH 85/88] linUCB(issue-158): remove confidencenridge.lean for now --- LeanMachineLearning.lean | 1 + .../Algorithms/LinUCB/ConfidenceEvents.lean | 2 +- .../LinUCB/TextbookConfidenceBridge.lean | 354 ------------------ 3 files changed, 2 insertions(+), 355 deletions(-) delete mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 1e69f14a..657002af 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -21,6 +21,7 @@ public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.ConfidenceEvents public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Matrix public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean index 99830570..72a3f14d 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConfidenceEvents.lean @@ -1,5 +1,5 @@ /- -Copyright (c) 2026. All rights reserved. +Copyright (c) 2026 Fawad Haider. All rights reserved. Released under Apache 2.0 license as described in the file LICENSE. Authors: OpenAI, Fawad Haider -/ diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean deleted file mode 100644 index 5a500a62..00000000 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/TextbookConfidenceBridge.lean +++ /dev/null @@ -1,354 +0,0 @@ -/- -Copyright (c) 2026. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: OpenAI, Fawad Haider --/ -module - -public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.TextbookMixture - -/-! -# LinUCB Textbook Confidence Bridge - -Deterministic bridges from centered-noise and textbook self-normalized events to -LinUCB prediction-confidence events. --/ - -@[expose] public section - -open MeasureTheory ProbabilityTheory Filter Real Finset Learning - -open scoped ENNReal NNReal Matrix MatrixOrder - -namespace Bandits - -variable {K d : ℕ} - -namespace LinUCB - -variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} - {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] - {Ω : Type*} {mΩ : MeasurableSpace Ω} - {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} - {n : ℕ} {ω : Ω} - -section AlgorithmBehavior - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Horizon-local coordinate-wise centered-noise event. - -This is a conservative finite-dimensional interface for reusing scalar projection concentration: -if every coordinate of the centered response vector is bounded, then the centered-noise quadratic -form is bounded through `centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg`. -/ -def LinUCBCenteredNoiseCoordinateBoundEventUpTo - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := - ∀ t, t ∈ range n → t ≠ 0 → ∀ i, - |centeredResponseVector A R ν x t ω i| ≤ coordBudget (t + 1) - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Upper-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ -def centeredNoiseCoordinateUpperFailure - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := - {ω | coordBudget (t + 1) < centeredResponseVector A R ν x t ω i} - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Lower-tail failure for one coordinate of the finite-horizon centered-noise vector. -/ -def centeredNoiseCoordinateLowerFailure - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (coordBudget : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (t : ℕ) (i : Fin d) : Set Ω := - {ω | centeredResponseVector A R ν x t ω i < -coordBudget (t + 1)} - -omit [IsMarkovKernel ν] in -/-- A global centered-noise-plus-bias confidence event implies its finite-horizon restriction. -/ -lemma LinUCBCenteredNoiseBiasConfidenceEvent.toUpTo - (θ : Feature d) - (h_noise : LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by - intro t _ht ht0 - exact h_noise t ht0 - -omit [IsMarkovKernel ν] in -/-- A coordinate-wise centered-noise event implies the centered-noise quadratic-form event under the -corresponding conservative budget. -/ -lemma LinUCBCenteredNoiseConfidenceEventUpTo.of_coordinateBound - (hreg_pos : 0 < reg) - {coordBudget noiseBudget : ℕ → ℝ} - (hcoord : - LinUCBCenteredNoiseCoordinateBoundEventUpTo A R coordBudget x ν n ω) - (hcoord_nonneg : ∀ t, t ∈ range n → t ≠ 0 → 0 ≤ coordBudget (t + 1)) - (h_budget : ∀ t, t ∈ range n → t ≠ 0 → - (d : ℝ) * coordBudget (t + 1) ^ 2 / reg ≤ noiseBudget (t + 1)) : - LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω := by - intro t ht ht0 - exact - (centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) hreg_pos - (hcoord_nonneg t ht ht0) (hcoord t ht ht0)).trans - (h_budget t ht ht0) - -omit [IsMarkovKernel ν] in -/-- A horizon-local centered-noise event plus a parameter norm bound implies the -centered-noise-plus-ridge-bias event. - -This is the finite-horizon deterministic half of the textbook confidence-set proof: -the future self-normalized concentration theorem only has to control -`centeredNoiseQuadraticForm`; the ridge-bias contribution is bounded here by -`reg * S2`. -/ -lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_centeredNoise - (θ : Feature d) (S2 : ℝ) - (hreg_pos : 0 < reg) - (hθ : ParameterSqNormBound θ S2) - {noiseBudget : ℕ → ℝ} - (h_noise : - LinUCBCenteredNoiseConfidenceEventUpTo A R reg noiseBudget x ν n ω) - (h_budget : ∀ t, t ∈ range n → t ≠ 0 → - (√(noiseBudget (t + 1)) + √(reg * S2)) ^ 2 ≤ β (t + 1)) : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by - intro t ht ht0 - exact - (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos - (noiseBudget (t + 1)) (reg * S2) (h_noise t ht ht0) - (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) - (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans - (h_budget t ht ht0) - -omit [IsMarkovKernel ν] in -/-- The textbook determinant-ratio self-normalized noise event plus the ridge-bias radius implies -the existing centered-noise-plus-bias confidence event. - -This is the deterministic bridge from the future Gaussian-mixture concentration theorem to the -confidence event already consumed by the LinUCB regret proof. -/ -lemma LinUCBCenteredNoiseBiasConfidenceEventUpTo.of_textbookSelfNormalizedNoise - (θ : Feature d) (S2 : ℝ) {σ2 : ℝ≥0} {δ : ℝ} - (hreg_pos : 0 < reg) - (hθ : ParameterSqNormBound θ S2) - (h_noise : - LinUCBTextbookSelfNormalizedNoiseEventUpTo A R reg σ2 δ x ν n ω) - (h_budget : ∀ t, t ∈ range n → t ≠ 0 → - (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + - √(reg * S2)) ^ 2 ≤ β (t + 1)) : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω := by - intro t ht ht0 - exact - (centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ hreg_pos - (textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) - (reg * S2) (h_noise t ht ht0) - (regularizationBiasQuadraticForm_le_of_parameterSqNormBound (A := A) - (reg := reg) (x := x) (n := t) (ω := ω) θ hreg_pos hθ)).trans - (h_budget t ht ht0) - -omit [IsMarkovKernel ν] in -/-- The centered-noise-plus-bias confidence event implies the textbook parameter ellipsoid event -under linear realizability and positive regularization. -/ -lemma LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_noise : - LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω) : - LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω := by - intro t ht - rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] - exact h_noise t ht - -omit [IsMarkovKernel ν] in -/-- The horizon-local centered-noise-plus-bias event implies the horizon-local parameter ellipsoid -event under linear realizability and positive regularization. -/ -lemma LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_noise : - LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : - LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω := by - intro t ht ht0 - rw [parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] - exact h_noise t ht ht0 - -omit [IsMarkovKernel ν] in -/-- Under linear realizability and positive regularization, the parameter ellipsoid event is exactly -the centered-noise-plus-bias confidence event. -/ -lemma linUCBParameterEllipsoidConfidenceEvent_iff_centeredNoiseBiasConfidenceEvent - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) : - LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω ↔ - LinUCBCenteredNoiseBiasConfidenceEvent A R reg β x ν θ ω := by - constructor - · intro h_ellipsoid t ht - rw [← parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm (A := A) (R := R) - (reg := reg) (x := x) (ν := ν) (n := t) (ω := ω) θ h_linear hreg_pos] - exact h_ellipsoid t ht - · intro h_noise - exact LinUCBParameterEllipsoidConfidenceEvent.of_centeredNoiseBias (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear hreg_pos h_noise - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Action-wise prediction-error confidence event around a linear parameter `θ`. - -This is the finite-action consequence of the textbook ellipsoid event after applying the -matrix Cauchy-Schwarz inequality to each arm. -/ -def LinUCBParameterPredictionConfidenceEvent - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (θ : Feature d) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → ∀ a, - |dotProduct (thetaHat A R reg x t ω - θ) (x a)| ≤ - √(β (t + 1)) * width A reg x a t ω - -/-- Uniform self-normalized prediction-error event for finite-action LinUCB. - -This is the event that a future self-normalized martingale concentration theorem should prove with -high probability, for a concrete textbook choice of `β`. It says that, at every positive time and -for every finite action, the least-squares prediction error is bounded by the LinUCB confidence -radius. -/ -def LinUCBSelfNormalizedConfidenceEvent - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (ω : Ω) : Prop := - ∀ t, t ≠ 0 → ∀ a, - |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ - √(β (t + 1)) * width A reg x a t ω - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- Horizon-local self-normalized prediction-error event. - -This is the finite-horizon version of `LinUCBSelfNormalizedConfidenceEvent`. For regret through -time `n`, the proof only needs prediction confidence for positive `t ∈ range n`. -/ -def LinUCBSelfNormalizedConfidenceEventUpTo - (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) - (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) - (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := - ∀ t, t ∈ range n → t ≠ 0 → ∀ a, - |estimatedReward A R reg x a t ω - (ν a)[id]| ≤ - √(β (t + 1)) * width A reg x a t ω - -omit [IsMarkovKernel ν] in -/-- A global self-normalized prediction-error event implies its finite-horizon restriction. -/ -lemma LinUCBSelfNormalizedConfidenceEvent.toUpTo - (h_self : LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω) : - LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by - intro t _ht ht0 a - exact h_self t ht0 a - -omit [IsMarkovKernel ν] in -/-- A parameter prediction-confidence event implies the self-normalized confidence event once the -arm means are realized by that parameter. -/ -lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (h_param : LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω) : - LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := by - intro t ht a - have h_param_t := h_param t ht a - have h_error : - estimatedReward A R reg x a t ω - (ν a)[id] = - dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - calc - estimatedReward A R reg x a t ω - (ν a)[id] - = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by - rw [estimatedReward, h_linear a] - _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - rw [sub_dotProduct] - rw [h_error] - simpa [sub_dotProduct] using h_param_t - -omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in -/-- The textbook ellipsoid event implies action-wise parameter prediction confidence under positive -regularization, by the matrix Cauchy-Schwarz step. -/ -lemma LinUCBParameterPredictionConfidenceEvent.of_ellipsoid - (θ : Feature d) - (hreg_pos : 0 < reg) - (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : - LinUCBParameterPredictionConfidenceEvent A R reg β x θ ω := by - intro t ht a - have h_cauchy_t := - linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) - hreg_pos (thetaHat A R reg x t ω - θ) a t ω - have h_radius : - √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := - Real.sqrt_le_sqrt (h_ellipsoid t ht) - calc - |dotProduct (thetaHat A R reg x t ω - θ) (x a)| - ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by - simpa [parameterErrorQuadraticForm] using h_cauchy_t - _ ≤ √(β (t + 1)) * width A reg x a t ω := by - exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) - -omit [IsMarkovKernel ν] in -/-- The textbook ellipsoid event implies the self-normalized prediction-error event under positive -regularization and linear realizability. -/ -lemma LinUCBSelfNormalizedConfidenceEvent.of_parameterEllipsoid - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : - LinUCBSelfNormalizedConfidenceEvent A R reg β x ν ω := - LinUCBSelfNormalizedConfidenceEvent.of_parameterPrediction (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (ω := ω) θ h_linear - (LinUCBParameterPredictionConfidenceEvent.of_ellipsoid (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ω := ω) θ hreg_pos h_ellipsoid) - -omit [IsMarkovKernel ν] in -/-- The horizon-local textbook ellipsoid event implies the horizon-local self-normalized -prediction-error event under positive regularization and linear realizability. -/ -lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω) : - LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := by - intro t ht ht0 a - have h_cauchy_t := - linUCBPredictionErrorCauchySchwarz_of_reg_pos (A := A) (reg := reg) (x := x) - hreg_pos (thetaHat A R reg x t ω - θ) a t ω - have h_radius : - √(parameterErrorQuadraticForm A R reg x θ t ω) ≤ √(β (t + 1)) := - Real.sqrt_le_sqrt (h_ellipsoid t ht ht0) - have h_error : - estimatedReward A R reg x a t ω - (ν a)[id] = - dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - calc - estimatedReward A R reg x a t ω - (ν a)[id] - = dotProduct (thetaHat A R reg x t ω) (x a) - dotProduct θ (x a) := by - rw [estimatedReward, h_linear a] - _ = dotProduct (thetaHat A R reg x t ω - θ) (x a) := by - rw [sub_dotProduct] - rw [h_error] - calc - |dotProduct (thetaHat A R reg x t ω - θ) (x a)| - ≤ √(parameterErrorQuadraticForm A R reg x θ t ω) * width A reg x a t ω := by - simpa [parameterErrorQuadraticForm] using h_cauchy_t - _ ≤ √(β (t + 1)) * width A reg x a t ω := by - exact mul_le_mul_of_nonneg_right h_radius (Real.sqrt_nonneg _) - -omit [IsMarkovKernel ν] in -/-- The horizon-local centered-noise-plus-bias event implies the horizon-local self-normalized -prediction-error event under linear realizability and positive regularization. -/ -lemma LinUCBSelfNormalizedConfidenceEventUpTo.of_centeredNoiseBias - (θ : Feature d) - (h_linear : LinearMeanModel ν x θ) - (hreg_pos : 0 < reg) - (h_noise : LinUCBCenteredNoiseBiasConfidenceEventUpTo A R reg β x ν θ n ω) : - LinUCBSelfNormalizedConfidenceEventUpTo A R reg β x ν n ω := - LinUCBSelfNormalizedConfidenceEventUpTo.of_parameterEllipsoid (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) θ h_linear hreg_pos - (LinUCBParameterEllipsoidConfidenceEventUpTo.of_centeredNoiseBias (A := A) (R := R) - (reg := reg) (β := β) (x := x) (ν := ν) (n := n) (ω := ω) - θ h_linear hreg_pos h_noise) - -end AlgorithmBehavior - -end LinUCB - -end Bandits From 6e0131f003acf264c190e1787fc42c3cd4a2df77 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 11:33:14 -0400 Subject: [PATCH 86/88] linUCB(issue-158): fix linter issue --- LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean index 23e19aac..d572c50e 100644 --- a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/Matrix.lean @@ -2402,7 +2402,7 @@ lemma index_zero_eq_initial_quadratic_form (A : ℕ → Ω → Fin K) (R : ℕ simp [index_zero, width_zero] /-- The finite-action LinUCB process starts from the deterministic default arm. -/ -lemma arm_zero [Nonempty (Fin K)] +lemma arm_zero (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) : A 0 =ᵐ[P] fun _ ↦ ⟨0, hK⟩ := by exact h.action_zero_detAlgorithm From ff055d70a9b5fb56a8e319be341255ced952fbe8 Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 14:46:10 -0400 Subject: [PATCH 87/88] Concentration Infrastructure proofs --- LeanMachineLearning.lean | 1 + 1 file changed, 1 insertion(+) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 657002af..397e3eb9 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -21,6 +21,7 @@ public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Basic +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.ConcentrationCore public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.ConfidenceEvents public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.Matrix public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS From afa08d37169d35d1e382c2e836833cb59827bb1f Mon Sep 17 00:00:00 2001 From: fawad haider <153737+FawadHa1der@users.noreply.github.com> Date: Fri, 26 Jun 2026 14:46:28 -0400 Subject: [PATCH 88/88] Concentration Infrastructure proofs --- .../Algorithms/LinUCB/ConcentrationCore.lean | 4215 +++++++++++++++++ 1 file changed, 4215 insertions(+) create mode 100644 LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConcentrationCore.lean diff --git a/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConcentrationCore.lean b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConcentrationCore.lean new file mode 100644 index 00000000..14fc663f --- /dev/null +++ b/LeanMachineLearning/Online/Bandit/Algorithms/LinUCB/ConcentrationCore.lean @@ -0,0 +1,4215 @@ +/- +Copyright (c) 2026 OpenAI, Fawad Haider. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: OpenAI, Fawad Haider +-/ +module + +public import LeanMachineLearning.Online.Bandit.Algorithms.LinUCB.ConfidenceEvents +public import Mathlib.Probability.ConditionalExpectation + +/-! +# LinUCB Scalar And Projected Concentration Core + +Reward-noise kernels, scalar projected-noise concentration, and fixed-direction +exponential-process lemmas for finite-action LinUCB. +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Filter Real Finset Learning + +open scoped ENNReal NNReal Matrix MatrixOrder + +namespace Bandits + +variable {K d : ℕ} + +namespace LinUCB + +variable {hK : 0 < K} {reg : ℝ} {β : ℕ → ℝ} {x : Fin K → Feature d} + {ν : Kernel (Fin K) ℝ} [IsMarkovKernel ν] + {Ω : Type*} {mΩ : MeasurableSpace Ω} + {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → Fin K} {R : ℕ → Ω → ℝ} + {n : ℕ} {ω : Ω} + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A pointwise-equal presentation of a real-valued supermartingale is again a supermartingale. + +This small helper keeps later textbook-facing concentration lemmas from repeating the three +definition fields of `Supermartingale`: adaptedness, conditional expectation monotonicity, and +integrability. -/ +lemma supermartingale_congr_eq + {f g : ℕ → Ω → ℝ} {ℱ : Filtration ℕ mΩ} + (hf : Supermartingale f ℱ P) (hfg : ∀ n, f n = g n) : + Supermartingale g ℱ P := by + refine ⟨?_, ?_, ?_⟩ + · intro i + simpa [← hfg i] using hf.stronglyAdapted i + · intro i j hij + have h_cond_eq : P[g j | ℱ i] =ᵐ[P] P[f j | ℱ i] := by + exact condExp_congr_ae (Filter.Eventually.of_forall fun ω ↦ by + rw [← hfg j]) + filter_upwards [h_cond_eq, hf.condExp_ae_le hij] with ω h_eq h_le + rw [h_eq] + simpa [hfg i] using h_le + · intro i + exact (hf.integrable i).congr (Filter.Eventually.of_forall fun ω ↦ by + rw [hfg i]) + +section AlgorithmBehavior + + + +omit [IsMarkovKernel ν] in +/-- Project the arm-wise reward-noise subgaussian assumption to a single arm. -/ +lemma RewardNoiseSubgaussian.apply + {σ2 : ℝ≥0} + (hν : RewardNoiseSubgaussian (K := K) ν σ2) (a : Fin K) : + HasSubgaussianMGF (fun r ↦ r - (ν a)[id]) σ2 (ν a) := + hν a + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A subgaussian MGF bound remains true when the variance proxy is enlarged. -/ +lemma hasSubgaussianMGF_mono_varianceProxy + {X : Ω → ℝ} {c c' : ℝ≥0} + (hX : HasSubgaussianMGF X c P) (hc : c ≤ c') : + HasSubgaussianMGF X c' P where + integrable_exp_mul := hX.integrable_exp_mul + mgf_le t := by + refine (hX.mgf_le t).trans ?_ + exact Real.exp_le_exp.mpr (by + have hc' : (c : ℝ) ≤ (c' : ℝ) := by exact_mod_cast hc + gcongr) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A scalar subgaussian MGF bound gives the normalized exponential-process bound at one time. -/ +lemma hasSubgaussianMGF_integral_exp_sub_le_one + {X : Ω → ℝ} {c : ℝ≥0} (hX : HasSubgaussianMGF X c P) (u : ℝ) : + ∫ ω, Real.exp (u * X ω - (c : ℝ) * u ^ 2 / 2) ∂P ≤ 1 := by + let b : ℝ := (c : ℝ) * u ^ 2 / 2 + have h_eq : + (∫ ω, Real.exp (u * X ω - (c : ℝ) * u ^ 2 / 2) ∂P) = + (∫ ω, Real.exp (u * X ω) ∂P) / Real.exp b := by + simp only [b] + simp_rw [Real.exp_sub] + rw [MeasureTheory.integral_div] + have h_mgf_le : + (∫ ω, Real.exp (u * X ω) ∂P) ≤ Real.exp b := by + simpa [mgf, b] using hX.mgf_le u + calc + (∫ ω, Real.exp (u * X ω - (c : ℝ) * u ^ 2 / 2) ∂P) + = (∫ ω, Real.exp (u * X ω) ∂P) / Real.exp b := h_eq + _ ≤ Real.exp b / Real.exp b := + div_le_div_of_nonneg_right h_mgf_le (le_of_lt (Real.exp_pos b)) + _ = 1 := div_self (Real.exp_ne_zero b) + +omit [IsMarkovKernel ν] in +/-- Measurability of the arm-mean map on the finite action space. -/ +lemma measurable_rewardMean (ν : Kernel (Fin K) ℝ) : + Measurable fun a : Fin K ↦ (ν a)[id] := + measurable_of_countable _ + +omit [IsMarkovKernel ν] in +/-- Deterministic centering map sending `(a, r)` to the reward noise `r - μ(a)`. -/ +noncomputable def centerReward (ν : Kernel (Fin K) ℝ) (p : Fin K × ℝ) : ℝ := + p.2 - (ν p.1)[id] + +omit [IsMarkovKernel ν] in +/-- The centered-reward map is measurable. -/ +lemma measurable_centerReward (ν : Kernel (Fin K) ℝ) : + Measurable (centerReward ν) := + measurable_snd.sub ((measurable_rewardMean ν).comp measurable_fst) + +omit [IsMarkovKernel ν] in +/-- Conditional kernel of centered reward noise given the selected action. + +For action `a`, this is the reward law `ν a` pushed forward by `r ↦ r - μ(a)`. It is the +Markov-kernel form of the scalar martingale noise process used by the future self-normalized +concentration theorem. -/ +noncomputable def rewardNoiseKernel (ν : Kernel (Fin K) ℝ) : Kernel (Fin K) ℝ := + ⟨fun a ↦ (ν a).map (fun r ↦ r - (ν a)[id]), measurable_of_countable _⟩ + +instance rewardNoiseKernel.instIsMarkovKernel : + IsMarkovKernel (rewardNoiseKernel (K := K) ν) := by + constructor + intro a + simpa [rewardNoiseKernel] using + Measure.isProbabilityMeasure_map (μ := ν a) (measurable_id.sub measurable_const).aemeasurable + +omit [IsMarkovKernel ν] in +/-- At a fixed action, `rewardNoiseKernel` is exactly the reward law pushed forward by centering at +that action's mean. -/ +lemma rewardNoiseKernel_apply (ν : Kernel (Fin K) ℝ) (a : Fin K) : + rewardNoiseKernel ν a = (ν a).map (fun r ↦ r - (ν a)[id]) := + rfl + +omit [IsMarkovKernel ν] in +/-- Arm-wise subgaussianity stated on the centered-noise kernel itself. -/ +def RewardNoiseKernelSubgaussian (ν : Kernel (Fin K) ℝ) (σ2 : ℝ≥0) : Prop := + ∀ a, HasSubgaussianMGF id σ2 (rewardNoiseKernel ν a) + +omit [IsMarkovKernel ν] in +/-- The UCB-style arm-wise centered reward assumption implies subgaussianity of the centered-noise +kernel. -/ +lemma RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian + {σ2 : ℝ≥0} + (hν : RewardNoiseSubgaussian (K := K) ν σ2) : + RewardNoiseKernelSubgaussian (K := K) ν σ2 := by + intro a + rw [rewardNoiseKernel_apply (ν := ν) a] + exact (HasSubgaussianMGF.id_map_iff + ((measurable_id.sub measurable_const).aemeasurable)).mpr (hν a) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Arm-wise subgaussianity of the centered-noise kernel packages into Mathlib's kernel-level +subgaussian MGF predicate over any finite action law. -/ +lemma RewardNoiseKernelSubgaussian.hasKernelSubgaussian + {σ2 : ℝ≥0} + (hν : RewardNoiseKernelSubgaussian (K := K) ν σ2) + (μ : Measure (Fin K)) [IsFiniteMeasure μ] : + Kernel.HasSubgaussianMGF id σ2 (rewardNoiseKernel ν) μ := by + constructor + · intro t + have hf : + AEStronglyMeasurable (fun η : ℝ ↦ Real.exp (t * η)) ((rewardNoiseKernel ν) ∘ₘ μ) := by + fun_prop + change Integrable (fun η : ℝ ↦ Real.exp (t * η)) ((rewardNoiseKernel ν) ∘ₘ μ) + rw [Measure.integrable_comp_iff hf] + constructor + · exact Filter.Eventually.of_forall fun a ↦ (hν a).integrable_exp_mul t + · have h_meas : + AEStronglyMeasurable + (fun a : Fin K ↦ ∫ η, ‖Real.exp (t * η)‖ ∂rewardNoiseKernel ν a) μ := by + exact (measurable_of_countable _).aestronglyMeasurable + refine integrable_of_le_of_le h_meas ?_ ?_ (integrable_const 0) + (integrable_const (Real.exp (σ2 * t ^ 2 / 2))) + · exact Filter.Eventually.of_forall fun a ↦ integral_nonneg fun η ↦ norm_nonneg _ + · refine Filter.Eventually.of_forall fun a ↦ ?_ + have h_int := (hν a).integrable_exp_mul t + have h_mgf := (hν a).mgf_le t + simpa [mgf, Real.norm_of_nonneg (Real.exp_nonneg _)] using h_mgf + · exact Filter.Eventually.of_forall fun a t ↦ (hν a).mgf_le t + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The history-action version of `RewardNoiseKernelSubgaussian.hasKernelSubgaussian`: adding an +ignored history coordinate to the conditioning variable preserves the kernel-level subgaussian MGF +property. -/ +lemma RewardNoiseKernelSubgaussian.hasKernelSubgaussian_prodMkLeft + {γ : Type*} [MeasurableSpace γ] {σ2 : ℝ≥0} + (hν : RewardNoiseKernelSubgaussian (K := K) ν σ2) + (μ : Measure (γ × Fin K)) [IsFiniteMeasure μ] : + Kernel.HasSubgaussianMGF id σ2 ((rewardNoiseKernel ν).prodMkLeft γ) μ := by + constructor + · intro t + have hf : + AEStronglyMeasurable (fun η : ℝ ↦ Real.exp (t * η)) + (((rewardNoiseKernel ν).prodMkLeft γ) ∘ₘ μ) := by + fun_prop + change Integrable (fun η : ℝ ↦ Real.exp (t * η)) + (((rewardNoiseKernel ν).prodMkLeft γ) ∘ₘ μ) + rw [Measure.integrable_comp_iff hf] + constructor + · exact Filter.Eventually.of_forall fun z ↦ by + simpa [Kernel.prodMkLeft_apply] using (hν z.2).integrable_exp_mul t + · have h_meas : + AEStronglyMeasurable + (fun z : γ × Fin K ↦ + ∫ η, ‖Real.exp (t * η)‖ ∂(rewardNoiseKernel ν).prodMkLeft γ z) μ := by + refine ((measurable_of_countable + (fun a : Fin K ↦ ∫ η, ‖Real.exp (t * η)‖ ∂rewardNoiseKernel ν a)).comp + measurable_snd).aestronglyMeasurable.congr ?_ + exact Filter.Eventually.of_forall fun z ↦ by simp [Kernel.prodMkLeft_apply] + refine integrable_of_le_of_le h_meas ?_ ?_ (integrable_const 0) + (integrable_const (Real.exp (σ2 * t ^ 2 / 2))) + · exact Filter.Eventually.of_forall fun z ↦ integral_nonneg fun η ↦ norm_nonneg _ + · refine Filter.Eventually.of_forall fun z ↦ ?_ + have h_mgf := (hν z.2).mgf_le t + simpa [Kernel.prodMkLeft_apply, mgf, Real.norm_of_nonneg (Real.exp_nonneg _)] using h_mgf + · exact Filter.Eventually.of_forall fun z t ↦ by + simpa [Kernel.prodMkLeft_apply] using (hν z.2).mgf_le t + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The UCB-style arm-wise reward-noise assumption gives the kernel-level subgaussian MGF package +for `rewardNoiseKernel`. -/ +lemma RewardNoiseSubgaussian.hasKernelSubgaussian + {σ2 : ℝ≥0} + (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (μ : Measure (Fin K)) [IsFiniteMeasure μ] : + Kernel.HasSubgaussianMGF id σ2 (rewardNoiseKernel ν) μ := + RewardNoiseKernelSubgaussian.hasKernelSubgaussian + (ν := ν) (σ2 := σ2) + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian (K := K) (ν := ν) hν) μ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The UCB-style arm-wise reward-noise assumption gives the kernel-level subgaussian MGF package +after adding an ignored history coordinate. -/ +lemma RewardNoiseSubgaussian.hasKernelSubgaussian_prodMkLeft + {γ : Type*} [MeasurableSpace γ] {σ2 : ℝ≥0} + (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (μ : Measure (γ × Fin K)) [IsFiniteMeasure μ] : + Kernel.HasSubgaussianMGF id σ2 ((rewardNoiseKernel ν).prodMkLeft γ) μ := + RewardNoiseKernelSubgaussian.hasKernelSubgaussian_prodMkLeft + (ν := ν) (σ2 := σ2) + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian (K := K) (ν := ν) hν) μ + +omit [IsProbabilityMeasure P] in +/-- Mapping an action-reward joint law by reward centering is the same as pairing the action law +with `rewardNoiseKernel`. -/ +lemma compProd_map_centerReward_eq_compProd_rewardNoiseKernel + (μ : Measure (Fin K)) [SFinite μ] : + (μ ⊗ₘ ν).map (fun p : Fin K × ℝ ↦ (p.1, centerReward ν p)) = + μ ⊗ₘ rewardNoiseKernel ν := by + ext s hs + let g : Fin K × ℝ → Fin K × ℝ := fun p ↦ (p.1, centerReward ν p) + have hg : Measurable g := by + dsimp [g] + exact Measurable.prodMk measurable_fst (measurable_centerReward ν) + change (Measure.map g (μ ⊗ₘ ν)) s = (μ ⊗ₘ rewardNoiseKernel ν) s + rw [Measure.map_apply hg hs, Measure.compProd_apply (hg hs), Measure.compProd_apply hs] + refine lintegral_congr_ae ?_ + refine Filter.Eventually.of_forall fun a ↦ ?_ + change (ν a) (Prod.mk a ⁻¹' (g ⁻¹' s)) = + (rewardNoiseKernel ν a) (Prod.mk a ⁻¹' s) + rw [rewardNoiseKernel_apply (ν := ν) a] + let f : ℝ → ℝ := fun r ↦ r - (ν a)[id] + have hf : Measurable f := by + dsimp [f] + exact measurable_id.sub measurable_const + change (ν a) (Prod.mk a ⁻¹' (g ⁻¹' s)) = (Measure.map f (ν a)) (Prod.mk a ⁻¹' s) + rw [Measure.map_apply hf (measurable_prodMk_left hs)] + congr 1 + +omit [IsProbabilityMeasure P] in +/-- Mapping a history/action-reward joint law by reward centering is the same as pairing the +history/action law with `rewardNoiseKernel`, ignoring the history coordinate. + +This is the history-action version of `compProd_map_centerReward_eq_compProd_rewardNoiseKernel`. +It is the measure identity used to show that reward noise is conditionally centered-subgaussian +given the past history and the current selected action. -/ +lemma compProd_map_centerReward_historyAction_eq_compProd_rewardNoiseKernel_prodMkLeft + {γ : Type*} [MeasurableSpace γ] + (μ : Measure (γ × Fin K)) [SFinite μ] : + (μ ⊗ₘ ν.prodMkLeft γ).map + (fun p : (γ × Fin K) × ℝ ↦ (p.1, p.2 - (ν p.1.2)[id])) = + μ ⊗ₘ (rewardNoiseKernel ν).prodMkLeft γ := by + ext s hs + let g : (γ × Fin K) × ℝ → (γ × Fin K) × ℝ := + fun p ↦ (p.1, p.2 - (ν p.1.2)[id]) + have hg : Measurable g := by + dsimp [g] + refine Measurable.prodMk measurable_fst ?_ + exact measurable_snd.sub ((measurable_rewardMean ν).comp (measurable_snd.comp measurable_fst)) + change (Measure.map g (μ ⊗ₘ ν.prodMkLeft γ)) s = + (μ ⊗ₘ (rewardNoiseKernel ν).prodMkLeft γ) s + rw [Measure.map_apply hg hs, Measure.compProd_apply (hg hs), Measure.compProd_apply hs] + refine lintegral_congr_ae ?_ + refine Filter.Eventually.of_forall fun ha ↦ ?_ + change (ν.prodMkLeft γ ha) (Prod.mk ha ⁻¹' (g ⁻¹' s)) = + ((rewardNoiseKernel ν).prodMkLeft γ ha) (Prod.mk ha ⁻¹' s) + rw [Kernel.prodMkLeft_apply, Kernel.prodMkLeft_apply, rewardNoiseKernel_apply (ν := ν) ha.2] + let f : ℝ → ℝ := fun r ↦ r - (ν ha.2)[id] + have hf : Measurable f := by + dsimp [f] + exact measurable_id.sub measurable_const + change (ν ha.2) (Prod.mk ha ⁻¹' (g ⁻¹' s)) = + (Measure.map f (ν ha.2)) (Prod.mk ha ⁻¹' s) + rw [Measure.map_apply hf (measurable_prodMk_left hs)] + congr 1 + +/-- In a stationary bandit environment, the scalar centered reward noise at time `t`, conditioned +on the selected action, has conditional kernel `rewardNoiseKernel ν`. + +This is the formal bridge from the repository's Markov-kernel environment model to the martingale +noise process used in the textbook LinUCB self-normalized concentration proof. -/ +lemma hasCondDistrib_rewardNoise_action {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) (t : ℕ) : + HasCondDistrib (rewardNoise A R ν t) (A t) (rewardNoiseKernel ν) P := by + have hR : HasCondDistrib (R t) (A t) ν P := + h.hasCondDistrib_feedback_stationaryEnv t + have h_noise_meas : + Measurable fun ω ↦ R t ω - (ν (A t ω))[id] := + (h.measurable_feedback t).sub ((measurable_rewardMean ν).comp (h.measurable_action t)) + have h_noise_ae : AEMeasurable (rewardNoise A R ν t) P := by + change AEMeasurable (fun ω ↦ R t ω - (ν (A t ω))[id]) P + exact h_noise_meas.aemeasurable + have h_pair_ae : AEMeasurable (fun ω ↦ (A t ω, R t ω)) P := + (Measurable.prodMk (h.measurable_action t) (h.measurable_feedback t)).aemeasurable + have h_center_ae : + AEMeasurable (fun p : Fin K × ℝ ↦ (p.1, centerReward ν p)) + (P.map fun ω ↦ (A t ω, R t ω)) := + (Measurable.prodMk measurable_fst (measurable_centerReward ν)).aemeasurable + refine ⟨?_, ?_⟩ + · exact (h.measurable_action t).aemeasurable.prodMk h_noise_ae + · have h_eq := hR.map_eq + calc P.map (fun ω ↦ (A t ω, rewardNoise A R ν t ω)) + _ = (P.map (fun ω ↦ (A t ω, R t ω))).map + (fun p : Fin K × ℝ ↦ (p.1, centerReward ν p)) := by + rw [AEMeasurable.map_map_of_aemeasurable h_center_ae h_pair_ae] + · rfl + _ = (P.map (A t) ⊗ₘ ν).map + (fun p : Fin K × ℝ ↦ (p.1, centerReward ν p)) := by + rw [h_eq] + _ = P.map (A t) ⊗ₘ rewardNoiseKernel ν := + compProd_map_centerReward_eq_compProd_rewardNoiseKernel (ν := ν) (μ := P.map (A t)) + +/-- In a stationary bandit environment, the scalar centered reward noise at a positive time, +conditioned on the previous history and the selected action, has conditional kernel +`rewardNoiseKernel ν`, ignoring the history coordinate. + +This is the martingale-noise conditional-law statement needed before applying a future +self-normalized concentration theorem: after the algorithm chooses `A t` from the past, the +remaining centered reward noise has the centered law of that selected arm. -/ +lemma hasCondDistrib_rewardNoise_history_action {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) {t : ℕ} (ht : t ≠ 0) : + HasCondDistrib (rewardNoise A R ν t) + (fun ω ↦ (history A R (t - 1) ω, A t ω)) + ((rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ)) P := by + cases t with + | zero => exact (ht rfl).elim + | succ n => + have hR : HasCondDistrib (R (n + 1)) + (fun ω ↦ (history A R n ω, A (n + 1) ω)) + (ν.prodMkLeft (Iic n → Fin K × ℝ)) P := by + simpa using h.hasCondDistrib_feedback n + have h_noise_meas : + Measurable fun ω ↦ R (n + 1) ω - (ν (A (n + 1) ω))[id] := + (h.measurable_feedback (n + 1)).sub + ((measurable_rewardMean ν).comp (h.measurable_action (n + 1))) + have h_noise_ae : AEMeasurable (rewardNoise A R ν (n + 1)) P := by + change AEMeasurable + (fun ω ↦ R (n + 1) ω - (ν (A (n + 1) ω))[id]) P + exact h_noise_meas.aemeasurable + have h_pair_ae : + AEMeasurable + (fun ω ↦ ((history A R n ω, A (n + 1) ω), R (n + 1) ω)) P := + (Measurable.prodMk + (Measurable.prodMk + (h.measurable_history n) + (h.measurable_action (n + 1))) + (h.measurable_feedback (n + 1))).aemeasurable + have h_center_ae : + AEMeasurable + (fun p : ((Iic n → Fin K × ℝ) × Fin K) × ℝ ↦ + (p.1, p.2 - (ν p.1.2)[id])) + (P.map fun ω ↦ + ((history A R n ω, A (n + 1) ω), R (n + 1) ω)) := by + refine (Measurable.prodMk measurable_fst ?_).aemeasurable + exact measurable_snd.sub + ((measurable_rewardMean ν).comp (measurable_snd.comp measurable_fst)) + refine ⟨?_, ?_⟩ + · exact ((Measurable.prodMk + (h.measurable_history n) + (h.measurable_action (n + 1))).aemeasurable).prodMk h_noise_ae + · have h_eq := hR.map_eq + calc + P.map + (fun ω ↦ + ((history A R n ω, A (n + 1) ω), + rewardNoise A R ν (n + 1) ω)) + = (P.map + (fun ω ↦ + ((history A R n ω, A (n + 1) ω), R (n + 1) ω))).map + (fun p : ((Iic n → Fin K × ℝ) × Fin K) × ℝ ↦ + (p.1, p.2 - (ν p.1.2)[id])) := by + rw [AEMeasurable.map_map_of_aemeasurable h_center_ae h_pair_ae] + rfl + _ = (P.map (fun ω ↦ (history A R n ω, A (n + 1) ω)) ⊗ₘ + ν.prodMkLeft (Iic n → Fin K × ℝ)).map + (fun p : ((Iic n → Fin K × ℝ) × Fin K) × ℝ ↦ + (p.1, p.2 - (ν p.1.2)[id])) := by + rw [h_eq] + _ = P.map (fun ω ↦ (history A R n ω, A (n + 1) ω)) ⊗ₘ + (rewardNoiseKernel ν).prodMkLeft (Iic n → Fin K × ℝ) := + compProd_map_centerReward_historyAction_eq_compProd_rewardNoiseKernel_prodMkLeft + (ν := ν) + (μ := P.map fun ω ↦ (history A R n ω, A (n + 1) ω)) + +omit [IsMarkovKernel ν] in +/-- Arm-wise subgaussianity of `rewardNoiseKernel` remains true after adding an ignored history +coordinate to the conditioning variable. -/ +lemma RewardNoiseKernelSubgaussian.prodMkLeft + {γ : Type*} [MeasurableSpace γ] {σ2 : ℝ≥0} + (hν : RewardNoiseKernelSubgaussian (K := K) ν σ2) : + ∀ z : γ × Fin K, + HasSubgaussianMGF id σ2 ((rewardNoiseKernel ν).prodMkLeft γ z) := by + intro z + simpa [Kernel.prodMkLeft_apply] using hν z.2 + +omit [IsMarkovKernel ν] in +/-- Pointwise predictable scalar projections of the centered-noise kernel are subgaussian with the +variance proxy multiplied by the squared scalar. + +For a later vector concentration proof, `q z` is the predictable coefficient obtained by projecting +the selected feature vector onto a fixed direction. -/ +lemma RewardNoiseKernelSubgaussian.prodMkLeft_constMul + {γ : Type*} [MeasurableSpace γ] {σ2 : ℝ≥0} + (hν : RewardNoiseKernelSubgaussian (K := K) ν σ2) (q : γ × Fin K → ℝ) : + ∀ z : γ × Fin K, + HasSubgaussianMGF (fun η ↦ q z * η) + (⟨q z ^ 2, sq_nonneg (q z)⟩ * σ2) + ((rewardNoiseKernel ν).prodMkLeft γ z) := by + intro z + simpa only [id_eq] using (hν.prodMkLeft z).const_mul (q z) + +/-- Under the UCB-style arm-wise reward-noise assumption, the conditional law of the positive-time +LinUCB reward noise given history and selected action is subgaussian. + +This is the scalar probabilistic input that a future vector self-normalized concentration theorem +should consume for predictable projections of `η_t x_{A_t}`. -/ +lemma rewardNoise_condDistrib_history_action_subgaussian {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) : + ∀ᵐ z ∂P.map (fun ω ↦ (history A R (t - 1) ω, A t ω)), + HasSubgaussianMGF id σ2 + (condDistrib (rewardNoise A R ν t) + (fun ω ↦ (history A R (t - 1) ω, A t ω)) P z) := by + have h_cond := hasCondDistrib_rewardNoise_history_action + (A := A) (R := R) (ν := ν) h ht + have h_kernel : + ∀ z : (Iic (t - 1) → Fin K × ℝ) × Fin K, + HasSubgaussianMGF id σ2 + ((rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) z) := + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian + (K := K) (ν := ν) hν).prodMkLeft + filter_upwards [h_cond.condDistrib_eq] with z hz + rw [hz] + exact h_kernel z + +omit [IsMarkovKernel ν] in +/-- The centered reward-noise kernel, viewed over the realized history/action conditioning law, +satisfies Mathlib's kernel-level subgaussian MGF predicate. + +This packages the previous pointwise conditional-law statement with the global exponential +integrability required by `Kernel.HasSubgaussianMGF`. -/ +lemma rewardNoise_history_action_kernelSubgaussian + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) (t : ℕ) : + Kernel.HasSubgaussianMGF id σ2 + ((rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ)) + (P.map fun ω ↦ (history A R (t - 1) ω, A t ω)) := + hν.hasKernelSubgaussian_prodMkLeft + (P.map fun ω ↦ (history A R (t - 1) ω, A t ω)) + +/-- The marginal law of the realized positive-time reward noise is the measure obtained by first +sampling the realized history/action pair and then sampling from the centered reward-noise kernel. + +This is the process-level law identity behind the conditional-law statement +`hasCondDistrib_rewardNoise_history_action`. It lets later concentration lemmas move between the +actual random variable `rewardNoise A R ν t` and the Markov-kernel description of its conditional +law. -/ +lemma rewardNoise_map_eq_historyAction_kernel_comp {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {t : ℕ} (ht : t ≠ 0) : + P.map (rewardNoise A R ν t) = + ((rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ)) ∘ₘ + P.map (fun ω ↦ (history A R (t - 1) ω, A t ω)) := by + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let Y : Ω → ℝ := rewardNoise A R ν t + let κ : Kernel ((Iic (t - 1) → Fin K × ℝ) × Fin K) ℝ := + (rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) + have h_cond : HasCondDistrib Y X κ P := by + simpa [X, Y, κ] using + hasCondDistrib_rewardNoise_history_action (A := A) (R := R) (ν := ν) h ht + have hX_meas : Measurable X := by + dsimp [X] + exact Measurable.prodMk + ((h.measurable_history (t - 1))) + (h.measurable_action t) + have hX_law : HasLaw X (P.map X) P := ⟨hX_meas.aemeasurable, rfl⟩ + have h_pair := HasLaw.prod_of_hasCondDistrib hX_law h_cond + have h_snd := congrArg Measure.snd h_pair.map_eq + rw [Measure.snd_map_prodMk₀ hX_meas.aemeasurable, Measure.snd_compProd] at h_snd + simpa [X, Y, κ] using h_snd + +/-- Exponential integrability of the realized positive-time reward noise follows from the +arm-wise subgaussian reward-noise assumption. + +The proof first packages the centered reward-noise law as a kernel-level subgaussian MGF statement, +then uses `rewardNoise_map_eq_historyAction_kernel_comp` to transport the resulting integrability +back to the actual process variable. -/ +lemma rewardNoise_integrable_exp_mul_history_action {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (u : ℝ) : + Integrable (fun ω ↦ Real.exp (u * rewardNoise A R ν t ω)) P := by + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let κ : Kernel ((Iic (t - 1) → Fin K × ℝ) × Fin K) ℝ := + (rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) + have h_kernel : + Kernel.HasSubgaussianMGF id σ2 κ (P.map X) := by + simpa [X, κ] using + rewardNoise_history_action_kernelSubgaussian (A := A) (R := R) (P := P) + (ν := ν) hν t + have h_int := h_kernel.integrable_exp_mul u + have h_law : + P.map (rewardNoise A R ν t) = κ ∘ₘ P.map X := by + simpa [X, κ] using + rewardNoise_map_eq_historyAction_kernel_comp (A := A) (R := R) (ν := ν) h ht + rw [← h_law] at h_int + have hY_ae : AEMeasurable (rewardNoise A R ν t) P := + (hasCondDistrib_rewardNoise_history_action (A := A) (R := R) (ν := ν) h ht).aemeasurable_snd + simpa [Function.comp_def] using + (integrable_map_measure (by fun_prop) hY_ae).mp h_int + +/-- Conditional MGF bound for the realized positive-time LinUCB reward noise. + +Given the previous history and the selected action at time `t`, the conditional expectation of +`exp (u * η_t)` is bounded by the arm-wise subgaussian MGF bound +`exp (σ2 * u^2 / 2)`, where `η_t = R_t - μ(A_t)`. + +This is the scalar conditional-subgaussian statement that the textbook vector self-normalized +argument uses for predictable projections of the noise process. -/ +lemma rewardNoise_ae_condExp_exp_le_history_action {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (u : ℝ) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp (u * rewardNoise A R ν t ω') | + (inferInstance : MeasurableSpace ((Iic (t - 1) → Fin K × ℝ) × Fin K)).comap + (fun ω ↦ (history A R (t - 1) ω, A t ω))] ω + ≤ Real.exp (σ2 * u ^ 2 / 2) := by + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let Y : Ω → ℝ := rewardNoise A R ν t + let κ : Kernel ((Iic (t - 1) → Fin K × ℝ) × Fin K) ℝ := + (rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) + have h_cond : HasCondDistrib Y X κ P := by + simpa [X, Y, κ] using + hasCondDistrib_rewardNoise_history_action (A := A) (R := R) (ν := ν) h ht + have hX_meas : Measurable X := by + dsimp [X] + exact Measurable.prodMk + ((h.measurable_history (t - 1))) + (h.measurable_action t) + have hY_ae : AEMeasurable Y P := h_cond.aemeasurable_snd + have hf : StronglyMeasurable (fun η : ℝ ↦ Real.exp (u * η)) := by + fun_prop + have h_int : Integrable (fun ω ↦ Real.exp (u * Y ω)) P := by + simpa [Y] using + rewardNoise_integrable_exp_mul_history_action (A := A) (R := R) (ν := ν) h hν ht u + have h_ce : + P[fun ω ↦ Real.exp (u * Y ω) | + (inferInstance : MeasurableSpace ((Iic (t - 1) → Fin K × ℝ) × Fin K)).comap X] + =ᵐ[P] fun ω ↦ ∫ η, Real.exp (u * η) ∂condDistrib Y X P (X ω) := by + exact condExp_ae_eq_integral_condDistrib (μ := P) (X := X) (Y := Y) + hX_meas hY_ae hf h_int + have h_kernel_eq : ∀ᵐ ω ∂P, condDistrib Y X P (X ω) = κ (X ω) := + ae_of_ae_map hX_meas.aemeasurable h_cond.condDistrib_eq + have h_kernel_subG : + ∀ z : (Iic (t - 1) → Fin K × ℝ) × Fin K, + HasSubgaussianMGF id σ2 (κ z) := by + intro z + simpa [κ] using + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian + (K := K) (ν := ν) hν).prodMkLeft z + filter_upwards [h_ce, h_kernel_eq] with ω h_ceω hκω + rw [h_ceω, hκω] + simpa [X, Y, κ, mgf] using (h_kernel_subG (X ω)).mgf_le u + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Variance proxy for multiplying a subgaussian scalar by a deterministic coefficient. -/ +def scalarProjectionVariance (σ2 : ℝ≥0) (q : ℝ) : ℝ≥0 := + ⟨q ^ 2, sq_nonneg q⟩ * σ2 + +/-- Conditional MGF bound for predictable scalar projections of positive-time LinUCB reward noise. + +If `q` is a measurable function of the previous history and current selected action, then +`q(history, action) * η_t` is conditionally subgaussian given that same history/action information, +with variance proxy multiplied by `q(history, action)^2`. + +The explicit integrability hypothesis is needed for this fully general statement because an +unbounded predictable coefficient does not automatically preserve global exponential +integrability. Bounded-coefficient corollaries can discharge that hypothesis separately. -/ +lemma rewardNoise_constMul_ae_condExp_exp_le_history_action {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) + (q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ) (hq : Measurable q) (u : ℝ) + (h_int : + Integrable + (fun ω ↦ Real.exp + (u * (q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω))) P) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp + (u * (q (history A R (t - 1) ω', A t ω') * + rewardNoise A R ν t ω')) | + (inferInstance : MeasurableSpace ((Iic (t - 1) → Fin K × ℝ) × Fin K)).comap + (fun ω ↦ (history A R (t - 1) ω, A t ω))] ω + ≤ Real.exp (scalarProjectionVariance σ2 + (q (history A R (t - 1) ω, A t ω)) * u ^ 2 / 2) := by + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let Y : Ω → ℝ := rewardNoise A R ν t + let κ : Kernel ((Iic (t - 1) → Fin K × ℝ) × Fin K) ℝ := + (rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) + let f : ((Iic (t - 1) → Fin K × ℝ) × Fin K) × ℝ → ℝ := + fun p ↦ Real.exp (u * (q p.1 * p.2)) + have h_cond : HasCondDistrib Y X κ P := by + simpa [X, Y, κ] using + hasCondDistrib_rewardNoise_history_action (A := A) (R := R) (ν := ν) h ht + have hX_meas : Measurable X := by + dsimp [X] + exact Measurable.prodMk + ((h.measurable_history (t - 1))) + (h.measurable_action t) + have hY_ae : AEMeasurable Y P := h_cond.aemeasurable_snd + have hf : StronglyMeasurable f := by + dsimp [f] + fun_prop + have h_ce : + P[fun ω ↦ f (X ω, Y ω) | + (inferInstance : MeasurableSpace ((Iic (t - 1) → Fin K × ℝ) × Fin K)).comap X] + =ᵐ[P] fun ω ↦ ∫ η, f (X ω, η) ∂condDistrib Y X P (X ω) := by + exact condExp_prod_ae_eq_integral_condDistrib (μ := P) (X := X) (Y := Y) + hX_meas hY_ae hf (by simpa [X, Y, f] using h_int) + have h_kernel_eq : ∀ᵐ ω ∂P, condDistrib Y X P (X ω) = κ (X ω) := + ae_of_ae_map hX_meas.aemeasurable h_cond.condDistrib_eq + have h_kernel_subG : + ∀ z : (Iic (t - 1) → Fin K × ℝ) × Fin K, + HasSubgaussianMGF (fun η ↦ q z * η) (scalarProjectionVariance σ2 (q z)) (κ z) := by + intro z + simpa [scalarProjectionVariance, κ] using + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian + (K := K) (ν := ν) hν).prodMkLeft_constMul q z + filter_upwards [h_ce, h_kernel_eq] with ω h_ceω hκω + rw [h_ceω, hκω] + simpa [X, Y, κ, f, mgf] using (h_kernel_subG (X ω)).mgf_le u + +/-- Conditional exponential-supermartingale one-step bound with the realized predictable variance. + +This is the local scalar ingredient needed by the textbook self-normalized LinUCB proof. Unlike +the later `HasCondSubgaussianMGF` wrappers, it does not replace the predictable coefficient +`q(history, action)` by a uniform deterministic bound. The variance penalty remains the realized +quantity `q(history, action)^2 * σ2`. -/ +lemma rewardNoise_constMul_ae_condExp_exp_sub_realizedVariance_le_one + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) + (q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ) (hq : Measurable q) (u : ℝ) + (h_int : + Integrable + (fun ω ↦ Real.exp + (u * (q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω))) P) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp + (u * (q (history A R (t - 1) ω', A t ω') * + rewardNoise A R ν t ω') - + scalarProjectionVariance σ2 + (q (history A R (t - 1) ω', A t ω')) * u ^ 2 / 2) | + (inferInstance : MeasurableSpace ((Iic (t - 1) → Fin K × ℝ) × Fin K)).comap + (fun ω ↦ (history A R (t - 1) ω, A t ω))] ω + ≤ 1 := by + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let Y : Ω → ℝ := rewardNoise A R ν t + let κ : Kernel ((Iic (t - 1) → Fin K × ℝ) × Fin K) ℝ := + (rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) + let f : ((Iic (t - 1) → Fin K × ℝ) × Fin K) × ℝ → ℝ := + fun p ↦ Real.exp + (u * (q p.1 * p.2) - scalarProjectionVariance σ2 (q p.1) * u ^ 2 / 2) + have h_cond : HasCondDistrib Y X κ P := by + simpa [X, Y, κ] using + hasCondDistrib_rewardNoise_history_action (A := A) (R := R) (ν := ν) h ht + have hX_meas : Measurable X := by + dsimp [X] + exact Measurable.prodMk + ((h.measurable_history (t - 1))) + (h.measurable_action t) + have hY_ae : AEMeasurable Y P := h_cond.aemeasurable_snd + have hf_meas : Measurable f := by + dsimp [f] + simp only [scalarProjectionVariance, NNReal.coe_mul] + fun_prop + have hf : StronglyMeasurable f := hf_meas.stronglyMeasurable + have h_int_sub : Integrable (fun ω ↦ f (X ω, Y ω)) P := by + have h_target : + AEStronglyMeasurable (fun ω ↦ f (X ω, Y ω)) P := by + exact (hf_meas.comp_aemeasurable (hX_meas.aemeasurable.prodMk hY_ae)).aestronglyMeasurable + refine Integrable.mono h_int h_target ?_ + refine Filter.Eventually.of_forall fun ω ↦ ?_ + have hvar_nonneg : + 0 ≤ (scalarProjectionVariance σ2 (q (X ω)) : ℝ) * u ^ 2 / 2 := by + positivity + simp only [f, X, Y, Real.norm_of_nonneg (Real.exp_nonneg _)] + exact Real.exp_le_exp.mpr (by linarith) + have h_ce : + P[fun ω ↦ f (X ω, Y ω) | + (inferInstance : MeasurableSpace ((Iic (t - 1) → Fin K × ℝ) × Fin K)).comap X] + =ᵐ[P] fun ω ↦ ∫ η, f (X ω, η) ∂condDistrib Y X P (X ω) := by + exact condExp_prod_ae_eq_integral_condDistrib (μ := P) (X := X) (Y := Y) + hX_meas hY_ae hf h_int_sub + have h_kernel_eq : ∀ᵐ ω ∂P, condDistrib Y X P (X ω) = κ (X ω) := + ae_of_ae_map hX_meas.aemeasurable h_cond.condDistrib_eq + have h_kernel_subG : + ∀ z : (Iic (t - 1) → Fin K × ℝ) × Fin K, + HasSubgaussianMGF (fun η ↦ q z * η) (scalarProjectionVariance σ2 (q z)) (κ z) := by + intro z + simpa [scalarProjectionVariance, κ] using + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian + (K := K) (ν := ν) hν).prodMkLeft_constMul q z + filter_upwards [h_ce, h_kernel_eq] with ω h_ceω hκω + rw [h_ceω, hκω] + let z : (Iic (t - 1) → Fin K × ℝ) × Fin K := X ω + let c : ℝ := (scalarProjectionVariance σ2 (q z) : ℝ) * u ^ 2 / 2 + have h_int_mgf : Integrable (fun η ↦ Real.exp (u * (q z * η))) (κ z) := + (h_kernel_subG z).integrable_exp_mul u + have h_integral_eq : + (∫ η, f (X ω, η) ∂κ (X ω)) = + (∫ η, Real.exp (u * (q z * η)) ∂κ z) / Real.exp c := by + simp only [z, c, f] + simp_rw [Real.exp_sub] + rw [MeasureTheory.integral_div] + have h_mgf_le : + (∫ η, Real.exp (u * (q z * η)) ∂κ z) ≤ Real.exp c := by + simpa [mgf, c] using (h_kernel_subG z).mgf_le u + calc + (∫ η, f (X ω, η) ∂κ (X ω)) + = (∫ η, Real.exp (u * (q z * η)) ∂κ z) / Real.exp c := h_integral_eq + _ ≤ Real.exp c / Real.exp c := + div_le_div_of_nonneg_right h_mgf_le (le_of_lt (Real.exp_pos c)) + _ = 1 := by + exact div_self (Real.exp_ne_zero c) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If an event is measurable in a smaller sigma-algebra, then a `μ`-a.e. proof of the event can +be viewed as a `μ.trim hm`-a.e. proof. + +This is a small measure-theory bridge used below to package conditional expectation bounds into +Mathlib's `HasCondSubgaussianMGF`, whose MGF bound is stated over the trimmed conditioning +measure. -/ +lemma ae_trim_of_ae_of_measurableSet (μ : Measure Ω) {m' : MeasurableSpace Ω} (hm' : m' ≤ mΩ) + {p : Ω → Prop} (hp : @MeasurableSet Ω m' {ω | p ω}) (hμ : ∀ᵐ ω ∂μ, p ω) : + ∀ᵐ ω ∂μ.trim hm', p ω := by + rw [ae_iff] at hμ ⊢ + have hp_compl : @MeasurableSet Ω m' {ω | ¬ p ω} := by + simpa only [Set.compl_setOf] using hp.compl + rw [trim_measurableSet_eq hm' hp_compl] + exact hμ + +/-- Positive-time LinUCB reward noise is conditionally subgaussian with respect to the filtration +generated by the previous history and the current selected action. + +This repackages `rewardNoise_ae_condExp_exp_le_history_action` into Mathlib's +`HasCondSubgaussianMGF` API, which is the API used by the existing martingale-sum concentration +lemmas. -/ +lemma rewardNoise_hasCondSubgaussianMGF_filtrationAction {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) : + HasCondSubgaussianMGF + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t) + ((IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback).le t) + (rewardNoise A R ν t) σ2 P := by + let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let mX : MeasurableSpace Ω := ℱ t + have hmX : mX ≤ mΩ := ℱ.le t + let Y : Ω → ℝ := rewardNoise A R ν t + change Kernel.HasSubgaussianMGF Y σ2 + (@condExpKernel Ω mΩ _ P _ mX) (@Measure.trim Ω mX mΩ P hmX) + refine Kernel.HasSubgaussianMGF.of_rat (X := Y) (c := σ2) + (κ := @condExpKernel Ω mΩ _ P _ mX) (ν := @Measure.trim Ω mX mΩ P hmX) ?_ ?_ + · intro u + rw [condExpKernel_comp_trim (Ω := Ω) (m := mX) (mΩ := mΩ) (μ := P) hmX] + simpa [Y] using + rewardNoise_integrable_exp_mul_history_action (A := A) (R := R) (ν := ν) h hν ht u + · intro q + let u : ℝ := q + have h_int : Integrable (fun ω ↦ Real.exp (u * Y ω)) P := by + simpa [Y] using + rewardNoise_integrable_exp_mul_history_action (A := A) (R := R) (ν := ν) h hν ht u + have h_condExp_eq : + P[fun ω ↦ Real.exp (u * Y ω) | mX] + =ᵐ[P.trim hmX] fun ω ↦ + ∫ y, Real.exp (u * Y y) ∂(@condExpKernel Ω mΩ _ P _ mX) ω := by + exact condExp_ae_eq_trim_integral_condExpKernel (Ω := Ω) (m := mX) + (mΩ := mΩ) (μ := P) hmX h_int + have h_condExp_le_P : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp (u * Y ω') | mX] ω ≤ Real.exp (σ2 * u ^ 2 / 2) := by + have h_le := rewardNoise_ae_condExp_exp_le_history_action (A := A) (R := R) + (ν := ν) h hν ht u + simpa [Y, mX, ℱ, u, IsAlgEnvSeq.filtrationAction_eq_comap + (A := A) (Y := R) t ht] using h_le + have h_event_meas : + @MeasurableSet Ω mX + {ω | P[fun ω' ↦ Real.exp (u * Y ω') | mX] ω ≤ Real.exp (σ2 * u ^ 2 / 2)} := by + exact measurableSet_le stronglyMeasurable_condExp.measurable measurable_const + have h_condExp_le_trim : + ∀ᵐ ω ∂P.trim hmX, + P[fun ω' ↦ Real.exp (u * Y ω') | mX] ω ≤ Real.exp (σ2 * u ^ 2 / 2) := + ae_trim_of_ae_of_measurableSet P hmX h_event_meas h_condExp_le_P + filter_upwards [h_condExp_eq, h_condExp_le_trim] with ω h_eq h_le + change (∫ y, Real.exp (u * Y y) ∂(@condExpKernel Ω mΩ _ P _ mX) ω) ≤ + Real.exp (σ2 * u ^ 2 / 2) + rw [← h_eq] + exact h_le + +/-- Bounded predictable scalar coefficients preserve exponential integrability of the projected +reward noise. + +If `|q(history, action)| ≤ Q`, then `exp (u * q_t * η_t)` is dominated by +`exp (|u| Q * η_t) + exp (-|u| Q * η_t)`, and both endpoint exponentials are integrable by the +arm-wise reward-noise subgaussian assumption. -/ +lemma rewardNoise_constMul_integrable_exp_mul_history_action_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) + (q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ) (hq : Measurable q) + (Q : ℝ) (hQ : 0 ≤ Q) (hq_bound : ∀ z, |q z| ≤ Q) (u : ℝ) : + Integrable + (fun ω ↦ Real.exp + (u * (q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω))) P := by + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let Y : Ω → ℝ := rewardNoise A R ν t + let c : ℝ := |u| * Q + have hX_meas : Measurable X := by + dsimp [X] + exact Measurable.prodMk + ((h.measurable_history (t - 1))) + (h.measurable_action t) + have hY_meas : Measurable Y := by + dsimp [Y, rewardNoise] + exact (h.measurable_feedback t).sub ((measurable_rewardMean ν).comp (h.measurable_action t)) + have h_target : + AEStronglyMeasurable + (fun ω ↦ Real.exp (u * (q (X ω) * Y ω))) P := by + exact (measurable_exp.comp + (measurable_const.mul (((hq.comp hX_meas).mul hY_meas)))).aestronglyMeasurable + have h_pos : Integrable (fun ω ↦ Real.exp (c * Y ω)) P := by + simpa [Y, c] using + rewardNoise_integrable_exp_mul_history_action (A := A) (R := R) (ν := ν) h hν ht c + have h_neg : Integrable (fun ω ↦ Real.exp ((-c) * Y ω)) P := by + simpa [Y, c] using + rewardNoise_integrable_exp_mul_history_action (A := A) (R := R) (ν := ν) h hν ht (-c) + refine Integrable.mono (h_pos.add h_neg) h_target ?_ + refine Filter.Eventually.of_forall fun ω ↦ ?_ + have hc_nonneg : 0 ≤ c := mul_nonneg (abs_nonneg u) hQ + have harg_abs : + |u * (q (X ω) * Y ω)| ≤ |c * Y ω| := by + calc + |u * (q (X ω) * Y ω)| + = |u| * |q (X ω)| * |Y ω| := by + rw [abs_mul, abs_mul, mul_assoc] + _ ≤ |u| * Q * |Y ω| := by + gcongr + exact hq_bound (X ω) + _ = |c * Y ω| := by + rw [abs_mul, abs_of_nonneg hc_nonneg] + have harg_le_abs : u * (q (X ω) * Y ω) ≤ |c * Y ω| := + (le_abs_self _).trans harg_abs + calc + ‖Real.exp (u * (q (X ω) * Y ω))‖ + = Real.exp (u * (q (X ω) * Y ω)) := Real.norm_of_nonneg (Real.exp_nonneg _) + _ ≤ Real.exp |c * Y ω| := Real.exp_le_exp.mpr harg_le_abs + _ ≤ Real.exp (c * Y ω) + Real.exp ((-c) * Y ω) := by + simpa [neg_mul] using Real.exp_abs_le (c * Y ω) + _ = ‖((fun ω ↦ Real.exp (c * Y ω)) + fun ω ↦ Real.exp ((-c) * Y ω)) ω‖ := by + rw [Pi.add_apply, Real.norm_of_nonneg] + positivity + +/-- Bounded predictable scalar projections of positive-time LinUCB reward noise are conditionally +subgaussian in Mathlib's standard `HasCondSubgaussianMGF` API. + +The coefficient `q` may depend on the previous history and current selected action. If its scaled +variance proxy `q^2 * σ2` is bounded by a deterministic `c`, and the projected exponentials are +integrable, then the projected noise process is conditionally subgaussian with parameter `c`. + +The remaining explicit integrability assumption is the only extra analytic side condition in this +wrapper. It is unavoidable at this level of generality for unbounded predictable coefficients. -/ +lemma rewardNoise_constMul_hasCondSubgaussianMGF_filtrationAction_of_integrable + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 c : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) + (q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ) (hq : Measurable q) + (hc : ∀ z, scalarProjectionVariance σ2 (q z) ≤ c) + (h_int : ∀ u : ℝ, + Integrable + (fun ω ↦ Real.exp + (u * (q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω))) P) : + HasCondSubgaussianMGF + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t) + ((IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback).le t) + (fun ω ↦ q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω) + c P := by + let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let mX : MeasurableSpace Ω := ℱ t + have hmX : mX ≤ mΩ := ℱ.le t + let X : Ω → (Iic (t - 1) → Fin K × ℝ) × Fin K := + fun ω ↦ (history A R (t - 1) ω, A t ω) + let Z : Ω → ℝ := fun ω ↦ q (X ω) * rewardNoise A R ν t ω + change Kernel.HasSubgaussianMGF Z c + (@condExpKernel Ω mΩ _ P _ mX) (@Measure.trim Ω mX mΩ P hmX) + refine Kernel.HasSubgaussianMGF.of_rat (X := Z) (c := c) + (κ := @condExpKernel Ω mΩ _ P _ mX) (ν := @Measure.trim Ω mX mΩ P hmX) ?_ ?_ + · intro u + rw [condExpKernel_comp_trim (Ω := Ω) (m := mX) (mΩ := mΩ) (μ := P) hmX] + simpa [X, Z] using h_int u + · intro r + let u : ℝ := r + have h_int_u : Integrable (fun ω ↦ Real.exp (u * Z ω)) P := by + simpa [X, Z] using h_int u + have h_condExp_eq : + P[fun ω ↦ Real.exp (u * Z ω) | mX] + =ᵐ[P.trim hmX] fun ω ↦ + ∫ y, Real.exp (u * Z y) ∂(@condExpKernel Ω mΩ _ P _ mX) ω := by + exact condExp_ae_eq_trim_integral_condExpKernel (Ω := Ω) (m := mX) + (mΩ := mΩ) (μ := P) hmX h_int_u + have h_condExp_le_P : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp (u * Z ω') | mX] ω ≤ Real.exp (c * u ^ 2 / 2) := by + have h_le : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp (u * Z ω') | mX] ω ≤ + Real.exp (scalarProjectionVariance σ2 (q (X ω)) * u ^ 2 / 2) := by + simpa [X, Z, mX, ℱ, IsAlgEnvSeq.filtrationAction_eq_comap + (A := A) (Y := R) t ht] using + rewardNoise_constMul_ae_condExp_exp_le_history_action + (A := A) (R := R) (ν := ν) h hν ht q hq u (by simpa [X, Z] using h_int_u) + filter_upwards [h_le] with ω hω + refine hω.trans ?_ + have hcω : (scalarProjectionVariance σ2 (q (X ω)) : ℝ) ≤ (c : ℝ) := by + exact_mod_cast hc (X ω) + gcongr + have h_event_meas : + @MeasurableSet Ω mX + {ω | P[fun ω' ↦ Real.exp (u * Z ω') | mX] ω ≤ Real.exp (c * u ^ 2 / 2)} := by + exact measurableSet_le stronglyMeasurable_condExp.measurable measurable_const + have h_condExp_le_trim : + ∀ᵐ ω ∂P.trim hmX, + P[fun ω' ↦ Real.exp (u * Z ω') | mX] ω ≤ Real.exp (c * u ^ 2 / 2) := + ae_trim_of_ae_of_measurableSet P hmX h_event_meas h_condExp_le_P + filter_upwards [h_condExp_eq, h_condExp_le_trim] with ω h_eq h_le + change (∫ y, Real.exp (u * Z y) ∂(@condExpKernel Ω mΩ _ P _ mX) ω) ≤ + Real.exp (c * u ^ 2 / 2) + rw [← h_eq] + exact h_le + +/-- Bounded predictable scalar projections of positive-time LinUCB reward noise are conditionally +subgaussian without any separate integrability hypothesis. + +This is the bounded-coefficient version of +`rewardNoise_constMul_hasCondSubgaussianMGF_filtrationAction_of_integrable`. If +`|q(history, action)| ≤ Q`, then the variance proxy is bounded by `Q^2 * σ2`, and exponential +integrability follows from the arm-wise reward-noise subgaussian assumption. -/ +lemma rewardNoise_constMul_hasCondSubgaussianMGF_filtrationAction_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) + (q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ) (hq : Measurable q) + (Q : ℝ) (hQ : 0 ≤ Q) (hq_bound : ∀ z, |q z| ≤ Q) : + HasCondSubgaussianMGF + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t) + ((IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback).le t) + (fun ω ↦ q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω) + (⟨Q ^ 2, sq_nonneg Q⟩ * σ2) P := by + refine rewardNoise_constMul_hasCondSubgaussianMGF_filtrationAction_of_integrable + (A := A) (R := R) (ν := ν) h hν ht q hq ?_ ?_ + · intro z + dsimp [scalarProjectionVariance] + gcongr + have hQ_le_abs : Q ≤ |Q| := by + rw [abs_of_nonneg hQ] + exact sq_le_sq.2 ((hq_bound z).trans hQ_le_abs) + · intro u + exact rewardNoise_constMul_integrable_exp_mul_history_action_of_abs_le + (A := A) (R := R) (ν := ν) h hν ht q hq Q hQ hq_bound u + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A fixed-direction scalar projection of the LinUCB reward-feature noise. + +For positive time `t`, this is +`⟪v, x_{A_t}⟫ * η_t`, where `η_t = R_t - μ(A_t)`. The `t = 0` value is set to zero because the +repository's algorithm/environment sequence gives the clean history/action conditional law for +positive times. -/ +noncomputable def projectedRewardFeatureNoise + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (v : Feature d) (t : ℕ) (ω : Ω) : ℝ := + if t = 0 then 0 else dotProduct v (x (A t ω)) * rewardNoise A R ν t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Realized variance proxy of the fixed-direction positive-time reward-feature noise. + +At positive time `t`, this is `σ2 * ⟪v, x_{A_t}⟫²`. The time-zero value is set to zero to match +`projectedRewardFeatureNoise`, which also omits time zero from the positive-time martingale route. +-/ +noncomputable def projectedRewardFeatureNoiseRealizedVariance + (A : ℕ → Ω → Fin K) (σ2 : ℝ≥0) + (x : Fin K → Feature d) (v : Feature d) (t : ℕ) (ω : Ω) : ℝ := + if t = 0 then 0 else (scalarProjectionVariance σ2 (dotProduct v (x (A t ω))) : ℝ) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Exponential-supermartingale increment for a fixed direction of LinUCB reward-feature noise. + +This is `exp(u Y_t - u² V_t / 2)`, where `Y_t` is the fixed-direction projected noise and `V_t` +is its realized subgaussian variance proxy. -/ +noncomputable def projectedRewardFeatureNoiseExpIncrement + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (σ2 : ℝ≥0) (x : Fin K → Feature d) (v : Feature d) (u : ℝ) + (t : ℕ) (ω : Ω) : ℝ := + Real.exp + (u * projectedRewardFeatureNoise A R ν x v t ω - + projectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω * u ^ 2 / 2) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Cumulative realized variance proxy for a fixed direction of LinUCB reward-feature noise. -/ +noncomputable def projectedRewardFeatureNoiseRealizedVarianceSum + (A : ℕ → Ω → Fin K) (σ2 : ℝ≥0) + (x : Fin K → Feature d) (v : Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, projectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Finite-horizon fixed-direction exponential process with realized variance. + +This is the scalar process +`exp(u * ∑_{t 0` it is measurable with respect to +`filtrationAction n`. The `n = 0` case is constant. -/ +lemma stronglyAdapted_projectedRewardFeatureNoiseExpProcess_filtrationAction + (hA : ∀ n, Measurable (A n)) (hR : ∀ n, Measurable (R n)) + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) : + StronglyAdapted (IsAlgEnvSeq.filtrationAction hA hR) + (fun n ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) := by + intro n + by_cases hn : n = 0 + · subst n + simpa [projectedRewardFeatureNoiseExpProcess_zero] using + (stronglyMeasurable_const : + StronglyMeasurable[IsAlgEnvSeq.filtrationAction hA hR 0] + (fun _ : Ω ↦ (1 : ℝ))) + · have hsm := + stronglyMeasurable_projectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (A := A) (R := R) (ν := ν) (x := x) hA hR σ2 v u n + have hpost : + postActionFiltration (A := A) (R := R) hA hR (n - 1) = + IsAlgEnvSeq.filtrationAction hA hR n := by + simp [postActionFiltration, Nat.sub_add_cancel (Nat.pos_of_ne_zero hn)] + rw [← hpost] + exact hsm + +/-- A bounded fixed-direction projection of LinUCB reward-feature noise is conditionally +subgaussian. + +This is the scalar concentration input immediately before a vector self-normalized theorem. For a +fixed direction `v`, if `|⟪v, x_a⟫| ≤ Q` for every finite action `a`, then the positive-time +projected noise `⟪v, x_{A_t}⟫η_t` is conditionally subgaussian with variance proxy `Q^2 σ2`. -/ +lemma projectedRewardFeatureNoise_hasCondSubgaussianMGF_filtrationAction_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (v : Feature d) + (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) : + HasCondSubgaussianMGF + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t) + ((IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback).le t) + (projectedRewardFeatureNoise A R ν x v t) + (⟨Q ^ 2, sq_nonneg Q⟩ * σ2) P := by + rw [show projectedRewardFeatureNoise A R ν x v t = + fun ω ↦ dotProduct v (x (A t ω)) * rewardNoise A R ν t ω by + funext ω + simp [projectedRewardFeatureNoise, ht]] + exact rewardNoise_constMul_hasCondSubgaussianMGF_filtrationAction_of_abs_le + (A := A) (R := R) (ν := ν) h hν ht + (fun z : (Iic (t - 1) → Fin K × ℝ) × Fin K ↦ dotProduct v (x z.2)) + (measurable_projectedRewardFeatureCoeff (x := x) v) + Q hQ (fun z ↦ hQ_bound z.2) + +/-- One-step fixed-direction exponential bound with realized variance. + +For a fixed direction `v`, the positive-time projected noise +`⟪v, x_{A_t}⟫η_t` satisfies the exponential-supermartingale increment inequality with the +realized variance proxy `σ2 * ⟪v, x_{A_t}⟫²`. This is the fixed-direction form of the adaptive +variance step used before the Gaussian-mixture/self-normalized argument in the textbook proof. -/ +lemma projectedRewardFeatureNoise_ae_condExp_exp_sub_realizedVariance_le_one_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (v : Feature d) + (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ Real.exp + (u * projectedRewardFeatureNoise A R ν x v t ω' - + scalarProjectionVariance σ2 (dotProduct v (x (A t ω'))) * u ^ 2 / 2) | + IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t] ω + ≤ 1 := by + let q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ := + fun z ↦ dotProduct v (x z.2) + have hq : Measurable q := measurable_projectedRewardFeatureCoeff (x := x) v + have h_int : + Integrable + (fun ω ↦ Real.exp + (u * (q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω))) P := + rewardNoise_constMul_integrable_exp_mul_history_action_of_abs_le + (A := A) (R := R) (ν := ν) h hν ht q hq Q hQ (fun z ↦ hQ_bound z.2) u + have h_step := + rewardNoise_constMul_ae_condExp_exp_sub_realizedVariance_le_one + (A := A) (R := R) (ν := ν) h hν ht q hq u h_int + simpa [q, projectedRewardFeatureNoise, ht, + IsAlgEnvSeq.filtrationAction_eq_comap (A := A) (Y := R) t ht] using h_step + +/-- Integrability of a fixed-direction exponential-supermartingale increment under a bounded +predictable projection. -/ +lemma projectedRewardFeatureNoiseExpIncrement_integrable_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (v : Feature d) + (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + Integrable + (fun ω ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) P := by + let q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ := + fun z ↦ dotProduct v (x z.2) + have hq : Measurable q := measurable_projectedRewardFeatureCoeff (x := x) v + have h_base : + Integrable + (fun ω ↦ Real.exp + (u * (q (history A R (t - 1) ω, A t ω) * + rewardNoise A R ν t ω))) P := + rewardNoise_constMul_integrable_exp_mul_history_action_of_abs_le + (A := A) (R := R) (ν := ν) h hν ht q hq Q hQ (fun z ↦ hQ_bound z.2) u + have h_coeff_meas : Measurable fun ω ↦ dotProduct v (x (A t ω)) := + (measurable_of_countable (fun a : Fin K ↦ dotProduct v (x a))).comp + (h.measurable_action t) + have h_noise_meas : Measurable (rewardNoise A R ν t) := by + change Measurable (fun ω ↦ R t ω - (ν (A t ω))[id]) + exact (h.measurable_feedback t).sub ((measurable_rewardMean ν).comp (h.measurable_action t)) + have h_proj_meas : + Measurable fun ω ↦ projectedRewardFeatureNoise A R ν x v t ω := by + simpa [projectedRewardFeatureNoise, ht] using h_coeff_meas.mul h_noise_meas + have h_var_meas : + Measurable fun ω ↦ projectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω := by + simpa [projectedRewardFeatureNoiseRealizedVariance, ht, Function.comp_def] using + (measurable_of_countable + (fun a : Fin K ↦ (scalarProjectionVariance σ2 (dotProduct v (x a)) : ℝ))).comp + (h.measurable_action t) + have h_target : + AEStronglyMeasurable + (fun ω ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) P := by + refine (measurable_exp.comp ?_).aestronglyMeasurable + exact (measurable_const.mul h_proj_meas).sub + ((h_var_meas.mul measurable_const).div_const 2) + refine Integrable.mono h_base h_target ?_ + refine Filter.Eventually.of_forall fun ω ↦ ?_ + have hpenalty_nonneg : + 0 ≤ (scalarProjectionVariance σ2 (dotProduct v (x (A t ω))) : ℝ) * u ^ 2 / 2 := by + positivity + simp only [projectedRewardFeatureNoiseExpIncrement, projectedRewardFeatureNoise, + projectedRewardFeatureNoiseRealizedVariance, ht, if_false, q, + Real.norm_of_nonneg (Real.exp_nonneg _)] + exact Real.exp_le_exp.mpr (by linarith [hpenalty_nonneg]) + +/-- One-step conditional expectation bound for the named fixed-direction exponential increment. -/ +lemma projectedRewardFeatureNoiseExpIncrement_ae_condExp_le_one_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (v : Feature d) + (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | + IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t] ω + ≤ 1 := by + simpa [projectedRewardFeatureNoiseExpIncrement, + projectedRewardFeatureNoiseRealizedVariance, ht] using + projectedRewardFeatureNoise_ae_condExp_exp_sub_realizedVariance_le_one_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν ht v Q hQ hQ_bound u + +/-- Predictable-multiplier form of the one-step exponential-supermartingale bound. + +This is the induction step needed for a finite-horizon fixed-direction exponential process. If +`M` is nonnegative and measurable at the current post-action sigma-algebra, then multiplying the +next realized-variance exponential increment by `M` keeps conditional expectation bounded by `M`. +-/ +lemma projectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (v : Feature d) + (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) + (M : Ω → ℝ) + (hM_meas : AEStronglyMeasurable[IsAlgEnvSeq.filtrationAction + h.measurable_action h.measurable_feedback t] M P) + (hM_nonneg : 0 ≤ᵐ[P] M) + (hM_mul_int : + Integrable + (fun ω ↦ M ω * projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) P) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | + IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t] ω + ≤ M ω := by + let ℱt := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t + have h_inc_int : + Integrable + (fun ω ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) P := + projectedRewardFeatureNoiseExpIncrement_integrable_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν ht v Q hQ hQ_bound u + have h_pull : + P[fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω | ℱt] + =ᵐ[P] fun ω ↦ + M ω * + P[fun ω' ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | + ℱt] ω := by + exact condExp_mul_of_aestronglyMeasurable_left hM_meas hM_mul_int h_inc_int + have h_step : + ∀ᵐ ω ∂P, + P[fun ω' ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | ℱt] ω + ≤ 1 := by + simpa [ℱt] using + projectedRewardFeatureNoiseExpIncrement_ae_condExp_le_one_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν ht v Q hQ hQ_bound u + filter_upwards [h_pull, h_step, hM_nonneg] with ω h_pullω h_stepω hMω + rw [h_pullω] + calc + M ω * + P[fun ω' ↦ projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | ℱt] ω + ≤ M ω * 1 := mul_le_mul_of_nonneg_left h_stepω hMω + _ = M ω := by simp + +/-- A bounded fixed-direction projection of the accumulated LinUCB reward-feature noise is +subgaussian by Mathlib's scalar martingale-sum theorem. + +This is not the full vector self-normalized concentration theorem. It is the scalar fixed-direction +ingredient: after choosing a direction `v`, the sum of +`⟪v, x_{A_t}⟫η_t` is subgaussian with variance proxy equal to the sum of the per-time proxy bounds. +-/ +lemma projectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) : + HasSubgaussianMGF + (fun ω ↦ ∑ t ∈ range n, projectedRewardFeatureNoise A R ν x v t ω) + (∑ t ∈ range n, if t = 0 then 0 else (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)) P := by + let ℱ := postActionFiltration (A := A) (R := R) h.measurable_action h.measurable_feedback + let Y : ℕ → Ω → ℝ := projectedRewardFeatureNoise A R ν x v + let cY : ℕ → ℝ≥0 := + fun t ↦ if t = 0 then 0 else ⟨Q ^ 2, sq_nonneg Q⟩ * σ2 + have h_adapted : StronglyAdapted ℱ Y := by + simpa [ℱ, Y] using + stronglyAdapted_projectedRewardFeatureNoise_postActionFiltration + (A := A) (R := R) (ν := ν) (x := x) h.measurable_action h.measurable_feedback v + have h0 : HasSubgaussianMGF (Y 0) (cY 0) P := by + have hY0 : Y 0 = (0 : Ω → ℝ) := by + funext ω + simp [Y, projectedRewardFeatureNoise] + have hcY0 : cY 0 = 0 := by + simp [cY] + rw [hY0, hcY0] + exact HasSubgaussianMGF.zero + have h_subG : + ∀ i < n - 1, HasCondSubgaussianMGF (ℱ i) (ℱ.le i) (Y (i + 1)) (cY (i + 1)) P := by + intro i _hi + simpa [ℱ, Y, cY, postActionFiltration] using + projectedRewardFeatureNoise_hasCondSubgaussianMGF_filtrationAction_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν (Nat.succ_ne_zero i) + v Q hQ hQ_bound + simpa [Y, cY] using + HasSubgaussianMGF.sum_of_hasCondSubgaussianMGF (μ := P) (ℱ := ℱ) (Y := Y) (cY := cY) + h_adapted h0 n h_subG + +/-- Integrability of the finite-horizon fixed-direction exponential process. + +The realized-variance penalty is nonnegative, so this process is pointwise bounded by +`exp(u * ∑ projectedRewardFeatureNoise)`, whose integrability follows from the existing scalar +subgaussian martingale-sum theorem. -/ +lemma projectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) (n : ℕ) : + Integrable + (fun ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) P := by + have h_base : + Integrable + (fun ω ↦ Real.exp + (u * (∑ t ∈ range n, projectedRewardFeatureNoise A R ν x v t ω))) P := + (projectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound n).integrable_exp_mul u + have h_target : + AEStronglyMeasurable + (fun ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) P := + ((stronglyMeasurable_projectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (A := A) (R := R) (ν := ν) (x := x) h.measurable_action h.measurable_feedback + σ2 v u n).mono + ((postActionFiltration (A := A) (R := R) h.measurable_action h.measurable_feedback).le + (n - 1))).aestronglyMeasurable + refine Integrable.mono h_base h_target ?_ + refine Filter.Eventually.of_forall fun ω ↦ ?_ + have hpenalty_nonneg : + 0 ≤ projectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω * u ^ 2 / 2 := by + have hsum_nonneg : + 0 ≤ projectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω := + projectedRewardFeatureNoiseRealizedVarianceSum_nonneg (A := A) (σ2 := σ2) + (x := x) (v := v) (n := n) (ω := ω) + positivity + simp only [projectedRewardFeatureNoiseExpProcess, Real.norm_of_nonneg (Real.exp_nonneg _)] + exact Real.exp_le_exp.mpr (by linarith [hpenalty_nonneg]) + +/-- Finite-horizon fixed-direction exponential-supermartingale bound. + +For every fixed direction `v` and scalar `u`, the realized-variance exponential process has +expectation at most one: +`E exp(u ∑ Y_t - u²/2 ∑ V_t) ≤ 1`. + +This is still scalar. The remaining textbook self-normalized LinUCB step is the Gaussian-mixture +argument that integrates this inequality over directions and converts it into the determinant +self-normalized confidence radius. -/ +lemma integral_projectedRewardFeatureNoiseExpProcess_le_one_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + ∫ ω, projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω ∂P ≤ 1 := by + induction n with + | zero => + simp [projectedRewardFeatureNoiseExpProcess_zero] + | succ n ih => + by_cases hn : n = 0 + · subst n + simp [projectedRewardFeatureNoiseExpProcess_succ, + projectedRewardFeatureNoiseExpProcess_zero, + projectedRewardFeatureNoiseExpIncrement_zero] + · let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let M : Ω → ℝ := fun ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω + have hM_meas : AEStronglyMeasurable[ℱ n] M P := by + have hsm := + stronglyMeasurable_projectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (A := A) (R := R) (ν := ν) (x := x) h.measurable_action h.measurable_feedback + σ2 v u n + have hpost : + postActionFiltration (A := A) (R := R) h.measurable_action h.measurable_feedback + (n - 1) = ℱ n := by + simp [postActionFiltration, ℱ, Nat.sub_add_cancel (Nat.pos_of_ne_zero hn)] + have hsm_F : StronglyMeasurable[ℱ n] M := by + rw [← hpost] + simpa [M] using hsm + exact hsm_F.aestronglyMeasurable + have hM_nonneg : 0 ≤ᵐ[P] M := + Filter.Eventually.of_forall fun ω ↦ + projectedRewardFeatureNoiseExpProcess_nonneg (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) (ω := ω) + have hM_int : Integrable M P := + projectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound u n + have hM_mul_int : + Integrable + (fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω) P := by + have hnext : + Integrable + (fun ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u (n + 1) ω) + P := + projectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound u (n + 1) + refine hnext.congr ?_ + exact Filter.Eventually.of_forall fun ω ↦ by + simpa [M] using + projectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) (ω := ω) + have h_cond : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω + ≤ M ω := by + simpa [ℱ] using + projectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν hn v Q hQ hQ_bound u + M hM_meas hM_nonneg hM_mul_int + have h_cond_int : + Integrable + (fun ω ↦ + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω) + P := by + exact integrable_condExp + have h_cond_integral_le : + (∫ ω, + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω ∂P) + ≤ ∫ ω, M ω ∂P := + integral_mono_ae h_cond_int hM_int h_cond + calc + ∫ ω, projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u (n + 1) ω ∂P + = ∫ ω, M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω ∂P := by + refine integral_congr_ae ?_ + exact Filter.Eventually.of_forall fun ω ↦ by + simpa [M] using + projectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) + (ν := ν) (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) + (ω := ω) + _ = ∫ ω, + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω ∂P := by + rw [integral_condExp (μ := P) (m := ℱ n) (hm := ℱ.le n) + (f := fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω)] + _ ≤ ∫ ω, M ω ∂P := h_cond_integral_le + _ ≤ 1 := ih + +/-- Fixed-direction exponential process as a Mathlib `Supermartingale`. + +This is the process-level form of +`integral_projectedRewardFeatureNoiseExpProcess_le_one_of_abs_le`. It keeps the same bounded +projection assumption and packages the existing one-step conditional bound into the interface used +by optional-stopping/Ville arguments. The statement is still scalar and fixed-direction; the +remaining textbook step is to integrate these scalar exponentials over Gaussian directions to +obtain the self-normalized determinant confidence event. -/ +lemma supermartingale_projectedRewardFeatureNoiseExpProcess_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + Supermartingale + (fun n ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback) P := by + let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let Mproc : ℕ → Ω → ℝ := + fun n ω ↦ projectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω + have h_adapted : StronglyAdapted ℱ Mproc := by + simpa [ℱ, Mproc] using + stronglyAdapted_projectedRewardFeatureNoiseExpProcess_filtrationAction + (A := A) (R := R) (ν := ν) (x := x) + h.measurable_action h.measurable_feedback σ2 v u + have h_integrable : ∀ i, Integrable (Mproc i) P := by + intro i + simpa [Mproc] using + projectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound u i + refine supermartingale_nat (𝒢 := ℱ) (μ := P) h_adapted h_integrable ?_ + intro i + by_cases hi : i = 0 + · subst i + change P[Mproc 1 | ℱ 0] ≤ᵐ[P] Mproc 0 + have hM0 : Mproc 0 = fun _ : Ω ↦ (1 : ℝ) := by + funext ω + simp [Mproc, projectedRewardFeatureNoiseExpProcess_zero] + have hM1 : Mproc 1 = fun _ : Ω ↦ (1 : ℝ) := by + funext ω + simp [Mproc, projectedRewardFeatureNoiseExpProcess_one] + rw [hM0, hM1, condExp_const (ℱ.le 0)] + · let M : Ω → ℝ := Mproc i + have hM_meas : AEStronglyMeasurable[ℱ i] M P := + (h_adapted i).aestronglyMeasurable + have hM_nonneg : 0 ≤ᵐ[P] M := + Filter.Eventually.of_forall fun ω ↦ by + exact projectedRewardFeatureNoiseExpProcess_nonneg (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := i) (ω := ω) + have hM_mul_int : + Integrable + (fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω) P := by + have hnext : Integrable (Mproc (i + 1)) P := h_integrable (i + 1) + refine hnext.congr ?_ + exact Filter.Eventually.of_forall fun ω ↦ by + simpa [Mproc, M] using + projectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := i) (ω := ω) + have h_cond : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω' | ℱ i] ω + ≤ M ω := by + simpa [ℱ] using + projectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν hi v Q hQ hQ_bound u + M hM_meas hM_nonneg hM_mul_int + have h_succ_eq : + (fun ω ↦ Mproc (i + 1) ω) =ᵐ[P] + fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω := + Filter.Eventually.of_forall fun ω ↦ by + simpa [Mproc, M] using + projectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := i) (ω := ω) + have h_cond_eq : + P[Mproc (i + 1) | ℱ i] =ᵐ[P] + P[fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω | ℱ i] := + condExp_congr_ae h_succ_eq + filter_upwards [h_cond_eq, h_cond] with ω h_eq h_le + rw [h_eq] + exact h_le + +/-- One-sided tail bound for a bounded fixed-direction projection of the accumulated LinUCB +reward-feature noise. + +This is the direct probability form of +`projectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le`, matching the way `UCB.lean` exposes +its scalar concentration facts. -/ +lemma probReal_projectedRewardFeatureNoise_sum_ge_le_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) + {ε : ℝ} (hε : 0 ≤ ε) : + P.real + {ω | + ε ≤ ∑ t ∈ range n, projectedRewardFeatureNoise A R ν x v t ω} + ≤ Real.exp + (-ε ^ 2 / + (2 * (∑ t ∈ range n, + if t = 0 then 0 else (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)))) := + (projectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound n).measure_ge_le hε + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive-time centered response vector. + +This is the vector martingale term that the scalar projected-noise concentration lemmas above +control directly: +`∑_{1 ≤ t < n} η_t x_{A_t}`. The full `centeredResponseVector` also contains the time-zero +centered reward term, whose law is handled separately by the initial distribution in the +algorithm/environment model. -/ +noncomputable def positiveTimeCenteredResponseVector + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + ∑ t ∈ range n, if t = 0 then 0 else rewardNoise A R ν t ω • x (A t ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- No positive-time centered response has accumulated at horizon zero. -/ +lemma positiveTimeCenteredResponseVector_zero : + positiveTimeCenteredResponseVector A R ν x 0 ω = 0 := by + simp [positiveTimeCenteredResponseVector] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Advancing the horizon adds the next positive-time centered reward-feature vector. -/ +lemma positiveTimeCenteredResponseVector_succ : + positiveTimeCenteredResponseVector A R ν x (n + 1) ω = + positiveTimeCenteredResponseVector A R ν x n ω + + if n = 0 then 0 else rewardNoise A R ν n ω • x (A n ω) := by + simp [positiveTimeCenteredResponseVector, sum_range_succ] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Time-zero contribution to the centered response vector, present exactly when the horizon is +positive. -/ +noncomputable def initialCenteredResponseVector + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (n : ℕ) (ω : Ω) : Feature d := + if n = 0 then 0 else rewardNoise A R ν 0 ω • x (A 0 ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projecting the time-zero centered response vector gives the scalar time-zero projected noise. -/ +lemma dotProduct_initialCenteredResponseVector + (v : Feature d) : + dotProduct v (initialCenteredResponseVector A R ν x n ω) = + if n = 0 then 0 else dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω := by + by_cases hn : n = 0 + · simp [initialCenteredResponseVector, hn] + · simp only [initialCenteredResponseVector, hn, if_false, Pi.smul_apply, smul_eq_mul, + dotProduct] + rw [Finset.sum_mul] + refine Finset.sum_congr rfl ?_ + intro i _hi + ring + +/-- The initial scalar reward noise is subgaussian for the deterministic initial LinUCB arm. -/ +lemma initialRewardNoise_hasSubgaussianMGF + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) : + HasSubgaussianMGF (rewardNoise A R ν 0) σ2 P := by + let a0 : Fin K := ⟨0, hK⟩ + have h_pair : + HasLaw (fun ω ↦ (A 0 ω, R 0 ω)) (Measure.dirac a0 ⊗ₘ ν) P := by + have h_step_pair : (fun ω ↦ (A 0 ω, R 0 ω)) =ᵐ[P] Learning.step A R 0 := + Filter.Eventually.of_forall fun _ ↦ rfl + simpa [linUCBAlgorithm, a0] using + h.hasLaw_step_zero.congr h_step_pair + have h_snd : + HasLaw (fun p : Fin K × ℝ ↦ p.2) (ν a0) (Measure.dirac a0 ⊗ₘ ν) := by + refine ⟨measurable_snd.aemeasurable, ?_⟩ + ext s hs + rw [Measure.map_apply measurable_snd hs, Measure.dirac_compProd_apply (measurable_snd hs)] + change ν a0 s = ν a0 s + rfl + have hR_law : HasLaw (R 0) (ν a0) P := by + simpa [Function.comp_def] using h_snd.comp h_pair + have h_ident : + IdentDistrib (fun r : ℝ ↦ r - (ν a0)[id]) + (fun ω ↦ R 0 ω - (ν a0)[id]) (ν a0) P := by + exact ((hR_law.identDistrib HasLaw.id).comp + (measurable_id.sub measurable_const)).symm + have h_subG : + HasSubgaussianMGF (fun ω ↦ R 0 ω - (ν a0)[id]) σ2 P := + (hν a0).congr_identDistrib h_ident + refine h_subG.congr ?_ + filter_upwards [arm_zero (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) h] with ω hA0 + simp [rewardNoise, a0, hA0] + +/-- The time-zero fixed-direction projected reward-feature noise is subgaussian. -/ +lemma initialProjectedRewardFeatureNoise_hasSubgaussianMGF + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω) + (scalarProjectionVariance σ2 (dotProduct v (x ⟨0, hK⟩))) P := by + let a0 : Fin K := ⟨0, hK⟩ + let q0 : ℝ := dotProduct v (x a0) + have h_exact : + HasSubgaussianMGF (fun ω ↦ q0 * rewardNoise A R ν 0 ω) + (⟨q0 ^ 2, sq_nonneg q0⟩ * σ2) P := + (initialRewardNoise_hasSubgaussianMGF (A := A) (R := R) (reg := reg) + (β := β) (x := x) (ν := ν) h hν).const_mul q0 + have h_congr : + (fun ω ↦ q0 * rewardNoise A R ν 0 ω) =ᵐ[P] + fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω := by + filter_upwards [arm_zero (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) h] with ω hA0 + simp [q0, a0, hA0] + simpa [scalarProjectionVariance, q0, a0] using h_exact.congr h_congr + +/-- The time-zero fixed-direction projected reward-feature noise is subgaussian under a uniform +projection bound. -/ +lemma initialProjectedRewardFeatureNoise_hasSubgaussianMGF_of_abs_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω) + (⟨Q ^ 2, sq_nonneg Q⟩ * σ2) P := by + let a0 : Fin K := ⟨0, hK⟩ + let q0 : ℝ := dotProduct v (x a0) + have hq0_sq_le : (⟨q0 ^ 2, sq_nonneg q0⟩ : ℝ≥0) ≤ ⟨Q ^ 2, sq_nonneg Q⟩ := by + exact_mod_cast sq_le_sq.2 ((hQ_bound a0).trans (by simp [abs_of_nonneg hQ])) + have h_exact : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω) + (⟨q0 ^ 2, sq_nonneg q0⟩ * σ2) P := by + simpa [scalarProjectionVariance, q0, a0] using + initialProjectedRewardFeatureNoise_hasSubgaussianMGF + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h hν v + have h_bound : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω) + (⟨Q ^ 2, sq_nonneg Q⟩ * σ2) P := + hasSubgaussianMGF_mono_varianceProxy (P := P) h_exact + (mul_le_mul_left hq0_sq_le σ2) + exact h_bound + +/-- The initial centered response vector has the expected fixed-direction scalar subgaussian +bound. At horizon zero the vector is zero; at every positive horizon it is the time-zero projected +reward-feature noise. -/ +lemma dotProduct_initialCenteredResponseVector_hasSubgaussianMGF_of_abs_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (initialCenteredResponseVector A R ν x n ω)) + (if n = 0 then 0 else (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)) P := by + by_cases hn : n = 0 + · rw [if_pos hn] + have h_zero : + (fun ω ↦ dotProduct v (initialCenteredResponseVector A R ν x n ω)) = + (0 : Ω → ℝ) := by + funext ω + simp [dotProduct_initialCenteredResponseVector, hn] + rw [h_zero] + exact HasSubgaussianMGF.zero + · rw [if_neg hn] + exact (initialProjectedRewardFeatureNoise_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h hν + v Q hQ hQ_bound).congr + (Filter.Eventually.of_forall fun ω ↦ by + simp [dotProduct_initialCenteredResponseVector, hn]) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A fixed-direction scalar projection of the full LinUCB centered reward-feature noise. -/ +noncomputable def fullProjectedRewardFeatureNoise + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (x : Fin K → Feature d) (v : Feature d) (t : ℕ) (ω : Ω) : ℝ := + dotProduct v (x (A t ω)) * rewardNoise A R ν t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At positive times, the full projected reward-feature noise agrees with the positive-time +projected-noise process. -/ +lemma fullProjectedRewardFeatureNoise_eq_projectedRewardFeatureNoise_of_ne_zero + (v : Feature d) {t : ℕ} (ht : t ≠ 0) : + fullProjectedRewardFeatureNoise A R ν x v t ω = + projectedRewardFeatureNoise A R ν x v t ω := by + simp [fullProjectedRewardFeatureNoise, projectedRewardFeatureNoise, ht] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Realized variance proxy for the full fixed-direction reward-feature noise. -/ +noncomputable def fullProjectedRewardFeatureNoiseRealizedVariance + (A : ℕ → Ω → Fin K) (σ2 : ℝ≥0) + (x : Fin K → Feature d) (v : Feature d) (t : ℕ) (ω : Ω) : ℝ := + (scalarProjectionVariance σ2 (dotProduct v (x (A t ω))) : ℝ) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At positive times, the full realized variance agrees with the positive-time realized variance +used by the conditional-law exponential step. -/ +lemma fullProjectedRewardFeatureNoiseRealizedVariance_eq_projected_of_ne_zero + (σ2 : ℝ≥0) (v : Feature d) {t : ℕ} (ht : t ≠ 0) : + fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω = + projectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω := by + simp [fullProjectedRewardFeatureNoiseRealizedVariance, + projectedRewardFeatureNoiseRealizedVariance, ht] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Exponential-supermartingale increment for the full fixed-direction reward-feature noise. -/ +noncomputable def fullProjectedRewardFeatureNoiseExpIncrement + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (σ2 : ℝ≥0) (x : Fin K → Feature d) (v : Feature d) (u : ℝ) + (t : ℕ) (ω : Ω) : ℝ := + Real.exp + (u * fullProjectedRewardFeatureNoise A R ν x v t ω - + fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω * u ^ 2 / 2) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- At positive times, the full exponential increment agrees with the positive-time increment. -/ +lemma fullProjectedRewardFeatureNoiseExpIncrement_eq_projected_of_ne_zero + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) {t : ℕ} (ht : t ≠ 0) : + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω = + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω := by + simp [fullProjectedRewardFeatureNoiseExpIncrement, projectedRewardFeatureNoiseExpIncrement, + fullProjectedRewardFeatureNoise, projectedRewardFeatureNoise, + fullProjectedRewardFeatureNoiseRealizedVariance, projectedRewardFeatureNoiseRealizedVariance, + ht] + +/-- At time zero, the full fixed-direction exponential increment has conditional expectation at +most one given the initial action. + +The only extra point compared with positive times is conditioning on `filtrationAction 0`, which is +generated by `A 0`. For `linUCBAlgorithm`, `A 0` is deterministic almost surely, so this +conditioning sigma-algebra is independent of the initial exponential increment. The proof then +reduces the conditional expectation to the ordinary expectation and applies the initial +subgaussian MGF bound. -/ +lemma fullProjectedRewardFeatureNoiseExpIncrement_zero_ae_condExp_le_one + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (u : ℝ) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u 0 ω' | + IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback 0] ω + ≤ 1 := by + let a0 : Fin K := ⟨0, hK⟩ + let F : Ω → ℝ := + fun ω ↦ fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u 0 ω + let X : Ω → ℝ := + fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω + let X0 : Ω → ℝ := + fun ω ↦ dotProduct v (x a0) * rewardNoise A R ν 0 ω + let c0 : ℝ := (scalarProjectionVariance σ2 (dotProduct v (x a0)) : ℝ) + let G : Ω → ℝ := fun ω ↦ Real.exp (u * X0 ω - c0 * u ^ 2 / 2) + have hcoeff_meas : Measurable fun ω ↦ dotProduct v (x (A 0 ω)) := + (measurable_of_countable fun a : Fin K ↦ dotProduct v (x a)).comp + (h.measurable_action 0) + have hnoise_meas : Measurable (rewardNoise A R ν 0) := by + change Measurable (fun ω ↦ R 0 ω - (ν (A 0 ω))[id]) + exact (h.measurable_feedback 0).sub + ((measurable_rewardMean ν).comp (h.measurable_action 0)) + have hproj_meas : Measurable X := hcoeff_meas.mul hnoise_meas + have hvar_meas : + Measurable fun ω ↦ fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v 0 ω := + (measurable_of_countable + (fun a : Fin K ↦ (scalarProjectionVariance σ2 (dotProduct v (x a)) : ℝ))).comp + (h.measurable_action 0) + have hF_meas : Measurable F := by + dsimp [F, fullProjectedRewardFeatureNoiseExpIncrement] + exact Real.measurable_exp.comp + ((measurable_const.mul hproj_meas).sub ((hvar_meas.mul measurable_const).div_const 2)) + have h_initial : + HasSubgaussianMGF X0 (scalarProjectionVariance σ2 (dotProduct v (x a0))) P := by + simpa [X0, scalarProjectionVariance, a0] using + (initialRewardNoise_hasSubgaussianMGF + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h hν + ).const_mul (dotProduct v (x a0)) + have hG_integral_le : (∫ ω, G ω ∂P) ≤ 1 := by + simpa [G, X0, c0] using hasSubgaussianMGF_integral_exp_sub_le_one h_initial u + have hF_eq_G : F =ᵐ[P] G := by + filter_upwards [arm_zero (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) h] with ω hA0 + simp [F, G, X0, c0, fullProjectedRewardFeatureNoiseExpIncrement, + fullProjectedRewardFeatureNoise, fullProjectedRewardFeatureNoiseRealizedVariance, + a0, hA0] + have hF_integral_le : (∫ ω, F ω ∂P) ≤ 1 := by + calc + (∫ ω, F ω ∂P) = ∫ ω, G ω ∂P := integral_congr_ae hF_eq_G + _ ≤ 1 := hG_integral_le + let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + have hA0_const : A 0 =ᵐ[P] fun _ : Ω ↦ a0 := + arm_zero (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h + have h_indep_fun : F ⟂ᵢ[P] A 0 := by + exact (indepFun_const_right F a0).congr EventuallyEq.rfl hA0_const.symm + have h_indep : + Indep (MeasurableSpace.comap F inferInstance) (ℱ 0) P := by + rw [IsAlgEnvSeq.filtrationAction_zero_eq_comap] + simpa [ProbabilityTheory.IndepFun, ProbabilityTheory.Kernel.IndepFun, + ProbabilityTheory.Indep] using h_indep_fun + have h_cond_eq : + P[F | ℱ 0] =ᵐ[P] fun _ : Ω ↦ ∫ ω, F ω ∂P := by + exact condExp_indep_eq (m₁ := MeasurableSpace.comap F inferInstance) + (m₂ := ℱ 0) (μ := P) + hF_meas.comap_le (ℱ.le 0) + (Measurable.of_comap_le le_rfl).stronglyMeasurable h_indep + filter_upwards [h_cond_eq] with ω hω + rw [hω] + exact hF_integral_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Cumulative realized variance proxy for the full fixed-direction reward-feature noise. -/ +noncomputable def fullProjectedRewardFeatureNoiseRealizedVarianceSum + (A : ℕ → Ω → Fin K) (σ2 : ℝ≥0) + (x : Fin K → Feature d) (v : Feature d) (n : ℕ) (ω : Ω) : ℝ := + ∑ t ∈ range n, fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Full finite-horizon fixed-direction exponential process with realized variance. -/ +noncomputable def fullProjectedRewardFeatureNoiseExpProcess + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (σ2 : ℝ≥0) (x : Fin K → Feature d) (v : Feature d) (u : ℝ) + (n : ℕ) (ω : Ω) : ℝ := + Real.exp + (u * (∑ t ∈ range n, fullProjectedRewardFeatureNoise A R ν x v t ω) - + fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω * u ^ 2 / 2) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Full realized variance proxies are nonnegative. -/ +lemma fullProjectedRewardFeatureNoiseRealizedVariance_nonneg + (σ2 : ℝ≥0) (v : Feature d) (t : ℕ) : + 0 ≤ fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω := by + simp [fullProjectedRewardFeatureNoiseRealizedVariance] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The cumulative full realized variance proxy is nonnegative. -/ +lemma fullProjectedRewardFeatureNoiseRealizedVarianceSum_nonneg + (σ2 : ℝ≥0) (v : Feature d) (n : ℕ) : + 0 ≤ fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω := by + rw [fullProjectedRewardFeatureNoiseRealizedVarianceSum] + exact sum_nonneg fun t _ht ↦ + fullProjectedRewardFeatureNoiseRealizedVariance_nonneg (A := A) (σ2 := σ2) + (x := x) (v := v) (t := t) (ω := ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A rank-one feature matrix contributes the square of the projected feature to the corresponding +quadratic form. -/ +lemma dotProduct_vecMulVec_mulVec_eq_sq (v y : Feature d) : + dotProduct v (Matrix.mulVec (Matrix.vecMulVec y y) v) = (dotProduct v y) ^ 2 := by + simp [Matrix.vecMulVec_mulVec, dotProduct, pow_two, Finset.sum_mul, mul_assoc, + mul_comm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The observed rank-one feature-matrix sum has quadratic form +`∑ (vᵀx_{A_t})²`. -/ +lemma dotProduct_sum_vecMulVec_mulVec_eq_sum_sq (v : Feature d) : + dotProduct v + (Matrix.mulVec + (∑ t ∈ range n, Matrix.vecMulVec (x (A t ω)) (x (A t ω))) v) = + ∑ t ∈ range n, (dotProduct v (x (A t ω))) ^ 2 := by + simp only [Matrix.sum_mulVec, dotProduct_sum] + refine Finset.sum_congr rfl ?_ + intro t _ht + exact dotProduct_vecMulVec_mulVec_eq_sq (v := v) (y := x (A t ω)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The design-matrix quadratic form decomposes into the ridge part plus the observed projected +feature squares. -/ +lemma dotProduct_designMatrix_mulVec_eq_reg_add_sum_sq (v : Feature d) : + dotProduct v (Matrix.mulVec (designMatrix A reg x n ω) v) = + reg * dotProduct v v + ∑ t ∈ range n, (dotProduct v (x (A t ω))) ^ 2 := by + simp only [designMatrix, Matrix.add_mulVec, Matrix.smul_mulVec, Matrix.one_mulVec, + dotProduct_add] + rw [dotProduct_sum_vecMulVec_mulVec_eq_sum_sq (A := A) (x := x) (n := n) + (ω := ω) v] + simp [dotProduct_smul] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- After removing the ridge part of the design matrix, the quadratic form is exactly the observed +projected feature-square sum. -/ +lemma dotProduct_designMatrix_sub_reg_smul_one_mulVec_eq_sum_sq (v : Feature d) : + dotProduct v + (Matrix.mulVec + (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)) v) = + ∑ t ∈ range n, (dotProduct v (x (A t ω))) ^ 2 := by + rw [Matrix.sub_mulVec, dotProduct_sub, + dotProduct_designMatrix_mulVec_eq_reg_add_sum_sq (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) v] + simp [Matrix.smul_mulVec, Matrix.one_mulVec, dotProduct_smul] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full realized-variance sum is `σ²` times the observed projected feature-square sum. -/ +lemma fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_sum_sq + (σ2 : ℝ≥0) (v : Feature d) : + fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω = + (σ2 : ℝ) * ∑ t ∈ range n, (dotProduct v (x (A t ω))) ^ 2 := by + rw [fullProjectedRewardFeatureNoiseRealizedVarianceSum] + calc + ∑ t ∈ range n, fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v t ω + = ∑ t ∈ range n, (σ2 : ℝ) * (dotProduct v (x (A t ω))) ^ 2 := by + refine Finset.sum_congr rfl ?_ + intro t _ht + change ((⟨(dotProduct v (x (A t ω))) ^ 2, + sq_nonneg (dotProduct v (x (A t ω)))⟩ : ℝ≥0) * σ2 : ℝ) = + (σ2 : ℝ) * (dotProduct v (x (A t ω))) ^ 2 + change (dotProduct v (x (A t ω))) ^ 2 * (σ2 : ℝ) = + (σ2 : ℝ) * (dotProduct v (x (A t ω))) ^ 2 + ring + _ = (σ2 : ℝ) * ∑ t ∈ range n, (dotProduct v (x (A t ω))) ^ 2 := by + rw [Finset.mul_sum] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full realized-variance sum is `σ²` times the quadratic form of the non-ridge part of the +design matrix. -/ +lemma fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_designMatrix_sub_reg + (σ2 : ℝ≥0) (v : Feature d) : + fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω = + (σ2 : ℝ) * + dotProduct v + (Matrix.mulVec + (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)) v) := by + rw [fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_sum_sq + (A := A) (σ2 := σ2) (x := x) (v := v) (n := n) (ω := ω), + dotProduct_designMatrix_sub_reg_smul_one_mulVec_eq_sum_sq (A := A) + (reg := reg) (x := x) (n := n) (ω := ω) v] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Equivalent design-matrix form of the full realized-variance sum: +`σ² * (vᵀV_n v - reg * vᵀv)`. -/ +lemma fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_designMatrix_minus_reg_norm + (σ2 : ℝ≥0) (v : Feature d) : + fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω = + (σ2 : ℝ) * + (dotProduct v (Matrix.mulVec (designMatrix A reg x n ω) v) - + reg * dotProduct v v) := by + rw [fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_sum_sq + (A := A) (σ2 := σ2) (x := x) (v := v) (n := n) (ω := ω), + dotProduct_designMatrix_mulVec_eq_reg_add_sum_sq (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) v] + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Full exponential increments are nonnegative. -/ +lemma fullProjectedRewardFeatureNoiseExpIncrement_nonneg + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) (t : ℕ) : + 0 ≤ fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω := by + exact Real.exp_nonneg _ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full finite-horizon exponential process starts at one. -/ +lemma fullProjectedRewardFeatureNoiseExpProcess_zero + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) : + fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u 0 ω = 1 := by + simp [fullProjectedRewardFeatureNoiseExpProcess, + fullProjectedRewardFeatureNoiseRealizedVarianceSum] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full finite-horizon exponential process is nonnegative. -/ +lemma fullProjectedRewardFeatureNoiseExpProcess_nonneg + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) (n : ℕ) : + 0 ≤ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω := by + exact Real.exp_nonneg _ + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full exponential process factors into the previous process times the next increment. -/ +lemma fullProjectedRewardFeatureNoiseExpProcess_succ + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) (n : ℕ) : + fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u (n + 1) ω = + fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω := by + simp only [fullProjectedRewardFeatureNoiseExpProcess, + fullProjectedRewardFeatureNoiseExpIncrement, + fullProjectedRewardFeatureNoiseRealizedVarianceSum, sum_range_succ] + rw [← Real.exp_add] + congr 1 + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full fixed-direction projected reward-feature noise is adapted to `postActionFiltration`. -/ +lemma stronglyAdapted_fullProjectedRewardFeatureNoise_postActionFiltration + (hA : ∀ n, Measurable (A n)) (hR : ∀ n, Measurable (R n)) + (v : Feature d) : + StronglyAdapted (postActionFiltration (A := A) (R := R) hA hR) + (fullProjectedRewardFeatureNoise A R ν x v) := by + intro t + have hfil_le : + IsAlgEnvSeq.filtration hA hR t ≤ postActionFiltration (A := A) (R := R) hA hR t := + filtration_le_postActionFiltration (A := A) (R := R) hA hR t + have hA_t : Measurable[postActionFiltration (A := A) (R := R) hA hR t] (A t) := + (IsAlgEnvSeq.adapted_action hA hR t).mono hfil_le le_rfl + have hR_t : Measurable[postActionFiltration (A := A) (R := R) hA hR t] (R t) := + (IsAlgEnvSeq.adapted_feedback hA hR t).mono hfil_le le_rfl + have hη_t : + Measurable[postActionFiltration (A := A) (R := R) hA hR t] + (rewardNoise A R ν t) := by + change Measurable[postActionFiltration (A := A) (R := R) hA hR t] + (fun ω ↦ R t ω - (ν (A t ω))[id]) + exact hR_t.sub ((measurable_rewardMean ν).comp hA_t) + have hq_t : + Measurable[postActionFiltration (A := A) (R := R) hA hR t] + (fun ω ↦ dotProduct v (x (A t ω))) := + (measurable_of_countable (fun a : Fin K ↦ dotProduct v (x a))).comp hA_t + exact (hq_t.mul hη_t).stronglyMeasurable + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full fixed-direction realized variance proxy is adapted to `postActionFiltration`. -/ +lemma stronglyAdapted_fullProjectedRewardFeatureNoiseRealizedVariance_postActionFiltration + (hA : ∀ n, Measurable (A n)) (hR : ∀ n, Measurable (R n)) + (σ2 : ℝ≥0) (v : Feature d) : + StronglyAdapted (postActionFiltration (A := A) (R := R) hA hR) + (fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v) := by + intro t + have hfil_le : + IsAlgEnvSeq.filtration hA hR t ≤ postActionFiltration (A := A) (R := R) hA hR t := + filtration_le_postActionFiltration (A := A) (R := R) hA hR t + have hA_t : Measurable[postActionFiltration (A := A) (R := R) hA hR t] (A t) := + (IsAlgEnvSeq.adapted_action hA hR t).mono hfil_le le_rfl + have hvar_t : + Measurable[postActionFiltration (A := A) (R := R) hA hR t] + (fun ω ↦ (scalarProjectionVariance σ2 (dotProduct v (x (A t ω))) : ℝ)) := + (measurable_of_countable + (fun a : Fin K ↦ (scalarProjectionVariance σ2 (dotProduct v (x a)) : ℝ))).comp hA_t + change StronglyMeasurable[postActionFiltration (A := A) (R := R) hA hR t] + (fun ω ↦ (scalarProjectionVariance σ2 (dotProduct v (x (A t ω))) : ℝ)) + exact hvar_t.stronglyMeasurable + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full fixed-direction exponential process up to a positive horizon is measurable with +respect to the post-action sigma-algebra at the previous time. -/ +lemma stronglyMeasurable_fullProjectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (hA : ∀ n, Measurable (A n)) (hR : ∀ n, Measurable (R n)) + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) (n : ℕ) : + StronglyMeasurable[postActionFiltration (A := A) (R := R) hA hR (n - 1)] + (fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) := by + let ℱ := postActionFiltration (A := A) (R := R) hA hR + have hY_adapt : StronglyAdapted ℱ (fullProjectedRewardFeatureNoise A R ν x v) := by + simpa [ℱ] using + stronglyAdapted_fullProjectedRewardFeatureNoise_postActionFiltration + (A := A) (R := R) (ν := ν) (x := x) hA hR v + have hV_adapt : + StronglyAdapted ℱ (fullProjectedRewardFeatureNoiseRealizedVariance A σ2 x v) := by + simpa [ℱ] using + stronglyAdapted_fullProjectedRewardFeatureNoiseRealizedVariance_postActionFiltration + (A := A) (R := R) (x := x) hA hR σ2 v + have hY_sum : + Measurable[ℱ (n - 1)] + (fun ω ↦ ∑ t ∈ range n, fullProjectedRewardFeatureNoise A R ν x v t ω) := by + refine Finset.measurable_fun_sum (range n) ?_ + intro t ht + have ht_le : t ≤ n - 1 := by + exact Nat.le_pred_of_lt (by simpa using ht) + exact ((hY_adapt t).mono (ℱ.mono ht_le)).measurable + have hV_sum : + Measurable[ℱ (n - 1)] + (fun ω ↦ fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω) := by + simp only [fullProjectedRewardFeatureNoiseRealizedVarianceSum] + refine Finset.measurable_fun_sum (range n) ?_ + intro t ht + have ht_le : t ≤ n - 1 := by + exact Nat.le_pred_of_lt (by simpa using ht) + exact ((hV_adapt t).mono (ℱ.mono ht_le)).measurable + refine (measurable_exp.comp ?_).stronglyMeasurable + exact (measurable_const.mul hY_sum).sub ((hV_sum.mul measurable_const).div_const 2) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full fixed-direction exponential process is adapted to the unshifted action filtration at +its horizon. -/ +lemma stronglyAdapted_fullProjectedRewardFeatureNoiseExpProcess_filtrationAction + (hA : ∀ n, Measurable (A n)) (hR : ∀ n, Measurable (R n)) + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) : + StronglyAdapted (IsAlgEnvSeq.filtrationAction hA hR) + (fun n ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) := by + intro n + by_cases hn : n = 0 + · subst n + simpa [fullProjectedRewardFeatureNoiseExpProcess_zero] using + (stronglyMeasurable_const : + StronglyMeasurable[IsAlgEnvSeq.filtrationAction hA hR 0] + (fun _ : Ω ↦ (1 : ℝ))) + · have hsm := + stronglyMeasurable_fullProjectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (A := A) (R := R) (ν := ν) (x := x) hA hR σ2 v u n + have hpost : + postActionFiltration (A := A) (R := R) hA hR (n - 1) = + IsAlgEnvSeq.filtrationAction hA hR n := by + simp [postActionFiltration, Nat.sub_add_cancel (Nat.pos_of_ne_zero hn)] + rw [← hpost] + exact hsm + +/-- The full fixed-direction projected centered response is a scalar subgaussian martingale sum. -/ +lemma fullProjectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) : + HasSubgaussianMGF + (fun ω ↦ ∑ t ∈ range n, fullProjectedRewardFeatureNoise A R ν x v t ω) + (∑ _t ∈ range n, (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)) P := by + let ℱ := postActionFiltration (A := A) (R := R) h.measurable_action h.measurable_feedback + let Y : ℕ → Ω → ℝ := fullProjectedRewardFeatureNoise A R ν x v + let cY : ℕ → ℝ≥0 := fun _ ↦ ⟨Q ^ 2, sq_nonneg Q⟩ * σ2 + have h_adapted : StronglyAdapted ℱ Y := by + simpa [ℱ, Y] using + stronglyAdapted_fullProjectedRewardFeatureNoise_postActionFiltration + (A := A) (R := R) (ν := ν) (x := x) h.measurable_action h.measurable_feedback v + have h0 : HasSubgaussianMGF (Y 0) (cY 0) P := by + have hY0 : + Y 0 = fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω := by + funext ω + simp [Y, fullProjectedRewardFeatureNoise] + rw [hY0] + simpa [cY] using + initialProjectedRewardFeatureNoise_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound + have h_subG : + ∀ i < n - 1, HasCondSubgaussianMGF (ℱ i) (ℱ.le i) (Y (i + 1)) (cY (i + 1)) P := by + intro i _hi + have h_eq : + Y (i + 1) = projectedRewardFeatureNoise A R ν x v (i + 1) := by + funext ω + simp [Y, fullProjectedRewardFeatureNoise_eq_projectedRewardFeatureNoise_of_ne_zero + (A := A) (R := R) (ν := ν) (x := x) (v := v) (t := i + 1) + (ω := ω) (Nat.succ_ne_zero i)] + rw [h_eq] + simpa [ℱ, cY, postActionFiltration] using + projectedRewardFeatureNoise_hasCondSubgaussianMGF_filtrationAction_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν (Nat.succ_ne_zero i) + v Q hQ hQ_bound + simpa [Y, cY] using + HasSubgaussianMGF.sum_of_hasCondSubgaussianMGF (μ := P) (ℱ := ℱ) (Y := Y) (cY := cY) + h_adapted h0 n h_subG + +/-- Predictable-multiplier form of the one-step bound for the full fixed-direction exponential +increment at positive times. -/ +lemma fullProjectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) (v : Feature d) + (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) + (M : Ω → ℝ) + (hM_meas : AEStronglyMeasurable[IsAlgEnvSeq.filtrationAction + h.measurable_action h.measurable_feedback t] M P) + (hM_nonneg : 0 ≤ᵐ[P] M) + (hM_mul_int : + Integrable + (fun ω ↦ M ω * fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) P) : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | + IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t] ω + ≤ M ω := by + let ℱt := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback t + have h_mul_eq : + (fun ω ↦ M ω * fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) =ᵐ[P] + fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω := + Filter.Eventually.of_forall fun ω ↦ by + change M ω * fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω = + M ω * projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω + rw [fullProjectedRewardFeatureNoiseExpIncrement_eq_projected_of_ne_zero + (A := A) (R := R) (ν := ν) (σ2 := σ2) (x := x) (v := v) (u := u) + (t := t) (ω := ω) ht] + have hM_mul_int_projected : + Integrable + (fun ω ↦ M ω * projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω) P := + hM_mul_int.congr h_mul_eq + have h_cond_projected : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω' | ℱt] ω + ≤ M ω := by + simpa [ℱt] using + projectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν ht v Q hQ hQ_bound u + M hM_meas hM_nonneg hM_mul_int_projected + have h_cond_eq : + P[fun ω ↦ M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω | ℱt] + =ᵐ[P] + P[fun ω ↦ M ω * + projectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u t ω | ℱt] := + condExp_congr_ae h_mul_eq + filter_upwards [h_cond_eq, h_cond_projected] with ω h_eq h_le + rw [h_eq] + exact h_le + +/-- Integrability of the full finite-horizon fixed-direction exponential process. + +The realized-variance penalty is nonnegative, so this process is pointwise bounded by the ordinary +exponential of the fixed-direction full centered reward-feature sum. -/ +lemma fullProjectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) (n : ℕ) : + Integrable + (fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) P := by + have h_base : + Integrable + (fun ω ↦ Real.exp + (u * (∑ t ∈ range n, fullProjectedRewardFeatureNoise A R ν x v t ω))) P := + (fullProjectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound n).integrable_exp_mul u + have h_target : + AEStronglyMeasurable + (fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) P := + ((stronglyMeasurable_fullProjectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (A := A) (R := R) (ν := ν) (x := x) h.measurable_action h.measurable_feedback + σ2 v u n).mono + ((postActionFiltration (A := A) (R := R) h.measurable_action h.measurable_feedback).le + (n - 1))).aestronglyMeasurable + refine Integrable.mono h_base h_target ?_ + refine Filter.Eventually.of_forall fun ω ↦ ?_ + have hpenalty_nonneg : + 0 ≤ fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω * u ^ 2 / 2 := by + have hsum_nonneg : + 0 ≤ fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω := + fullProjectedRewardFeatureNoiseRealizedVarianceSum_nonneg (A := A) (σ2 := σ2) + (x := x) (v := v) (n := n) (ω := ω) + positivity + simp only [fullProjectedRewardFeatureNoiseExpProcess, + Real.norm_of_nonneg (Real.exp_nonneg _)] + exact Real.exp_le_exp.mpr (by linarith [hpenalty_nonneg]) + +/-- Finite-horizon fixed-direction exponential-supermartingale bound for the full centered +reward-feature noise. + +This is the closest scalar theorem before the vector self-normalized method-of-mixtures step: +for every fixed direction `v` and scalar `u`, +`E exp(u * ⟪v, S_n⟫ - u² / 2 * ∑_{t + simp [fullProjectedRewardFeatureNoiseExpProcess_zero] + | succ n ih => + by_cases hn : n = 0 + · subst n + let a0 : Fin K := ⟨0, hK⟩ + let X : Ω → ℝ := + fun ω ↦ dotProduct v (x (A 0 ω)) * rewardNoise A R ν 0 ω + have h_initial : + HasSubgaussianMGF X + (scalarProjectionVariance σ2 (dotProduct v (x a0))) P := by + simpa [X, a0] using + initialProjectedRewardFeatureNoise_hasSubgaussianMGF + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v + have h_integral_exact : + ∫ ω, Real.exp + (u * X ω - + (scalarProjectionVariance σ2 (dotProduct v (x a0)) : ℝ) * u ^ 2 / 2) ∂P + ≤ 1 := + hasSubgaussianMGF_integral_exp_sub_le_one h_initial u + have h_process_eq : + (fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u 1 ω) + =ᵐ[P] + fun ω ↦ Real.exp + (u * X ω - + (scalarProjectionVariance σ2 (dotProduct v (x a0)) : ℝ) * u ^ 2 / 2) := by + filter_upwards [arm_zero (A := A) (R := R) (reg := reg) (β := β) (x := x) + (ν := ν) h] with ω hA0 + simp [fullProjectedRewardFeatureNoiseExpProcess, + fullProjectedRewardFeatureNoiseRealizedVarianceSum, + fullProjectedRewardFeatureNoiseRealizedVariance, + fullProjectedRewardFeatureNoise, X, a0, hA0] + calc + ∫ ω, fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u 1 ω ∂P + = ∫ ω, Real.exp + (u * X ω - + (scalarProjectionVariance σ2 (dotProduct v (x a0)) : ℝ) * u ^ 2 / 2) ∂P := + integral_congr_ae h_process_eq + _ ≤ 1 := h_integral_exact + · let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let M : Ω → ℝ := fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω + have hM_meas : AEStronglyMeasurable[ℱ n] M P := by + have hsm := + stronglyMeasurable_fullProjectedRewardFeatureNoiseExpProcess_postActionFiltration_pred + (A := A) (R := R) (ν := ν) (x := x) h.measurable_action h.measurable_feedback + σ2 v u n + have hpost : + postActionFiltration (A := A) (R := R) h.measurable_action h.measurable_feedback + (n - 1) = ℱ n := by + simp [postActionFiltration, ℱ, Nat.sub_add_cancel (Nat.pos_of_ne_zero hn)] + have hsm_F : StronglyMeasurable[ℱ n] M := by + rw [← hpost] + simpa [M] using hsm + exact hsm_F.aestronglyMeasurable + have hM_nonneg : 0 ≤ᵐ[P] M := + Filter.Eventually.of_forall fun ω ↦ + fullProjectedRewardFeatureNoiseExpProcess_nonneg (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) (ω := ω) + have hM_int : Integrable M P := + fullProjectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound u n + have hM_mul_int : + Integrable + (fun ω ↦ M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω) P := by + have hnext : + Integrable + (fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u (n + 1) ω) + P := + fullProjectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound u (n + 1) + refine hnext.congr ?_ + exact Filter.Eventually.of_forall fun ω ↦ by + simpa [M] using + fullProjectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) (ω := ω) + have h_cond : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω + ≤ M ω := by + simpa [ℱ] using + fullProjectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν hn v Q hQ hQ_bound u M hM_meas hM_nonneg hM_mul_int + have h_cond_int : + Integrable + (fun ω ↦ + P[fun ω' ↦ M ω' * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω) + P := by + exact integrable_condExp + have h_cond_integral_le : + (∫ ω, + P[fun ω' ↦ M ω' * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | ℱ n] ω ∂P) + ≤ ∫ ω, M ω ∂P := + integral_mono_ae h_cond_int hM_int h_cond + calc + ∫ ω, fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u (n + 1) ω ∂P + = ∫ ω, M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω ∂P := by + refine integral_congr_ae ?_ + exact Filter.Eventually.of_forall fun ω ↦ by + simpa [M] using + fullProjectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) + (ν := ν) (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) + (ω := ω) + _ = ∫ ω, + P[fun ω' ↦ M ω' * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω' | + ℱ n] ω ∂P := by + rw [integral_condExp (μ := P) (m := ℱ n) (hm := ℱ.le n) + (f := fun ω ↦ M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u n ω)] + _ ≤ ∫ ω, M ω ∂P := h_cond_integral_le + _ ≤ 1 := ih + +/-- Full fixed-direction exponential process as a Mathlib `Supermartingale`. + +This is the scalar process used immediately before the textbook Gaussian-mixture argument, now +including the time-zero centered reward-feature contribution. The zero-time conditional step uses +the deterministic initial action of `linUCBAlgorithm`; positive times reuse the standard +history/action conditional-subgaussian increment bound. -/ +lemma supermartingale_fullProjectedRewardFeatureNoiseExpProcess_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + Supermartingale + (fun n ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback) P := by + let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let Mproc : ℕ → Ω → ℝ := + fun n ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω + have h_adapted : StronglyAdapted ℱ Mproc := by + simpa [ℱ, Mproc] using + stronglyAdapted_fullProjectedRewardFeatureNoiseExpProcess_filtrationAction + (A := A) (R := R) (ν := ν) (x := x) + h.measurable_action h.measurable_feedback σ2 v u + have h_integrable : ∀ i, Integrable (Mproc i) P := by + intro i + simpa [Mproc] using + fullProjectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound u i + refine supermartingale_nat (𝒢 := ℱ) (μ := P) h_adapted h_integrable ?_ + intro i + by_cases hi : i = 0 + · subst i + change P[Mproc 1 | ℱ 0] ≤ᵐ[P] Mproc 0 + let Inc0 : Ω → ℝ := + fun ω ↦ fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u 0 ω + have h_succ_eq : Mproc 1 = Inc0 := by + funext ω + simp [Mproc, Inc0, fullProjectedRewardFeatureNoiseExpProcess_succ, + fullProjectedRewardFeatureNoiseExpProcess_zero] + have h_cond : + ∀ᵐ ω ∂P, P[Inc0 | ℱ 0] ω ≤ 1 := by + simpa [ℱ, Inc0] using + fullProjectedRewardFeatureNoiseExpIncrement_zero_ae_condExp_le_one + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) h hν v u + have h_cond_eq : P[Mproc 1 | ℱ 0] =ᵐ[P] P[Inc0 | ℱ 0] := by + exact condExp_congr_ae (Filter.Eventually.of_forall fun ω ↦ by rw [h_succ_eq]) + have hM0 : Mproc 0 = fun _ : Ω ↦ (1 : ℝ) := by + funext ω + simp [Mproc, fullProjectedRewardFeatureNoiseExpProcess_zero] + filter_upwards [h_cond_eq, h_cond] with ω h_eq h_le + rw [h_eq, hM0] + exact h_le + · let M : Ω → ℝ := Mproc i + have hM_meas : AEStronglyMeasurable[ℱ i] M P := + (h_adapted i).aestronglyMeasurable + have hM_nonneg : 0 ≤ᵐ[P] M := + Filter.Eventually.of_forall fun ω ↦ by + exact fullProjectedRewardFeatureNoiseExpProcess_nonneg (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := i) (ω := ω) + have hM_mul_int : + Integrable + (fun ω ↦ M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω) P := by + have hnext : Integrable (Mproc (i + 1)) P := h_integrable (i + 1) + refine hnext.congr ?_ + exact Filter.Eventually.of_forall fun ω ↦ by + simpa [Mproc, M] using + fullProjectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := i) (ω := ω) + have h_cond : + ∀ᵐ ω ∂P, + P[fun ω' ↦ M ω' * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω' | ℱ i] ω + ≤ M ω := by + simpa [ℱ] using + fullProjectedRewardFeatureNoiseExpIncrement_ae_condExp_mul_le_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν hi v Q hQ hQ_bound u M hM_meas hM_nonneg hM_mul_int + have h_succ_eq : + (fun ω ↦ Mproc (i + 1) ω) =ᵐ[P] + fun ω ↦ M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω := + Filter.Eventually.of_forall fun ω ↦ by + simpa [Mproc, M] using + fullProjectedRewardFeatureNoiseExpProcess_succ (A := A) (R := R) (ν := ν) + (σ2 := σ2) (x := x) (v := v) (u := u) (n := i) (ω := ω) + have h_cond_eq : + P[Mproc (i + 1) | ℱ i] =ᵐ[P] + P[fun ω ↦ M ω * + fullProjectedRewardFeatureNoiseExpIncrement A R ν σ2 x v u i ω | ℱ i] := + condExp_congr_ae h_succ_eq + filter_upwards [h_cond_eq, h_cond] with ω h_eq h_le + rw [h_eq] + exact h_le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Projecting the full centered response vector gives the sum of full projected reward-feature +noise terms. -/ +lemma dotProduct_centeredResponseVector_eq_fullProjectedRewardFeatureNoise_sum + (v : Feature d) : + dotProduct v (centeredResponseVector A R ν x n ω) = + ∑ t ∈ range n, fullProjectedRewardFeatureNoise A R ν x v t ω := by + have hcenter : + centeredResponseVector A R ν x n ω = + ∑ t ∈ range n, rewardNoise A R ν t ω • x (A t ω) := by + ext i + simp [centeredResponseVector, responseVector, meanResponseVector, rewardNoise, + Finset.sum_sub_distrib, smul_eq_mul, sub_mul] + rw [hcenter] + simp only [fullProjectedRewardFeatureNoise] + change dotProduct v + (∑ t ∈ range n, rewardNoise A R ν t ω • x (A t ω)) = + ∑ t ∈ range n, dotProduct v (x (A t ω)) * rewardNoise A R ν t ω + simp only [dotProduct, Finset.sum_apply] + calc + ∑ i, v i * + (∑ t ∈ range n, (rewardNoise A R ν t ω • x (A t ω)) i) + = ∑ i, ∑ t ∈ range n, + v i * (rewardNoise A R ν t ω • x (A t ω)) i := by + refine Finset.sum_congr rfl ?_ + intro i _hi + rw [Finset.mul_sum] + _ = ∑ t ∈ range n, ∑ i, + v i * (rewardNoise A R ν t ω • x (A t ω)) i := by + rw [Finset.sum_comm] + _ = ∑ t ∈ range n, dotProduct v (x (A t ω)) * rewardNoise A R ν t ω := by + refine Finset.sum_congr rfl ?_ + intro t _ht + simp only [Pi.smul_apply, smul_eq_mul, dotProduct] + rw [Finset.sum_mul] + refine Finset.sum_congr rfl ?_ + intro i _hi + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The full fixed-direction exponential process is exactly the normalized exponential of the +projection of the centered response vector. -/ +lemma fullProjectedRewardFeatureNoiseExpProcess_eq_centeredResponseVector + (σ2 : ℝ≥0) (v : Feature d) (u : ℝ) : + fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω = + Real.exp + (u * dotProduct v (centeredResponseVector A R ν x n ω) - + fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω * u ^ 2 / 2) := by + rw [fullProjectedRewardFeatureNoiseExpProcess, + dotProduct_centeredResponseVector_eq_fullProjectedRewardFeatureNoise_sum + (A := A) (R := R) (ν := ν) (x := x) (n := n) (ω := ω) v] + +/-- Fixed-direction exponential-supermartingale bound stated directly for the centered response +vector. This is the scalar form immediately before the textbook Gaussian-mixture argument. -/ +lemma integral_exp_dotProduct_centeredResponseVector_sub_fullRealizedVariance_le_one_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + ∫ ω, Real.exp + (u * dotProduct v (centeredResponseVector A R ν x n ω) - + fullProjectedRewardFeatureNoiseRealizedVarianceSum A σ2 x v n ω * u ^ 2 / 2) ∂P + ≤ 1 := by + simpa [fullProjectedRewardFeatureNoiseExpProcess_eq_centeredResponseVector + (A := A) (R := R) (ν := ν) (x := x) (n := n) (σ2 := σ2) (v := v) (u := u)] using + integral_fullProjectedRewardFeatureNoiseExpProcess_le_one_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u + +/-- Fixed-direction exponential-supermartingale bound with the realized variance written as the +quadratic form of the non-ridge part of the design matrix. -/ +lemma integral_exp_dotProduct_centeredResponseVector_sub_designMatrix_sub_reg_le_one_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + ∫ ω, Real.exp + (u * dotProduct v (centeredResponseVector A R ν x n ω) - + ((σ2 : ℝ) * + dotProduct v + (Matrix.mulVec + (designMatrix A reg x n ω - reg • (1 : Matrix (Fin d) (Fin d) ℝ)) v)) * + u ^ 2 / 2) ∂P + ≤ 1 := by + simpa [fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_designMatrix_sub_reg + (A := A) (reg := reg) (x := x) (σ2 := σ2) (v := v) (n := n)] using + integral_exp_dotProduct_centeredResponseVector_sub_fullRealizedVariance_le_one_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u + +/-- Fixed-direction exponential-supermartingale bound with the realized variance written as +`σ² * (vᵀV_n v - reg * vᵀv)`. This is the algebraic form immediately before adding the Gaussian +mixture's ridge term. -/ +lemma integral_exp_centeredResponse_sub_designMatrix_minus_reg_norm_le_one_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) : + ∫ ω, Real.exp + (u * dotProduct v (centeredResponseVector A R ν x n ω) - + ((σ2 : ℝ) * + (dotProduct v (Matrix.mulVec (designMatrix A reg x n ω) v) - + reg * dotProduct v v)) * + u ^ 2 / 2) ∂P + ≤ 1 := by + simpa [fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_designMatrix_minus_reg_norm + (A := A) (reg := reg) (x := x) (σ2 := σ2) (v := v) (n := n)] using + integral_exp_dotProduct_centeredResponseVector_sub_fullRealizedVariance_le_one_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u + +/-- Fixed-direction exponential-supermartingale bound with the finite-action projection bound +chosen automatically. This is the scalar theorem used immediately before the textbook +Gaussian-mixture step. -/ +lemma integral_exp_centeredResponse_sub_designMatrix_minus_reg_norm_le_one + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (u : ℝ) : + ∫ ω, Real.exp + (u * dotProduct v (centeredResponseVector A R ν x n ω) - + ((σ2 : ℝ) * + (dotProduct v (Matrix.mulVec (designMatrix A reg x n ω) v) - + reg * dotProduct v v)) * + u ^ 2 / 2) ∂P + ≤ 1 := by + obtain ⟨Q, hQ, hQ_bound⟩ := exists_abs_dotProduct_feature_bound x v + exact integral_exp_centeredResponse_sub_designMatrix_minus_reg_norm_le_one_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Fixed-vector exponent from the scalar self-normalized proof. + +For a fixed direction `λ`, this is +`λᵀ S_t - (σ² / 2) * (λᵀ V_t λ - reg * λᵀλ)`, where `S_t` is the centered +response vector and `V_t` is the regularized design matrix. This is the quantity that the +Gaussian-mixture proof integrates over `λ`. -/ +noncomputable def centeredResponseDirectionalExponent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) (ν : Kernel (Fin K) ℝ) + (reg : ℝ) (σ2 : ℝ≥0) (x : Fin K → Feature d) (n : ℕ) (ω : Ω) + (lambda : Feature d) : ℝ := + dotProduct lambda (centeredResponseVector A R ν x n ω) - + ((σ2 : ℝ) * + (dotProduct lambda (Matrix.mulVec (designMatrix A reg x n ω) lambda) - + reg * dotProduct lambda lambda)) / 2 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The Gaussian prior penalty cancels the ridge-regularization term in the fixed-direction +exponent. + +In the textbook method-of-mixtures proof, the fixed-direction supermartingale contributes +`λᵀS_t - (σ² / 2) * λᵀ(V_t - reg I)λ`, while the Gaussian mixing density contributes +`-(σ² * reg / 2) * λᵀλ`. Adding those two exponents leaves the clean quadratic +`λᵀS_t - (σ² / 2) * λᵀV_tλ`, which is the expression that can be completed into a square. -/ +lemma centeredResponseDirectionalExponent_sub_priorPenalty_eq + (σ2 : ℝ≥0) (lambda : Feature d) : + centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda - + ((σ2 : ℝ) * reg * dotProduct lambda lambda) / 2 = + dotProduct lambda (centeredResponseVector A R ν x n ω) - + ((σ2 : ℝ) * + dotProduct lambda (Matrix.mulVec (designMatrix A reg x n ω) lambda)) / 2 := by + unfold centeredResponseDirectionalExponent + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The existing full fixed-direction exponential process at multiplier `1` is exactly the +exponential of the named textbook directional exponent. -/ +lemma fullProjectedRewardFeatureNoiseExpProcess_one_eq_exp_centeredResponseDirectionalExponent + (σ2 : ℝ≥0) (lambda : Feature d) : + fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x lambda (1 : ℝ) n ω = + Real.exp (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda) := by + rw [fullProjectedRewardFeatureNoiseExpProcess_eq_centeredResponseVector] + simp [centeredResponseDirectionalExponent, + fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_designMatrix_minus_reg_norm + (A := A) (reg := reg) (x := x) (σ2 := σ2) (v := lambda) (n := n)] + +/-- Integrability of the named fixed-vector exponential integrand used by the textbook +Gaussian-mixture proof. -/ +lemma integrable_exp_centeredResponseDirectionalExponent + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (lambda : Feature d) : + Integrable + (fun ω ↦ + Real.exp (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda)) P := by + obtain ⟨Q, hQ, hQ_bound⟩ := exists_abs_dotProduct_feature_bound x lambda + exact (fullProjectedRewardFeatureNoiseExpProcess_integrable_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν lambda Q hQ hQ_bound (1 : ℝ)).congr + (Filter.Eventually.of_forall fun ω ↦ by + exact fullProjectedRewardFeatureNoiseExpProcess_one_eq_exp_centeredResponseDirectionalExponent + (A := A) (R := R) (reg := reg) (x := x) (ν := ν) (n := n) (ω := ω) + σ2 lambda) + +/-- Fixed-vector textbook exponential process as a Mathlib `Supermartingale`. + +For each fixed direction `λ`, this packages the scalar exponential-supermartingale result in the +notation used by the textbook mixture proof: +`exp(λᵀ S_t - (σ² / 2) * λᵀ(V_t - reg I)λ)`. + +The remaining textbook Gaussian-mixture step has to integrate this fixed-`λ` supermartingale over +`λ`; this lemma supplies the pointwise-in-`λ` process without exposing the internal projected-noise +process names. -/ +lemma supermartingale_exp_centeredResponseDirectionalExponent + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (lambda : Feature d) : + Supermartingale + (fun n ω ↦ + Real.exp (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda)) + (IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback) P := by + obtain ⟨Q, hQ, hQ_bound⟩ := exists_abs_dotProduct_feature_bound x lambda + let ℱ := IsAlgEnvSeq.filtrationAction h.measurable_action h.measurable_feedback + let M : ℕ → Ω → ℝ := + fun n ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x lambda (1 : ℝ) n ω + let G : ℕ → Ω → ℝ := + fun n ω ↦ + Real.exp (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda) + have hM : Supermartingale M ℱ P := by + simpa [M, ℱ] using + supermartingale_fullProjectedRewardFeatureNoiseExpProcess_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν lambda Q hQ hQ_bound (1 : ℝ) + have hMG : ∀ n, M n = G n := by + intro n + funext ω + exact fullProjectedRewardFeatureNoiseExpProcess_one_eq_exp_centeredResponseDirectionalExponent + (A := A) (R := R) (reg := reg) (x := x) (ν := ν) (n := n) (ω := ω) + σ2 lambda + simpa [G, ℱ] using supermartingale_congr_eq (P := P) (ℱ := ℱ) hM hMG + +/-- Fixed-vector exponential bound in the named form used by the textbook Gaussian-mixture proof. + +The scalar concentration core already proves this bound for every fixed direction and scalar +multiplier. This theorem packages the `u = 1` case around +`centeredResponseDirectionalExponent`, which is the exact integrand exponent that the future +Gaussian-mixture theorem must integrate over `λ`. -/ +lemma integral_exp_centeredResponseDirectionalExponent_le_one + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (lambda : Feature d) : + ∫ ω, + Real.exp (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda) ∂P ≤ 1 := by + simpa [centeredResponseDirectionalExponent, mul_assoc] using + integral_exp_centeredResponse_sub_designMatrix_minus_reg_norm_le_one + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν lambda (1 : ℝ) + +/-- Markov tail bound for the fixed-direction exponential process, with an explicit projection +bound. + +This is the scalar probability step immediately after the exponential-supermartingale integral +bound and immediately before the textbook Gaussian-mixture argument. -/ +lemma probReal_fullProjectedRewardFeatureNoiseExpProcess_ge_le_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) + {threshold : ℝ} (hthreshold : 0 < threshold) : + P.real {ω | + threshold ≤ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω} ≤ + 1 / threshold := by + have hmarkov : + threshold * + P.real {ω | + threshold ≤ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω} ≤ + ∫ ω, fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω ∂P := + mul_meas_ge_le_integral_of_nonneg + (μ := P) + (f := fun ω ↦ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω) + (Filter.Eventually.of_forall fun ω ↦ + fullProjectedRewardFeatureNoiseExpProcess_nonneg (A := A) (R := R) + (ν := ν) (σ2 := σ2) (x := x) (v := v) (u := u) (n := n) (ω := ω)) + (fullProjectedRewardFeatureNoiseExpProcess_integrable_of_abs_le (A := A) + (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u) + threshold + have hmul : + threshold * + P.real {ω | + threshold ≤ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω} ≤ + 1 := + hmarkov.trans + (integral_fullProjectedRewardFeatureNoiseExpProcess_le_one_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u) + rw [le_div_iff₀ hthreshold] + simpa [mul_comm] using hmul + +/-- Fixed-direction exponential-process tail bound at threshold `1 / δ`, with an explicit +projection bound. -/ +lemma probReal_fullProjectedRewardFeatureNoiseExpProcess_ge_inv_delta_le_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (u : ℝ) + {δ : ℝ} (hδ_pos : 0 < δ) : + P.real {ω | + 1 / δ ≤ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω} ≤ δ := by + have htail := + probReal_fullProjectedRewardFeatureNoiseExpProcess_ge_le_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u (one_div_pos.mpr hδ_pos) + have hinv : 1 / (1 / δ) = δ := by + field_simp [hδ_pos.ne'] + simpa [hinv] using htail + +/-- Fixed-direction exponential-process tail bound with the finite-action projection bound chosen +automatically. -/ +lemma probReal_fullProjectedRewardFeatureNoiseExpProcess_ge_inv_delta_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (u : ℝ) {δ : ℝ} (hδ_pos : 0 < δ) : + P.real {ω | + 1 / δ ≤ fullProjectedRewardFeatureNoiseExpProcess A R ν σ2 x v u n ω} ≤ δ := by + obtain ⟨Q, hQ, hQ_bound⟩ := exists_abs_dotProduct_feature_bound x v + exact probReal_fullProjectedRewardFeatureNoiseExpProcess_ge_inv_delta_le_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v Q hQ hQ_bound u hδ_pos + +/-- Fixed-direction exponential-process tail bound in the centered-response/design-matrix form +used by the textbook Gaussian-mixture proof. -/ +lemma probReal_exp_centeredResponse_sub_designMatrix_minus_reg_norm_ge_inv_delta_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (u : ℝ) {δ : ℝ} (hδ_pos : 0 < δ) : + P.real {ω | + 1 / δ ≤ + Real.exp + (u * dotProduct v (centeredResponseVector A R ν x n ω) - + ((σ2 : ℝ) * + (dotProduct v (Matrix.mulVec (designMatrix A reg x n ω) v) - + reg * dotProduct v v)) * + u ^ 2 / 2)} ≤ δ := by + simpa [fullProjectedRewardFeatureNoiseExpProcess_eq_centeredResponseVector + (A := A) (R := R) (ν := ν) (x := x) (n := n) (σ2 := σ2) (v := v) (u := u), + fullProjectedRewardFeatureNoiseRealizedVarianceSum_eq_sigma_mul_designMatrix_minus_reg_norm + (A := A) (reg := reg) (x := x) (σ2 := σ2) (v := v) (n := n)] using + probReal_fullProjectedRewardFeatureNoiseExpProcess_ge_inv_delta_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) (n := n) + h hν v u hδ_pos + +/-- Fixed-direction subgaussianity of the full centered response vector. -/ +lemma dotProduct_centeredResponseVector_hasSubgaussianMGF_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (centeredResponseVector A R ν x n ω)) + (∑ _t ∈ range n, (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)) P := by + exact (fullProjectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound n).congr + (Filter.Eventually.of_forall fun ω ↦ + (dotProduct_centeredResponseVector_eq_fullProjectedRewardFeatureNoise_sum + (A := A) (R := R) (ν := ν) (x := x) (n := n) (ω := ω) v).symm) + +/-- One-sided tail bound for a fixed direction of the full centered response vector. -/ +lemma probReal_dotProduct_centeredResponseVector_ge_le_of_abs_le + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R (linUCBAlgorithm hK reg β x) (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) + {ε : ℝ} (hε : 0 ≤ ε) : + P.real + {ω | ε ≤ dotProduct v (centeredResponseVector A R ν x n ω)} + ≤ Real.exp + (-ε ^ 2 / + (2 * (∑ _t ∈ range n, (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)))) := + (dotProduct_centeredResponseVector_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (reg := reg) (β := β) (x := x) (ν := ν) + h hν v Q hQ hQ_bound n).measure_ge_le hε + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The scalar projected-noise sum is exactly the dot product of the direction with the +positive-time centered response vector. -/ +lemma dotProduct_positiveTimeCenteredResponseVector_eq_projectedRewardFeatureNoise_sum + (v : Feature d) : + dotProduct v (positiveTimeCenteredResponseVector A R ν x n ω) = + ∑ t ∈ range n, projectedRewardFeatureNoise A R ν x v t ω := by + change dotProduct v + (∑ t ∈ range n, if t = 0 then 0 else rewardNoise A R ν t ω • x (A t ω)) = + ∑ t ∈ range n, + (if t = 0 then 0 else dotProduct v (x (A t ω)) * rewardNoise A R ν t ω) + simp only [dotProduct, Finset.sum_apply] + calc + ∑ i, v i * + (∑ t ∈ range n, (if t = 0 then 0 else rewardNoise A R ν t ω • x (A t ω)) i) + = ∑ i, ∑ t ∈ range n, + v i * (if t = 0 then 0 else rewardNoise A R ν t ω • x (A t ω)) i := by + refine Finset.sum_congr rfl ?_ + intro i _hi + rw [Finset.mul_sum] + _ = ∑ t ∈ range n, ∑ i, + v i * (if t = 0 then 0 else rewardNoise A R ν t ω • x (A t ω)) i := by + rw [Finset.sum_comm] + _ = ∑ t ∈ range n, + if t = 0 then 0 else dotProduct v (x (A t ω)) * rewardNoise A R ν t ω := by + refine Finset.sum_congr rfl ?_ + intro t _ht + by_cases ht0 : t = 0 + · simp [ht0] + · simp only [ht0, if_false, Pi.smul_apply, smul_eq_mul, dotProduct] + rw [Finset.sum_mul] + refine Finset.sum_congr rfl ?_ + intro i _hi + ring + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The projected-noise sum can be rewritten as a dot product with the positive-time centered +response vector. -/ +lemma projectedRewardFeatureNoise_sum_eq_dotProduct_positiveTimeCenteredResponseVector + (v : Feature d) : + (∑ t ∈ range n, projectedRewardFeatureNoise A R ν x v t ω) = + dotProduct v (positiveTimeCenteredResponseVector A R ν x n ω) := by + rw [dotProduct_positiveTimeCenteredResponseVector_eq_projectedRewardFeatureNoise_sum] + +/-- Fixed-direction subgaussianity of the positive-time centered response vector. -/ +lemma dotProduct_positiveTimeCenteredResponseVector_hasSubgaussianMGF_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) : + HasSubgaussianMGF + (fun ω ↦ dotProduct v (positiveTimeCenteredResponseVector A R ν x n ω)) + (∑ t ∈ range n, if t = 0 then 0 else (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)) P := by + exact (projectedRewardFeatureNoise_sum_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound n).congr + (Filter.Eventually.of_forall fun ω ↦ + projectedRewardFeatureNoise_sum_eq_dotProduct_positiveTimeCenteredResponseVector + (A := A) (R := R) (ν := ν) (x := x) (n := n) (ω := ω) v) + +/-- One-sided tail bound for a fixed direction of the positive-time centered response vector. -/ +lemma probReal_dotProduct_positiveTimeCenteredResponseVector_ge_le_of_abs_le + {alg : Algorithm (Fin K) ℝ} + [StandardBorelSpace Ω] [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + (v : Feature d) (Q : ℝ) (hQ : 0 ≤ Q) + (hQ_bound : ∀ a, |dotProduct v (x a)| ≤ Q) (n : ℕ) + {ε : ℝ} (hε : 0 ≤ ε) : + P.real + {ω | + ε ≤ dotProduct v (positiveTimeCenteredResponseVector A R ν x n ω)} + ≤ Real.exp + (-ε ^ 2 / + (2 * (∑ t ∈ range n, + if t = 0 then 0 else (⟨Q ^ 2, sq_nonneg Q⟩ * σ2 : ℝ≥0)))) := + (dotProduct_positiveTimeCenteredResponseVector_hasSubgaussianMGF_of_abs_le + (A := A) (R := R) (ν := ν) (x := x) h hν v Q hQ hQ_bound n).measure_ge_le hε + +/-- Under the UCB-style arm-wise reward-noise assumption, every predictable scalar projection of the +positive-time reward-noise conditional law is subgaussian with the squared projection coefficient. + +This is still a conditional-law statement, not yet the final martingale concentration theorem. It +is the local scalar ingredient that the vector self-normalized argument must combine over time and +directions. -/ +lemma rewardNoise_condDistrib_history_action_constMul_subgaussian + {alg : Algorithm (Fin K) ℝ} + [Nonempty (Fin K)] + (h : IsAlgEnvSeq A R alg (stationaryEnv ν) P) + {σ2 : ℝ≥0} (hν : RewardNoiseSubgaussian (K := K) ν σ2) + {t : ℕ} (ht : t ≠ 0) + (q : (Iic (t - 1) → Fin K × ℝ) × Fin K → ℝ) : + ∀ᵐ z ∂P.map (fun ω ↦ (history A R (t - 1) ω, A t ω)), + HasSubgaussianMGF (fun η ↦ q z * η) + (⟨q z ^ 2, sq_nonneg (q z)⟩ * σ2) + (condDistrib (rewardNoise A R ν t) + (fun ω ↦ (history A R (t - 1) ω, A t ω)) P z) := by + have h_cond := hasCondDistrib_rewardNoise_history_action + (A := A) (R := R) (ν := ν) h ht + have h_kernel : + ∀ z : (Iic (t - 1) → Fin K × ℝ) × Fin K, + HasSubgaussianMGF (fun η ↦ q z * η) + (⟨q z ^ 2, sq_nonneg (q z)⟩ * σ2) + ((rewardNoiseKernel ν).prodMkLeft (Iic (t - 1) → Fin K × ℝ) z) := + (RewardNoiseKernelSubgaussian.of_rewardNoiseSubgaussian + (K := K) (ν := ν) hν).prodMkLeft_constMul q + filter_upwards [h_cond.condDistrib_eq] with z hz + rw [hz] + exact h_kernel z + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The centered response vector is the accumulated reward noise times the selected feature +vectors: `∑_{s.mp hV.isUnit + rw [Matrix.mulVec_sub] + have htheta : + Matrix.mulVec V (thetaHat A R reg x n ω) = responseVector A R x n ω := by + simp [thetaHat, V, Matrix.mulVec_mulVec, Matrix.mul_nonsing_inv _ hVdet] + rw [htheta] + rw [show Matrix.mulVec V θ = reg • θ + meanResponseVector A ν x n ω by + subst V + exact designMatrix_mulVec_linearMeanParameter (A := A) (reg := reg) (x := x) + (ν := ν) (n := n) (ω := ω) θ h_linear] + simp [centeredResponseVector, sub_eq_add_neg, add_assoc, add_comm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Squared ellipsoid error of the least-squares estimate around a candidate linear parameter. + +This is the process-level quantity `‖θHat_t - θ‖²_{V_t}` from the textbook confidence-set proof, +written as `(θHat_t - θ)ᵀ V_t (θHat_t - θ)`. -/ +noncomputable def parameterErrorQuadraticForm + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (x : Fin K → Feature d) (θ : Feature d) + (n : ℕ) (ω : Ω) : ℝ := + dotProduct (thetaHat A R reg x n ω - θ) + (Matrix.mulVec (designMatrix A reg x n ω) (thetaHat A R reg x n ω - θ)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The inverse-design quadratic form of the centered reward-feature vector plus the +regularization bias. + +For `z_t = ∑_{s.mp hV.isUnit + have hVy : Matrix.mulVec V y = S := by + simp [V, y, Matrix.mulVec_mulVec, Matrix.mul_nonsing_inv _ hVdet] + letI : SeminormedAddCommGroup (Feature d) := Matrix.toSeminormedAddCommGroup V hV.posSemidef + letI : InnerProductSpace ℝ (Feature d) := Matrix.toInnerProductSpace V hV.posSemidef + have h_dot_lambda : + dotProduct lambda (Matrix.mulVec V lambda) = inner ℝ lambda lambda := by + change dotProduct lambda (V *ᵥ lambda) = (V *ᵥ lambda) ⬝ᵥ lambda + rw [dotProduct_comm] + have h_dot_shift : + dotProduct (lambda - m) (Matrix.mulVec V (lambda - m)) = + inner ℝ (lambda - m) (lambda - m) := by + change dotProduct (lambda - m) (V *ᵥ (lambda - m)) = + (V *ᵥ (lambda - m)) ⬝ᵥ (lambda - m) + rw [dotProduct_comm] + have h_inner_lambda_y : inner ℝ lambda y = dotProduct lambda S := by + change (V *ᵥ y) ⬝ᵥ lambda = dotProduct lambda S + rw [hVy, dotProduct_comm] + have h_inner_y_y : + inner ℝ y y = centeredNoiseQuadraticForm A R ν reg x n ω := by + change (V *ᵥ y) ⬝ᵥ y = + dotProduct S (Matrix.mulVec V⁻¹ S) + rw [hVy] + have h_inner_shift : + inner ℝ (lambda - m) (lambda - m) = + inner ℝ lambda lambda - 2 * σ⁻¹ * dotProduct lambda S + + σ⁻¹ * σ⁻¹ * centeredNoiseQuadraticForm A R ν reg x n ω := by + rw [show m = σ⁻¹ • y by rfl] + rw [inner_sub_sub_self] + have h_left : + inner ℝ lambda (σ⁻¹ • y) = σ⁻¹ * dotProduct lambda S := by + rw [real_inner_smul_right, h_inner_lambda_y] + have h_right : + inner ℝ (σ⁻¹ • y) lambda = σ⁻¹ * dotProduct lambda S := by + rw [real_inner_smul_left, real_inner_comm, h_inner_lambda_y] + have h_self : + inner ℝ (σ⁻¹ • y) (σ⁻¹ • y) = + σ⁻¹ * σ⁻¹ * centeredNoiseQuadraticForm A R ν reg x n ω := by + rw [real_inner_smul_left, real_inner_smul_right, h_inner_y_y] + ring + rw [h_left, h_right, h_self] + ring + have h_cancel : + centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda - + (σ * reg * dotProduct lambda lambda) / 2 = + dotProduct lambda S - + (σ * dotProduct lambda (Matrix.mulVec V lambda)) / 2 := by + simpa [σ, V, S] using + centeredResponseDirectionalExponent_sub_priorPenalty_eq + (A := A) (R := R) (ν := ν) (reg := reg) (x := x) (n := n) (ω := ω) + σ2 lambda + calc + centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda - + ((σ2 : ℝ) * reg * dotProduct lambda lambda) / 2 + = dotProduct lambda S - (σ * dotProduct lambda (Matrix.mulVec V lambda)) / 2 := by + simpa [σ] using h_cancel + _ = + centeredNoiseQuadraticForm A R ν reg x n ω / (2 * σ) - + (σ / 2) * dotProduct (lambda - m) (Matrix.mulVec V (lambda - m)) := by + rw [h_dot_lambda, h_dot_shift, h_inner_shift] + field_simp [hσ2_ne, σ] + ring + _ = + centeredNoiseQuadraticForm A R ν reg x n ω / (2 * (σ2 : ℝ)) - + ((σ2 : ℝ) / 2) * + dotProduct + (lambda - (σ2 : ℝ)⁻¹ • + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ + (centeredResponseVector A R ν x n ω)) + (Matrix.mulVec (designMatrix A reg x n ω) + (lambda - (σ2 : ℝ)⁻¹ • + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ + (centeredResponseVector A R ν x n ω))) := by + rfl + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Exponential form of the completed-square identity. + +This is the integrand shape needed by the Gaussian-mixture proof: after multiplying the +fixed-direction supermartingale integrand by the Gaussian prior density's ridge penalty, the +dependence on `lambda` is only the translated quadratic kernel, while the +`centeredNoiseQuadraticForm` factor is independent of `lambda`. -/ +lemma exp_centeredResponseDirectionalExponent_sub_priorPenalty_eq_completedSquare + (σ2 : ℝ≥0) (lambda : Feature d) + (hreg_pos : 0 < reg) (hσ2_ne : (σ2 : ℝ) ≠ 0) : + Real.exp + (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda - + ((σ2 : ℝ) * reg * dotProduct lambda lambda) / 2) = + Real.exp (centeredNoiseQuadraticForm A R ν reg x n ω / (2 * (σ2 : ℝ))) * + Real.exp + (-(((σ2 : ℝ) / 2) * + dotProduct + (lambda - (σ2 : ℝ)⁻¹ • + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ + (centeredResponseVector A R ν x n ω)) + (Matrix.mulVec (designMatrix A reg x n ω) + (lambda - (σ2 : ℝ)⁻¹ • + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ + (centeredResponseVector A R ν x n ω))))) := by + rw [centeredResponseDirectionalExponent_sub_priorPenalty_eq_completedSquare + (A := A) (R := R) (ν := ν) (reg := reg) (x := x) (n := n) (ω := ω) + σ2 lambda hreg_pos hσ2_ne] + rw [sub_eq_add_neg, Real.exp_add] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Multiplicative form of the completed-square identity used by the Gaussian-mixture proof. + +The left-hand side is the product of the fixed-direction supermartingale integrand and the +unnormalized Gaussian prior penalty. The right-hand side separates the `lambda`-independent +self-normalized factor from the translated Gaussian kernel. -/ +lemma exp_centeredResponseDirectionalExponent_mul_exp_neg_priorPenalty_eq_completedSquare + (σ2 : ℝ≥0) (lambda : Feature d) + (hreg_pos : 0 < reg) (hσ2_ne : (σ2 : ℝ) ≠ 0) : + Real.exp (centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda) * + Real.exp (-(((σ2 : ℝ) * reg * dotProduct lambda lambda) / 2)) = + Real.exp (centeredNoiseQuadraticForm A R ν reg x n ω / (2 * (σ2 : ℝ))) * + Real.exp + (-(((σ2 : ℝ) / 2) * + dotProduct + (lambda - (σ2 : ℝ)⁻¹ • + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ + (centeredResponseVector A R ν x n ω)) + (Matrix.mulVec (designMatrix A reg x n ω) + (lambda - (σ2 : ℝ)⁻¹ • + Matrix.mulVec (designMatrix A reg x n ω)⁻¹ + (centeredResponseVector A R ν x n ω))))) := by + rw [← Real.exp_add] + have h_arg : + centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda + + -(((σ2 : ℝ) * reg * dotProduct lambda lambda) / 2) = + centeredResponseDirectionalExponent A R ν reg σ2 x n ω lambda - + ((σ2 : ℝ) * reg * dotProduct lambda lambda) / 2 := by + ring + rw [h_arg] + exact exp_centeredResponseDirectionalExponent_sub_priorPenalty_eq_completedSquare + (A := A) (R := R) (ν := ν) (reg := reg) (x := x) (n := n) (ω := ω) + σ2 lambda hreg_pos hσ2_ne + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- A coordinate-wise absolute-value bound controls the squared Euclidean norm of a feature +vector. -/ +lemma dotProduct_self_le_nat_mul_sq_of_abs_le + (u : Feature d) {B : ℝ} (hB_nonneg : 0 ≤ B) + (hu : ∀ i, |u i| ≤ B) : + dotProduct u u ≤ (d : ℝ) * B ^ 2 := by + rw [dotProduct] + calc + ∑ i, u i * u i ≤ ∑ _i : Fin d, B ^ 2 := by + refine Finset.sum_le_sum fun i _hi ↦ ?_ + have hsq : u i ^ 2 ≤ B ^ 2 := by + exact sq_le_sq.2 (by simpa [abs_of_nonneg hB_nonneg] using hu i) + simpa [pow_two] using hsq + _ = (d : ℝ) * B ^ 2 := by + simp [Finset.sum_const, nsmul_eq_mul] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization bounds the centered-noise inverse-design quadratic form by the +ordinary squared Euclidean norm divided by `reg`. -/ +lemma centeredNoiseQuadraticForm_le_sqNorm_div_reg + (hreg_pos : 0 < reg) : + centeredNoiseQuadraticForm A R ν reg x n ω ≤ + dotProduct (centeredResponseVector A R ν x n ω) + (centeredResponseVector A R ν x n ω) / reg := by + calc + centeredNoiseQuadraticForm A R ν reg x n ω + ≤ dotProduct (centeredResponseVector A R ν x n ω) + (Matrix.mulVec (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ + (centeredResponseVector A R ν x n ω)) := + dotProduct_mulVec_le_of_matrix_le + ((DesignMatrixInvLeRegInv.of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos).apply n ω) + (centeredResponseVector A R ν x n ω) + _ = dotProduct (centeredResponseVector A R ν x n ω) + (centeredResponseVector A R ν x n ω) / reg := + dotProduct_reg_smul_one_inv_mulVec (reg := reg) hreg_pos.ne' + (centeredResponseVector A R ν x n ω) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Coordinate-wise control of the centered response vector gives a conservative bound on the +centered-noise inverse-design quadratic form. -/ +lemma centeredNoiseQuadraticForm_le_nat_mul_coord_sq_div_reg + (hreg_pos : 0 < reg) {B : ℝ} (hB_nonneg : 0 ≤ B) + (hcoord : ∀ i, |centeredResponseVector A R ν x n ω i| ≤ B) : + centeredNoiseQuadraticForm A R ν reg x n ω ≤ (d : ℝ) * B ^ 2 / reg := by + refine (centeredNoiseQuadraticForm_le_sqNorm_div_reg (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := n) (ω := ω) hreg_pos).trans ?_ + exact div_le_div_of_nonneg_right + (dotProduct_self_le_nat_mul_sq_of_abs_le + (centeredResponseVector A R ν x n ω) hB_nonneg hcoord) + hreg_pos.le + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Nonnegative regularization makes the centered-noise inverse-design quadratic form +nonnegative. -/ +lemma centeredNoiseQuadraticForm_nonneg_of_reg_nonneg + (hreg_nonneg : 0 ≤ reg) : + 0 ≤ centeredNoiseQuadraticForm A R ν reg x n ω := by + simpa [centeredNoiseQuadraticForm] using + ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg).inv.dotProduct_mulVec_nonneg (centeredResponseVector A R ν x n ω)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Nonnegative regularization makes the ridge-bias inverse-design quadratic form nonnegative. -/ +lemma regularizationBiasQuadraticForm_nonneg_of_reg_nonneg + (θ : Feature d) (hreg_nonneg : 0 ≤ reg) : + 0 ≤ regularizationBiasQuadraticForm A reg x θ n ω := by + simpa [regularizationBiasQuadraticForm] using + ((designMatrix_posSemidef (A := A) (reg := reg) (x := x) (n := n) (ω := ω) + hreg_nonneg).inv.dotProduct_mulVec_nonneg (reg • θ)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The combined centered-noise-plus-bias quadratic form is bounded by the square of the sum of +the separate noise and ridge-bias inverse-design norms. + +This is the deterministic triangle-inequality step in the textbook proof: +`‖S_t - reg θ‖_{V_t⁻¹} ≤ ‖S_t‖_{V_t⁻¹} + ‖reg θ‖_{V_t⁻¹}`. -/ +lemma centeredNoiseBiasQuadraticForm_le_sqrt_add_sqrt_sq + (θ : Feature d) + (hreg_pos : 0 < reg) : + centeredNoiseBiasQuadraticForm A R ν reg x θ n ω ≤ + (√(centeredNoiseQuadraticForm A R ν reg x n ω) + + √(regularizationBiasQuadraticForm A reg x θ n ω)) ^ 2 := by + let V : Matrix (Fin d) (Fin d) ℝ := designMatrix A reg x n ω + let M : Matrix (Fin d) (Fin d) ℝ := V⁻¹ + let u : Feature d := centeredResponseVector A R ν x n ω + let b : Feature d := reg • θ + have hV : V.PosDef := designMatrix_posDef (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) hreg_pos + have hM : M.PosDef := hV.inv + letI : SeminormedAddCommGroup (Feature d) := Matrix.toSeminormedAddCommGroup M hM.posSemidef + letI : InnerProductSpace ℝ (Feature d) := Matrix.toInnerProductSpace M hM.posSemidef + have hq_sub : + centeredNoiseBiasQuadraticForm A R ν reg x θ n ω = + inner ℝ (u - b) (u - b) := by + change dotProduct (u - b) (Matrix.mulVec M (u - b)) = inner ℝ (u - b) (u - b) + change dotProduct (u - b) (M *ᵥ (u - b)) = (M *ᵥ (u - b)) ⬝ᵥ (u - b) + rw [dotProduct_comm] + have hq_u : + centeredNoiseQuadraticForm A R ν reg x n ω = inner ℝ u u := by + change dotProduct u (Matrix.mulVec M u) = inner ℝ u u + change dotProduct u (M *ᵥ u) = (M *ᵥ u) ⬝ᵥ u + rw [dotProduct_comm] + have hq_b : + regularizationBiasQuadraticForm A reg x θ n ω = inner ℝ b b := by + change dotProduct b (Matrix.mulVec M b) = inner ℝ b b + change dotProduct b (M *ᵥ b) = (M *ᵥ b) ⬝ᵥ b + rw [dotProduct_comm] + have h_cross_sq := real_inner_mul_inner_self_le u b + have h_cross_sq_pow : + (inner ℝ u b) ^ 2 ≤ inner ℝ u u * inner ℝ b b := by + simpa [pow_two] using h_cross_sq + have h_cross_abs : + |inner ℝ u b| ≤ √(inner ℝ u u * inner ℝ b b) := + Real.abs_le_sqrt h_cross_sq_pow + have h_sqrt_mul : + √(inner ℝ u u * inner ℝ b b) = + √(inner ℝ u u) * √(inner ℝ b b) := + Real.sqrt_mul (real_inner_self_nonneg (x := u)) (inner ℝ b b) + have h_sqrt_u_mul : + √(inner ℝ u u) * √(inner ℝ u u) = inner ℝ u u := by + rw [← sq, Real.sq_sqrt (real_inner_self_nonneg (x := u))] + have h_sqrt_b_mul : + √(inner ℝ b b) * √(inner ℝ b b) = inner ℝ b b := by + rw [← sq, Real.sq_sqrt (real_inner_self_nonneg (x := b))] + have h_neg_cross : + - inner ℝ u b ≤ √(inner ℝ u u) * √(inner ℝ b b) := by + exact (neg_le_abs (inner ℝ u b)).trans (h_cross_abs.trans_eq h_sqrt_mul) + have h_cross_comm : inner ℝ b u = inner ℝ u b := by + rw [real_inner_comm] + calc + centeredNoiseBiasQuadraticForm A R ν reg x θ n ω + = inner ℝ (u - b) (u - b) := hq_sub + _ = inner ℝ u u - inner ℝ u b - inner ℝ b u + inner ℝ b b := by + rw [inner_sub_sub_self] + _ = inner ℝ u u + inner ℝ b b - 2 * inner ℝ u b := by + rw [h_cross_comm] + ring + _ ≤ inner ℝ u u + inner ℝ b b + + 2 * (√(inner ℝ u u) * √(inner ℝ b b)) := by + nlinarith [h_neg_cross] + _ = (√(inner ℝ u u) + √(inner ℝ b b)) ^ 2 := by + nlinarith [h_sqrt_u_mul, h_sqrt_b_mul] + _ = (√(centeredNoiseQuadraticForm A R ν reg x n ω) + + √(regularizationBiasQuadraticForm A reg x θ n ω)) ^ 2 := by + rw [hq_u, hq_b] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Bound form of `centeredNoiseBiasQuadraticForm_le_sqrt_add_sqrt_sq`: if the random +self-normalized noise term is at most `B` and the ridge-bias term is at most `C`, then the combined +least-squares error term is at most `(√B + √C)²`. -/ +lemma centeredNoiseBiasQuadraticForm_le_sqrt_bounds_sq + (θ : Feature d) (hreg_pos : 0 < reg) (B C : ℝ) + (hB : centeredNoiseQuadraticForm A R ν reg x n ω ≤ B) + (hC : regularizationBiasQuadraticForm A reg x θ n ω ≤ C) : + centeredNoiseBiasQuadraticForm A R ν reg x θ n ω ≤ (√B + √C) ^ 2 := by + refine (centeredNoiseBiasQuadraticForm_le_sqrt_add_sqrt_sq (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := n) (ω := ω) θ hreg_pos).trans ?_ + gcongr + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under linear realizability and positive regularization, the parameter ellipsoid quantity is the +inverse-design quadratic form of the centered reward-feature vector plus the regularization bias. + +This is the deterministic algebraic identity immediately before the textbook self-normalized +concentration step: if `z_t = ∑_{s.mp hV.isUnit + have hz : Matrix.mulVec V u = z := by + simp only [V, u, z] + exact designMatrix_mulVec_thetaHat_sub_linearMeanParameter (A := A) (R := R) + (reg := reg) (x := x) (ν := ν) (n := n) (ω := ω) θ h_linear hreg_pos + have hu : u = Matrix.mulVec V⁻¹ z := by + calc + u = Matrix.mulVec 1 u := by simp + _ = Matrix.mulVec (V⁻¹ * V) u := by rw [Matrix.nonsing_inv_mul _ hVdet] + _ = Matrix.mulVec V⁻¹ (Matrix.mulVec V u) := by rw [Matrix.mulVec_mulVec] + _ = Matrix.mulVec V⁻¹ z := by rw [hz] + change dotProduct u (Matrix.mulVec V u) = dotProduct z (Matrix.mulVec V⁻¹ z) + rw [hz, hu, dotProduct_comm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Packaged form of +`parameterErrorQuadraticForm_eq_centeredResponseVector_sub_reg_inv_designMatrix` using +`centeredNoiseBiasQuadraticForm`. -/ +lemma parameterErrorQuadraticForm_eq_centeredNoiseBiasQuadraticForm + (θ : Feature d) + (h_linear : LinearMeanModel ν x θ) + (hreg_pos : 0 < reg) : + parameterErrorQuadraticForm A R reg x θ n ω = + centeredNoiseBiasQuadraticForm A R ν reg x θ n ω := by + simpa [centeredNoiseBiasQuadraticForm] using + parameterErrorQuadraticForm_eq_centeredResponseVector_sub_reg_inv_designMatrix + (A := A) (R := R) (reg := reg) (x := x) (ν := ν) (n := n) (ω := ω) + θ h_linear hreg_pos + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization controls the deterministic regularization-bias term in the +inverse-design norm: +`(reg θ)ᵀ V_t⁻¹ (reg θ) ≤ reg * ‖θ‖²`. + +This is the deterministic bias half of the textbook confidence-radius proof, paired later with a +self-normalized concentration bound for the centered reward-feature noise. -/ +lemma regularizationBias_invDesign_quadraticForm_le + (θ : Feature d) + (hreg_pos : 0 < reg) : + dotProduct (reg • θ) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (reg • θ)) ≤ + reg * dotProduct θ θ := by + calc + dotProduct (reg • θ) + (Matrix.mulVec (designMatrix A reg x n ω)⁻¹ (reg • θ)) + ≤ dotProduct (reg • θ) + (Matrix.mulVec (reg • (1 : Matrix (Fin d) (Fin d) ℝ))⁻¹ (reg • θ)) := + dotProduct_mulVec_le_of_matrix_le + ((DesignMatrixInvLeRegInv.of_reg_pos (A := A) (reg := reg) (x := x) + hreg_pos).apply n ω) + (reg • θ) + _ = reg * dotProduct θ θ := by + rw [dotProduct_reg_smul_one_inv_mulVec (reg := reg) hreg_pos.ne' (reg • θ)] + simp [dotProduct, smul_eq_mul] + field_simp [hreg_pos.ne'] + rw [← Finset.mul_sum] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Packaged version of `regularizationBias_invDesign_quadraticForm_le` using +`regularizationBiasQuadraticForm`. -/ +lemma regularizationBiasQuadraticForm_le + (θ : Feature d) + (hreg_pos : 0 < reg) : + regularizationBiasQuadraticForm A reg x θ n ω ≤ reg * dotProduct θ θ := by + simpa [regularizationBiasQuadraticForm] using + regularizationBias_invDesign_quadraticForm_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) θ hreg_pos + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- If `θᵀθ ≤ S2`, then the deterministic ridge-bias contribution is bounded by `reg * S2`. -/ +lemma regularizationBiasQuadraticForm_le_of_parameterSqNormBound + (θ : Feature d) (hreg_pos : 0 < reg) {S2 : ℝ} + (hθ : ParameterSqNormBound θ S2) : + regularizationBiasQuadraticForm A reg x θ n ω ≤ reg * S2 := + (regularizationBiasQuadraticForm_le (A := A) (reg := reg) (x := x) + (n := n) (ω := ω) θ hreg_pos).trans + (mul_le_mul_of_nonneg_left hθ hreg_pos.le) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Matrix Cauchy-Schwarz property for LinUCB prediction errors. + +Mathematically, for `V_t = designMatrix A reg x t ω`, this is +`|uᵀ x_a| ≤ sqrt(uᵀ V_t u) * sqrt(x_aᵀ V_t⁻¹ x_a)`. + +It is isolated as a named proposition because it is the linear-algebra step that turns the +textbook parameter ellipsoid confidence set into action-wise prediction confidence intervals. The +lemma `linUCBPredictionErrorCauchySchwarz_of_reg_pos` below proves it from positive +regularization. -/ +def LinUCBPredictionErrorCauchySchwarz + (A : ℕ → Ω → Fin K) (reg : ℝ) (x : Fin K → Feature d) : Prop := + ∀ (u : Feature d) (a : Fin K) (t : ℕ) (ω : Ω), + |dotProduct u (x a)| ≤ + √(dotProduct u (Matrix.mulVec (designMatrix A reg x t ω) u)) * + width A reg x a t ω + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Positive regularization gives the matrix Cauchy-Schwarz property used to pass from the +textbook parameter ellipsoid to arm-wise prediction intervals. + +The proof equips the feature space with the inner product induced by +`V_t = designMatrix A reg x t ω`. Cauchy-Schwarz in that inner product gives +`|uᵀ V_t (V_t⁻¹ x_a)|² ≤ (uᵀ V_t u) (x_aᵀ V_t⁻¹ x_a)`, and positive regularization makes +`V_t` positive definite, so `V_t V_t⁻¹ = I`. -/ +lemma linUCBPredictionErrorCauchySchwarz_of_reg_pos + (hreg_pos : 0 < reg) : + LinUCBPredictionErrorCauchySchwarz A reg x := by + intro u a t ω + let V : Matrix (Fin d) (Fin d) ℝ := designMatrix A reg x t ω + have hV : V.PosDef := designMatrix_posDef (A := A) (reg := reg) (x := x) + (n := t) (ω := ω) hreg_pos + have hVdet : IsUnit V.det := + (Matrix.isUnit_iff_isUnit_det (A := V)).mp hV.isUnit + let v : Feature d := Matrix.mulVec V⁻¹ (x a) + letI : SeminormedAddCommGroup (Feature d) := Matrix.toSeminormedAddCommGroup V hV.posSemidef + letI : InnerProductSpace ℝ (Feature d) := Matrix.toInnerProductSpace V hV.posSemidef + have h_inner_uv : + inner ℝ u v = dotProduct u (Matrix.mulVec V v) := by + change (V *ᵥ v) ⬝ᵥ u = dotProduct u (Matrix.mulVec V v) + rw [dotProduct_comm] + have h_inner_uu : + inner ℝ u u = dotProduct u (Matrix.mulVec V u) := by + change (V *ᵥ u) ⬝ᵥ u = dotProduct u (Matrix.mulVec V u) + rw [dotProduct_comm] + have h_inner_vv : + inner ℝ v v = dotProduct v (Matrix.mulVec V v) := by + change (V *ᵥ v) ⬝ᵥ v = dotProduct v (Matrix.mulVec V v) + rw [dotProduct_comm] + have hsq := real_inner_mul_inner_self_le u v + rw [h_inner_uv, h_inner_uu, h_inner_vv] at hsq + have hsq' : + dotProduct u (x a) ^ 2 ≤ + dotProduct u (Matrix.mulVec V u) * + dotProduct (x a) (Matrix.mulVec V⁻¹ (x a)) := by + simpa [v, Matrix.mulVec_mulVec, Matrix.mul_nonsing_inv _ hVdet, dotProduct_comm, sq] using + hsq + have h_abs := Real.abs_le_sqrt hsq' + calc + |dotProduct u (x a)| + ≤ √(dotProduct u (Matrix.mulVec V u) * + dotProduct (x a) (Matrix.mulVec V⁻¹ (x a))) := h_abs + _ = √(dotProduct u (Matrix.mulVec V u)) * + √(dotProduct (x a) (Matrix.mulVec V⁻¹ (x a))) := by + rw [Real.sqrt_mul (by simpa using hV.posSemidef.dotProduct_mulVec_nonneg u)] + _ = √(dotProduct u (Matrix.mulVec (designMatrix A reg x t ω) u)) * + width A reg x a t ω := by + simp [V, width, widthQuadraticForm] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Textbook-shaped parameter confidence event for finite-action LinUCB. + +This is the event that the true linear parameter `θ` lies in every positive-time confidence +ellipsoid: `‖θHat_t - θ‖²_{V_t} ≤ β_{t+1}`. The later self-normalized concentration theorem should +prove this event, or a theorem immediately implying it, with high probability for the textbook +choice of `β`. -/ +def LinUCBParameterEllipsoidConfidenceEvent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (θ : Feature d) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → parameterErrorQuadraticForm A R reg x θ t ω ≤ β (t + 1) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local parameter confidence event for finite-action LinUCB. -/ +def LinUCBParameterEllipsoidConfidenceEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (θ : Feature d) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → + parameterErrorQuadraticForm A R reg x θ t ω ≤ β (t + 1) + +omit [IsMarkovKernel ν] in +/-- A global parameter ellipsoid confidence event implies its finite-horizon restriction. -/ +lemma LinUCBParameterEllipsoidConfidenceEvent.toUpTo + (θ : Feature d) + (h_ellipsoid : LinUCBParameterEllipsoidConfidenceEvent A R reg β x θ ω) : + LinUCBParameterEllipsoidConfidenceEventUpTo A R reg β x θ n ω := by + intro t _ht ht0 + exact h_ellipsoid t ht0 + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Textbook self-normalized confidence event stated at the centered-noise-plus-bias level. + +This is the event naturally exposed by the least-squares decomposition proved above: +`‖θHat_t - θ‖²_{V_t}` is equal to `centeredNoiseBiasQuadraticForm`, so controlling this event is +enough to put the true parameter in every LinUCB confidence ellipsoid. -/ +def LinUCBCenteredNoiseBiasConfidenceEvent + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (θ : Feature d) (ω : Ω) : Prop := + ∀ t, t ≠ 0 → centeredNoiseBiasQuadraticForm A R ν reg x θ t ω ≤ β (t + 1) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local centered-noise-plus-bias confidence event for finite-action LinUCB. -/ +def LinUCBCenteredNoiseBiasConfidenceEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (β : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (θ : Feature d) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → + centeredNoiseBiasQuadraticForm A R ν reg x θ t ω ≤ β (t + 1) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Horizon-local self-normalized event for only the random centered reward-feature vector. + +This separates the probabilistic martingale term from the deterministic ridge-bias term. A +textbook self-normalized concentration theorem should prove this event for a concrete +`noiseBudget`; the deterministic lemmas above then add the ridge-bias radius. -/ +def LinUCBCenteredNoiseConfidenceEventUpTo + (A : ℕ → Ω → Fin K) (R : ℕ → Ω → ℝ) + (reg : ℝ) (noiseBudget : ℕ → ℝ) (x : Fin K → Feature d) + (ν : Kernel (Fin K) ℝ) (n : ℕ) (ω : Ω) : Prop := + ∀ t, t ∈ range n → t ≠ 0 → + centeredNoiseQuadraticForm A R ν reg x t ω ≤ noiseBudget (t + 1) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Textbook determinant-ratio self-normalized noise bound. + +This is the scalar radius appearing before the ridge-bias term in the LinUCB confidence proof: +`2 σ² log(√detRatio / δ)`. The future Gaussian-mixture theorem should prove that the centered +noise quadratic form is bounded by this expression with high probability. -/ +noncomputable def textbookSelfNormalizedNoiseBound + (σ2 : ℝ≥0) (δ : ℝ) (detRatio : ℝ) : ℝ := + 2 * (σ2 : ℝ) * Real.log (√detRatio / δ) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The textbook self-normalized noise bound is monotone in the determinant ratio. -/ +lemma textbookSelfNormalizedNoiseBound_mono_detRatio + {σ2 : ℝ≥0} {δ ratio D : ℝ} + (hratio_pos : 0 < ratio) (hratio_le : ratio ≤ D) (hδ_pos : 0 < δ) : + textbookSelfNormalizedNoiseBound σ2 δ ratio ≤ + textbookSelfNormalizedNoiseBound σ2 δ D := by + have hsqrt_ratio_pos : 0 < √ratio := Real.sqrt_pos.2 hratio_pos + have hdiv_pos : 0 < √ratio / δ := div_pos hsqrt_ratio_pos hδ_pos + have hdiv_le : √ratio / δ ≤ √D / δ := + div_le_div_of_nonneg_right (Real.sqrt_le_sqrt hratio_le) hδ_pos.le + have hlog : Real.log (√ratio / δ) ≤ Real.log (√D / δ) := + Real.log_le_log hdiv_pos hdiv_le + have hcoef_nonneg : 0 ≤ 2 * (σ2 : ℝ) := by positivity + simpa [textbookSelfNormalizedNoiseBound, mul_assoc] using + mul_le_mul_of_nonneg_left hlog hcoef_nonneg + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Finite-horizon textbook LinUCB confidence radius, squared. + +This is the deterministic radius used by `textbookLinUCBBeta` on the target horizon. -/ +noncomputable def textbookLinUCBConfidenceRadius + (d : ℕ) (reg S2 : ℝ) (σ2 : ℝ≥0) (L2 : ℝ) (n : ℕ) (δ : ℝ) : ℝ := + max 1 + ((√(textbookSelfNormalizedNoiseBound σ2 δ + (textbookDesignDetRatioTraceBound d reg L2 n)) + √(reg * S2)) ^ 2) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Finite-horizon textbook LinUCB beta schedule. + +The inner expression is the textbook radius +`(sqrt(2 σ² log(sqrt(D_n)/δ)) + sqrt(reg * S²))²`, where +`D_n = (1 + n L² / (reg d))^d` is the terminal determinant-ratio trace budget. +The outer `max 1` packages the schedule assumptions used by the regret proof. The schedule equals +this textbook radius through the target horizon `n` and is extended monotonically after `n`. -/ +noncomputable def textbookLinUCBBeta + (d : ℕ) (reg S2 : ℝ) (σ2 : ℝ≥0) (L2 : ℝ) (n : ℕ) (δ : ℝ) (m : ℕ) : ℝ := + textbookLinUCBConfidenceRadius d reg S2 σ2 L2 n δ + max 0 ((m : ℝ) - (n : ℝ)) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- On the target horizon, the finite-horizon textbook beta schedule is exactly the textbook +confidence radius. -/ +lemma textbookLinUCBBeta_at_horizon + (d : ℕ) (reg S2 : ℝ) (σ2 : ℝ≥0) (L2 : ℝ) (n : ℕ) (δ : ℝ) : + textbookLinUCBBeta d reg S2 σ2 L2 n δ n = + textbookLinUCBConfidenceRadius d reg S2 σ2 L2 n δ := by + simp [textbookLinUCBBeta] + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- The textbook beta schedule dominates the realized determinant-ratio confidence radius whenever +the realized determinant ratio is below the terminal trace budget. -/ +lemma textbookLinUCBBeta_budget_of_designDetRatio_le + (S2 L2 : ℝ) {σ2 : ℝ≥0} {δ : ℝ} {t m : ℕ} + (hratio_pos : 0 < designDetRatio A reg x t ω) + (hratio_le : + designDetRatio A reg x t ω ≤ textbookDesignDetRatioTraceBound d reg L2 n) + (hδ_pos : 0 < δ) : + (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2)) ^ 2 ≤ + textbookLinUCBBeta d reg S2 σ2 L2 n δ m := by + have hnoise_le : + textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω) ≤ + textbookSelfNormalizedNoiseBound σ2 δ (textbookDesignDetRatioTraceBound d reg L2 n) := + textbookSelfNormalizedNoiseBound_mono_detRatio hratio_pos hratio_le hδ_pos + have hsum_le : + √(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2) ≤ + √(textbookSelfNormalizedNoiseBound σ2 δ + (textbookDesignDetRatioTraceBound d reg L2 n)) + √(reg * S2) := + add_le_add (Real.sqrt_le_sqrt hnoise_le) le_rfl + have hleft_nonneg : + 0 ≤ √(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2) := + add_nonneg (Real.sqrt_nonneg _) (Real.sqrt_nonneg _) + have hright_nonneg : + 0 ≤ √(textbookSelfNormalizedNoiseBound σ2 δ + (textbookDesignDetRatioTraceBound d reg L2 n)) + √(reg * S2) := + add_nonneg (Real.sqrt_nonneg _) (Real.sqrt_nonneg _) + have hbase : + (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2)) ^ 2 ≤ + textbookLinUCBConfidenceRadius d reg S2 σ2 L2 n δ := by + rw [textbookLinUCBConfidenceRadius] + exact ((sq_le_sq₀ hleft_nonneg hright_nonneg).mpr hsum_le).trans (le_max_right _ _) + exact hbase.trans (by simp [textbookLinUCBBeta]) + +omit [IsMarkovKernel ν] [IsProbabilityMeasure P] in +/-- Under bounded features, the textbook beta schedule almost surely dominates every realized +finite-horizon determinant-ratio confidence radius. -/ +lemma textbookLinUCBBeta_budget_ae_of_featureSqNorm_bound + [Nonempty (Fin K)] + (S2 L2 : ℝ) {σ2 : ℝ≥0} {δ : ℝ} + (hreg_pos : 0 < reg) (hd : d ≠ 0) + (hL2 : FeatureSqNormBound x L2) (hδ_pos : 0 < δ) : + ∀ᵐ ω ∂P, ∀ t, t ∈ range n → t ≠ 0 → + (√(textbookSelfNormalizedNoiseBound σ2 δ (designDetRatio A reg x t ω)) + + √(reg * S2)) ^ 2 ≤ + textbookLinUCBBeta d reg S2 σ2 L2 n δ (t + 1) := by + filter_upwards + [designDetRatio_ae_all_le_textbookTraceBound_of_featureSqNorm_bound + (A := A) (reg := reg) (x := x) (n := n) (P := P) + L2 hreg_pos hd hL2 matrixDetLeTraceAveragePow] with ω hratioω + intro t ht _ht0 + exact textbookLinUCBBeta_budget_of_designDetRatio_le (A := A) (reg := reg) + (x := x) (n := n) (ω := ω) (S2 := S2) (L2 := L2) (σ2 := σ2) + (δ := δ) (t := t) (m := t + 1) + (designDetRatio_pos_of_reg_pos (A := A) (reg := reg) (x := x) (n := t) + (ω := ω) hreg_pos) + (hratioω t ht) hδ_pos + +end AlgorithmBehavior + +end LinUCB + +end Bandits