From 138d0d1e83b3599bb17ea75510e6b8a94f3d1c84 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 9 Mar 2026 18:28:39 +0100 Subject: [PATCH 01/49] Add some global optimization algorithms --- LeanBandits.lean | 6 + LeanBandits/OptimizationAlgorithms/LIPO.lean | 104 ++++++++++ LeanBandits/OptimizationAlgorithms/PRS.lean | 25 +++ .../OptimizationAlgorithms/RankOpt.lean | 181 ++++++++++++++++++ .../Utils/EuclideanSpace.lean | 14 ++ .../OptimizationAlgorithms/Utils/Tuple.lean | 56 ++++++ .../OptimizationAlgorithms/Utils/Uniform.lean | 25 +++ 7 files changed, 411 insertions(+) create mode 100644 LeanBandits/OptimizationAlgorithms/LIPO.lean create mode 100644 LeanBandits/OptimizationAlgorithms/PRS.lean create mode 100644 LeanBandits/OptimizationAlgorithms/RankOpt.lean create mode 100644 LeanBandits/OptimizationAlgorithms/Utils/EuclideanSpace.lean create mode 100644 LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean create mode 100644 LeanBandits/OptimizationAlgorithms/Utils/Uniform.lean diff --git a/LeanBandits.lean b/LeanBandits.lean index 1bf5134c..a0694fad 100644 --- a/LeanBandits.lean +++ b/LeanBandits.lean @@ -24,3 +24,9 @@ import LeanBandits.SequentialLearning.Deterministic import LeanBandits.SequentialLearning.FiniteActions import LeanBandits.SequentialLearning.IonescuTulceaSpace import LeanBandits.SequentialLearning.StationaryEnv +import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace +import LeanBandits.OptimizationAlgorithms.Utils.Tuple +import LeanBandits.OptimizationAlgorithms.Utils.Uniform +import LeanBandits.OptimizationAlgorithms.LIPO +import LeanBandits.OptimizationAlgorithms.PRS +import LeanBandits.OptimizationAlgorithms.RankOpt diff --git a/LeanBandits/OptimizationAlgorithms/LIPO.lean b/LeanBandits/OptimizationAlgorithms/LIPO.lean new file mode 100644 index 00000000..b72ae961 --- /dev/null +++ b/LeanBandits/OptimizationAlgorithms/LIPO.lean @@ -0,0 +1,104 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import LeanBandits.OptimizationAlgorithms.Utils.Tuple +import LeanBandits.OptimizationAlgorithms.Utils.Uniform +import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace +import LeanBandits.SequentialLearning.Algorithm + + +open MeasureTheory ProbabilityTheory Finset NNReal Learning + +/-! +# LIPO: Lipschitz Optimization +Implementation of the _LIPO_ algorithm +[(_Global optimization of Lipschitz functions_, Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) +defined on a measurable subset of a Euclidean space, with finite and non-zero measure. +The algorithm samples from the uniform distribution on the set of potential maximizers of +the function at each iteration. +-/ + +variable {d : ℕ} {α : Set (ℝᵈ d)} (mes_α : MeasurableSet α) (mα₁ : ℙ α ≠ ⊤) (κ : ℝ≥0) + +namespace LIPO + +noncomputable instance : MeasureSpace α := Measure.Subtype.measureSpace + +instance : MeasurableSpace α := by infer_instance + +instance i₁ : IsFiniteMeasure (ℙ : Measure α) := by + rw [isFiniteMeasure_iff ℙ, Measure.Subtype.volume_univ] + · exact mα₁.lt_top + · exact mes_α.nullMeasurableSet + +variable {n : ℕ} (data : Iic n → α × ℝ) + +/-- The set of potential maximizers for the LIPO algorithm. +Given observed data points and function values, this set contains all points `x` where +the maximum observed value is at most the minimum Lipschitz upper bound across all observations. +The upper bound at `x` from observation `i` is `f(xᵢ) + κ · d(xᵢ, x)`, where `κ` is the +Lipschitz constant. -/ +def potential_max : Set α := + {x | Tuple.max (fun i ↦ (data i).2) ≤ Tuple.min (fun i ↦ (data i).2 + κ * dist (data i).1 x)} + +lemma measurableSet_potential_max_prod : + MeasurableSet {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max κ p.1} := by + unfold potential_max + simp only [Set.mem_setOf_eq, measurableSet_setOf] + refine Measurable.le' ?_ ?_ + · fun_prop + · fun_prop + +include mes_α mα₁ in +lemma measurable_volume_potential_max_inter (s : Set α) (hs : MeasurableSet s) : + Measurable (fun data : Iic n → α × ℝ ↦ ℙ (potential_max κ data ∩ s)) := by + set E := {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max κ p.1 ∩ s} + have hE_meas : MeasurableSet E := + (measurableSet_potential_max_prod κ).inter (measurableSet_preimage measurable_snd hs) + have := i₁ mes_α mα₁ + exact measurable_measure_prodMk_left hE_meas + +/-- Markov kernel that samples uniformly from the set of potential maximizers. +This kernel forms the core sampling strategy of LIPO: at each iteration, given the observed +data, it samples the next query point uniformly from `potential_max`. -/ +noncomputable def potential_max_kernel : Kernel (Iic n → α × ℝ) α := by + refine ⟨fun data ↦ uniform <| potential_max κ data, ?_⟩ + rw [Measure.measurable_measure] + intro s hs + simp only [Measure.smul_apply, MeasureTheory.Measure.restrict_apply hs, smul_eq_mul] + refine Measurable.mul ?_ ?_ + · refine Measurable.inv ?_ + convert measurable_volume_potential_max_inter mes_α mα₁ κ Set.univ (MeasurableSet.univ) + simp [Set.inter_univ] + · convert measurable_volume_potential_max_inter mes_α mα₁ κ s hs using 1 + simp [Set.inter_comm] + +end LIPO + +open LIPO + +variable (mα₀ : ℙ α ≠ 0) + +/- We suppose that the set of potential maximizers has non-zero measure at each iteration, +ensuring that the algorithm can sample from it. -/ +variable (h : ∀ n (data : Iic n → α × ℝ), ℙ (potential_max κ data) ≠ 0) + +/-- The LIPO (LIPschitz Optimization) algorithm for global optimization. +This algorithm optimizes an unknown function assuming only that it has a finite Lipschitz +constant `κ`. It starts with a uniform initial distribution and iteratively samples from +the set of potential maximizers, ensuring consistency and convergence to the global optimum +[(Malherbe et al., 2017)](https://arxiv.org/abs/1703.02628). -/ +noncomputable def LIPO : Algorithm α ℝ where + policy _ := potential_max_kernel mes_α mα₁ κ + p0 := uniform Set.univ + hp0 := by + have := i₁ mes_α mα₁ + refine uniform_is_prob_measure ?_ + rwa [Measure.Subtype.volume_univ mes_α.nullMeasurableSet] + h_policy n := by + refine ⟨fun data => ?_⟩ + have := i₁ mes_α mα₁ + exact uniform_is_prob_measure <| h n data diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean new file mode 100644 index 00000000..4ba3f8c2 --- /dev/null +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -0,0 +1,25 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import LeanBandits.SequentialLearning.Algorithm + +open MeasureTheory ProbabilityTheory Learning Set + +/-! +# PRS: Pure Random Search +Implementation of the _Pure Random Search_ algorithm, which samples from the uniform +distribution on the input space at each iteration. +-/ + +variable {α β : Type*} [MeasureSpace α] [IsFiniteMeasure (ℙ : Measure α)] + [NeZero (ℙ : Measure α)] [MeasurableSpace β] + +noncomputable +def PRS : Algorithm α β where + policy _ := Kernel.const _ ((ℙ (univ : Set α))⁻¹ • ℙ) + p0 := (ℙ (univ : Set α))⁻¹ • ℙ + h_policy _ := ⟨fun _ ↦ by simp [isProbabilityMeasure_iff]⟩ + hp0 := by simp [isProbabilityMeasure_iff] diff --git a/LeanBandits/OptimizationAlgorithms/RankOpt.lean b/LeanBandits/OptimizationAlgorithms/RankOpt.lean new file mode 100644 index 00000000..8d272c89 --- /dev/null +++ b/LeanBandits/OptimizationAlgorithms/RankOpt.lean @@ -0,0 +1,181 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace +import LeanBandits.OptimizationAlgorithms.Utils.Tuple +import LeanBandits.OptimizationAlgorithms.Utils.Uniform +import LeanBandits.SequentialLearning.Algorithm + + +open MeasureTheory ProbabilityTheory Finset NNReal Learning + +/-! +# RankOpt: A Ranking Approach to Global Optimization +Implementation of the _RankOpt_ algorithm +[(_A Ranking Approach to Global Optimization_, Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) +defined on a measurable subset of a Euclidean space, with finite and non-zero measure. +The algorithm samples from the uniform distribution on the set of potential maximizers of +the function at each iteration. +-/ + +section RankRule + +/-- A rank rule is a measurable function that compares pairs of points. +It returns 1 if the first point is ranked higher, -1 if lower, and 0 if equal. -/ +-- ANCHOR: RankRule +def RankRule (α : Type) [MeasurableSpace α] := + {f : α → α → ({-1, 0, 1} : Set ℝ) // Measurable <| Function.uncurry f} +-- ANCHOR_END: RankRule + +end RankRule + +variable {α : Type} [MeasurableSpace α] {d : ℕ} {α : Set (ℝᵈ d)} + (mes_α : MeasurableSet α) (mα₁ : ℙ α ≠ ⊤) + +namespace RankOpt + +noncomputable instance : MeasureSpace α := Measure.Subtype.measureSpace + +instance : MeasurableSpace α := by infer_instance + +instance i₁ : IsFiniteMeasure (ℙ : Measure α) := by + rw [isFiniteMeasure_iff ℙ, Measure.Subtype.volume_univ] + · exact mα₁.lt_top + · exact mes_α.nullMeasurableSet + +instance : MeasurableSpace (RankRule α) := Subtype.instMeasurableSpace + +/-- Computes the ranking from observed function values. +Returns 1 if `y₁ > y₂`, 0 if `y₁ = y₂`, and -1 if `y₁ < y₂`. -/ +noncomputable def ranking_data (y₁ y₂ : ℝ) := + if y₂ < y₁ then 1 else if y₂ = y₁ then 0 else -1 + +/-- Indicator function checking if two rankings agree. +Returns 1 if both values are equal, 0 otherwise. -/ +noncomputable abbrev rindicator (r₁ r₂ : ℝ) := + if r₁ = r₂ then (1 : ℝ) else 0 + +variable {n : ℕ} (data : Iic n → α × ℝ) + +abbrev s := {(i, j) : Iic n × Iic n | i ≤ j} + +/-- Computes the ranking loss for a rank rule. +Measures the agreement between a candidate rule `r` and the rankings induced by the observed +function values on all pairs of data points, normalized by the number of pairs. -/ +noncomputable def ranking_loss (r : RankRule α) := + 2 * (n * (n + 1) : ℝ)⁻¹ * ∑ ij ∈ s, + rindicator (r.1 (data ij.1).1 (data ij.2).1) (ranking_data (data ij.1).2 (data ij.2).2) + +/-- The point in the observed data with the maximum function value. -/ +noncomputable abbrev argmax_f := (data <| Tuple.argmax (fun i ↦ (data i).2)).1 + +/-- The set of potential maximizers for the RankOpt algorithm. +Contains all points `x` for which there exists a ranking rule `r` in the hypothesis class `𝓡` +that: (1) has zero ranking loss (perfectly consistent with the observed data), +and (2) ranks `x` at least as high as the current best observed point. -/ +def potential_max (𝓡 : Set (RankRule α)) := + {x | ∃ (r : 𝓡), ranking_loss data r = 0 ∧ 0 ≤ (r.1.1 x (argmax_f data)).1} + +lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) : + MeasurableSet {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max p.1 𝓡} := by + simp only [potential_max, Set.mem_setOf_eq, measurableSet_setOf] + have : Countable (𝓡) := h𝓡.to_subtype + refine Measurable.exists fun r ↦ (.and ?_ ?_) + · simp only [ranking_loss] + refine Measurable.eq ?_ measurable_const + refine Measurable.const_mul (measurable_sum _ fun i hi ↦ ?_) _ + simp only [rindicator] + refine Measurable.ite (measurableSet_eq_fun ?_ ?_) measurable_const measurable_const + · have := r.1.2 + fun_prop + · simp only [ranking_data] + have : Measurable (fun (z : ℤ) ↦ (z : ℝ)) := by fun_prop + refine this.comp ?_ + refine Measurable.ite ?_ measurable_const <| .ite ?_ measurable_const measurable_const + · measurability + · measurability + · refine Measurable.le' measurable_const ?_ + have : Measurable (fun x : ({-1, 0, 1} : Set ℝ) ↦ (x : ℝ)) := by fun_prop + refine this.comp (r.1.2.comp (measurable_snd.prodMk ?_)) + suffices Measurable (fun p : Iic n → α × ℝ ↦ (p <| Tuple.argmax (fun i ↦ (p i).2)).1) by + exact this.comp measurable_fst + have h_eval : Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ (p.1 p.2).1) := by + sorry + refine h_eval.comp (Measurable.prodMk ?_ ?_) + · fun_prop + · change Measurable (fun p : Iic n → α × ℝ ↦ Tuple.argmax (fun i ↦ (p i).2)) + suffices Measurable (fun u : Iic n → ℝ ↦ Tuple.argmax u) by + fun_prop + refine measurable_to_countable' fun i ↦ ?_ + simp only [Set.preimage, Set.mem_singleton_iff] + let Maximizers {n : ℕ} (u : Iic n → ℝ) : Set (Iic n) := {i | u i = Tuple.max u} + have : {u : Iic n → ℝ | Tuple.argmax u = i} = ⋃ (S) + (hS : ∀ x, Maximizers x = S → Tuple.argmax x = i), {u | Maximizers u = S} := by + ext u + simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] + constructor + · intro hu x hx + rw [← hu] + unfold Tuple.argmax + exact Classical.choose.congr_simp hx (Tuple.exists_argmax x) + · intro h + exact h u rfl + rw [this] + refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) + exact measurableSet_eq_fun (by fun_prop) measurable_const + +include mes_α mα₁ in +lemma measurable_volume_potential_max_inter {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) + (s : Set α) (hs : MeasurableSet s) : + Measurable (fun data : Iic n → α × ℝ ↦ ℙ (potential_max data 𝓡 ∩ s)) := by + set E := {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max p.1 𝓡 ∩ s} + have hE_meas : MeasurableSet E := + (measurableSet_potential_max_prod h𝓡).inter (measurableSet_preimage measurable_snd hs) + have := i₁ mes_α mα₁ + exact measurable_measure_prodMk_left hE_meas + +/-- Markov kernel that samples uniformly from the set of potential maximizers. +This kernel forms the core sampling strategy of RankOpt: at each iteration, given the observed +data, it samples the next query point uniformly from `potential_max`. -/ +noncomputable def potential_max_kernel {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) : + Kernel (Iic n → α × ℝ) α := by + refine ⟨fun data ↦ uniform <| @potential_max d α n data 𝓡, ?_⟩ + rw [Measure.measurable_measure] + intro s hs + simp only [Measure.smul_apply, MeasureTheory.Measure.restrict_apply hs, smul_eq_mul] + refine Measurable.mul ?_ ?_ + · refine Measurable.inv ?_ + convert measurable_volume_potential_max_inter mes_α mα₁ h𝓡 Set.univ (MeasurableSet.univ) + simp [Set.inter_univ] + · convert measurable_volume_potential_max_inter mes_α mα₁ h𝓡 s hs using 1 + simp [Set.inter_comm] + +end RankOpt + +open RankOpt + +variable (mα₀ : ℙ α ≠ 0) {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) + +/- We suppose that the set of potential maximizers has non-zero measure at each iteration, +ensuring that the algorithm can sample from it. -/ +variable (h : ∀ n (data : Iic n → α × ℝ), ℙ (potential_max data 𝓡) ≠ 0) + +/-- The RankOpt algorithm for global optimization. +This algorithm uses a ranking approach to optimize an unknown function. It maintains a hypothesis +class `𝓡` of ranking rules. At each iteration, it samples from the set of points that could be +optimal according to ranking rules consistent with the observed data +[(Malherbe et al., 2017)](https://arxiv.org/pdf/1603.04381). -/ +noncomputable def RankOpt : Algorithm α ℝ where + policy _ := potential_max_kernel mes_α mα₁ h𝓡 + p0 := uniform Set.univ + hp0 := by + have := i₁ mes_α mα₁ + refine uniform_is_prob_measure ?_ + rwa [Measure.Subtype.volume_univ mes_α.nullMeasurableSet] + h_policy n := by + refine ⟨fun data => ?_⟩ + have := i₁ mes_α mα₁ + exact uniform_is_prob_measure <| h n data diff --git a/LeanBandits/OptimizationAlgorithms/Utils/EuclideanSpace.lean b/LeanBandits/OptimizationAlgorithms/Utils/EuclideanSpace.lean new file mode 100644 index 00000000..6e27ccfd --- /dev/null +++ b/LeanBandits/OptimizationAlgorithms/Utils/EuclideanSpace.lean @@ -0,0 +1,14 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import Mathlib.Analysis.InnerProductSpace.PiL2 + +/-- Euclidean space of dimension `d`. +Used as the domain for LIPO optimization problems. -/ +abbrev ED (d : ℕ) := EuclideanSpace ℝ (Fin d) + +@[inherit_doc ED] +notation3 "ℝᵈ " d => ED d diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean new file mode 100644 index 00000000..b4f8df8b --- /dev/null +++ b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean @@ -0,0 +1,56 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import Mathlib + +open Finset + +namespace Tuple + +variable {ι α : Type*} [LinearOrder α] [Fintype ι] [Nonempty ι] (f : ι → α) + +def max : α := univ.sup' (by simp) f + +def min : α := univ.inf' (by simp) f + +instance {n : ℕ} : Nonempty (Iic n) := Nonempty.intro ⟨0, insert_eq_self.mp rfl⟩ + +lemma exists_argmax {n : ℕ} (u : Iic n → α) : ∃ i, u i = max u := by + have : Nonempty (Iic n) := inferInstance + unfold max + let A := u '' Set.univ + suffices h : univ.sup' (by simp) u ∈ A by + obtain ⟨x, -, h⟩ := h + exact ⟨x, h⟩ + refine sup'_mem A (fun x hx y hy ↦ ?_) _ _ u fun i _ ↦ ?_ + · cases max_choice x y with + | inl l => simp_all + | inr r => simp_all + · simp [A] + +noncomputable def argmax {n : ℕ} (u : Iic n → α) := (exists_argmax u).choose + +lemma argmax_spec {n : ℕ} (u : Iic n → α) : u (argmax u) = max u := + (exists_argmax u).choose_spec + +variable [MeasurableSpace α] + +@[fun_prop] +lemma measurable_max {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.max t) := by + unfold Tuple.max + have : Nonempty (Iic n) := inferInstance + simp_all only [mem_Iic, nonempty_subtype] + fun_prop + +@[fun_prop] +lemma measurable_min_fst {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.min t) := by + unfold Tuple.min + have : Nonempty (Iic n) := inferInstance + simp_all only [mem_Iic, nonempty_subtype] + fun_prop + + +end Tuple diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Uniform.lean b/LeanBandits/OptimizationAlgorithms/Utils/Uniform.lean new file mode 100644 index 00000000..6ec66172 --- /dev/null +++ b/LeanBandits/OptimizationAlgorithms/Utils/Uniform.lean @@ -0,0 +1,25 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import Mathlib.Probability.Notation + +open MeasureTheory Set ProbabilityTheory + +/-! +# Uniform distribution +This file defines the uniform distribution on a set of a finite measure space as the normalized +restriction of the original measure to the set. +-/ + +variable {α β : Type*} [MeasureSpace α] [IsFiniteMeasure (ℙ : Measure α)] [MeasurableSpace β] + +/-- The uniform distribution on a set `s` is defined as the normalized restriction of the original +measure to `s`. -/ +noncomputable abbrev uniform (s : Set α) : Measure α := (ℙ s)⁻¹ • (ℙ).restrict s + +instance uniform_is_prob_measure {s : Set α} (hs : ℙ s ≠ 0) : IsProbabilityMeasure (uniform s) := by + rw [isProbabilityMeasure_iff] + simp [ENNReal.inv_mul_cancel, hs] From 8fb80b23d3bdc0246efd2e55af8714e2550d82a9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 9 Mar 2026 18:36:48 +0100 Subject: [PATCH 02/49] no more sorry --- LeanBandits/OptimizationAlgorithms/RankOpt.lean | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/LeanBandits/OptimizationAlgorithms/RankOpt.lean b/LeanBandits/OptimizationAlgorithms/RankOpt.lean index 8d272c89..d726d02b 100644 --- a/LeanBandits/OptimizationAlgorithms/RankOpt.lean +++ b/LeanBandits/OptimizationAlgorithms/RankOpt.lean @@ -103,7 +103,10 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. suffices Measurable (fun p : Iic n → α × ℝ ↦ (p <| Tuple.argmax (fun i ↦ (p i).2)).1) by exact this.comp measurable_fst have h_eval : Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ (p.1 p.2).1) := by - sorry + suffices Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ p.1 p.2) by + fun_prop + refine measurable_from_prod_countable_left fun i ↦ ?_ + exact measurable_pi_apply i refine h_eval.comp (Measurable.prodMk ?_ ?_) · fun_prop · change Measurable (fun p : Iic n → α × ℝ ↦ Tuple.argmax (fun i ↦ (p i).2)) From 622beb24bba5cd89d53ce3db69be87ff56cc3be2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 9 Mar 2026 18:37:03 +0100 Subject: [PATCH 03/49] golf --- LeanBandits/OptimizationAlgorithms/RankOpt.lean | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/RankOpt.lean b/LeanBandits/OptimizationAlgorithms/RankOpt.lean index d726d02b..eb7fffa8 100644 --- a/LeanBandits/OptimizationAlgorithms/RankOpt.lean +++ b/LeanBandits/OptimizationAlgorithms/RankOpt.lean @@ -105,8 +105,7 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. have h_eval : Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ (p.1 p.2).1) := by suffices Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ p.1 p.2) by fun_prop - refine measurable_from_prod_countable_left fun i ↦ ?_ - exact measurable_pi_apply i + exact measurable_from_prod_countable_left fun i ↦ measurable_pi_apply i refine h_eval.comp (Measurable.prodMk ?_ ?_) · fun_prop · change Measurable (fun p : Iic n → α × ℝ ↦ Tuple.argmax (fun i ↦ (p i).2)) From a0a04895f2b444b601b9ab0c73a3b30527cceb58 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Wed, 1 Apr 2026 12:40:48 +0200 Subject: [PATCH 04/49] `EvaluationEnv` --- LeanBandits/OptimizationAlgorithms/LIPO.lean | 1 - LeanBandits/OptimizationAlgorithms/PRS.lean | 51 ++++++++++++++++++- .../SequentialLearning/EvaluationEnv.lean | 24 +++++++++ 3 files changed, 74 insertions(+), 2 deletions(-) create mode 100644 LeanBandits/SequentialLearning/EvaluationEnv.lean diff --git a/LeanBandits/OptimizationAlgorithms/LIPO.lean b/LeanBandits/OptimizationAlgorithms/LIPO.lean index b72ae961..146c2e5f 100644 --- a/LeanBandits/OptimizationAlgorithms/LIPO.lean +++ b/LeanBandits/OptimizationAlgorithms/LIPO.lean @@ -9,7 +9,6 @@ import LeanBandits.OptimizationAlgorithms.Utils.Uniform import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace import LeanBandits.SequentialLearning.Algorithm - open MeasureTheory ProbabilityTheory Finset NNReal Learning /-! diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index 4ba3f8c2..0e0b62a5 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -5,8 +5,10 @@ Authors: Gaëtan Serré -/ import LeanBandits.SequentialLearning.Algorithm +import LeanBandits.OptimizationAlgorithms.Utils.Tuple +import LeanBandits.SequentialLearning.EvaluationEnv -open MeasureTheory ProbabilityTheory Learning Set +open MeasureTheory ProbabilityTheory Learning /-! # PRS: Pure Random Search @@ -17,9 +19,56 @@ distribution on the input space at each iteration. variable {α β : Type*} [MeasureSpace α] [IsFiniteMeasure (ℙ : Measure α)] [NeZero (ℙ : Measure α)] [MeasurableSpace β] +open Set in noncomputable def PRS : Algorithm α β where policy _ := Kernel.const _ ((ℙ (univ : Set α))⁻¹ • ℙ) p0 := (ℙ (univ : Set α))⁻¹ • ℙ h_policy _ := ⟨fun _ ↦ by simp [isProbabilityMeasure_iff]⟩ hp0 := by simp [isProbabilityMeasure_iff] + +open Finset + +variable {f : ℝ → ℝ} (hf : Continuous f) (c : ℝ) (hfc : ∀ x, f x ≤ f c) + (A : ℕ → ℝ → ℝ) (R : ℕ → ℝ → ℝ) + +noncomputable abbrev IsAlgEnvSeq.argmax (A : ℕ → ℝ → ℝ) (R : ℕ → ℝ → ℝ) (n : ℕ) (ω : ℝ) : ℝ := + A (Tuple.argmax (fun (i : Iic n) ↦ R i ω)) ω + +noncomputable abbrev IsAlgEnvSeq.max (R : ℕ → ℝ → ℝ) (n : ℕ) (ω : ℝ) : ℝ := + R (Tuple.argmax (fun (i : Iic n) ↦ R i ω)) ω + +open Filter Topology ENNReal in +lemma ENNReal.tendsto_zero_le {α : Type*} {f g : α → ℝ≥0∞} {ι : Filter α} + (hg : Tendsto g ι (𝓝 0)) (h : f ≤ g) : Tendsto f ι (𝓝 0) := by + refine tendsto_of_tendsto_of_tendsto_of_le_of_le (g := fun _ ↦ 0) tendsto_const_nhds hg ?_ h + intro + simp + +variable [IsFiniteMeasure (ℙ : Measure ℝ)] -- False, but assumed now for simplicity + +open Filter Topology ENNReal in +example (h' : IsAlgEnvSeq A R PRS (evalEnv hf.measurable) ((ℙ (Set.univ : Set ℝ))⁻¹ • ℙ)) : + TendstoInMeasure ((ℙ (Set.univ : Set α))⁻¹ • ℙ) (IsAlgEnvSeq.max R) atTop (fun _ ↦ f c) := by + set μ := ((ℙ (Set.univ : Set α))⁻¹ • (ℙ : Measure ℝ)) + rw [tendstoInMeasure_iff_dist] + intro ε₁ hε₁ + rw [Metric.continuous_iff] at hf + specialize hf c ε₁ hε₁ + obtain ⟨δ, hδ, hf⟩ := hf + let h : ℕ → ℝ≥0∞ := fun n ↦ μ {x | δ ≤ dist (IsAlgEnvSeq.argmax A R n x) c} + refine tendsto_zero_le (g := h) ?_ ?_ + · sorry + · intro n + simp only [h] + refine measure_mono ?_ + simp only [Set.setOf_subset_setOf] + intro a + by_contra! h'' + specialize hf (IsAlgEnvSeq.argmax A R n a) h''.2 + have := h''.1 + suffices IsAlgEnvSeq.max R n a = f (IsAlgEnvSeq.argmax A R n a) by + rw [← this] at hf + linarith + + sorry diff --git a/LeanBandits/SequentialLearning/EvaluationEnv.lean b/LeanBandits/SequentialLearning/EvaluationEnv.lean new file mode 100644 index 00000000..d915d234 --- /dev/null +++ b/LeanBandits/SequentialLearning/EvaluationEnv.lean @@ -0,0 +1,24 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import LeanBandits.SequentialLearning.Algorithm + +/-! +# Function evaluation environments +-/ + +open MeasureTheory ProbabilityTheory + +namespace Learning + +variable {α R : Type*} [MeasurableSpace α] [MeasurableSpace R] + +@[simps] +noncomputable def evalEnv {f : α → R} (hf : Measurable f) : Environment α R where + ν0 := Kernel.deterministic f hf + feedback _ := Kernel.deterministic (f ∘ Prod.snd) <| hf.comp measurable_snd + +end Learning From 4f202954451cdda6f52ee3acc48b60c692822344 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Thu, 2 Apr 2026 17:09:27 +0200 Subject: [PATCH 05/49] partial PRS convergence proof --- LeanBandits/OptimizationAlgorithms/PRS.lean | 153 +++++++++++++----- .../OptimizationAlgorithms/Utils/Tuple.lean | 2 +- .../SequentialLearning/EvaluationEnv.lean | 8 +- 3 files changed, 113 insertions(+), 50 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index 0e0b62a5..cf296ab9 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -7,6 +7,8 @@ Authors: Gaëtan Serré import LeanBandits.SequentialLearning.Algorithm import LeanBandits.OptimizationAlgorithms.Utils.Tuple import LeanBandits.SequentialLearning.EvaluationEnv +import LeanBandits.ForMathlib.IndepFun +import Mathlib open MeasureTheory ProbabilityTheory Learning @@ -16,59 +18,122 @@ Implementation of the _Pure Random Search_ algorithm, which samples from the uni distribution on the input space at each iteration. -/ -variable {α β : Type*} [MeasureSpace α] [IsFiniteMeasure (ℙ : Measure α)] - [NeZero (ℙ : Measure α)] [MeasurableSpace β] +section + +lemma hasLaw_of_hasCondDistrib_const {β' Ω' Ω'' : Type*} + [MeasurableSpace β'] [MeasurableSpace Ω'] [StandardBorelSpace Ω'] [Nonempty Ω'] + [MeasurableSpace Ω''] + {X : Ω'' → β'} {Y : Ω'' → Ω'} {μ : Measure Ω'} + {Q : Measure Ω''} [IsProbabilityMeasure Q] [SFinite μ] + (h : HasCondDistrib Y X (Kernel.const _ μ) Q) : HasLaw Y μ Q := by + obtain ⟨hY, hX, h⟩ := h + refine ⟨hY, ?_⟩ + have h_snd : (Q.map (fun ω => (X ω, Y ω))).snd = μ := by + have h_map : Q.map (fun ω => (X ω, Y ω)) = (Q.map X) ⊗ₘ (Kernel.const _ μ) := + have h_map : Q.map (fun ω => (X ω, Y ω)) = (Q.map X) ⊗ₘ (condDistrib Y X Q) := + (compProd_map_condDistrib hY).symm + h_map.trans (Measure.compProd_congr h) + rw [h_map, MeasureTheory.Measure.snd_compProd] + simp [MeasureTheory.Measure.map_apply_of_aemeasurable hX] + rwa [Measure.snd_map_prodMk₀ hX] at h_snd + +open ENNReal Filter Topology in +lemma ENNReal.tendsto_zero_le {α : Type*} {f g : α → ℝ≥0∞} {ι : Filter α} + (hg : Tendsto g ι (𝓝 0)) (h : f ≤ g) : Tendsto f ι (𝓝 0) := by + refine tendsto_of_tendsto_of_tendsto_of_le_of_le (g := fun _ ↦ 0) tendsto_const_nhds hg ?_ h + intro + simp + +end + +variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] (μ : Measure α) + [IsProbabilityMeasure μ] open Set in +@[simps] noncomputable def PRS : Algorithm α β where - policy _ := Kernel.const _ ((ℙ (univ : Set α))⁻¹ • ℙ) - p0 := (ℙ (univ : Set α))⁻¹ • ℙ - h_policy _ := ⟨fun _ ↦ by simp [isProbabilityMeasure_iff]⟩ - hp0 := by simp [isProbabilityMeasure_iff] + policy _ := Kernel.const _ μ + p0 := μ + h_policy _ := ⟨fun _ ↦ inferInstance⟩ + hp0 := inferInstance -open Finset +namespace PRS -variable {f : ℝ → ℝ} (hf : Continuous f) (c : ℝ) (hfc : ∀ x, f x ≤ f c) - (A : ℕ → ℝ → ℝ) (R : ℕ → ℝ → ℝ) +variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace β] [Nonempty β] + {Ω : Type*} [MeasurableSpace Ω] (P : Measure Ω) [IsProbabilityMeasure P] + (A : ℕ → Ω → α) (R : ℕ → Ω → β) {f : α → β} (hf : Measurable f) -noncomputable abbrev IsAlgEnvSeq.argmax (A : ℕ → ℝ → ℝ) (R : ℕ → ℝ → ℝ) (n : ℕ) (ω : ℝ) : ℝ := - A (Tuple.argmax (fun (i : Iic n) ↦ R i ω)) ω +lemma hasLaw_action + (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) + (n : ℕ) : HasLaw (A n) (PRS (β := β) μ).p0 P := by + by_cases hn : n = 0 + · rw [hn] + exact h'.hasLaw_action_zero + · push_neg at hn + obtain ⟨k, rfl⟩ := Nat.exists_eq_succ_of_ne_zero hn + exact hasLaw_of_hasCondDistrib_const <| h'.hasCondDistrib_action k -noncomputable abbrev IsAlgEnvSeq.max (R : ℕ → ℝ → ℝ) (n : ℕ) (ω : ℝ) : ℝ := - R (Tuple.argmax (fun (i : Iic n) ↦ R i ω)) ω +lemma iIndep_actions (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : + iIndepFun A P := by + have hA := h'.measurable_A + set PRS_alg := PRS (β := β) μ + rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)] + intro n + sorry -open Filter Topology ENNReal in -lemma ENNReal.tendsto_zero_le {α : Type*} {f g : α → ℝ≥0∞} {ι : Filter α} - (hg : Tendsto g ι (𝓝 0)) (h : f ≤ g) : Tendsto f ι (𝓝 0) := by - refine tendsto_of_tendsto_of_tendsto_of_le_of_le (g := fun _ ↦ 0) tendsto_const_nhds hg ?_ h - intro - simp +variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] + [μ.IsOpenPosMeasure] -variable [IsFiniteMeasure (ℙ : Measure ℝ)] -- False, but assumed now for simplicity - -open Filter Topology ENNReal in -example (h' : IsAlgEnvSeq A R PRS (evalEnv hf.measurable) ((ℙ (Set.univ : Set ℝ))⁻¹ • ℙ)) : - TendstoInMeasure ((ℙ (Set.univ : Set α))⁻¹ • ℙ) (IsAlgEnvSeq.max R) atTop (fun _ ↦ f c) := by - set μ := ((ℙ (Set.univ : Set α))⁻¹ • (ℙ : Measure ℝ)) - rw [tendstoInMeasure_iff_dist] - intro ε₁ hε₁ - rw [Metric.continuous_iff] at hf - specialize hf c ε₁ hε₁ - obtain ⟨δ, hδ, hf⟩ := hf - let h : ℕ → ℝ≥0∞ := fun n ↦ μ {x | δ ≤ dist (IsAlgEnvSeq.argmax A R n x) c} - refine tendsto_zero_le (g := h) ?_ ?_ - · sorry +open Finset Preorder Filter Topology ENNReal in +/-- The probability of sampling points at distance at least ε from a given point goes to zero +as the number of samples goes to infinity. -/ +theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : + ∀ ε, 0 < ε → Tendsto (fun i => P + {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (A j.1 x) a)}) atTop (𝓝 0) := by + set PRS_alg := PRS (β := β) μ + intro ε hε + refine tendsto_zero_le (g := fun n ↦ P (⋂ i ∈ Iic n, {x | ε ≤ dist (A i x) a})) ?_ ?_ + · have inter_prod (n : ℕ) : P (⋂ j ∈ Iic n, {x | ε ≤ dist (A j x) a}) = + ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} := by + refine iIndepSet.meas_biInter ?_ _ + rw [iIndepSet_iff_meas_biInter fun i ↦ ?_] + · intro s + have iIndep_actions := PRS.iIndep_actions μ P A R hf h' + rw [iIndepFun_iff_measure_inter_preimage_eq_mul] at iIndep_actions + have meas_dist : ∀ i ∈ s, MeasurableSet {x | ε ≤ dist x a} := by + intro i hs + measurability + specialize iIndep_actions s meas_dist + simpa only [Set.preimage] using iIndep_actions + · have hAi := h'.measurable_A i + measurability + simp_rw [inter_prod] + have prod_law (n : ℕ) : ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} = + ∏ j ∈ Iic n, μ {x | ε ≤ dist x a} := by + refine prod_congr rfl fun j hj ↦ ?_ + have hlaw (n : ℕ) : HasLaw (A n) μ P := PRS.hasLaw_action μ P A R hf h' n + rw [← (hlaw j).map_eq, P.map_apply] + · simp + · exact h'.measurable_A j + · measurability + simp_rw [prod_law] + simp only [prod_const, Nat.card_Iic] + suffices μ {x | ε ≤ dist x a} < 1 by + refine Tendsto.comp ?_ <| tendsto_add_atTop_nat 1 + exact tendsto_pow_atTop_nhds_zero_of_lt_one this + have compl : {x | ε ≤ dist x a} = {x | dist x a < ε}ᶜ := by + ext a + simp + rw [compl] + rw [measure_compl (by measurability) (by simp), measure_univ] + refine ENNReal.sub_lt_self (by simp) (by simp) ?_ + exact (Metric.measure_ball_pos μ a hε).ne' · intro n - simp only [h] refine measure_mono ?_ - simp only [Set.setOf_subset_setOf] - intro a - by_contra! h'' - specialize hf (IsAlgEnvSeq.argmax A R n a) h''.2 - have := h''.1 - suffices IsAlgEnvSeq.max R n a = f (IsAlgEnvSeq.argmax A R n a) by - rw [← this] at hf - linarith - - sorry + simp only [mem_Iic, Set.subset_iInter_iff, Set.setOf_subset_setOf] + intro i hi ω (hω : ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (A j.1 ω) a)) + unfold Tuple.min at hω + simp_all + +end PRS diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean index b4f8df8b..1a7cb62f 100644 --- a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean +++ b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean @@ -46,7 +46,7 @@ lemma measurable_max {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.ma fun_prop @[fun_prop] -lemma measurable_min_fst {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.min t) := by +lemma measurable_min {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.min t) := by unfold Tuple.min have : Nonempty (Iic n) := inferInstance simp_all only [mem_Iic, nonempty_subtype] diff --git a/LeanBandits/SequentialLearning/EvaluationEnv.lean b/LeanBandits/SequentialLearning/EvaluationEnv.lean index d915d234..d6a63950 100644 --- a/LeanBandits/SequentialLearning/EvaluationEnv.lean +++ b/LeanBandits/SequentialLearning/EvaluationEnv.lean @@ -4,7 +4,7 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanBandits.SequentialLearning.Algorithm +import LeanBandits.SequentialLearning.StationaryEnv /-! # Function evaluation environments @@ -16,9 +16,7 @@ namespace Learning variable {α R : Type*} [MeasurableSpace α] [MeasurableSpace R] -@[simps] -noncomputable def evalEnv {f : α → R} (hf : Measurable f) : Environment α R where - ν0 := Kernel.deterministic f hf - feedback _ := Kernel.deterministic (f ∘ Prod.snd) <| hf.comp measurable_snd +noncomputable def evalEnv {f : α → R} (hf : Measurable f) := + stationaryEnv <| Kernel.deterministic f hf end Learning From 15b3488fdc069d1442864d293349900c2dd9775d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Thu, 2 Apr 2026 17:12:55 +0200 Subject: [PATCH 06/49] cleaning --- LeanBandits/OptimizationAlgorithms/PRS.lean | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index cf296ab9..5d1b61b1 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -119,14 +119,11 @@ theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : · measurability simp_rw [prod_law] simp only [prod_const, Nat.card_Iic] - suffices μ {x | ε ≤ dist x a} < 1 by - refine Tendsto.comp ?_ <| tendsto_add_atTop_nat 1 - exact tendsto_pow_atTop_nhds_zero_of_lt_one this + refine Tendsto.comp (tendsto_pow_atTop_nhds_zero_of_lt_one ?_) (tendsto_add_atTop_nat 1) have compl : {x | ε ≤ dist x a} = {x | dist x a < ε}ᶜ := by ext a simp - rw [compl] - rw [measure_compl (by measurability) (by simp), measure_univ] + rw [compl, measure_compl (by measurability) (by simp), measure_univ] refine ENNReal.sub_lt_self (by simp) (by simp) ?_ exact (Metric.measure_ball_pos μ a hε).ne' · intro n From 2ab18dabff9fafa94814d6916ac4ae1e2d9508db Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Thu, 2 Apr 2026 17:50:04 +0200 Subject: [PATCH 07/49] remove useless instances --- LeanBandits/OptimizationAlgorithms/PRS.lean | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index 5d1b61b1..6cf24e59 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -18,7 +18,7 @@ Implementation of the _Pure Random Search_ algorithm, which samples from the uni distribution on the input space at each iteration. -/ -section +section -- Move this somewhere else lemma hasLaw_of_hasCondDistrib_const {β' Ω' Ω'' : Type*} [MeasurableSpace β'] [MeasurableSpace Ω'] [StandardBorelSpace Ω'] [Nonempty Ω'] @@ -55,8 +55,6 @@ noncomputable def PRS : Algorithm α β where policy _ := Kernel.const _ μ p0 := μ - h_policy _ := ⟨fun _ ↦ inferInstance⟩ - hp0 := inferInstance namespace PRS From 863db30cd2440016826126be2404516ba60e35d6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Thu, 2 Apr 2026 18:25:46 +0200 Subject: [PATCH 08/49] `iIndep_actions` --- LeanBandits/OptimizationAlgorithms/PRS.lean | 25 +++++++++++++-------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index 6cf24e59..512dbf5f 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -46,21 +46,21 @@ lemma ENNReal.tendsto_zero_le {α : Type*} {f g : α → ℝ≥0∞} {ι : Filte end -variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] (μ : Measure α) - [IsProbabilityMeasure μ] +variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] open Set in @[simps] noncomputable -def PRS : Algorithm α β where +def PRS (μ : Measure α) [IsProbabilityMeasure μ] : Algorithm α β where policy _ := Kernel.const _ μ p0 := μ namespace PRS variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace β] [Nonempty β] - {Ω : Type*} [MeasurableSpace Ω] (P : Measure Ω) [IsProbabilityMeasure P] - (A : ℕ → Ω → α) (R : ℕ → Ω → β) {f : α → β} (hf : Measurable f) + {Ω : Type*} [MeasurableSpace Ω] {P : Measure Ω} [IsProbabilityMeasure P] + {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) {μ : Measure α} + [IsProbabilityMeasure μ] lemma hasLaw_action (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) @@ -72,13 +72,20 @@ lemma hasLaw_action obtain ⟨k, rfl⟩ := Nat.exists_eq_succ_of_ne_zero hn exact hasLaw_of_hasCondDistrib_const <| h'.hasCondDistrib_action k +open Finset in lemma iIndep_actions (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : iIndepFun A P := by have hA := h'.measurable_A - set PRS_alg := PRS (β := β) μ rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)] intro n - sorry + have condDistrib_eq := (h'.hasCondDistrib_action n).condDistrib_eq + have law_eq := (hasLaw_action hf h' (n + 1)).map_eq + simp only [PRS_p0, PRS_policy] at law_eq condDistrib_eq + rw [← law_eq, ← indepFun_iff_condDistrib_eq_const ?_ (by fun_prop)] at condDistrib_eq + · have meas_fst : Measurable (fun (f : Iic n → α × β) ↦ (fun i ↦ (f i).1)) := by + fun_prop + exact (condDistrib_eq.comp meas_fst measurable_id).symm + · exact (IsAlgEnvSeq.measurable_hist (h'.measurable_A) (h'.measurable_R) n).aemeasurable variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] [μ.IsOpenPosMeasure] @@ -97,7 +104,7 @@ theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : refine iIndepSet.meas_biInter ?_ _ rw [iIndepSet_iff_meas_biInter fun i ↦ ?_] · intro s - have iIndep_actions := PRS.iIndep_actions μ P A R hf h' + have iIndep_actions := PRS.iIndep_actions hf h' rw [iIndepFun_iff_measure_inter_preimage_eq_mul] at iIndep_actions have meas_dist : ∀ i ∈ s, MeasurableSet {x | ε ≤ dist x a} := by intro i hs @@ -110,7 +117,7 @@ theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : have prod_law (n : ℕ) : ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} = ∏ j ∈ Iic n, μ {x | ε ≤ dist x a} := by refine prod_congr rfl fun j hj ↦ ?_ - have hlaw (n : ℕ) : HasLaw (A n) μ P := PRS.hasLaw_action μ P A R hf h' n + have hlaw (n : ℕ) : HasLaw (A n) μ P := PRS.hasLaw_action hf h' n rw [← (hlaw j).map_eq, P.map_apply] · simp · exact h'.measurable_A j From 1e279eb473accfd9dc165abc37e76ca7f5c9d422 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Thu, 2 Apr 2026 18:27:02 +0200 Subject: [PATCH 09/49] refactor --- LeanBandits/OptimizationAlgorithms/PRS.lean | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index 512dbf5f..dd5f4a32 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -62,9 +62,8 @@ variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace β] [Nonempty {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) {μ : Measure α} [IsProbabilityMeasure μ] -lemma hasLaw_action - (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) - (n : ℕ) : HasLaw (A n) (PRS (β := β) μ).p0 P := by +lemma hasLaw_action (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : + HasLaw (A n) (PRS (β := β) μ).p0 P := by by_cases hn : n = 0 · rw [hn] exact h'.hasLaw_action_zero From 6eccf5141c456fe779440335d6ffe22060a2860e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 3 Apr 2026 01:31:55 +0200 Subject: [PATCH 10/49] ENNReal lemma --- LeanBandits/ForMathlib/ENNReal.lean | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 LeanBandits/ForMathlib/ENNReal.lean diff --git a/LeanBandits/ForMathlib/ENNReal.lean b/LeanBandits/ForMathlib/ENNReal.lean new file mode 100644 index 00000000..05b3e00d --- /dev/null +++ b/LeanBandits/ForMathlib/ENNReal.lean @@ -0,0 +1,17 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ + +import Mathlib.Analysis.Normed.Group.Basic + +open ENNReal Filter + +open scoped Topology + +lemma ENNReal.tendsto_zero_le {α : Type*} {f g : α → ℝ≥0∞} {ι : Filter α} + (hg : Tendsto g ι (𝓝 0)) (h : f ≤ g) : Tendsto f ι (𝓝 0) := by + refine tendsto_of_tendsto_of_tendsto_of_le_of_le (g := fun _ ↦ 0) tendsto_const_nhds hg ?_ h + intro + simp From 5a377ec0b8d9f60303d0fcf598d80e7ccc696b82 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 3 Apr 2026 01:32:03 +0200 Subject: [PATCH 11/49] `hasLaw_of_hasCondDistrib_const` --- LeanBandits/ForMathlib/HasCondDistrib.lean | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/LeanBandits/ForMathlib/HasCondDistrib.lean b/LeanBandits/ForMathlib/HasCondDistrib.lean index a151124e..227387ae 100644 --- a/LeanBandits/ForMathlib/HasCondDistrib.lean +++ b/LeanBandits/ForMathlib/HasCondDistrib.lean @@ -216,4 +216,17 @@ lemma HasCondDistrib.prod [IsFiniteMeasure μ] [IsFiniteKernel κ] AEMeasurable.map_map_of_aemeasurable (by fun_prop) (by fun_prop)] rfl +lemma hasLaw_of_hasCondDistrib_const [IsProbabilityMeasure μ] {Q : Measure Ω} [SFinite Q] + (h : HasCondDistrib Y X (Kernel.const _ Q) μ) : HasLaw Y Q μ := by + obtain ⟨hY, hX, h⟩ := h + refine ⟨hY, ?_⟩ + have h_snd : (μ.map (fun ω => (X ω, Y ω))).snd = Q := by + have h_map : μ.map (fun ω => (X ω, Y ω)) = (μ.map X) ⊗ₘ (Kernel.const _ Q) := + have h_map : μ.map (fun ω => (X ω, Y ω)) = (μ.map X) ⊗ₘ (condDistrib Y X μ) := + (compProd_map_condDistrib hY).symm + h_map.trans (Measure.compProd_congr h) + rw [h_map, MeasureTheory.Measure.snd_compProd] + simp [MeasureTheory.Measure.map_apply_of_aemeasurable hX] + rwa [Measure.snd_map_prodMk₀ hX] at h_snd + end ProbabilityTheory From 288ae492726b80b9da151b6df1d691318508095f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 3 Apr 2026 01:33:06 +0200 Subject: [PATCH 12/49] golf and `tendsto_{max, min}` --- LeanBandits/OptimizationAlgorithms/PRS.lean | 156 +++++++++++------- .../OptimizationAlgorithms/Utils/Tuple.lean | 56 +++++-- .../SequentialLearning/EvaluationEnv.lean | 36 ++++ 3 files changed, 181 insertions(+), 67 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index dd5f4a32..07114deb 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -4,48 +4,21 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanBandits.SequentialLearning.Algorithm +import LeanBandits.ForMathlib.ENNReal +import LeanBandits.ForMathlib.IndepFun import LeanBandits.OptimizationAlgorithms.Utils.Tuple import LeanBandits.SequentialLearning.EvaluationEnv -import LeanBandits.ForMathlib.IndepFun -import Mathlib -open MeasureTheory ProbabilityTheory Learning +open MeasureTheory ProbabilityTheory Learning Finset ENNReal Filter + +open scoped Topology /-! # PRS: Pure Random Search -Implementation of the _Pure Random Search_ algorithm, which samples from the uniform -distribution on the input space at each iteration. +Implementation of the _Pure Random Search_ algorithm, which samples from a fixe probability measure +at each iteration. -/ -section -- Move this somewhere else - -lemma hasLaw_of_hasCondDistrib_const {β' Ω' Ω'' : Type*} - [MeasurableSpace β'] [MeasurableSpace Ω'] [StandardBorelSpace Ω'] [Nonempty Ω'] - [MeasurableSpace Ω''] - {X : Ω'' → β'} {Y : Ω'' → Ω'} {μ : Measure Ω'} - {Q : Measure Ω''} [IsProbabilityMeasure Q] [SFinite μ] - (h : HasCondDistrib Y X (Kernel.const _ μ) Q) : HasLaw Y μ Q := by - obtain ⟨hY, hX, h⟩ := h - refine ⟨hY, ?_⟩ - have h_snd : (Q.map (fun ω => (X ω, Y ω))).snd = μ := by - have h_map : Q.map (fun ω => (X ω, Y ω)) = (Q.map X) ⊗ₘ (Kernel.const _ μ) := - have h_map : Q.map (fun ω => (X ω, Y ω)) = (Q.map X) ⊗ₘ (condDistrib Y X Q) := - (compProd_map_condDistrib hY).symm - h_map.trans (Measure.compProd_congr h) - rw [h_map, MeasureTheory.Measure.snd_compProd] - simp [MeasureTheory.Measure.map_apply_of_aemeasurable hX] - rwa [Measure.snd_map_prodMk₀ hX] at h_snd - -open ENNReal Filter Topology in -lemma ENNReal.tendsto_zero_le {α : Type*} {f g : α → ℝ≥0∞} {ι : Filter α} - (hg : Tendsto g ι (𝓝 0)) (h : f ≤ g) : Tendsto f ι (𝓝 0) := by - refine tendsto_of_tendsto_of_tendsto_of_le_of_le (g := fun _ ↦ 0) tendsto_const_nhds hg ?_ h - intro - simp - -end - variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] open Set in @@ -57,42 +30,46 @@ def PRS (μ : Measure α) [IsProbabilityMeasure μ] : Algorithm α β where namespace PRS +section + variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace β] [Nonempty β] {Ω : Type*} [MeasurableSpace Ω] {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) {μ : Measure α} - [IsProbabilityMeasure μ] + [IsProbabilityMeasure μ] (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) -lemma hasLaw_action (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : - HasLaw (A n) (PRS (β := β) μ).p0 P := by +lemma hasLaw_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : HasLaw (A n) μ P := by by_cases hn : n = 0 · rw [hn] - exact h'.hasLaw_action_zero + exact h.hasLaw_action_zero · push_neg at hn obtain ⟨k, rfl⟩ := Nat.exists_eq_succ_of_ne_zero hn - exact hasLaw_of_hasCondDistrib_const <| h'.hasCondDistrib_action k + exact hasLaw_of_hasCondDistrib_const <| h.hasCondDistrib_action k + +lemma hasLaw_rewards (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : + HasLaw (R n) (μ.map f) P := by + refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_eq_eval_action hf h n) + have hA := h.measurable_A n + refine ⟨by fun_prop, ?_⟩ + rw [← Measure.map_map hf hA, (hasLaw_actions hf h n).map_eq] -open Finset in -lemma iIndep_actions (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : +lemma iIndep_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : iIndepFun A P := by - have hA := h'.measurable_A + have hA := h.measurable_A rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)] intro n - have condDistrib_eq := (h'.hasCondDistrib_action n).condDistrib_eq - have law_eq := (hasLaw_action hf h' (n + 1)).map_eq - simp only [PRS_p0, PRS_policy] at law_eq condDistrib_eq + have condDistrib_eq := (h.hasCondDistrib_action n).condDistrib_eq + have law_eq := (hasLaw_actions hf h (n + 1)).map_eq + simp only [PRS_policy] at condDistrib_eq rw [← law_eq, ← indepFun_iff_condDistrib_eq_const ?_ (by fun_prop)] at condDistrib_eq · have meas_fst : Measurable (fun (f : Iic n → α × β) ↦ (fun i ↦ (f i).1)) := by fun_prop exact (condDistrib_eq.comp meas_fst measurable_id).symm - · exact (IsAlgEnvSeq.measurable_hist (h'.measurable_A) (h'.measurable_R) n).aemeasurable + · exact (IsAlgEnvSeq.measurable_hist (h.measurable_A) (h.measurable_R) n).aemeasurable variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] [μ.IsOpenPosMeasure] -open Finset Preorder Filter Topology ENNReal in -/-- The probability of sampling points at distance at least ε from a given point goes to zero -as the number of samples goes to infinity. -/ -theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : +theorem tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (A j.1 x) a)}) atTop (𝓝 0) := by set PRS_alg := PRS (β := β) μ @@ -103,27 +80,27 @@ theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : refine iIndepSet.meas_biInter ?_ _ rw [iIndepSet_iff_meas_biInter fun i ↦ ?_] · intro s - have iIndep_actions := PRS.iIndep_actions hf h' + have iIndep_actions := PRS.iIndep_actions hf h rw [iIndepFun_iff_measure_inter_preimage_eq_mul] at iIndep_actions have meas_dist : ∀ i ∈ s, MeasurableSet {x | ε ≤ dist x a} := by intro i hs measurability specialize iIndep_actions s meas_dist simpa only [Set.preimage] using iIndep_actions - · have hAi := h'.measurable_A i + · have hAi := h.measurable_A i measurability simp_rw [inter_prod] have prod_law (n : ℕ) : ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} = ∏ j ∈ Iic n, μ {x | ε ≤ dist x a} := by refine prod_congr rfl fun j hj ↦ ?_ - have hlaw (n : ℕ) : HasLaw (A n) μ P := PRS.hasLaw_action hf h' n + have hlaw (n : ℕ) : HasLaw (A n) μ P := PRS.hasLaw_actions hf h n rw [← (hlaw j).map_eq, P.map_apply] · simp - · exact h'.measurable_A j + · exact h.measurable_A j · measurability simp_rw [prod_law] simp only [prod_const, Nat.card_Iic] - refine Tendsto.comp (tendsto_pow_atTop_nhds_zero_of_lt_one ?_) (tendsto_add_atTop_nat 1) + refine tendsto_pow_atTop_nhds_zero_of_lt_one ?_ |> Tendsto.comp <| tendsto_add_atTop_nat 1 have compl : {x | ε ≤ dist x a} = {x | dist x a < ε}ᶜ := by ext a simp @@ -134,7 +111,72 @@ theorem convergence (h' : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : refine measure_mono ?_ simp only [mem_Iic, Set.subset_iInter_iff, Set.setOf_subset_setOf] intro i hi ω (hω : ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (A j.1 ω) a)) - unfold Tuple.min at hω - simp_all + simp_all only [univ_eq_attach, le_inf'_iff, mem_attach, forall_const, Subtype.forall, mem_Iic] + +end + +section Real + +variable [StandardBorelSpace α] [Nonempty α] [PseudoMetricSpace α] [OpensMeasurableSpace α] + [SecondCountableTopology α] {Ω : Type*} [MeasurableSpace Ω] {P : Measure Ω} + [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R : ℕ → Ω → ℝ} {f : α → ℝ} (hfc : Continuous f) + {μ : Measure α} [IsProbabilityMeasure μ] [μ.IsOpenPosMeasure] + (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) {a : α} + +lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) + (hf_min : ∀ x, f a ≤ f x) : + TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by + rw [tendstoInMeasure_iff_dist] + intro ε hε + have hf := hfc.measurable + rw [Metric.continuous_iff] at hfc + obtain ⟨δ, hδ, hfc⟩ := hfc a ε hε + have (n : ℕ) : P {x | ε ≤ dist (Tuple.min fun (i : Iic n) ↦ R i x) (f a)} = + P {x | ε ≤ dist (Tuple.min fun (i : Iic n) ↦ f (A i x)) (f a)} := by + refine measure_congr ?_ + filter_upwards [IsAlgEnvSeq.reward_eq_evals_actions_comp hf h Tuple.min] with ω hω + simp only [eq_iff_iff] + change ε ≤ dist (Tuple.min fun (i : Iic n) ↦ R (↑i) ω) (f a) ↔ + ε ≤ dist (Tuple.min fun (i : Iic n) ↦ f (A ↑i ω)) (f a) + rw [hω] + simp_rw [this] + refine tendsto_any hf h a δ hδ |> tendsto_zero_le <| ?_ + intro n + refine measure_mono ?_ + simp only [Set.setOf_subset_setOf] + intro ω hω + rw [← Tuple.argmin_spec] + set j := Tuple.argmin (fun (i : Iic n) ↦ dist (A i ω) a) + have : dist (Tuple.min fun (i : Iic n) ↦ f (A i ω)) (f a) ≤ dist (f (A j ω)) (f a) := by + rw [← Tuple.argmin_spec] + set k := Tuple.argmin (fun (i : Iic n) ↦ f (A i ω)) + have := hf_min (A k ω) + have : f (A k ω) ≤ f (A j ω) := + Tuple.argmin_le (fun (i : Iic n) ↦ f (A i ω)) j + simp [Real.dist_eq] + grind + have := hω.trans this + by_contra! h_contra + specialize hfc (A j ω) h_contra + linarith + +lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) + (hf_max : ∀ x, f x ≤ f a) : + TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by + have hmf_min (x : α) : -f a ≤ -f x := by + specialize hf_max x + linarith + have := tendsto_min (continuous_neg_iff.mpr hfc) (IsAlgEnvSeq.neg hfc.measurable h) hmf_min + rw [tendstoInMeasure_iff_dist] at this ⊢ + intro ε hε + specialize this ε hε + have dist_neg (n : ℕ) : {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R i x) (f a)} = + {x | ε ≤ dist (-(Tuple.max fun (i : Iic n) ↦ R i x)) (-f a)} := by + simp [dist_neg_neg] + simp_rw [dist_neg] + convert this with n ω + exact Tuple.neg_max_eq_min_neg _ + +end Real end PRS diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean index 1a7cb62f..90b3a5db 100644 --- a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean +++ b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean @@ -12,15 +12,28 @@ namespace Tuple variable {ι α : Type*} [LinearOrder α] [Fintype ι] [Nonempty ι] (f : ι → α) -def max : α := univ.sup' (by simp) f +abbrev max : α := univ.sup' (by simp) f -def min : α := univ.inf' (by simp) f +abbrev min : α := univ.inf' (by simp) f + +lemma le_max (x : ι) : f x ≤ max f := by + simp only [le_sup'_iff, mem_univ, true_and] + exact ⟨x, le_refl _⟩ + +lemma min_le (x : ι) : min f ≤ f x := by + simp only [inf'_le_iff, mem_univ, true_and] + exact ⟨x, le_refl _⟩ instance {n : ℕ} : Nonempty (Iic n) := Nonempty.intro ⟨0, insert_eq_self.mp rfl⟩ -lemma exists_argmax {n : ℕ} (u : Iic n → α) : ∃ i, u i = max u := by +/-- TODO: generalize -/ +lemma neg_max_eq_min_neg {n : ℕ} (u : Iic n → ℝ) : -(max u) = min (-u) := by + sorry + +variable {n : ℕ} (u : Iic n → α) + +lemma exists_argmax : ∃ i, u i = max u := by have : Nonempty (Iic n) := inferInstance - unfold max let A := u '' Set.univ suffices h : univ.sup' (by simp) u ∈ A by obtain ⟨x, -, h⟩ := h @@ -31,23 +44,46 @@ lemma exists_argmax {n : ℕ} (u : Iic n → α) : ∃ i, u i = max u := by | inr r => simp_all · simp [A] -noncomputable def argmax {n : ℕ} (u : Iic n → α) := (exists_argmax u).choose +noncomputable def argmax := (exists_argmax u).choose -lemma argmax_spec {n : ℕ} (u : Iic n → α) : u (argmax u) = max u := +lemma argmax_spec : u (argmax u) = max u := (exists_argmax u).choose_spec +lemma le_argmax (x : Iic n) : u x ≤ u (argmax u) := by + rw [argmax_spec u] + exact le_max u x + +lemma exists_argmin : ∃ i, u i = min u := by + have : Nonempty (Iic n) := inferInstance + let A := u '' Set.univ + suffices h : univ.inf' (by simp) u ∈ A by + obtain ⟨x, -, h⟩ := h + exact ⟨x, h⟩ + refine inf'_mem A (fun x hx y hy ↦ ?_) _ _ u fun i _ ↦ ?_ + · cases min_choice x y with + | inl l => simp_all + | inr r => simp_all + · simp [A] + +noncomputable def argmin := (exists_argmin u).choose + +lemma argmin_spec : u (argmin u) = min u := + (exists_argmin u).choose_spec + +lemma argmin_le (x : Iic n) : u (argmin u) ≤ u x := by + rw [argmin_spec u] + exact min_le u x + variable [MeasurableSpace α] @[fun_prop] -lemma measurable_max {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.max t) := by - unfold Tuple.max +lemma measurable_max : Measurable (fun (t : Iic n → ℝ) => Tuple.max t) := by have : Nonempty (Iic n) := inferInstance simp_all only [mem_Iic, nonempty_subtype] fun_prop @[fun_prop] -lemma measurable_min {n : ℕ} : Measurable (fun (t : Iic n → ℝ) => Tuple.min t) := by - unfold Tuple.min +lemma measurable_min : Measurable (fun (t : Iic n → ℝ) => Tuple.min t) := by have : Nonempty (Iic n) := inferInstance simp_all only [mem_Iic, nonempty_subtype] fun_prop diff --git a/LeanBandits/SequentialLearning/EvaluationEnv.lean b/LeanBandits/SequentialLearning/EvaluationEnv.lean index d6a63950..1c2876d0 100644 --- a/LeanBandits/SequentialLearning/EvaluationEnv.lean +++ b/LeanBandits/SequentialLearning/EvaluationEnv.lean @@ -5,6 +5,7 @@ Authors: Gaëtan Serré -/ import LeanBandits.SequentialLearning.StationaryEnv +import LeanBandits.ForMathlib.CondDistrib /-! # Function evaluation environments @@ -19,4 +20,39 @@ variable {α R : Type*} [MeasurableSpace α] [MeasurableSpace R] noncomputable def evalEnv {f : α → R} (hf : Measurable f) := stationaryEnv <| Kernel.deterministic f hf +namespace IsAlgEnvSeq + +variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] + {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {f : α → R} (hf : Measurable f) + {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} + +lemma hascondDistrib_reward_evalEnv (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : + HasCondDistrib (R' n) (A n) (Kernel.deterministic f hf) P := + have hRn := h.measurable_R n + have hAn := h.measurable_A n + ⟨hRn.aemeasurable, hAn.aemeasurable, h.condDistrib_reward_stationaryEnv n⟩ + +lemma reward_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : + R' n =ᵐ[P] f ∘ A n := + ae_eq_of_condDistrib_eq_deterministic hf (h.measurable_A n).aemeasurable + (h.measurable_R n).aemeasurable (hascondDistrib_reward_evalEnv hf h n).condDistrib_eq + +lemma reward_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : + ∀ᵐ ω ∂P, ∀ n, R' n ω = f (A n ω) := by + rw [ae_all_iff] + intro n + exact reward_eq_eval_action hf h n + +open Finset in +lemma reward_eq_evals_actions_comp (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} + (g : (Iic n → R) → R) : ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by + filter_upwards [reward_eq_evals_actions hf h] with ω hω + simp_rw [hω] + +lemma neg [Neg R] [MeasurableNeg R] (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : + IsAlgEnvSeq A (-R') alg (evalEnv <| Measurable.neg hf) P := by + sorry + +end IsAlgEnvSeq + end Learning From 91ad3db89d00e5eff6dcd8d1c8b3db02fd540192 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 3 Apr 2026 11:30:13 +0200 Subject: [PATCH 13/49] Tuple --- LeanBandits/OptimizationAlgorithms/PRS.lean | 10 ++-- .../OptimizationAlgorithms/Utils/Tuple.lean | 60 +++++++++++-------- 2 files changed, 40 insertions(+), 30 deletions(-) diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanBandits/OptimizationAlgorithms/PRS.lean index 07114deb..2f34977b 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanBandits/OptimizationAlgorithms/PRS.lean @@ -41,7 +41,7 @@ lemma hasLaw_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : H by_cases hn : n = 0 · rw [hn] exact h.hasLaw_action_zero - · push_neg at hn + · push Not at hn obtain ⟨k, rfl⟩ := Nat.exists_eq_succ_of_ne_zero hn exact hasLaw_of_hasCondDistrib_const <| h.hasCondDistrib_action k @@ -166,15 +166,15 @@ lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) have hmf_min (x : α) : -f a ≤ -f x := by specialize hf_max x linarith - have := tendsto_min (continuous_neg_iff.mpr hfc) (IsAlgEnvSeq.neg hfc.measurable h) hmf_min - rw [tendstoInMeasure_iff_dist] at this ⊢ + have h' := tendsto_min (continuous_neg_iff.mpr hfc) (IsAlgEnvSeq.neg hfc.measurable h) hmf_min + rw [tendstoInMeasure_iff_dist] at h' ⊢ intro ε hε - specialize this ε hε + specialize h' ε hε have dist_neg (n : ℕ) : {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R i x) (f a)} = {x | ε ≤ dist (-(Tuple.max fun (i : Iic n) ↦ R i x)) (-f a)} := by simp [dist_neg_neg] simp_rw [dist_neg] - convert this with n ω + convert h' with n ω exact Tuple.neg_max_eq_min_neg _ end Real diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean index 90b3a5db..89cd4dcb 100644 --- a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean +++ b/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean @@ -4,7 +4,9 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import Mathlib +import Mathlib.Analysis.Normed.Order.Lattice +import Mathlib.CategoryTheory.Countable +import Mathlib.MeasureTheory.Constructions.BorelSpace.Basic open Finset @@ -18,31 +20,26 @@ abbrev min : α := univ.inf' (by simp) f lemma le_max (x : ι) : f x ≤ max f := by simp only [le_sup'_iff, mem_univ, true_and] - exact ⟨x, le_refl _⟩ + exact ⟨x, le_rfl⟩ lemma min_le (x : ι) : min f ≤ f x := by simp only [inf'_le_iff, mem_univ, true_and] - exact ⟨x, le_refl _⟩ + exact ⟨x, le_rfl⟩ instance {n : ℕ} : Nonempty (Iic n) := Nonempty.intro ⟨0, insert_eq_self.mp rfl⟩ -/-- TODO: generalize -/ -lemma neg_max_eq_min_neg {n : ℕ} (u : Iic n → ℝ) : -(max u) = min (-u) := by - sorry - variable {n : ℕ} (u : Iic n → α) lemma exists_argmax : ∃ i, u i = max u := by have : Nonempty (Iic n) := inferInstance - let A := u '' Set.univ - suffices h : univ.sup' (by simp) u ∈ A by - obtain ⟨x, -, h⟩ := h - exact ⟨x, h⟩ - refine sup'_mem A (fun x hx y hy ↦ ?_) _ _ u fun i _ ↦ ?_ - · cases max_choice x y with - | inl l => simp_all - | inr r => simp_all - · simp [A] + obtain ⟨i, -, hi⟩ := Finset.exists_max_image Finset.univ u (by simp) + refine ⟨i, ?_⟩ + refine le_antisymm ?_ ?_ + · simp only [le_sup'_iff, univ_eq_attach, mem_attach, true_and, Subtype.exists, mem_Iic] + exact ⟨i, by grind, le_rfl⟩ + · simp only [sup'_le_iff, univ_eq_attach, mem_attach, forall_const, Subtype.forall, mem_Iic] + intro j hj + exact hi ⟨j, mem_Iic.mpr hj⟩ (by simp) noncomputable def argmax := (exists_argmax u).choose @@ -55,15 +52,14 @@ lemma le_argmax (x : Iic n) : u x ≤ u (argmax u) := by lemma exists_argmin : ∃ i, u i = min u := by have : Nonempty (Iic n) := inferInstance - let A := u '' Set.univ - suffices h : univ.inf' (by simp) u ∈ A by - obtain ⟨x, -, h⟩ := h - exact ⟨x, h⟩ - refine inf'_mem A (fun x hx y hy ↦ ?_) _ _ u fun i _ ↦ ?_ - · cases min_choice x y with - | inl l => simp_all - | inr r => simp_all - · simp [A] + obtain ⟨i, -, hi⟩ := Finset.exists_min_image Finset.univ u (by simp) + refine ⟨i, ?_⟩ + refine le_antisymm ?_ ?_ + · simp only [le_inf'_iff, univ_eq_attach, mem_attach, forall_const, Subtype.forall, mem_Iic] + intro j hj + exact hi ⟨j, mem_Iic.mpr hj⟩ (by simp) + · simp only [inf'_le_iff, univ_eq_attach, mem_attach, true_and, Subtype.exists, mem_Iic] + exact ⟨i, by grind, le_rfl⟩ noncomputable def argmin := (exists_argmin u).choose @@ -74,6 +70,20 @@ lemma argmin_le (x : Iic n) : u (argmin u) ≤ u x := by rw [argmin_spec u] exact min_le u x +lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] {n : ℕ} (u : Iic n → α) : + -(max u) = min (-u) := by + simp only [max, univ_eq_attach, min, Pi.neg_apply] + refine le_antisymm ?_ ?_ + · simp only [le_inf'_iff, mem_attach, neg_le_neg_iff, le_sup'_iff, true_and, Subtype.exists, + mem_Iic, forall_const, Subtype.forall] + intro i hi + exact ⟨i, hi, le_rfl⟩ + · simp only [inf'_le_iff, mem_attach, neg_le_neg_iff, sup'_le_iff, forall_const, Subtype.forall, + mem_Iic, true_and, Subtype.exists] + refine ⟨argmax u, by grind, ?_⟩ + intro i hi + exact le_argmax u ⟨i, mem_Iic.mpr hi⟩ + variable [MeasurableSpace α] @[fun_prop] From 588f87de32aa36fc16ea0679cc4b3caec917ee69 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 3 Apr 2026 17:30:33 +0200 Subject: [PATCH 14/49] some progress on `neg` --- .../SequentialLearning/EvaluationEnv.lean | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/LeanBandits/SequentialLearning/EvaluationEnv.lean b/LeanBandits/SequentialLearning/EvaluationEnv.lean index 1c2876d0..2adaf3e7 100644 --- a/LeanBandits/SequentialLearning/EvaluationEnv.lean +++ b/LeanBandits/SequentialLearning/EvaluationEnv.lean @@ -50,8 +50,23 @@ lemma reward_eq_evals_actions_comp (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n simp_rw [hω] lemma neg [Neg R] [MeasurableNeg R] (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : - IsAlgEnvSeq A (-R') alg (evalEnv <| Measurable.neg hf) P := by - sorry + IsAlgEnvSeq A (-R') alg (evalEnv <| Measurable.neg hf) P where + measurable_A n := h.measurable_A n + measurable_R n := Measurable.neg <| h.measurable_R n + hasLaw_action_zero := h.hasLaw_action_zero + hasCondDistrib_reward_zero := by + have hA := h.measurable_A 0 + have hR := h.measurable_R 0 + refine ⟨(Measurable.neg <| h.measurable_R 0).aemeasurable, by fun_prop, ?_⟩ + have : (- R') 0 =ᶠ[ae P] (-f) ∘ A 0 := by + have := reward_eq_eval_action hf h 0 + filter_upwards [this] with ω hω + simp [hω] + rw [condDistrib_congr this (ae_eq_rfl)] + filter_upwards [condDistrib_comp_self (f := (-f)) (A 0) (Measurable.neg hf)] with ω hω + simpa using hω + hasCondDistrib_action := by sorry + hasCondDistrib_reward := by sorry end IsAlgEnvSeq From 7e36a1bea9e4f6b6b814ee94527646eba9719ac2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 19:34:56 +0200 Subject: [PATCH 15/49] remove `neg` --- .../SequentialLearning/EvaluationEnv.lean | 19 ----------- lake-manifest.json | 33 ++++++++++--------- lean-toolchain | 2 +- 3 files changed, 18 insertions(+), 36 deletions(-) diff --git a/LeanBandits/SequentialLearning/EvaluationEnv.lean b/LeanBandits/SequentialLearning/EvaluationEnv.lean index 2adaf3e7..536fe0da 100644 --- a/LeanBandits/SequentialLearning/EvaluationEnv.lean +++ b/LeanBandits/SequentialLearning/EvaluationEnv.lean @@ -49,25 +49,6 @@ lemma reward_eq_evals_actions_comp (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n filter_upwards [reward_eq_evals_actions hf h] with ω hω simp_rw [hω] -lemma neg [Neg R] [MeasurableNeg R] (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : - IsAlgEnvSeq A (-R') alg (evalEnv <| Measurable.neg hf) P where - measurable_A n := h.measurable_A n - measurable_R n := Measurable.neg <| h.measurable_R n - hasLaw_action_zero := h.hasLaw_action_zero - hasCondDistrib_reward_zero := by - have hA := h.measurable_A 0 - have hR := h.measurable_R 0 - refine ⟨(Measurable.neg <| h.measurable_R 0).aemeasurable, by fun_prop, ?_⟩ - have : (- R') 0 =ᶠ[ae P] (-f) ∘ A 0 := by - have := reward_eq_eval_action hf h 0 - filter_upwards [this] with ω hω - simp [hω] - rw [condDistrib_congr this (ae_eq_rfl)] - filter_upwards [condDistrib_comp_self (f := (-f)) (A 0) (Measurable.neg hf)] with ω hω - simpa using hω - hasCondDistrib_action := by sorry - hasCondDistrib_reward := by sorry - end IsAlgEnvSeq end Learning diff --git a/lake-manifest.json b/lake-manifest.json index eab658f0..6d2443be 100644 --- a/lake-manifest.json +++ b/lake-manifest.json @@ -1,4 +1,4 @@ -{"version": "1.1.0", +{"version": "1.2.0", "packagesDir": ".lake/packages", "packages": [{"url": "https://github.com/PatrickMassot/checkdecls.git", @@ -15,7 +15,7 @@ "type": "git", "subDir": null, "scope": "", - "rev": "62a82d6d0a3d3c54ca96bf00a861c623f3e0be44", + "rev": "c60f51652501f2f3bebdc2db3eb75a9268bc9879", "name": "mathlib", "manifestFile": "lake-manifest.json", "inputRev": null, @@ -25,7 +25,7 @@ "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "7311586e1a56af887b1081d05e80c11b6c41d212", + "rev": "264309b5c0c10e569025a53ab6440a45c03133e4", "name": "plausible", "manifestFile": "lake-manifest.json", "inputRev": "main", @@ -35,7 +35,7 @@ "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "5ce7f0a355f522a952a3d678d696bd563bb4fd28", + "rev": "c5d5b8fe6e5158def25cd28eb94e4141ad97c843", "name": "LeanSearchClient", "manifestFile": "lake-manifest.json", "inputRev": "main", @@ -45,7 +45,7 @@ "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "875ad9d88ed684e39c16bdea260e6ecfa15afd60", + "rev": "4411c5f89c797401c609b3a946c8874569e69731", "name": "importGraph", "manifestFile": "lake-manifest.json", "inputRev": "main", @@ -55,51 +55,52 @@ "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "6d65c6e0a25b8a52c13c3adeb63ecde3bfbb6294", + "rev": "82d457fb3bdd9efadbae06608ff337d689efdddf", "name": "proofwidgets", "manifestFile": "lake-manifest.json", - "inputRev": "v0.0.86", + "inputRev": "v0.0.97", "inherited": true, "configFile": "lakefile.lean"}, {"url": "https://github.com/leanprover-community/aesop", "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "f08e838d4f9aea519f3cde06260cfb686fd4bab0", + "rev": "f74c7555aaa94eadd7b7bff9170f7983f92aac21", "name": "aesop", "manifestFile": "lake-manifest.json", - "inputRev": "master", + "inputRev": "v4.30.0-rc1", "inherited": true, "configFile": "lakefile.toml"}, {"url": "https://github.com/leanprover-community/quote4", "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "23324752757bf28124a518ec284044c8db79fee5", + "rev": "7aa86cb20b8458748dc24d55dab2d7ea01161057", "name": "Qq", "manifestFile": "lake-manifest.json", - "inputRev": "master", + "inputRev": "v4.30.0-rc1", "inherited": true, "configFile": "lakefile.toml"}, {"url": "https://github.com/leanprover-community/batteries", "type": "git", "subDir": null, "scope": "leanprover-community", - "rev": "ab9f3956f91980e61bea324c0cf1e9e7d9c8518b", + "rev": "bf597c77bf9b8e66720d724928207f5911533113", "name": "batteries", "manifestFile": "lake-manifest.json", - "inputRev": "main", + "inputRev": "v4.30.0-rc1", "inherited": true, "configFile": "lakefile.toml"}, {"url": "https://github.com/leanprover/lean4-cli", "type": "git", "subDir": null, "scope": "leanprover", - "rev": "28e0856d4424863a85b18f38868c5420c55f9bae", + "rev": "f7d0ca7c926cdde0562af20394dd25d028b839a5", "name": "Cli", "manifestFile": "lake-manifest.json", - "inputRev": "v4.28.0-rc1", + "inputRev": "v4.30.0-rc1", "inherited": true, "configFile": "lakefile.toml"}], "name": "LeanBandits", - "lakeDir": ".lake"} + "lakeDir": ".lake", + "fixedToolchain": false} diff --git a/lean-toolchain b/lean-toolchain index 3e9b4e15..79ac861d 100644 --- a/lean-toolchain +++ b/lean-toolchain @@ -1 +1 @@ -leanprover/lean4:v4.28.0-rc1 \ No newline at end of file +leanprover/lean4:v4.30.0-rc1 \ No newline at end of file From eb6a9a8a8d5a16aa66c6ae20629a70936d22dd1f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 19:41:10 +0200 Subject: [PATCH 16/49] Add opti algos --- LeanBandits.lean | 32 ------------------- .../OptimizationAlgorithms/LIPO.lean | 8 ++--- .../OptimizationAlgorithms/PRS.lean | 8 ++--- .../OptimizationAlgorithms/RankOpt.lean | 8 ++--- .../Utils/EuclideanSpace.lean | 0 .../OptimizationAlgorithms/Utils/Tuple.lean | 0 .../OptimizationAlgorithms/Utils/Uniform.lean | 0 .../SequentialLearning/EvaluationEnv.lean | 4 +-- 8 files changed, 14 insertions(+), 46 deletions(-) delete mode 100644 LeanBandits.lean rename {LeanBandits => LeanMachineLearning}/OptimizationAlgorithms/LIPO.lean (94%) rename {LeanBandits => LeanMachineLearning}/OptimizationAlgorithms/PRS.lean (97%) rename {LeanBandits => LeanMachineLearning}/OptimizationAlgorithms/RankOpt.lean (97%) rename {LeanBandits => LeanMachineLearning}/OptimizationAlgorithms/Utils/EuclideanSpace.lean (100%) rename {LeanBandits => LeanMachineLearning}/OptimizationAlgorithms/Utils/Tuple.lean (100%) rename {LeanBandits => LeanMachineLearning}/OptimizationAlgorithms/Utils/Uniform.lean (100%) diff --git a/LeanBandits.lean b/LeanBandits.lean deleted file mode 100644 index a0694fad..00000000 --- a/LeanBandits.lean +++ /dev/null @@ -1,32 +0,0 @@ -import LeanBandits.Bandit.Bandit -import LeanBandits.Bandit.Regret -import LeanBandits.Bandit.RewardByCountMeasure -import LeanBandits.Bandit.SumRewards -import LeanBandits.BanditAlgorithms.AuxSums -import LeanBandits.BanditAlgorithms.ETC -import LeanBandits.BanditAlgorithms.RoundRobin -import LeanBandits.BanditAlgorithms.UCB -import LeanBandits.ForMathlib.CondDistrib -import LeanBandits.ForMathlib.CondIndepFun -import LeanBandits.ForMathlib.HasCondDistrib -import LeanBandits.ForMathlib.IndepFun -import LeanBandits.ForMathlib.IndepInfinitePi -import LeanBandits.ForMathlib.Integrable -import LeanBandits.ForMathlib.KernelRepresentation -import LeanBandits.ForMathlib.KernelSub -import LeanBandits.ForMathlib.Measurable -import LeanBandits.ForMathlib.MeasurableArgMax -import LeanBandits.ForMathlib.StandardBorel -import LeanBandits.ForMathlib.SubGaussian -import LeanBandits.ForMathlib.Traj -import LeanBandits.SequentialLearning.Algorithm -import LeanBandits.SequentialLearning.Deterministic -import LeanBandits.SequentialLearning.FiniteActions -import LeanBandits.SequentialLearning.IonescuTulceaSpace -import LeanBandits.SequentialLearning.StationaryEnv -import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace -import LeanBandits.OptimizationAlgorithms.Utils.Tuple -import LeanBandits.OptimizationAlgorithms.Utils.Uniform -import LeanBandits.OptimizationAlgorithms.LIPO -import LeanBandits.OptimizationAlgorithms.PRS -import LeanBandits.OptimizationAlgorithms.RankOpt diff --git a/LeanBandits/OptimizationAlgorithms/LIPO.lean b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean similarity index 94% rename from LeanBandits/OptimizationAlgorithms/LIPO.lean rename to LeanMachineLearning/OptimizationAlgorithms/LIPO.lean index 146c2e5f..6f4c4435 100644 --- a/LeanBandits/OptimizationAlgorithms/LIPO.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean @@ -4,10 +4,10 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanBandits.OptimizationAlgorithms.Utils.Tuple -import LeanBandits.OptimizationAlgorithms.Utils.Uniform -import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace -import LeanBandits.SequentialLearning.Algorithm +import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.OptimizationAlgorithms.Utils.Uniform +import LeanMachineLearning.OptimizationAlgorithms.Utils.EuclideanSpace +import LeanMachineLearning.SequentialLearning.Algorithm open MeasureTheory ProbabilityTheory Finset NNReal Learning diff --git a/LeanBandits/OptimizationAlgorithms/PRS.lean b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean similarity index 97% rename from LeanBandits/OptimizationAlgorithms/PRS.lean rename to LeanMachineLearning/OptimizationAlgorithms/PRS.lean index 2f34977b..1bbd9494 100644 --- a/LeanBandits/OptimizationAlgorithms/PRS.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean @@ -4,10 +4,10 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanBandits.ForMathlib.ENNReal -import LeanBandits.ForMathlib.IndepFun -import LeanBandits.OptimizationAlgorithms.Utils.Tuple -import LeanBandits.SequentialLearning.EvaluationEnv +import LeanMachineLearning.ForMathlib.ENNReal +import LeanMachineLearning.ForMathlib.IndepFun +import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.SequentialLearning.EvaluationEnv open MeasureTheory ProbabilityTheory Learning Finset ENNReal Filter diff --git a/LeanBandits/OptimizationAlgorithms/RankOpt.lean b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean similarity index 97% rename from LeanBandits/OptimizationAlgorithms/RankOpt.lean rename to LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean index eb7fffa8..a99cff4a 100644 --- a/LeanBandits/OptimizationAlgorithms/RankOpt.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean @@ -4,10 +4,10 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanBandits.OptimizationAlgorithms.Utils.EuclideanSpace -import LeanBandits.OptimizationAlgorithms.Utils.Tuple -import LeanBandits.OptimizationAlgorithms.Utils.Uniform -import LeanBandits.SequentialLearning.Algorithm +import LeanMachineLearning.OptimizationAlgorithms.Utils.EuclideanSpace +import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.OptimizationAlgorithms.Utils.Uniform +import LeanMachineLearning.SequentialLearning.Algorithm open MeasureTheory ProbabilityTheory Finset NNReal Learning diff --git a/LeanBandits/OptimizationAlgorithms/Utils/EuclideanSpace.lean b/LeanMachineLearning/OptimizationAlgorithms/Utils/EuclideanSpace.lean similarity index 100% rename from LeanBandits/OptimizationAlgorithms/Utils/EuclideanSpace.lean rename to LeanMachineLearning/OptimizationAlgorithms/Utils/EuclideanSpace.lean diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean b/LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean similarity index 100% rename from LeanBandits/OptimizationAlgorithms/Utils/Tuple.lean rename to LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean diff --git a/LeanBandits/OptimizationAlgorithms/Utils/Uniform.lean b/LeanMachineLearning/OptimizationAlgorithms/Utils/Uniform.lean similarity index 100% rename from LeanBandits/OptimizationAlgorithms/Utils/Uniform.lean rename to LeanMachineLearning/OptimizationAlgorithms/Utils/Uniform.lean diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index 536fe0da..8697d396 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -4,8 +4,8 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanBandits.SequentialLearning.StationaryEnv -import LeanBandits.ForMathlib.CondDistrib +import LeanMachineLearning.SequentialLearning.StationaryEnv +import LeanMachineLearning.ForMathlib.CondDistrib /-! # Function evaluation environments From da9fc2e18721c93600524011bda10ab5c003475d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 20:09:21 +0200 Subject: [PATCH 17/49] golf --- LeanMachineLearning.lean | 58 ++++++------ .../OptimizationAlgorithms/LIPO.lean | 62 ++++-------- .../OptimizationAlgorithms/PRS.lean | 44 ++++++--- .../OptimizationAlgorithms/RankOpt.lean | 94 +++++++------------ .../Utils/EuclideanSpace.lean | 14 --- .../OptimizationAlgorithms/Utils/Uniform.lean | 25 ----- 6 files changed, 118 insertions(+), 179 deletions(-) delete mode 100644 LeanMachineLearning/OptimizationAlgorithms/Utils/EuclideanSpace.lean delete mode 100644 LeanMachineLearning/OptimizationAlgorithms/Utils/Uniform.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index d4aa0eca..7c4b1fcd 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,27 +1,31 @@ -module - -public import LeanMachineLearning.Bandit.Bandit -public import LeanMachineLearning.Bandit.Regret -public import LeanMachineLearning.Bandit.RewardByCountMeasure -public import LeanMachineLearning.Bandit.SumRewards -public import LeanMachineLearning.BanditAlgorithms.AuxSums -public import LeanMachineLearning.BanditAlgorithms.ETC -public import LeanMachineLearning.BanditAlgorithms.RoundRobin -public import LeanMachineLearning.BanditAlgorithms.UCB -public import LeanMachineLearning.ForMathlib.CondDistrib -public import LeanMachineLearning.ForMathlib.CondIndepFun -public import LeanMachineLearning.ForMathlib.HasCondDistrib -public import LeanMachineLearning.ForMathlib.IndepFun -public import LeanMachineLearning.ForMathlib.IndepInfinitePi -public import LeanMachineLearning.ForMathlib.Integrable -public import LeanMachineLearning.ForMathlib.KernelSub -public import LeanMachineLearning.ForMathlib.Measurable -public import LeanMachineLearning.ForMathlib.MeasurableArgMax -public import LeanMachineLearning.ForMathlib.StandardBorel -public import LeanMachineLearning.ForMathlib.SubGaussian -public import LeanMachineLearning.ForMathlib.Traj -public import LeanMachineLearning.SequentialLearning.Algorithm -public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.SequentialLearning.FiniteActions -public import LeanMachineLearning.SequentialLearning.IonescuTulceaSpace -public import LeanMachineLearning.SequentialLearning.StationaryEnv +import LeanMachineLearning.Bandit.Bandit +import LeanMachineLearning.Bandit.Regret +import LeanMachineLearning.Bandit.RewardByCountMeasure +import LeanMachineLearning.Bandit.SumRewards +import LeanMachineLearning.BanditAlgorithms.AuxSums +import LeanMachineLearning.BanditAlgorithms.ETC +import LeanMachineLearning.BanditAlgorithms.RoundRobin +import LeanMachineLearning.BanditAlgorithms.UCB +import LeanMachineLearning.ForMathlib.CondDistrib +import LeanMachineLearning.ForMathlib.CondIndepFun +import LeanMachineLearning.ForMathlib.ENNReal +import LeanMachineLearning.ForMathlib.HasCondDistrib +import LeanMachineLearning.ForMathlib.IndepFun +import LeanMachineLearning.ForMathlib.IndepInfinitePi +import LeanMachineLearning.ForMathlib.Integrable +import LeanMachineLearning.ForMathlib.KernelSub +import LeanMachineLearning.ForMathlib.Measurable +import LeanMachineLearning.ForMathlib.MeasurableArgMax +import LeanMachineLearning.ForMathlib.StandardBorel +import LeanMachineLearning.ForMathlib.SubGaussian +import LeanMachineLearning.ForMathlib.Traj +import LeanMachineLearning.OptimizationAlgorithms.LIPO +import LeanMachineLearning.OptimizationAlgorithms.PRS +import LeanMachineLearning.OptimizationAlgorithms.RankOpt +import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.SequentialLearning.Algorithm +import LeanMachineLearning.SequentialLearning.Deterministic +import LeanMachineLearning.SequentialLearning.EvaluationEnv +import LeanMachineLearning.SequentialLearning.FiniteActions +import LeanMachineLearning.SequentialLearning.IonescuTulceaSpace +import LeanMachineLearning.SequentialLearning.StationaryEnv diff --git a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean index 6f4c4435..2c07f711 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean @@ -5,36 +5,26 @@ Authors: Gaëtan Serré -/ import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple -import LeanMachineLearning.OptimizationAlgorithms.Utils.Uniform -import LeanMachineLearning.OptimizationAlgorithms.Utils.EuclideanSpace import LeanMachineLearning.SequentialLearning.Algorithm open MeasureTheory ProbabilityTheory Finset NNReal Learning /-! # LIPO: Lipschitz Optimization + Implementation of the _LIPO_ algorithm -[(_Global optimization of Lipschitz functions_, Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) -defined on a measurable subset of a Euclidean space, with finite and non-zero measure. -The algorithm samples from the uniform distribution on the set of potential maximizers of -the function at each iteration. +[(_Global optimization of Lipschitz functions_, +Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) +defined on a measurable space with a metric. The algorithm samples from an arbitrary +probability measure on the set of potential maximizers of the function at each iteration. -/ -variable {d : ℕ} {α : Set (ℝᵈ d)} (mes_α : MeasurableSet α) (mα₁ : ℙ α ≠ ⊤) (κ : ℝ≥0) +variable {α : Type*} [PseudoMetricSpace α] [MeasurableSpace α] [BorelSpace α] + [SecondCountableTopology α] (μ : Measure α) [IsProbabilityMeasure μ] {n : ℕ} (κ : ℝ≥0) + (data : Iic n → α × ℝ) namespace LIPO -noncomputable instance : MeasureSpace α := Measure.Subtype.measureSpace - -instance : MeasurableSpace α := by infer_instance - -instance i₁ : IsFiniteMeasure (ℙ : Measure α) := by - rw [isFiniteMeasure_iff ℙ, Measure.Subtype.volume_univ] - · exact mα₁.lt_top - · exact mes_α.nullMeasurableSet - -variable {n : ℕ} (data : Iic n → α × ℝ) - /-- The set of potential maximizers for the LIPO algorithm. Given observed data points and function values, this set contains all points `x` where the maximum observed value is at most the minimum Lipschitz upper bound across all observations. @@ -51,39 +41,34 @@ lemma measurableSet_potential_max_prod : · fun_prop · fun_prop -include mes_α mα₁ in -lemma measurable_volume_potential_max_inter (s : Set α) (hs : MeasurableSet s) : - Measurable (fun data : Iic n → α × ℝ ↦ ℙ (potential_max κ data ∩ s)) := by +lemma measurable_potential_max_inter {s : Set α} (hs : MeasurableSet s) : + Measurable (fun data : Iic n → α × ℝ ↦ μ (potential_max κ data ∩ s)) := by set E := {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max κ p.1 ∩ s} have hE_meas : MeasurableSet E := (measurableSet_potential_max_prod κ).inter (measurableSet_preimage measurable_snd hs) - have := i₁ mes_α mα₁ exact measurable_measure_prodMk_left hE_meas -/-- Markov kernel that samples uniformly from the set of potential maximizers. -This kernel forms the core sampling strategy of LIPO: at each iteration, given the observed -data, it samples the next query point uniformly from `potential_max`. -/ +/-- Markov kernel sampling from the set of potential maximizers according to μ. -/ noncomputable def potential_max_kernel : Kernel (Iic n → α × ℝ) α := by - refine ⟨fun data ↦ uniform <| potential_max κ data, ?_⟩ + refine ⟨fun data ↦ cond μ <| potential_max κ data, ?_⟩ rw [Measure.measurable_measure] intro s hs - simp only [Measure.smul_apply, MeasureTheory.Measure.restrict_apply hs, smul_eq_mul] + simp only [ProbabilityTheory.cond, Measure.smul_apply, smul_eq_mul] refine Measurable.mul ?_ ?_ · refine Measurable.inv ?_ - convert measurable_volume_potential_max_inter mes_α mα₁ κ Set.univ (MeasurableSet.univ) + convert measurable_potential_max_inter μ κ (MeasurableSet.univ) simp [Set.inter_univ] - · convert measurable_volume_potential_max_inter mes_α mα₁ κ s hs using 1 + · simp_rw [μ.restrict_apply hs] + convert measurable_potential_max_inter μ κ hs using 1 simp [Set.inter_comm] end LIPO open LIPO -variable (mα₀ : ℙ α ≠ 0) - /- We suppose that the set of potential maximizers has non-zero measure at each iteration, ensuring that the algorithm can sample from it. -/ -variable (h : ∀ n (data : Iic n → α × ℝ), ℙ (potential_max κ data) ≠ 0) +variable (h : ∀ n (data : Iic n → α × ℝ), μ (potential_max κ data) ≠ 0) /-- The LIPO (LIPschitz Optimization) algorithm for global optimization. This algorithm optimizes an unknown function assuming only that it has a finite Lipschitz @@ -91,13 +76,6 @@ constant `κ`. It starts with a uniform initial distribution and iteratively sam the set of potential maximizers, ensuring consistency and convergence to the global optimum [(Malherbe et al., 2017)](https://arxiv.org/abs/1703.02628). -/ noncomputable def LIPO : Algorithm α ℝ where - policy _ := potential_max_kernel mes_α mα₁ κ - p0 := uniform Set.univ - hp0 := by - have := i₁ mes_α mα₁ - refine uniform_is_prob_measure ?_ - rwa [Measure.Subtype.volume_univ mes_α.nullMeasurableSet] - h_policy n := by - refine ⟨fun data => ?_⟩ - have := i₁ mes_α mα₁ - exact uniform_is_prob_measure <| h n data + policy _ := potential_max_kernel μ κ + p0 := μ + h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ diff --git a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean index 1bbd9494..efb997e2 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean @@ -163,19 +163,39 @@ lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by - have hmf_min (x : α) : -f a ≤ -f x := by - specialize hf_max x - linarith - have h' := tendsto_min (continuous_neg_iff.mpr hfc) (IsAlgEnvSeq.neg hfc.measurable h) hmf_min - rw [tendstoInMeasure_iff_dist] at h' ⊢ + rw [tendstoInMeasure_iff_dist] intro ε hε - specialize h' ε hε - have dist_neg (n : ℕ) : {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R i x) (f a)} = - {x | ε ≤ dist (-(Tuple.max fun (i : Iic n) ↦ R i x)) (-f a)} := by - simp [dist_neg_neg] - simp_rw [dist_neg] - convert h' with n ω - exact Tuple.neg_max_eq_min_neg _ + have hf := hfc.measurable + rw [Metric.continuous_iff] at hfc + obtain ⟨δ, hδ, hfc⟩ := hfc a ε hε + have (n : ℕ) : P {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R i x) (f a)} = + P {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ f (A i x)) (f a)} := by + refine measure_congr ?_ + filter_upwards [IsAlgEnvSeq.reward_eq_evals_actions_comp hf h Tuple.max] with ω hω + simp only [eq_iff_iff] + change ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R (↑i) ω) (f a) ↔ + ε ≤ dist (Tuple.max fun (i : Iic n) ↦ f (A ↑i ω)) (f a) + rw [hω] + simp_rw [this] + refine tendsto_any hf h a δ hδ |> tendsto_zero_le <| ?_ + intro n + refine measure_mono ?_ + simp only [Set.setOf_subset_setOf] + intro ω hω + rw [← Tuple.argmin_spec] + set j := Tuple.argmin (fun (i : Iic n) ↦ dist (A i ω) a) + have : dist (Tuple.max fun (i : Iic n) ↦ f (A i ω)) (f a) ≤ dist (f (A j ω)) (f a) := by + rw [← Tuple.argmax_spec] + set k := Tuple.argmax (fun (i : Iic n) ↦ f (A i ω)) + have := hf_max (A k ω) + have : f (A j ω) ≤ f (A k ω) := + Tuple.le_argmax (fun (i : Iic n) ↦ f (A i ω)) j + simp [Real.dist_eq] + grind + have := hω.trans this + by_contra! h_contra + specialize hfc (A j ω) h_contra + linarith end Real diff --git a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean index a99cff4a..aac60b4d 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean @@ -4,18 +4,18 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanMachineLearning.OptimizationAlgorithms.Utils.EuclideanSpace import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple -import LeanMachineLearning.OptimizationAlgorithms.Utils.Uniform import LeanMachineLearning.SequentialLearning.Algorithm open MeasureTheory ProbabilityTheory Finset NNReal Learning + /-! # RankOpt: A Ranking Approach to Global Optimization Implementation of the _RankOpt_ algorithm -[(_A Ranking Approach to Global Optimization_, Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) +[(_A Ranking Approach to Global Optimization_, +Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) defined on a measurable subset of a Euclidean space, with finite and non-zero measure. The algorithm samples from the uniform distribution on the set of potential maximizers of the function at each iteration. @@ -26,47 +26,32 @@ section RankRule /-- A rank rule is a measurable function that compares pairs of points. It returns 1 if the first point is ranked higher, -1 if lower, and 0 if equal. -/ -- ANCHOR: RankRule -def RankRule (α : Type) [MeasurableSpace α] := +def RankRule (α : Type*) [MeasurableSpace α] := {f : α → α → ({-1, 0, 1} : Set ℝ) // Measurable <| Function.uncurry f} -- ANCHOR_END: RankRule end RankRule -variable {α : Type} [MeasurableSpace α] {d : ℕ} {α : Set (ℝᵈ d)} - (mes_α : MeasurableSet α) (mα₁ : ℙ α ≠ ⊤) +variable {α β : Type*} [MeasurableSpace α] (μ : Measure α) [IsProbabilityMeasure μ] {n : ℕ} + [TopologicalSpace β] [MeasurableSpace β] [BorelSpace β] [LinearOrder β] + [SecondCountableTopology β] [OpensMeasurableSpace β] [OrderClosedTopology β] + (data : Iic n → α × β) namespace RankOpt -noncomputable instance : MeasureSpace α := Measure.Subtype.measureSpace - -instance : MeasurableSpace α := by infer_instance - -instance i₁ : IsFiniteMeasure (ℙ : Measure α) := by - rw [isFiniteMeasure_iff ℙ, Measure.Subtype.volume_univ] - · exact mα₁.lt_top - · exact mes_α.nullMeasurableSet - -instance : MeasurableSpace (RankRule α) := Subtype.instMeasurableSpace - /-- Computes the ranking from observed function values. Returns 1 if `y₁ > y₂`, 0 if `y₁ = y₂`, and -1 if `y₁ < y₂`. -/ -noncomputable def ranking_data (y₁ y₂ : ℝ) := - if y₂ < y₁ then 1 else if y₂ = y₁ then 0 else -1 +noncomputable def ranking_data (y₁ y₂ : β) := if y₂ < y₁ then 1 else if y₂ = y₁ then 0 else -1 /-- Indicator function checking if two rankings agree. Returns 1 if both values are equal, 0 otherwise. -/ -noncomputable abbrev rindicator (r₁ r₂ : ℝ) := - if r₁ = r₂ then (1 : ℝ) else 0 - -variable {n : ℕ} (data : Iic n → α × ℝ) - -abbrev s := {(i, j) : Iic n × Iic n | i ≤ j} +noncomputable abbrev rindicator (r₁ r₂ : ℝ) := if r₁ = r₂ then (1 : ℝ) else 0 /-- Computes the ranking loss for a rank rule. Measures the agreement between a candidate rule `r` and the rankings induced by the observed function values on all pairs of data points, normalized by the number of pairs. -/ noncomputable def ranking_loss (r : RankRule α) := - 2 * (n * (n + 1) : ℝ)⁻¹ * ∑ ij ∈ s, + 2 * (n * (n + 1) : ℝ)⁻¹ * ∑ ij ∈ {(i, j) : Finset.Iic n × Finset.Iic n | i ≤ j}, rindicator (r.1 (data ij.1).1 (data ij.2).1) (ranking_data (data ij.1).2 (data ij.2).2) /-- The point in the observed data with the maximum function value. -/ @@ -80,7 +65,7 @@ def potential_max (𝓡 : Set (RankRule α)) := {x | ∃ (r : 𝓡), ranking_loss data r = 0 ∧ 0 ≤ (r.1.1 x (argmax_f data)).1} lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) : - MeasurableSet {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max p.1 𝓡} := by + MeasurableSet {p : (Iic n → α × β) × α | p.2 ∈ potential_max p.1 𝓡} := by simp only [potential_max, Set.mem_setOf_eq, measurableSet_setOf] have : Countable (𝓡) := h𝓡.to_subtype refine Measurable.exists fun r ↦ (.and ?_ ?_) @@ -100,21 +85,21 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. · refine Measurable.le' measurable_const ?_ have : Measurable (fun x : ({-1, 0, 1} : Set ℝ) ↦ (x : ℝ)) := by fun_prop refine this.comp (r.1.2.comp (measurable_snd.prodMk ?_)) - suffices Measurable (fun p : Iic n → α × ℝ ↦ (p <| Tuple.argmax (fun i ↦ (p i).2)).1) by + suffices Measurable (fun p : Iic n → α × β ↦ (p <| Tuple.argmax (fun i ↦ (p i).2)).1) by exact this.comp measurable_fst - have h_eval : Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ (p.1 p.2).1) := by - suffices Measurable (fun p : (Iic n → α × ℝ) × Iic n ↦ p.1 p.2) by + have h_eval : Measurable (fun p : (Iic n → α × β) × Iic n ↦ (p.1 p.2).1) := by + suffices Measurable (fun p : (Iic n → α × β) × Iic n ↦ p.1 p.2) by fun_prop exact measurable_from_prod_countable_left fun i ↦ measurable_pi_apply i refine h_eval.comp (Measurable.prodMk ?_ ?_) · fun_prop - · change Measurable (fun p : Iic n → α × ℝ ↦ Tuple.argmax (fun i ↦ (p i).2)) - suffices Measurable (fun u : Iic n → ℝ ↦ Tuple.argmax u) by + · change Measurable (fun p : Iic n → α × β ↦ Tuple.argmax (fun i ↦ (p i).2)) + suffices Measurable (fun u : Iic n → β ↦ Tuple.argmax u) by fun_prop refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] - let Maximizers {n : ℕ} (u : Iic n → ℝ) : Set (Iic n) := {i | u i = Tuple.max u} - have : {u : Iic n → ℝ | Tuple.argmax u = i} = ⋃ (S) + let Maximizers {n : ℕ} (u : Iic n → β) : Set (Iic n) := {i | u i = Tuple.max u} + have : {u : Iic n → β | Tuple.argmax u = i} = ⋃ (S) (hS : ∀ x, Maximizers x = S → Tuple.argmax x = i), {u | Maximizers u = S} := by ext u simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] @@ -129,55 +114,46 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) exact measurableSet_eq_fun (by fun_prop) measurable_const -include mes_α mα₁ in -lemma measurable_volume_potential_max_inter {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) - (s : Set α) (hs : MeasurableSet s) : - Measurable (fun data : Iic n → α × ℝ ↦ ℙ (potential_max data 𝓡 ∩ s)) := by - set E := {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max p.1 𝓡 ∩ s} +lemma measurable_potential_max_inter {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) + {s : Set α} (hs : MeasurableSet s) : + Measurable (fun data : Iic n → α × β ↦ μ (potential_max data 𝓡 ∩ s)) := by + set E := {p : (Iic n → α × β) × α | p.2 ∈ potential_max p.1 𝓡 ∩ s} have hE_meas : MeasurableSet E := (measurableSet_potential_max_prod h𝓡).inter (measurableSet_preimage measurable_snd hs) - have := i₁ mes_α mα₁ exact measurable_measure_prodMk_left hE_meas /-- Markov kernel that samples uniformly from the set of potential maximizers. This kernel forms the core sampling strategy of RankOpt: at each iteration, given the observed data, it samples the next query point uniformly from `potential_max`. -/ noncomputable def potential_max_kernel {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) : - Kernel (Iic n → α × ℝ) α := by - refine ⟨fun data ↦ uniform <| @potential_max d α n data 𝓡, ?_⟩ + Kernel (Iic n → α × β) α := by + refine ⟨fun data ↦ cond μ <| potential_max data 𝓡, ?_⟩ rw [Measure.measurable_measure] intro s hs - simp only [Measure.smul_apply, MeasureTheory.Measure.restrict_apply hs, smul_eq_mul] + simp only [ProbabilityTheory.cond, Measure.smul_apply, smul_eq_mul] refine Measurable.mul ?_ ?_ · refine Measurable.inv ?_ - convert measurable_volume_potential_max_inter mes_α mα₁ h𝓡 Set.univ (MeasurableSet.univ) + convert measurable_potential_max_inter (β := β) μ h𝓡 (MeasurableSet.univ) simp [Set.inter_univ] - · convert measurable_volume_potential_max_inter mes_α mα₁ h𝓡 s hs using 1 + · simp_rw [μ.restrict_apply hs] + convert measurable_potential_max_inter (β := β) μ h𝓡 hs using 1 simp [Set.inter_comm] end RankOpt open RankOpt -variable (mα₀ : ℙ α ≠ 0) {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) - /- We suppose that the set of potential maximizers has non-zero measure at each iteration, ensuring that the algorithm can sample from it. -/ -variable (h : ∀ n (data : Iic n → α × ℝ), ℙ (potential_max data 𝓡) ≠ 0) +variable {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) + (h : ∀ n (data : Iic n → α × β), μ (potential_max data 𝓡) ≠ 0) /-- The RankOpt algorithm for global optimization. This algorithm uses a ranking approach to optimize an unknown function. It maintains a hypothesis class `𝓡` of ranking rules. At each iteration, it samples from the set of points that could be optimal according to ranking rules consistent with the observed data [(Malherbe et al., 2017)](https://arxiv.org/pdf/1603.04381). -/ -noncomputable def RankOpt : Algorithm α ℝ where - policy _ := potential_max_kernel mes_α mα₁ h𝓡 - p0 := uniform Set.univ - hp0 := by - have := i₁ mes_α mα₁ - refine uniform_is_prob_measure ?_ - rwa [Measure.Subtype.volume_univ mes_α.nullMeasurableSet] - h_policy n := by - refine ⟨fun data => ?_⟩ - have := i₁ mes_α mα₁ - exact uniform_is_prob_measure <| h n data +noncomputable def RankOpt : Algorithm α β where + policy _ := potential_max_kernel μ h𝓡 + p0 := μ + h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ diff --git a/LeanMachineLearning/OptimizationAlgorithms/Utils/EuclideanSpace.lean b/LeanMachineLearning/OptimizationAlgorithms/Utils/EuclideanSpace.lean deleted file mode 100644 index 6e27ccfd..00000000 --- a/LeanMachineLearning/OptimizationAlgorithms/Utils/EuclideanSpace.lean +++ /dev/null @@ -1,14 +0,0 @@ -/- -Copyright (c) 2026 Gaëtan Serré. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Gaëtan Serré --/ - -import Mathlib.Analysis.InnerProductSpace.PiL2 - -/-- Euclidean space of dimension `d`. -Used as the domain for LIPO optimization problems. -/ -abbrev ED (d : ℕ) := EuclideanSpace ℝ (Fin d) - -@[inherit_doc ED] -notation3 "ℝᵈ " d => ED d diff --git a/LeanMachineLearning/OptimizationAlgorithms/Utils/Uniform.lean b/LeanMachineLearning/OptimizationAlgorithms/Utils/Uniform.lean deleted file mode 100644 index 6ec66172..00000000 --- a/LeanMachineLearning/OptimizationAlgorithms/Utils/Uniform.lean +++ /dev/null @@ -1,25 +0,0 @@ -/- -Copyright (c) 2026 Gaëtan Serré. All rights reserved. -Released under Apache 2.0 as described in the file LICENSE. -Authors: Gaëtan Serré --/ - -import Mathlib.Probability.Notation - -open MeasureTheory Set ProbabilityTheory - -/-! -# Uniform distribution -This file defines the uniform distribution on a set of a finite measure space as the normalized -restriction of the original measure to the set. --/ - -variable {α β : Type*} [MeasureSpace α] [IsFiniteMeasure (ℙ : Measure α)] [MeasurableSpace β] - -/-- The uniform distribution on a set `s` is defined as the normalized restriction of the original -measure to `s`. -/ -noncomputable abbrev uniform (s : Set α) : Measure α := (ℙ s)⁻¹ • (ℙ).restrict s - -instance uniform_is_prob_measure {s : Set α} (hs : ℙ s ≠ 0) : IsProbabilityMeasure (uniform s) := by - rw [isProbabilityMeasure_iff] - simp [ENNReal.inv_mul_cancel, hs] From 3260cd76b087e956845d9a0fce45be08f9e121d7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 20:29:29 +0200 Subject: [PATCH 18/49] docstring --- LeanMachineLearning/OptimizationAlgorithms/PRS.lean | 13 +++++++++++++ .../SequentialLearning/EvaluationEnv.lean | 9 +++++++++ 2 files changed, 22 insertions(+) diff --git a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean index efb997e2..81729604 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean @@ -17,6 +17,19 @@ open scoped Topology # PRS: Pure Random Search Implementation of the _Pure Random Search_ algorithm, which samples from a fixe probability measure at each iteration. + +## Main definitions + +* `PRS μ`: The Pure Random Search algorithm with sampling measure `μ`. At each iteration, it samples + the next action from the fixed probability measure `μ`, independently of the past history. + +## Main statements + +* `iIndep_actions`: The actions taken by the PRS algorithm are independent random variables. +* `iIndep_rewards`: The rewards obtained by the PRS algorithm are independent random variables. +* `tendsto_any`: In a pseudo-metric space such as `μ` is an open positive measure, the actions + taken by PRS get arbitrarily close to any point with probability tending to 1. +* `tendsto_min`: If the reward function is continuous and has a global minimum at `a`, then the -/ variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index 8697d396..d3241ff7 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -9,6 +9,15 @@ import LeanMachineLearning.ForMathlib.CondDistrib /-! # Function evaluation environments + +A stationary environment where the reward is given by evaluating a fixed measurable function `f` at +the chosen action. + +## Main definitions + +* `evalEnv hf`: A stationary environment where the reward is given by a deterministic kernel that + evaluates a fixed measurable function `f` at the chosen action. + -/ open MeasureTheory ProbabilityTheory From 427129be777ed2bc094cf541b6accb176aceb29b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 20:45:39 +0200 Subject: [PATCH 19/49] golf --- .../OptimizationAlgorithms/LIPO.lean | 17 +- .../OptimizationAlgorithms/PRS.lean | 161 +++++++++++------- .../OptimizationAlgorithms/RankOpt.lean | 36 ++-- .../OptimizationAlgorithms/Utils/Tuple.lean | 4 + .../SequentialLearning/EvaluationEnv.lean | 18 +- 5 files changed, 143 insertions(+), 93 deletions(-) diff --git a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean index 2c07f711..84dd6725 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean @@ -13,10 +13,17 @@ open MeasureTheory ProbabilityTheory Finset NNReal Learning # LIPO: Lipschitz Optimization Implementation of the _LIPO_ algorithm -[(_Global optimization of Lipschitz functions_, -Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) +[(_Global optimization of Lipschitz functions_, Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) defined on a measurable space with a metric. The algorithm samples from an arbitrary probability measure on the set of potential maximizers of the function at each iteration. + +## Main definitions + +* `potential_max`: The set of potential maximizers for the LIPO algorithm. +* `potential_max_kernel`: The Markov kernel that samples from the set of potential maximizers + according to a given measure `μ`. +* `LIPO`: The LIPO algorithm that samples from the set of potential maximizers using a given + probability measure at each iteration. -/ variable {α : Type*} [PseudoMetricSpace α] [MeasurableSpace α] [BorelSpace α] @@ -72,9 +79,9 @@ variable (h : ∀ n (data : Iic n → α × ℝ), μ (potential_max κ data) ≠ /-- The LIPO (LIPschitz Optimization) algorithm for global optimization. This algorithm optimizes an unknown function assuming only that it has a finite Lipschitz -constant `κ`. It starts with a uniform initial distribution and iteratively samples from -the set of potential maximizers, ensuring consistency and convergence to the global optimum -[(Malherbe et al., 2017)](https://arxiv.org/abs/1703.02628). -/ +constant `κ`. It starts with an arbitrary probability measure `μ` as initial distribution and +iteratively samples from the set of potential maximizers, ensuring consistency and convergence to +the global optimum [(Malherbe et al., 2017)](https://arxiv.org/abs/1703.02628). -/ noncomputable def LIPO : Algorithm α ℝ where policy _ := potential_max_kernel μ κ p0 := μ diff --git a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean index 81729604..e60b3eaf 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean @@ -15,41 +15,42 @@ open scoped Topology /-! # PRS: Pure Random Search -Implementation of the _Pure Random Search_ algorithm, which samples from a fixe probability measure -at each iteration. + +Implementation of the _Pure Random Search_ algorithm, which samples from a fixed probability +measure at each iteration. ## Main definitions -* `PRS μ`: The Pure Random Search algorithm with sampling measure `μ`. At each iteration, it samples - the next action from the fixed probability measure `μ`, independently of the past history. +* `PRS`: The pure random search algorithm that samples from a fixed distribution at each iteration. -## Main statements +## Main results -* `iIndep_actions`: The actions taken by the PRS algorithm are independent random variables. -* `iIndep_rewards`: The rewards obtained by the PRS algorithm are independent random variables. -* `tendsto_any`: In a pseudo-metric space such as `μ` is an open positive measure, the actions - taken by PRS get arbitrarily close to any point with probability tending to 1. -* `tendsto_min`: If the reward function is continuous and has a global minimum at `a`, then the +- `hasLaw_actions`: Each action follows the distribution μ. +- `hasLaw_rewards`: Each reward follows the distribution μ.map f. +- `iIndep_actions`: Actions are mutually independent across time steps. +- `iIndep_rewards`: Rewards are mutually independent across time steps. +- `actions_tendsto_any`: The minimum distance from sampled actions to any point in α tends to zero. +- `rewards_tendsto_any`: The minimum distance from rewards to any value tends to zero. +- `tendsto_min`: The minimum reward converges in measure to the global minimum value. +- `tendsto_max`: The maximum reward converges in measure to the global maximum value. -/ -variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] +variable {α β Ω : Type*} [MeasurableSpace α] [MeasurableSpace β] [StandardBorelSpace α] [Nonempty α] + [StandardBorelSpace β] [Nonempty β] {μ : Measure α} [IsProbabilityMeasure μ] [MeasurableSpace Ω] + {P : Measure Ω} [IsProbabilityMeasure P] open Set in +/-- The Pure Random Search algorithm. -/ @[simps] -noncomputable -def PRS (μ : Measure α) [IsProbabilityMeasure μ] : Algorithm α β where +noncomputable def PRS (μ : Measure α) [IsProbabilityMeasure μ] : Algorithm α β where policy _ := Kernel.const _ μ p0 := μ namespace PRS -section - -variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace β] [Nonempty β] - {Ω : Type*} [MeasurableSpace Ω] {P : Measure Ω} [IsProbabilityMeasure P] - {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) {μ : Measure α} - [IsProbabilityMeasure μ] (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) +variable {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) +/-- Each action follows the distribution μ. -/ lemma hasLaw_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : HasLaw (A n) μ P := by by_cases hn : n = 0 · rw [hn] @@ -58,31 +59,41 @@ lemma hasLaw_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : H obtain ⟨k, rfl⟩ := Nat.exists_eq_succ_of_ne_zero hn exact hasLaw_of_hasCondDistrib_const <| h.hasCondDistrib_action k +/-- Each reward follows the distribution μ.map f. -/ lemma hasLaw_rewards (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : HasLaw (R n) (μ.map f) P := by - refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_eq_eval_action hf h n) + refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_ae_eq_eval_action hf h n) have hA := h.measurable_A n refine ⟨by fun_prop, ?_⟩ rw [← Measure.map_map hf hA, (hasLaw_actions hf h n).map_eq] +/-- Actions are mutually independent. -/ lemma iIndep_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : iIndepFun A P := by have hA := h.measurable_A rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)] intro n have condDistrib_eq := (h.hasCondDistrib_action n).condDistrib_eq - have law_eq := (hasLaw_actions hf h (n + 1)).map_eq simp only [PRS_policy] at condDistrib_eq + have law_eq := (hasLaw_actions hf h (n + 1)).map_eq rw [← law_eq, ← indepFun_iff_condDistrib_eq_const ?_ (by fun_prop)] at condDistrib_eq · have meas_fst : Measurable (fun (f : Iic n → α × β) ↦ (fun i ↦ (f i).1)) := by fun_prop exact (condDistrib_eq.comp meas_fst measurable_id).symm · exact (IsAlgEnvSeq.measurable_hist (h.measurable_A) (h.measurable_R) n).aemeasurable +/-- Rewards are mutually independent. -/ +lemma iIndep_rewards (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : + iIndepFun R P := + have (n : ℕ) : f ∘ A n =ᵐ[P] R n := + (IsAlgEnvSeq.reward_ae_eq_eval_action hf h n).symm + iIndepFun.congr this <| (iIndep_actions hf h).comp _ (fun _ ↦ hf) + variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] [μ.IsOpenPosMeasure] -theorem tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : +/-- The minimum distance from sampled actions to any point tends to zero. -/ +theorem actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (A j.1 x) a)}) atTop (𝓝 0) := by set PRS_alg := PRS (β := β) μ @@ -126,40 +137,57 @@ theorem tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : intro i hi ω (hω : ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (A j.1 ω) a)) simp_all only [univ_eq_attach, le_inf'_iff, mem_attach, forall_const, Subtype.forall, mem_Iic] -end - -section Real +variable [PseudoMetricSpace β] [BorelSpace β] (hfc : Continuous f) -variable [StandardBorelSpace α] [Nonempty α] [PseudoMetricSpace α] [OpensMeasurableSpace α] - [SecondCountableTopology α] {Ω : Type*} [MeasurableSpace Ω] {P : Measure Ω} - [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R : ℕ → Ω → ℝ} {f : α → ℝ} (hfc : Continuous f) - {μ : Measure α} [IsProbabilityMeasure μ] [μ.IsOpenPosMeasure] - (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) {a : α} - -lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) - (hf_min : ∀ x, f a ≤ f x) : - TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by - rw [tendstoInMeasure_iff_dist] +/-- The minimum distance from image of actions to any value tends to zero. -/ +lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) (a : α) : + ∀ ε, 0 < ε → Tendsto (fun i => P + {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (f (A j.1 x)) (f a))}) atTop (𝓝 0) := by intro ε hε have hf := hfc.measurable rw [Metric.continuous_iff] at hfc obtain ⟨δ, hδ, hfc⟩ := hfc a ε hε - have (n : ℕ) : P {x | ε ≤ dist (Tuple.min fun (i : Iic n) ↦ R i x) (f a)} = - P {x | ε ≤ dist (Tuple.min fun (i : Iic n) ↦ f (A i x)) (f a)} := by - refine measure_congr ?_ - filter_upwards [IsAlgEnvSeq.reward_eq_evals_actions_comp hf h Tuple.min] with ω hω - simp only [eq_iff_iff] - change ε ≤ dist (Tuple.min fun (i : Iic n) ↦ R (↑i) ω) (f a) ↔ - ε ≤ dist (Tuple.min fun (i : Iic n) ↦ f (A ↑i ω)) (f a) - rw [hω] - simp_rw [this] - refine tendsto_any hf h a δ hδ |> tendsto_zero_le <| ?_ + refine actions_tendsto_any hf h a δ hδ |> tendsto_zero_le <| ?_ + intro n + refine measure_mono ?_ + simp only [Set.setOf_subset_setOf] + intro ω hω + rw [← Tuple.argmin_spec] + set j := Tuple.argmin (fun (i : Iic n) ↦ dist (A i.1 ω) a) + by_contra! h_contra + specialize hfc (A j.1 ω) h_contra + have := Tuple.min_le (fun (j : Iic n) ↦ dist (f (A (j) ω)) (f a)) j + linarith + +/-- The minimum distance from rewards to any value tends to zero. -/ +lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) (a : α) : + ∀ ε, 0 < ε → Tendsto (fun i => P + {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (R j.1 x) (f a))}) atTop (𝓝 0) := by + intro ε hε + convert image_actions_tendsto_any hfc h a ε hε using 2 with n + refine measure_congr ?_ + let g : ((Iic n) → β) → ℝ := fun r ↦ Tuple.min (fun i ↦ dist (r i) (f a)) + filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h g] with ω hω + simp only [eq_iff_iff] + change ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (R j ω) (f a)) ↔ + ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (f (A j ω)) (f a)) + simp [g, hω] + +variable {R : ℕ → Ω → ℝ} {f : α → ℝ} (hfc : Continuous f) {a : α} + +/-- The minimum function value converges to the global minimum. -/ +lemma tendsto_min₀ (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) + (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ + Tuple.min (fun (i : Iic n) ↦ f (A i.1 ω))) atTop (fun _ ↦ f a) := by + rw [tendstoInMeasure_iff_dist] + intro ε hε + refine image_actions_tendsto_any hfc h a ε hε |> tendsto_zero_le <| ?_ intro n refine measure_mono ?_ simp only [Set.setOf_subset_setOf] intro ω hω rw [← Tuple.argmin_spec] - set j := Tuple.argmin (fun (i : Iic n) ↦ dist (A i ω) a) + set j := Tuple.argmin (fun (i : Iic n) ↦ dist (f (A i ω)) (f a)) have : dist (Tuple.min fun (i : Iic n) ↦ f (A i ω)) (f a) ≤ dist (f (A j ω)) (f a) := by rw [← Tuple.argmin_spec] set k := Tuple.argmin (fun (i : Iic n) ↦ f (A i ω)) @@ -170,33 +198,29 @@ lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) grind have := hω.trans this by_contra! h_contra - specialize hfc (A j ω) h_contra linarith -lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) - (hf_max : ∀ x, f x ≤ f a) : - TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by +/-- The minimum reward converges to the global minimum value. -/ +lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) + (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ + Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by + refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_min₀ hfc h hf_min + filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h Tuple.min] with ω hω + rw [← hω] + +/-- The maximum function value converges to the global maximum. -/ +lemma tendsto_max₀ (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) + (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ + Tuple.max (fun (i : Iic n) ↦ f (A i.1 ω))) atTop (fun _ ↦ f a) := by rw [tendstoInMeasure_iff_dist] intro ε hε - have hf := hfc.measurable - rw [Metric.continuous_iff] at hfc - obtain ⟨δ, hδ, hfc⟩ := hfc a ε hε - have (n : ℕ) : P {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R i x) (f a)} = - P {x | ε ≤ dist (Tuple.max fun (i : Iic n) ↦ f (A i x)) (f a)} := by - refine measure_congr ?_ - filter_upwards [IsAlgEnvSeq.reward_eq_evals_actions_comp hf h Tuple.max] with ω hω - simp only [eq_iff_iff] - change ε ≤ dist (Tuple.max fun (i : Iic n) ↦ R (↑i) ω) (f a) ↔ - ε ≤ dist (Tuple.max fun (i : Iic n) ↦ f (A ↑i ω)) (f a) - rw [hω] - simp_rw [this] - refine tendsto_any hf h a δ hδ |> tendsto_zero_le <| ?_ + refine image_actions_tendsto_any hfc h a ε hε |> tendsto_zero_le <| ?_ intro n refine measure_mono ?_ simp only [Set.setOf_subset_setOf] intro ω hω rw [← Tuple.argmin_spec] - set j := Tuple.argmin (fun (i : Iic n) ↦ dist (A i ω) a) + set j := Tuple.argmin (fun (i : Iic n) ↦ dist (f (A i ω)) (f a)) have : dist (Tuple.max fun (i : Iic n) ↦ f (A i ω)) (f a) ≤ dist (f (A j ω)) (f a) := by rw [← Tuple.argmax_spec] set k := Tuple.argmax (fun (i : Iic n) ↦ f (A i ω)) @@ -207,9 +231,14 @@ lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) grind have := hω.trans this by_contra! h_contra - specialize hfc (A j ω) h_contra linarith -end Real +/-- The maximum reward converges to the global maximum value. -/ +lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) + (hf_max : ∀ x, f x ≤ f a) : + TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by + refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_max₀ hfc h hf_max + filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h Tuple.max] with ω hω + rw [← hω] end PRS diff --git a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean index aac60b4d..be9436bd 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean @@ -10,25 +10,31 @@ import LeanMachineLearning.SequentialLearning.Algorithm open MeasureTheory ProbabilityTheory Finset NNReal Learning - /-! # RankOpt: A Ranking Approach to Global Optimization + Implementation of the _RankOpt_ algorithm -[(_A Ranking Approach to Global Optimization_, -Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) -defined on a measurable subset of a Euclidean space, with finite and non-zero measure. -The algorithm samples from the uniform distribution on the set of potential maximizers of -the function at each iteration. +[(_A Ranking Approach to Global Optimization_, Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) +defined on a measurable space. The algorithm samples from an arbitrary probability measure +on the set of potential maximizers of the function at each iteration. + +## Main definitions + +* `RankRule`: A rank rule is a measurable function that compares pairs of points. + It returns 1 if the first point is ranked higher, -1 if lower, and 0 if equal. +* `potential_max`: The set of potential maximizers for the RankOpt algorithm. +* `potential_max_kernel`: The Markov kernel that samples from the set of potential maximizers + according to a given measure `μ`. +* `RankOpt`: The RankOpt algorithm that samples from the set of potential maximizers using a given + probability measure at each iteration. -/ section RankRule /-- A rank rule is a measurable function that compares pairs of points. It returns 1 if the first point is ranked higher, -1 if lower, and 0 if equal. -/ --- ANCHOR: RankRule def RankRule (α : Type*) [MeasurableSpace α] := {f : α → α → ({-1, 0, 1} : Set ℝ) // Measurable <| Function.uncurry f} --- ANCHOR_END: RankRule end RankRule @@ -51,7 +57,7 @@ noncomputable abbrev rindicator (r₁ r₂ : ℝ) := if r₁ = r₂ then (1 : Measures the agreement between a candidate rule `r` and the rankings induced by the observed function values on all pairs of data points, normalized by the number of pairs. -/ noncomputable def ranking_loss (r : RankRule α) := - 2 * (n * (n + 1) : ℝ)⁻¹ * ∑ ij ∈ {(i, j) : Finset.Iic n × Finset.Iic n | i ≤ j}, + 2 * (n * (n + 1) : ℝ)⁻¹ * ∑ ij ∈ {(i, j) : Iic n × Iic n | i ≤ j}, rindicator (r.1 (data ij.1).1 (data ij.2).1) (ranking_data (data ij.1).2 (data ij.2).2) /-- The point in the observed data with the maximum function value. -/ @@ -112,7 +118,7 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. exact h u rfl rw [this] refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) - exact measurableSet_eq_fun (by fun_prop) measurable_const + refine measurableSet_eq_fun (by fun_prop) measurable_const lemma measurable_potential_max_inter {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) {s : Set α} (hs : MeasurableSet s) : @@ -122,9 +128,7 @@ lemma measurable_potential_max_inter {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Co (measurableSet_potential_max_prod h𝓡).inter (measurableSet_preimage measurable_snd hs) exact measurable_measure_prodMk_left hE_meas -/-- Markov kernel that samples uniformly from the set of potential maximizers. -This kernel forms the core sampling strategy of RankOpt: at each iteration, given the observed -data, it samples the next query point uniformly from `potential_max`. -/ +/-- Markov kernel sampling from the set of potential maximizers according to μ. -/ noncomputable def potential_max_kernel {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) : Kernel (Iic n → α × β) α := by refine ⟨fun data ↦ cond μ <| potential_max data 𝓡, ?_⟩ @@ -150,9 +154,9 @@ variable {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) /-- The RankOpt algorithm for global optimization. This algorithm uses a ranking approach to optimize an unknown function. It maintains a hypothesis -class `𝓡` of ranking rules. At each iteration, it samples from the set of points that could be -optimal according to ranking rules consistent with the observed data -[(Malherbe et al., 2017)](https://arxiv.org/pdf/1603.04381). -/ +class `𝓡` of ranking rules. It starts with an arbitrary probability measure `μ` as initial +distribution and samples from the set of points that could be optimal according to ranking rules +consistent with the observed data [(Malherbe et al., 2017)](https://arxiv.org/pdf/1603.04381). -/ noncomputable def RankOpt : Algorithm α β where policy _ := potential_max_kernel μ h𝓡 p0 := μ diff --git a/LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean b/LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean index 89cd4dcb..c2f413d7 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean @@ -14,8 +14,10 @@ namespace Tuple variable {ι α : Type*} [LinearOrder α] [Fintype ι] [Nonempty ι] (f : ι → α) +/-- The maximum value of a tuple. -/ abbrev max : α := univ.sup' (by simp) f +/-- The minimum value of a tuple. -/ abbrev min : α := univ.inf' (by simp) f lemma le_max (x : ι) : f x ≤ max f := by @@ -41,6 +43,7 @@ lemma exists_argmax : ∃ i, u i = max u := by intro j hj exact hi ⟨j, mem_Iic.mpr hj⟩ (by simp) +/-- The index of the maximum value of a tuple. -/ noncomputable def argmax := (exists_argmax u).choose lemma argmax_spec : u (argmax u) = max u := @@ -61,6 +64,7 @@ lemma exists_argmin : ∃ i, u i = min u := by · simp only [inf'_le_iff, univ_eq_attach, mem_attach, true_and, Subtype.exists, mem_Iic] exact ⟨i, by grind, le_rfl⟩ +/-- The index of the minimum value of a tuple. -/ noncomputable def argmin := (exists_argmin u).choose lemma argmin_spec : u (argmin u) = min u := diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index d3241ff7..db714886 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -18,6 +18,10 @@ the chosen action. * `evalEnv hf`: A stationary environment where the reward is given by a deterministic kernel that evaluates a fixed measurable function `f` at the chosen action. +## Main statements + +* `reward_ae_eq_evals_actions`: For almost all `ω`, the reward at time `n` is equal to `f` + evaluated at the action taken at time `n`. -/ open MeasureTheory ProbabilityTheory @@ -26,6 +30,8 @@ namespace Learning variable {α R : Type*} [MeasurableSpace α] [MeasurableSpace R] +/-- The evaluation environment where the reward is given by evaluating a fixed measurable function +`f` at the chosen action. -/ noncomputable def evalEnv {f : α → R} (hf : Measurable f) := stationaryEnv <| Kernel.deterministic f hf @@ -41,21 +47,21 @@ lemma hascondDistrib_reward_evalEnv (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n have hAn := h.measurable_A n ⟨hRn.aemeasurable, hAn.aemeasurable, h.condDistrib_reward_stationaryEnv n⟩ -lemma reward_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : +lemma reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : R' n =ᵐ[P] f ∘ A n := ae_eq_of_condDistrib_eq_deterministic hf (h.measurable_A n).aemeasurable (h.measurable_R n).aemeasurable (hascondDistrib_reward_evalEnv hf h n).condDistrib_eq -lemma reward_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : +lemma reward_ae_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : ∀ᵐ ω ∂P, ∀ n, R' n ω = f (A n ω) := by rw [ae_all_iff] intro n - exact reward_eq_eval_action hf h n + exact reward_ae_eq_eval_action hf h n open Finset in -lemma reward_eq_evals_actions_comp (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} - (g : (Iic n → R) → R) : ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by - filter_upwards [reward_eq_evals_actions hf h] with ω hω +lemma reward_ae_eq_evals_actions_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} + (g : (Iic n → R) → β) : ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by + filter_upwards [reward_ae_eq_evals_actions hf h] with ω hω simp_rw [hω] end IsAlgEnvSeq From 5ce4833549bb7a46474b0adf5037b787b06e40ea Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 20:56:11 +0200 Subject: [PATCH 20/49] docstring --- LeanMachineLearning/OptimizationAlgorithms/LIPO.lean | 3 ++- LeanMachineLearning/OptimizationAlgorithms/PRS.lean | 6 ++++-- LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean | 3 ++- LeanMachineLearning/SequentialLearning/EvaluationEnv.lean | 2 +- 4 files changed, 9 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean index 84dd6725..e4c9105b 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean @@ -13,7 +13,8 @@ open MeasureTheory ProbabilityTheory Finset NNReal Learning # LIPO: Lipschitz Optimization Implementation of the _LIPO_ algorithm -[(_Global optimization of Lipschitz functions_, Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) +[(_Global optimization of Lipschitz functions_, +Malherbe et al. 2017)](https://arxiv.org/pdf/1703.02628) defined on a measurable space with a metric. The algorithm samples from an arbitrary probability measure on the set of potential maximizers of the function at each iteration. diff --git a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean index e60b3eaf..40a2cc3a 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/PRS.lean @@ -29,8 +29,10 @@ measure at each iteration. - `hasLaw_rewards`: Each reward follows the distribution μ.map f. - `iIndep_actions`: Actions are mutually independent across time steps. - `iIndep_rewards`: Rewards are mutually independent across time steps. -- `actions_tendsto_any`: The minimum distance from sampled actions to any point in α tends to zero. -- `rewards_tendsto_any`: The minimum distance from rewards to any value tends to zero. +- `actions_tendsto_any`: The minimum distance from sampled actions to any point in α tends to zero + in measure. +- `rewards_tendsto_any`: The minimum distance from rewards to any value of `f` tends to zero in + measure. - `tendsto_min`: The minimum reward converges in measure to the global minimum value. - `tendsto_max`: The maximum reward converges in measure to the global maximum value. -/ diff --git a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean index be9436bd..7e0af523 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean +++ b/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean @@ -14,7 +14,8 @@ open MeasureTheory ProbabilityTheory Finset NNReal Learning # RankOpt: A Ranking Approach to Global Optimization Implementation of the _RankOpt_ algorithm -[(_A Ranking Approach to Global Optimization_, Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) +[(_A Ranking Approach to Global Optimization_, +Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) defined on a measurable space. The algorithm samples from an arbitrary probability measure on the set of potential maximizers of the function at each iteration. diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index db714886..47ab6131 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -16,7 +16,7 @@ the chosen action. ## Main definitions * `evalEnv hf`: A stationary environment where the reward is given by a deterministic kernel that - evaluates a fixed measurable function `f` at the chosen action. + evaluates a fixed measurable function at the chosen action. ## Main statements From bfea59775d8add57cf4bbe0b101396839b393d97 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 13 Apr 2026 20:56:36 +0200 Subject: [PATCH 21/49] lake-manifest.json --- lake-manifest.json | 255 ++++++++++++++++++++------------------------- 1 file changed, 115 insertions(+), 140 deletions(-) diff --git a/lake-manifest.json b/lake-manifest.json index 03e5ca2e..d9d0a85c 100644 --- a/lake-manifest.json +++ b/lake-manifest.json @@ -1,140 +1,115 @@ -{ - "version": "1.2.0", - "packagesDir": ".lake/packages", - "packages": [ - { - "url": "https://github.com/leanprover/subverso", - "type": "git", - "subDir": null, - "scope": "", - "rev": "52b9dfbd2658408e37ae6e8b72601ddeaaa25a0c", - "name": "subverso", - "manifestFile": "lake-manifest.json", - "inputRev": null, - "inherited": false, - "configFile": "lakefile.lean" - }, - { - "url": "https://github.com/PatrickMassot/checkdecls.git", - "type": "git", - "subDir": null, - "scope": "", - "rev": "3d425859e73fcfbef85b9638c2a91708ef4a22d4", - "name": "checkdecls", - "manifestFile": "lake-manifest.json", - "inputRev": null, - "inherited": false, - "configFile": "lakefile.lean" - }, - { - "url": "https://github.com/leanprover-community/mathlib4.git", - "type": "git", - "subDir": null, - "scope": "", - "rev": "8a178386ffc0f5fef0b77738bb5449d50efeea95", - "name": "mathlib", - "manifestFile": "lake-manifest.json", - "inputRev": "v4.29.0", - "inherited": false, - "configFile": "lakefile.lean" - }, - { - "url": "https://github.com/leanprover-community/plausible", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "83e90935a17ca19ebe4b7893c7f7066e266f50d3", - "name": "plausible", - "manifestFile": "lake-manifest.json", - "inputRev": "main", - "inherited": true, - "configFile": "lakefile.toml" - }, - { - "url": "https://github.com/leanprover-community/LeanSearchClient", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "c5d5b8fe6e5158def25cd28eb94e4141ad97c843", - "name": "LeanSearchClient", - "manifestFile": "lake-manifest.json", - "inputRev": "main", - "inherited": true, - "configFile": "lakefile.toml" - }, - { - "url": "https://github.com/leanprover-community/import-graph", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "48d5698bc464786347c1b0d859b18f938420f060", - "name": "importGraph", - "manifestFile": "lake-manifest.json", - "inputRev": "main", - "inherited": true, - "configFile": "lakefile.toml" - }, - { - "url": "https://github.com/leanprover-community/ProofWidgets4", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "3c52dee17f0cd89c1ec14de78920d1bdaa3d26b3", - "name": "proofwidgets", - "manifestFile": "lake-manifest.json", - "inputRev": "v0.0.95", - "inherited": true, - "configFile": "lakefile.lean" - }, - { - "url": "https://github.com/leanprover-community/aesop", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "7152850e7b216a0d409701617721b6e469d34bf6", - "name": "aesop", - "manifestFile": "lake-manifest.json", - "inputRev": "v4.30.0-rc1", - "inherited": true, - "configFile": "lakefile.toml" - }, - { - "url": "https://github.com/leanprover-community/quote4", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "707efb56d0696634e9e965523a1bbe9ac6ce141d", - "name": "Qq", - "manifestFile": "lake-manifest.json", - "inputRev": "v4.30.0-rc1", - "inherited": true, - "configFile": "lakefile.toml" - }, - { - "url": "https://github.com/leanprover-community/batteries", - "type": "git", - "subDir": null, - "scope": "leanprover-community", - "rev": "756e3321fd3b02a85ffda19fef789916223e578c", - "name": "batteries", - "manifestFile": "lake-manifest.json", - "inputRev": "v4.30.0-rc1", - "inherited": true, - "configFile": "lakefile.toml" - }, - { - "url": "https://github.com/leanprover/lean4-cli", - "type": "git", - "subDir": null, - "scope": "leanprover", - "rev": "7802da01beb530bf051ab657443f9cd9bc3e1a29", - "name": "Cli", - "manifestFile": "lake-manifest.json", - "inputRev": "v4.29.0", - "inherited": true, - "configFile": "lakefile.toml" - } - ], - "name": "LeanMachineLearning", - "lakeDir": ".lake" -} +{"version": "1.1.0", + "packagesDir": ".lake/packages", + "packages": + [{"url": "https://github.com/leanprover/subverso", + "type": "git", + "subDir": null, + "scope": "", + "rev": "52b9dfbd2658408e37ae6e8b72601ddeaaa25a0c", + "name": "subverso", + "manifestFile": "lake-manifest.json", + "inputRev": null, + "inherited": false, + "configFile": "lakefile.lean"}, + {"url": "https://github.com/PatrickMassot/checkdecls.git", + "type": "git", + "subDir": null, + "scope": "", + "rev": "3d425859e73fcfbef85b9638c2a91708ef4a22d4", + "name": "checkdecls", + "manifestFile": "lake-manifest.json", + "inputRev": null, + "inherited": false, + "configFile": "lakefile.lean"}, + {"url": "https://github.com/leanprover-community/mathlib4.git", + "type": "git", + "subDir": null, + "scope": "", + "rev": "8a178386ffc0f5fef0b77738bb5449d50efeea95", + "name": "mathlib", + "manifestFile": "lake-manifest.json", + "inputRev": "v4.29.0", + "inherited": false, + "configFile": "lakefile.lean"}, + {"url": "https://github.com/leanprover-community/plausible", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "83e90935a17ca19ebe4b7893c7f7066e266f50d3", + "name": "plausible", + "manifestFile": "lake-manifest.json", + "inputRev": "main", + "inherited": true, + "configFile": "lakefile.toml"}, + {"url": "https://github.com/leanprover-community/LeanSearchClient", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "c5d5b8fe6e5158def25cd28eb94e4141ad97c843", + "name": "LeanSearchClient", + "manifestFile": "lake-manifest.json", + "inputRev": "main", + "inherited": true, + "configFile": "lakefile.toml"}, + {"url": "https://github.com/leanprover-community/import-graph", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "48d5698bc464786347c1b0d859b18f938420f060", + "name": "importGraph", + "manifestFile": "lake-manifest.json", + "inputRev": "main", + "inherited": true, + "configFile": "lakefile.toml"}, + {"url": "https://github.com/leanprover-community/ProofWidgets4", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "3c52dee17f0cd89c1ec14de78920d1bdaa3d26b3", + "name": "proofwidgets", + "manifestFile": "lake-manifest.json", + "inputRev": "v0.0.95", + "inherited": true, + "configFile": "lakefile.lean"}, + {"url": "https://github.com/leanprover-community/aesop", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "7152850e7b216a0d409701617721b6e469d34bf6", + "name": "aesop", + "manifestFile": "lake-manifest.json", + "inputRev": "master", + "inherited": true, + "configFile": "lakefile.toml"}, + {"url": "https://github.com/leanprover-community/quote4", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "707efb56d0696634e9e965523a1bbe9ac6ce141d", + "name": "Qq", + "manifestFile": "lake-manifest.json", + "inputRev": "master", + "inherited": true, + "configFile": "lakefile.toml"}, + {"url": "https://github.com/leanprover-community/batteries", + "type": "git", + "subDir": null, + "scope": "leanprover-community", + "rev": "756e3321fd3b02a85ffda19fef789916223e578c", + "name": "batteries", + "manifestFile": "lake-manifest.json", + "inputRev": "main", + "inherited": true, + "configFile": "lakefile.toml"}, + {"url": "https://github.com/leanprover/lean4-cli", + "type": "git", + "subDir": null, + "scope": "leanprover", + "rev": "7802da01beb530bf051ab657443f9cd9bc3e1a29", + "name": "Cli", + "manifestFile": "lake-manifest.json", + "inputRev": "v4.29.0", + "inherited": true, + "configFile": "lakefile.toml"}], + "name": "LeanMachineLearning", + "lakeDir": ".lake"} From 02aa9b8985ee8e8d71006f6e5228df674e9388b7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Tue, 14 Apr 2026 14:53:16 +0200 Subject: [PATCH 22/49] Move files --- .../Algorithms}/LIPO.lean | 2 +- .../Algorithms}/RankOpt.lean | 2 +- .../Algorithms}/Utils/Tuple.lean | 0 .../Algorithms/RandomSampling.lean} | 51 ++++++++++--------- 4 files changed, 30 insertions(+), 25 deletions(-) rename LeanMachineLearning/{OptimizationAlgorithms => Optimization/Algorithms}/LIPO.lean (98%) rename LeanMachineLearning/{OptimizationAlgorithms => Optimization/Algorithms}/RankOpt.lean (99%) rename LeanMachineLearning/{OptimizationAlgorithms => Optimization/Algorithms}/Utils/Tuple.lean (100%) rename LeanMachineLearning/{OptimizationAlgorithms/PRS.lean => SequentialLearning/Algorithms/RandomSampling.lean} (83%) diff --git a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean similarity index 98% rename from LeanMachineLearning/OptimizationAlgorithms/LIPO.lean rename to LeanMachineLearning/Optimization/Algorithms/LIPO.lean index e4c9105b..da9ca70f 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/LIPO.lean +++ b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean @@ -4,7 +4,7 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple import LeanMachineLearning.SequentialLearning.Algorithm open MeasureTheory ProbabilityTheory Finset NNReal Learning diff --git a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean similarity index 99% rename from LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean rename to LeanMachineLearning/Optimization/Algorithms/RankOpt.lean index 7e0af523..ef9ddc82 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/RankOpt.lean +++ b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean @@ -4,7 +4,7 @@ Released under Apache 2.0 license as described in the file LICENSE. Authors: Gaëtan Serré -/ -import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple import LeanMachineLearning.SequentialLearning.Algorithm diff --git a/LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean similarity index 100% rename from LeanMachineLearning/OptimizationAlgorithms/Utils/Tuple.lean rename to LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean diff --git a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean similarity index 83% rename from LeanMachineLearning/OptimizationAlgorithms/PRS.lean rename to LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index 40a2cc3a..328b0a8a 100644 --- a/LeanMachineLearning/OptimizationAlgorithms/PRS.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -6,7 +6,7 @@ Authors: Gaëtan Serré import LeanMachineLearning.ForMathlib.ENNReal import LeanMachineLearning.ForMathlib.IndepFun -import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple import LeanMachineLearning.SequentialLearning.EvaluationEnv open MeasureTheory ProbabilityTheory Learning Finset ENNReal Filter @@ -14,16 +14,20 @@ open MeasureTheory ProbabilityTheory Learning Finset ENNReal Filter open scoped Topology /-! -# PRS: Pure Random Search +# Random Sampling -Implementation of the _Pure Random Search_ algorithm, which samples from a fixed probability +Implementation of the _Random Sampling_ algorithm, which samples from a fixed probability measure at each iteration. ## Main definitions -* `PRS`: The pure random search algorithm that samples from a fixed distribution at each iteration. +* `randomSampling`: The random sampling algorithm that samples from a fixed distribution at +each iteration. -## Main results +## Main statements + +The main results about the random sampling algorithm are stated using the `evalEnv` evaluation +environment, which rewards actions using a measurable function `f`. - `hasLaw_actions`: Each action follows the distribution μ. - `hasLaw_rewards`: Each reward follows the distribution μ.map f. @@ -44,16 +48,17 @@ variable {α β Ω : Type*} [MeasurableSpace α] [MeasurableSpace β] [StandardB open Set in /-- The Pure Random Search algorithm. -/ @[simps] -noncomputable def PRS (μ : Measure α) [IsProbabilityMeasure μ] : Algorithm α β where +noncomputable def randomSampling (μ : Measure α) [IsProbabilityMeasure μ] : Algorithm α β where policy _ := Kernel.const _ μ p0 := μ -namespace PRS +namespace randomSampling variable {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) /-- Each action follows the distribution μ. -/ -lemma hasLaw_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : HasLaw (A n) μ P := by +lemma hasLaw_actions (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n : ℕ) : + HasLaw (A n) μ P := by by_cases hn : n = 0 · rw [hn] exact h.hasLaw_action_zero @@ -62,7 +67,7 @@ lemma hasLaw_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : H exact hasLaw_of_hasCondDistrib_const <| h.hasCondDistrib_action k /-- Each reward follows the distribution μ.map f. -/ -lemma hasLaw_rewards (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : +lemma hasLaw_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n : ℕ) : HasLaw (R n) (μ.map f) P := by refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_ae_eq_eval_action hf h n) have hA := h.measurable_A n @@ -70,13 +75,13 @@ lemma hasLaw_rewards (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (n : ℕ) : rw [← Measure.map_map hf hA, (hasLaw_actions hf h n).map_eq] /-- Actions are mutually independent. -/ -lemma iIndep_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : +lemma iIndep_actions (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : iIndepFun A P := by have hA := h.measurable_A rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)] intro n have condDistrib_eq := (h.hasCondDistrib_action n).condDistrib_eq - simp only [PRS_policy] at condDistrib_eq + simp only [randomSampling_policy] at condDistrib_eq have law_eq := (hasLaw_actions hf h (n + 1)).map_eq rw [← law_eq, ← indepFun_iff_condDistrib_eq_const ?_ (by fun_prop)] at condDistrib_eq · have meas_fst : Measurable (fun (f : Iic n → α × β) ↦ (fun i ↦ (f i).1)) := by @@ -85,7 +90,7 @@ lemma iIndep_actions (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : · exact (IsAlgEnvSeq.measurable_hist (h.measurable_A) (h.measurable_R) n).aemeasurable /-- Rewards are mutually independent. -/ -lemma iIndep_rewards (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) : +lemma iIndep_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : iIndepFun R P := have (n : ℕ) : f ∘ A n =ᵐ[P] R n := (IsAlgEnvSeq.reward_ae_eq_eval_action hf h n).symm @@ -95,10 +100,10 @@ variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpa [μ.IsOpenPosMeasure] /-- The minimum distance from sampled actions to any point tends to zero. -/ -theorem actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : α) : +theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (A j.1 x) a)}) atTop (𝓝 0) := by - set PRS_alg := PRS (β := β) μ + set randomSampling_alg := randomSampling (β := β) μ intro ε hε refine tendsto_zero_le (g := fun n ↦ P (⋂ i ∈ Iic n, {x | ε ≤ dist (A i x) a})) ?_ ?_ · have inter_prod (n : ℕ) : P (⋂ j ∈ Iic n, {x | ε ≤ dist (A j x) a}) = @@ -106,7 +111,7 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : refine iIndepSet.meas_biInter ?_ _ rw [iIndepSet_iff_meas_biInter fun i ↦ ?_] · intro s - have iIndep_actions := PRS.iIndep_actions hf h + have iIndep_actions := randomSampling.iIndep_actions hf h rw [iIndepFun_iff_measure_inter_preimage_eq_mul] at iIndep_actions have meas_dist : ∀ i ∈ s, MeasurableSet {x | ε ≤ dist x a} := by intro i hs @@ -119,7 +124,7 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : have prod_law (n : ℕ) : ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} = ∏ j ∈ Iic n, μ {x | ε ≤ dist x a} := by refine prod_congr rfl fun j hj ↦ ?_ - have hlaw (n : ℕ) : HasLaw (A n) μ P := PRS.hasLaw_actions hf h n + have hlaw (n : ℕ) : HasLaw (A n) μ P := randomSampling.hasLaw_actions hf h n rw [← (hlaw j).map_eq, P.map_apply] · simp · exact h.measurable_A j @@ -142,7 +147,7 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hf) P) (a : variable [PseudoMetricSpace β] [BorelSpace β] (hfc : Continuous f) /-- The minimum distance from image of actions to any value tends to zero. -/ -lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) (a : α) : +lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (f (A j.1 x)) (f a))}) atTop (𝓝 0) := by intro ε hε @@ -162,7 +167,7 @@ lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measu linarith /-- The minimum distance from rewards to any value tends to zero. -/ -lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) (a : α) : +lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (R j.1 x) (f a))}) atTop (𝓝 0) := by intro ε hε @@ -178,7 +183,7 @@ lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) variable {R : ℕ → Ω → ℝ} {f : α → ℝ} (hfc : Continuous f) {a : α} /-- The minimum function value converges to the global minimum. -/ -lemma tendsto_min₀ (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) +lemma tendsto_min₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ f (A i.1 ω))) atTop (fun _ ↦ f a) := by rw [tendstoInMeasure_iff_dist] @@ -203,7 +208,7 @@ lemma tendsto_min₀ (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) linarith /-- The minimum reward converges to the global minimum value. -/ -lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) +lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_min₀ hfc h hf_min @@ -211,7 +216,7 @@ lemma tendsto_min (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) rw [← hω] /-- The maximum function value converges to the global maximum. -/ -lemma tendsto_max₀ (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) +lemma tendsto_max₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ f (A i.1 ω))) atTop (fun _ ↦ f a) := by rw [tendstoInMeasure_iff_dist] @@ -236,11 +241,11 @@ lemma tendsto_max₀ (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) linarith /-- The maximum reward converges to the global maximum value. -/ -lemma tendsto_max (h : IsAlgEnvSeq A R (PRS μ) (evalEnv hfc.measurable) P) +lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_max₀ hfc h hf_max filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h Tuple.max] with ω hω rw [← hω] -end PRS +end randomSampling From 953425d050203dc19589d036078c5c5257d6460d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Tue, 14 Apr 2026 21:26:35 +0200 Subject: [PATCH 23/49] Update LeanMachineLearning.lean --- LeanMachineLearning.lean | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 7c4b1fcd..f143a4c4 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -19,11 +19,11 @@ import LeanMachineLearning.ForMathlib.MeasurableArgMax import LeanMachineLearning.ForMathlib.StandardBorel import LeanMachineLearning.ForMathlib.SubGaussian import LeanMachineLearning.ForMathlib.Traj -import LeanMachineLearning.OptimizationAlgorithms.LIPO -import LeanMachineLearning.OptimizationAlgorithms.PRS -import LeanMachineLearning.OptimizationAlgorithms.RankOpt -import LeanMachineLearning.OptimizationAlgorithms.Utils.Tuple +import LeanMachineLearning.Optimization.Algorithms.LIPO +import LeanMachineLearning.Optimization.Algorithms.RankOpt +import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple import LeanMachineLearning.SequentialLearning.Algorithm +import LeanMachineLearning.SequentialLearning.Algorithms.RandomSampling import LeanMachineLearning.SequentialLearning.Deterministic import LeanMachineLearning.SequentialLearning.EvaluationEnv import LeanMachineLearning.SequentialLearning.FiniteActions From 7eab76ede20ee6174265478afce0420df1adbd91 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Tue, 14 Apr 2026 21:34:50 +0200 Subject: [PATCH 24/49] lint --- .../SequentialLearning/Algorithms/RandomSampling.lean | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index 328b0a8a..29629a86 100644 --- a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -147,8 +147,8 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf variable [PseudoMetricSpace β] [BorelSpace β] (hfc : Continuous f) /-- The minimum distance from image of actions to any value tends to zero. -/ -lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (a : α) : - ∀ ε, 0 < ε → Tendsto (fun i => P +lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) + (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (f (A j.1 x)) (f a))}) atTop (𝓝 0) := by intro ε hε have hf := hfc.measurable @@ -167,8 +167,8 @@ lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEn linarith /-- The minimum distance from rewards to any value tends to zero. -/ -lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) (a : α) : - ∀ ε, 0 < ε → Tendsto (fun i => P +lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) + (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (R j.1 x) (f a))}) atTop (𝓝 0) := by intro ε hε convert image_actions_tendsto_any hfc h a ε hε using 2 with n From e29e25c99f3a64132a51c8bf3bc4d8a1a51f6729 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Wed, 15 Apr 2026 18:59:57 +0200 Subject: [PATCH 25/49] arxiv links --- LeanMachineLearning/Optimization/Algorithms/LIPO.lean | 2 +- LeanMachineLearning/Optimization/Algorithms/RankOpt.lean | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean index da9ca70f..11962a89 100644 --- a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean +++ b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean @@ -14,7 +14,7 @@ open MeasureTheory ProbabilityTheory Finset NNReal Learning Implementation of the _LIPO_ algorithm [(_Global optimization of Lipschitz functions_, -Malherbe et al. 2017)](https://arxiv.org/pdf/1703.02628) +Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) defined on a measurable space with a metric. The algorithm samples from an arbitrary probability measure on the set of potential maximizers of the function at each iteration. diff --git a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean index ef9ddc82..93a4cf5b 100644 --- a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean +++ b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean @@ -15,7 +15,7 @@ open MeasureTheory ProbabilityTheory Finset NNReal Learning Implementation of the _RankOpt_ algorithm [(_A Ranking Approach to Global Optimization_, -Malherbe et al. 2017)](https://arxiv.org/pdf/1603.04381) +Malherbe et al. 2017)](https://arxiv.org/abs/1603.04381) defined on a measurable space. The algorithm samples from an arbitrary probability measure on the set of potential maximizers of the function at each iteration. @@ -157,7 +157,7 @@ variable {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) This algorithm uses a ranking approach to optimize an unknown function. It maintains a hypothesis class `𝓡` of ranking rules. It starts with an arbitrary probability measure `μ` as initial distribution and samples from the set of points that could be optimal according to ranking rules -consistent with the observed data [(Malherbe et al., 2017)](https://arxiv.org/pdf/1603.04381). -/ +consistent with the observed data [(Malherbe et al., 2017)](https://arxiv.org/abs/1603.04381). -/ noncomputable def RankOpt : Algorithm α β where policy _ := potential_max_kernel μ h𝓡 p0 := μ From e18dd6f352e6bf71d51414e830dea4625de66466 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 12:22:15 +0200 Subject: [PATCH 26/49] generalize --- .../Optimization/Algorithms/Utils/Tuple.lean | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index d7b6e6d8..625ec562 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -86,20 +86,19 @@ lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] {n : exact ⟨i, hi, le_rfl⟩ · simp only [inf'_le_iff, mem_attach, neg_le_neg_iff, sup'_le_iff, forall_const, Subtype.forall, mem_Iic, true_and, Subtype.exists] - refine ⟨argmax u, by grind, ?_⟩ - intro i hi - exact le_argmax u ⟨i, mem_Iic.mpr hi⟩ + exact ⟨argmax u, by grind, fun i hi ↦ le_argmax u ⟨i, mem_Iic.mpr hi⟩⟩ -variable [MeasurableSpace α] +variable [MeasurableSpace α] [TopologicalSpace α] [BorelSpace α] [OpensMeasurableSpace α] + [SecondCountableTopology α] @[fun_prop] -lemma measurable_max : Measurable (fun (t : Iic n → ℝ) => Tuple.max t) := by +lemma measurable_max [ContinuousSup α] : Measurable (fun (t : Iic n → α) => Tuple.max t) := by have : Nonempty (Iic n) := inferInstance simp_all only [mem_Iic, nonempty_subtype] fun_prop @[fun_prop] -lemma measurable_min : Measurable (fun (t : Iic n → ℝ) => Tuple.min t) := by +lemma measurable_min [ContinuousInf α] : Measurable (fun (t : Iic n → α) => Tuple.min t) := by have : Nonempty (Iic n) := inferInstance simp_all only [mem_Iic, nonempty_subtype] fun_prop From a2b3e134dedb3c78c11fbf2d44e57487387f2683 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 20 Apr 2026 14:22:35 +0200 Subject: [PATCH 27/49] golf --- .../Optimization/Algorithms/Utils/Tuple.lean | 32 ++++--------------- 1 file changed, 7 insertions(+), 25 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index 625ec562..f5c505c5 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -22,28 +22,17 @@ abbrev max : α := univ.sup' (by simp) f /-- The minimum value of a tuple. -/ abbrev min : α := univ.inf' (by simp) f -lemma le_max (x : ι) : f x ≤ max f := by - simp only [le_sup'_iff, mem_univ, true_and] - exact ⟨x, le_rfl⟩ +lemma le_max (x : ι) : f x ≤ max f := le_sup' _ (by simp) -lemma min_le (x : ι) : min f ≤ f x := by - simp only [inf'_le_iff, mem_univ, true_and] - exact ⟨x, le_rfl⟩ +lemma min_le (x : ι) : min f ≤ f x := inf'_le _ (by simp) -instance {n : ℕ} : Nonempty (Iic n) := Nonempty.intro ⟨0, insert_eq_self.mp rfl⟩ +instance {n : ℕ} : Nonempty (Iic n) := ⟨0, insert_eq_self.mp rfl⟩ variable {n : ℕ} (u : Iic n → α) lemma exists_argmax : ∃ i, u i = max u := by - have : Nonempty (Iic n) := inferInstance - obtain ⟨i, -, hi⟩ := Finset.exists_max_image Finset.univ u (by simp) - refine ⟨i, ?_⟩ - refine le_antisymm ?_ ?_ - · simp only [le_sup'_iff, univ_eq_attach, mem_attach, true_and, Subtype.exists, mem_Iic] - exact ⟨i, by grind, le_rfl⟩ - · simp only [sup'_le_iff, univ_eq_attach, mem_attach, forall_const, Subtype.forall, mem_Iic] - intro j hj - exact hi ⟨j, mem_Iic.mpr hj⟩ (by simp) + obtain ⟨i, _, hi⟩ := Finset.exists_mem_eq_sup' (by simp : Finset.univ.Nonempty) u + exact ⟨i, hi.symm⟩ /-- The index of the maximum value of a tuple. -/ noncomputable def argmax := (exists_argmax u).choose @@ -56,15 +45,8 @@ lemma le_argmax (x : Iic n) : u x ≤ u (argmax u) := by exact le_max u x lemma exists_argmin : ∃ i, u i = min u := by - have : Nonempty (Iic n) := inferInstance - obtain ⟨i, -, hi⟩ := Finset.exists_min_image Finset.univ u (by simp) - refine ⟨i, ?_⟩ - refine le_antisymm ?_ ?_ - · simp only [le_inf'_iff, univ_eq_attach, mem_attach, forall_const, Subtype.forall, mem_Iic] - intro j hj - exact hi ⟨j, mem_Iic.mpr hj⟩ (by simp) - · simp only [inf'_le_iff, univ_eq_attach, mem_attach, true_and, Subtype.exists, mem_Iic] - exact ⟨i, by grind, le_rfl⟩ + obtain ⟨i, _, hi⟩ := Finset.exists_mem_eq_inf' (by simp : Finset.univ.Nonempty) u + exact ⟨i, hi.symm⟩ /-- The index of the minimum value of a tuple. -/ noncomputable def argmin := (exists_argmin u).choose From e265e2e478c723b47fec1dace581ba58983910a9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 14:30:21 +0200 Subject: [PATCH 28/49] `Decision` interface --- .../Optimization/Algorithms/Decision.lean | 63 +++++++++++++++++++ .../Optimization/Algorithms/LIPO.lean | 33 ++-------- .../Optimization/Algorithms/RankOpt.lean | 37 ++--------- 3 files changed, 72 insertions(+), 61 deletions(-) create mode 100644 LeanMachineLearning/Optimization/Algorithms/Decision.lean diff --git a/LeanMachineLearning/Optimization/Algorithms/Decision.lean b/LeanMachineLearning/Optimization/Algorithms/Decision.lean new file mode 100644 index 00000000..d88e7baa --- /dev/null +++ b/LeanMachineLearning/Optimization/Algorithms/Decision.lean @@ -0,0 +1,63 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ +module + +public import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple +public import LeanMachineLearning.SequentialLearning.Algorithm + +/-! +# Decision-based Optimization Algorithms + +An interface for decision-based optimization algorithms, which sample from a set of potential +maximizers using a fixed probability measure at each iteration. This module defines the `Decision` +algorithm, which relies on a user-defined set of potential maximizers and a probability measure to +sample from it. + +## Main definitions + +* `potential_max_kernel`: The Markov kernel that samples from the set of potential maximizers +according to a given measure `μ`. +* `Decision`: The Decision algorithm that starts by sampling from the initial measure `μ` and then +samples from the set of potential maximizers at each iteration using the defined kernel. +-/ + +@[expose] public section + +open MeasureTheory ProbabilityTheory Finset Learning + +variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] + (μ : Measure α) [IsProbabilityMeasure μ] {potential_max : (n : ℕ) → ((Iic n) → α × β) → Set α} + (measurableSet_potential_max_prod : + ∀ n, MeasurableSet {p : (Iic n → α × β) × α | p.2 ∈ potential_max n p.1}) {n : ℕ} + +include measurableSet_potential_max_prod in +lemma measurable_potential_max_inter {s : Set α} (hs : MeasurableSet s) : + Measurable (fun data : Iic n → α × β ↦ μ (potential_max n data ∩ s)) := by + set E := {p : (Iic n → α × β) × α | p.2 ∈ potential_max n p.1 ∩ s} + have hE_meas : MeasurableSet E := + (measurableSet_potential_max_prod n).inter + <| measurableSet_preimage measurable_snd hs + exact measurable_measure_prodMk_left hE_meas + +noncomputable def potential_max_kernel : Kernel (Iic n → α × β) α := by + refine ⟨fun data ↦ cond μ <| potential_max n data, ?_⟩ + rw [Measure.measurable_measure] + intro s hs + simp only [ProbabilityTheory.cond, Measure.smul_apply, smul_eq_mul] + refine Measurable.mul ?_ ?_ + · refine Measurable.inv ?_ + convert measurable_potential_max_inter μ measurableSet_potential_max_prod (MeasurableSet.univ) + simp [Set.inter_univ] + · simp_rw [μ.restrict_apply hs] + convert measurable_potential_max_inter μ measurableSet_potential_max_prod hs using 1 + simp [Set.inter_comm] + +variable (h : ∀ n (data : Iic n → α × β), μ (potential_max n data) ≠ 0) + +noncomputable def Decision : Algorithm α β where + policy _ := potential_max_kernel μ measurableSet_potential_max_prod + p0 := μ + h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ diff --git a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean index d04a81fe..e90c7d07 100644 --- a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean +++ b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean @@ -5,8 +5,7 @@ Authors: Gaëtan Serré -/ module -public import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple -public import LeanMachineLearning.SequentialLearning.Algorithm +public import LeanMachineLearning.Optimization.Algorithms.Decision /-! # LIPO: Lipschitz Optimization @@ -16,12 +15,11 @@ Implementation of the _LIPO_ algorithm Malherbe et al. 2017)](https://arxiv.org/abs/1703.02628) defined on a measurable space with a metric. The algorithm samples from an arbitrary probability measure on the set of potential maximizers of the function at each iteration. +It is defined as a special case of the `Decision` algorithm. ## Main definitions * `potential_max`: The set of potential maximizers for the LIPO algorithm. -* `potential_max_kernel`: The Markov kernel that samples from the set of potential maximizers - according to a given measure `μ`. * `LIPO`: The LIPO algorithm that samples from the set of potential maximizers using a given probability measure at each iteration. -/ @@ -52,27 +50,6 @@ lemma measurableSet_potential_max_prod : · fun_prop · fun_prop -lemma measurable_potential_max_inter {s : Set α} (hs : MeasurableSet s) : - Measurable (fun data : Iic n → α × ℝ ↦ μ (potential_max κ data ∩ s)) := by - set E := {p : (Iic n → α × ℝ) × α | p.2 ∈ potential_max κ p.1 ∩ s} - have hE_meas : MeasurableSet E := - (measurableSet_potential_max_prod κ).inter (measurableSet_preimage measurable_snd hs) - exact measurable_measure_prodMk_left hE_meas - -/-- Markov kernel sampling from the set of potential maximizers according to μ. -/ -noncomputable def potential_max_kernel : Kernel (Iic n → α × ℝ) α := by - refine ⟨fun data ↦ cond μ <| potential_max κ data, ?_⟩ - rw [Measure.measurable_measure] - intro s hs - simp only [ProbabilityTheory.cond, Measure.smul_apply, smul_eq_mul] - refine Measurable.mul ?_ ?_ - · refine Measurable.inv ?_ - convert measurable_potential_max_inter μ κ (MeasurableSet.univ) - simp [Set.inter_univ] - · simp_rw [μ.restrict_apply hs] - convert measurable_potential_max_inter μ κ hs using 1 - simp [Set.inter_comm] - end LIPO open LIPO @@ -86,7 +63,5 @@ This algorithm optimizes an unknown function assuming only that it has a finite constant `κ`. It starts with an arbitrary probability measure `μ` as initial distribution and iteratively samples from the set of potential maximizers, ensuring consistency and convergence to the global optimum [(Malherbe et al., 2017)](https://arxiv.org/abs/1703.02628). -/ -noncomputable def LIPO : Algorithm α ℝ where - policy _ := potential_max_kernel μ κ - p0 := μ - h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ +noncomputable def LIPO : Algorithm α ℝ := + Decision μ (fun n ↦ measurableSet_potential_max_prod (n := n) κ) h diff --git a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean index 47658377..71abf954 100644 --- a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean +++ b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean @@ -5,8 +5,7 @@ Authors: Gaëtan Serré -/ module -public import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple -public import LeanMachineLearning.SequentialLearning.Algorithm +public import LeanMachineLearning.Optimization.Algorithms.Decision /-! # RankOpt: A Ranking Approach to Global Optimization @@ -15,15 +14,14 @@ Implementation of the _RankOpt_ algorithm [(_A Ranking Approach to Global Optimization_, Malherbe et al. 2017)](https://arxiv.org/abs/1603.04381) defined on a measurable space. The algorithm samples from an arbitrary probability measure -on the set of potential maximizers of the function at each iteration. +on the set of potential maximizers of the function at each iteration. It is defined as a special +case of the `Decision` algorithm. ## Main definitions * `RankRule`: A rank rule is a measurable function that compares pairs of points. It returns 1 if the first point is ranked higher, -1 if lower, and 0 if equal. * `potential_max`: The set of potential maximizers for the RankOpt algorithm. -* `potential_max_kernel`: The Markov kernel that samples from the set of potential maximizers - according to a given measure `μ`. * `RankOpt`: The RankOpt algorithm that samples from the set of potential maximizers using a given probability measure at each iteration. -/ @@ -123,29 +121,6 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) refine measurableSet_eq_fun (by fun_prop) measurable_const -lemma measurable_potential_max_inter {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) - {s : Set α} (hs : MeasurableSet s) : - Measurable (fun data : Iic n → α × β ↦ μ (potential_max data 𝓡 ∩ s)) := by - set E := {p : (Iic n → α × β) × α | p.2 ∈ potential_max p.1 𝓡 ∩ s} - have hE_meas : MeasurableSet E := - (measurableSet_potential_max_prod h𝓡).inter (measurableSet_preimage measurable_snd hs) - exact measurable_measure_prodMk_left hE_meas - -/-- Markov kernel sampling from the set of potential maximizers according to μ. -/ -noncomputable def potential_max_kernel {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) : - Kernel (Iic n → α × β) α := by - refine ⟨fun data ↦ cond μ <| potential_max data 𝓡, ?_⟩ - rw [Measure.measurable_measure] - intro s hs - simp only [ProbabilityTheory.cond, Measure.smul_apply, smul_eq_mul] - refine Measurable.mul ?_ ?_ - · refine Measurable.inv ?_ - convert measurable_potential_max_inter (β := β) μ h𝓡 (MeasurableSet.univ) - simp [Set.inter_univ] - · simp_rw [μ.restrict_apply hs] - convert measurable_potential_max_inter (β := β) μ h𝓡 hs using 1 - simp [Set.inter_comm] - end RankOpt open RankOpt @@ -160,7 +135,5 @@ This algorithm uses a ranking approach to optimize an unknown function. It maint class `𝓡` of ranking rules. It starts with an arbitrary probability measure `μ` as initial distribution and samples from the set of points that could be optimal according to ranking rules consistent with the observed data [(Malherbe et al., 2017)](https://arxiv.org/abs/1603.04381). -/ -noncomputable def RankOpt : Algorithm α β where - policy _ := potential_max_kernel μ h𝓡 - p0 := μ - h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ +noncomputable def RankOpt : Algorithm α β := + Decision μ (fun n ↦ measurableSet_potential_max_prod (n := n) h𝓡) h From 8ac49f2d098e18a6a9353bfdf64a90cb312494f1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 14:34:51 +0200 Subject: [PATCH 29/49] LeanMachineLearning.lean --- LeanMachineLearning.lean | 65 ++++++++++++++++++++-------------------- 1 file changed, 32 insertions(+), 33 deletions(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 92b63ac9..c1f21981 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,33 +1,32 @@ -module - -public import LeanMachineLearning.Online.Bandit.ArrayProbSpace -public import LeanMachineLearning.Online.Bandit.Regret -public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure -public import LeanMachineLearning.Online.Bandit.SumRewards -public import LeanMachineLearning.Online.Bandit.Algorithms.ETC -public import LeanMachineLearning.Online.Bandit.Algorithms.UCB -public import LeanMachineLearning.Probability.Independence.CondDistrib -public import LeanMachineLearning.Probability.Independence.CondIndepFun -public import LeanMachineLearning.Probability.HasCondDistrib -public import LeanMachineLearning.Probability.Independence.IndepFun -public import LeanMachineLearning.Probability.Independence.IndepInfinitePi -public import LeanMachineLearning.Probability.Integrable -public import LeanMachineLearning.Probability.Kernel.KernelSub -public import LeanMachineLearning.MeasureTheory.Measurable -public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax -public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel -public import LeanMachineLearning.Probability.Moments.SubGaussian -public import LeanMachineLearning.Probability.Kernel.IonescuTulcea.Traj -public import LeanMachineLearning.Optimization.Algorithms.LIPO -public import LeanMachineLearning.Optimization.Algorithms.RankOpt -public import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple -public import LeanMachineLearning.Optimization.ENNReal -public import LeanMachineLearning.SequentialLearning.Algorithm -public import LeanMachineLearning.SequentialLearning.Algorithms.AuxSums -public import LeanMachineLearning.SequentialLearning.Algorithms.RandomSampling -public import LeanMachineLearning.SequentialLearning.Algorithms.RoundRobin -public import LeanMachineLearning.SequentialLearning.Deterministic -public import LeanMachineLearning.SequentialLearning.EvaluationEnv -public import LeanMachineLearning.SequentialLearning.FiniteActions -public import LeanMachineLearning.SequentialLearning.IonescuTulceaSpace -public import LeanMachineLearning.SequentialLearning.StationaryEnv +import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax +import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel +import LeanMachineLearning.MeasureTheory.Measurable +import LeanMachineLearning.Online.Bandit.Algorithms.ETC +import LeanMachineLearning.Online.Bandit.Algorithms.UCB +import LeanMachineLearning.Online.Bandit.ArrayProbSpace +import LeanMachineLearning.Online.Bandit.Regret +import LeanMachineLearning.Online.Bandit.RewardByCountMeasure +import LeanMachineLearning.Online.Bandit.SumRewards +import LeanMachineLearning.Optimization.Algorithms.Decision +import LeanMachineLearning.Optimization.Algorithms.LIPO +import LeanMachineLearning.Optimization.Algorithms.RankOpt +import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple +import LeanMachineLearning.Optimization.ENNReal +import LeanMachineLearning.Probability.HasCondDistrib +import LeanMachineLearning.Probability.Independence.CondDistrib +import LeanMachineLearning.Probability.Independence.CondIndepFun +import LeanMachineLearning.Probability.Independence.IndepFun +import LeanMachineLearning.Probability.Independence.IndepInfinitePi +import LeanMachineLearning.Probability.Integrable +import LeanMachineLearning.Probability.Kernel.IonescuTulcea.Traj +import LeanMachineLearning.Probability.Kernel.KernelSub +import LeanMachineLearning.Probability.Moments.SubGaussian +import LeanMachineLearning.SequentialLearning.Algorithm +import LeanMachineLearning.SequentialLearning.Algorithms.AuxSums +import LeanMachineLearning.SequentialLearning.Algorithms.RandomSampling +import LeanMachineLearning.SequentialLearning.Algorithms.RoundRobin +import LeanMachineLearning.SequentialLearning.Deterministic +import LeanMachineLearning.SequentialLearning.EvaluationEnv +import LeanMachineLearning.SequentialLearning.FiniteActions +import LeanMachineLearning.SequentialLearning.IonescuTulceaSpace +import LeanMachineLearning.SequentialLearning.StationaryEnv From b5866f678f3291d2820ed3ac97f8eff94ba876db Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 14:34:55 +0200 Subject: [PATCH 30/49] docstring --- LeanMachineLearning/Optimization/Algorithms/Decision.lean | 5 +++++ LeanMachineLearning/Optimization/Algorithms/LIPO.lean | 2 +- LeanMachineLearning/Optimization/Algorithms/RankOpt.lean | 2 +- 3 files changed, 7 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Decision.lean b/LeanMachineLearning/Optimization/Algorithms/Decision.lean index d88e7baa..b923a779 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Decision.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Decision.lean @@ -42,6 +42,8 @@ lemma measurable_potential_max_inter {s : Set α} (hs : MeasurableSet s) : <| measurableSet_preimage measurable_snd hs exact measurable_measure_prodMk_left hE_meas +/-- The Markov kernel that samples from the set of potential maximizers according to a given +measure `μ`. -/ noncomputable def potential_max_kernel : Kernel (Iic n → α × β) α := by refine ⟨fun data ↦ cond μ <| potential_max n data, ?_⟩ rw [Measure.measurable_measure] @@ -55,8 +57,11 @@ noncomputable def potential_max_kernel : Kernel (Iic n → α × β) α := by convert measurable_potential_max_inter μ measurableSet_potential_max_prod hs using 1 simp [Set.inter_comm] +/- We need that the set of potential maximizers has non-zero measure at each iteration, +ensuring that the algorithm can sample from it. -/ variable (h : ∀ n (data : Iic n → α × β), μ (potential_max n data) ≠ 0) +/-- The interface for decision-based optimization algorithms. -/ noncomputable def Decision : Algorithm α β where policy _ := potential_max_kernel μ measurableSet_potential_max_prod p0 := μ diff --git a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean index e90c7d07..38e4f5d3 100644 --- a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean +++ b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean @@ -54,7 +54,7 @@ end LIPO open LIPO -/- We suppose that the set of potential maximizers has non-zero measure at each iteration, +/- We need that the set of potential maximizers has non-zero measure at each iteration, ensuring that the algorithm can sample from it. -/ variable (h : ∀ n (data : Iic n → α × ℝ), μ (potential_max κ data) ≠ 0) diff --git a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean index 71abf954..331c150f 100644 --- a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean +++ b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean @@ -125,7 +125,7 @@ end RankOpt open RankOpt -/- We suppose that the set of potential maximizers has non-zero measure at each iteration, +/- We need that the set of potential maximizers has non-zero measure at each iteration, ensuring that the algorithm can sample from it. -/ variable {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡.Countable) (h : ∀ n (data : Iic n → α × β), μ (potential_max data 𝓡) ≠ 0) From 0ff63dfc6774e7ee3f8f335103d98984ccee15fb Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 20 Apr 2026 14:43:30 +0200 Subject: [PATCH 31/49] golf --- LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean | 4 ---- 1 file changed, 4 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index f5c505c5..c2536ee1 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -75,14 +75,10 @@ variable [MeasurableSpace α] [TopologicalSpace α] [BorelSpace α] [OpensMeasur @[fun_prop] lemma measurable_max [ContinuousSup α] : Measurable (fun (t : Iic n → α) => Tuple.max t) := by - have : Nonempty (Iic n) := inferInstance - simp_all only [mem_Iic, nonempty_subtype] fun_prop @[fun_prop] lemma measurable_min [ContinuousInf α] : Measurable (fun (t : Iic n → α) => Tuple.min t) := by - have : Nonempty (Iic n) := inferInstance - simp_all only [mem_Iic, nonempty_subtype] fun_prop From 38a5e926acc505c0e0ea541784a83be9291ccd83 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 14:50:21 +0200 Subject: [PATCH 32/49] golf --- .../Algorithms/RandomSampling.lean | 38 ++++++++----------- 1 file changed, 16 insertions(+), 22 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index de29273a..e66be459 100644 --- a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -198,17 +198,14 @@ lemma tendsto_min₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measu intro ω hω rw [← Tuple.argmin_spec] set j := Tuple.argmin (fun (i : Iic n) ↦ dist (f (A i ω)) (f a)) - have : dist (Tuple.min fun (i : Iic n) ↦ f (A i ω)) (f a) ≤ dist (f (A j ω)) (f a) := by - rw [← Tuple.argmin_spec] - set k := Tuple.argmin (fun (i : Iic n) ↦ f (A i ω)) - have := hf_min (A k ω) - have : f (A k ω) ≤ f (A j ω) := - Tuple.argmin_le (fun (i : Iic n) ↦ f (A i ω)) j - simp [Real.dist_eq] - grind - have := hω.trans this - by_contra! h_contra - linarith + refine hω.trans ?_ + rw [← Tuple.argmin_spec] + set k := Tuple.argmin (fun (i : Iic n) ↦ f (A i ω)) + have := hf_min (A k ω) + have : f (A k ω) ≤ f (A j ω) := + Tuple.argmin_le (fun (i : Iic n) ↦ f (A i ω)) j + simp [Real.dist_eq] + grind /-- The minimum reward converges to the global minimum value. -/ lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) @@ -231,17 +228,14 @@ lemma tendsto_max₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measu intro ω hω rw [← Tuple.argmin_spec] set j := Tuple.argmin (fun (i : Iic n) ↦ dist (f (A i ω)) (f a)) - have : dist (Tuple.max fun (i : Iic n) ↦ f (A i ω)) (f a) ≤ dist (f (A j ω)) (f a) := by - rw [← Tuple.argmax_spec] - set k := Tuple.argmax (fun (i : Iic n) ↦ f (A i ω)) - have := hf_max (A k ω) - have : f (A j ω) ≤ f (A k ω) := - Tuple.le_argmax (fun (i : Iic n) ↦ f (A i ω)) j - simp [Real.dist_eq] - grind - have := hω.trans this - by_contra! h_contra - linarith + refine hω.trans ?_ + rw [← Tuple.argmax_spec] + set k := Tuple.argmax (fun (i : Iic n) ↦ f (A i ω)) + have := hf_max (A k ω) + have : f (A j ω) ≤ f (A k ω) := + Tuple.le_argmax (fun (i : Iic n) ↦ f (A i ω)) j + simp [Real.dist_eq] + grind /-- The maximum reward converges to the global maximum value. -/ lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) From 91ce64d6de0f71764dde0fc947b10c26b8e0d707 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 15:05:58 +0200 Subject: [PATCH 33/49] `Measurable_argmax` --- .../Optimization/Algorithms/RankOpt.lean | 25 ++-------- .../Optimization/Algorithms/Utils/Tuple.lean | 48 +++++++++++++++++-- 2 files changed, 47 insertions(+), 26 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean index 331c150f..4414715b 100644 --- a/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean +++ b/LeanMachineLearning/Optimization/Algorithms/RankOpt.lean @@ -98,28 +98,9 @@ lemma measurableSet_potential_max_prod {𝓡 : Set (RankRule α)} (h𝓡 : 𝓡. suffices Measurable (fun p : (Iic n → α × β) × Iic n ↦ p.1 p.2) by fun_prop exact measurable_from_prod_countable_left fun i ↦ measurable_pi_apply i - refine h_eval.comp (Measurable.prodMk ?_ ?_) - · fun_prop - · change Measurable (fun p : Iic n → α × β ↦ Tuple.argmax (fun i ↦ (p i).2)) - suffices Measurable (fun u : Iic n → β ↦ Tuple.argmax u) by - fun_prop - refine measurable_to_countable' fun i ↦ ?_ - simp only [Set.preimage, Set.mem_singleton_iff] - let Maximizers {n : ℕ} (u : Iic n → β) : Set (Iic n) := {i | u i = Tuple.max u} - have : {u : Iic n → β | Tuple.argmax u = i} = ⋃ (S) - (hS : ∀ x, Maximizers x = S → Tuple.argmax x = i), {u | Maximizers u = S} := by - ext u - simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] - constructor - · intro hu x hx - rw [← hu] - unfold Tuple.argmax - exact Classical.choose.congr_simp hx (Tuple.exists_argmax x) - · intro h - exact h u rfl - rw [this] - refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) - refine measurableSet_eq_fun (by fun_prop) measurable_const + refine h_eval.comp (Measurable.prodMk (by fun_prop) ?_) + change Measurable (fun p : Iic n → α × β ↦ Tuple.argmax (fun i ↦ (p i).2)) + fun_prop end RankOpt diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index c2536ee1..0fc79df7 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -70,16 +70,56 @@ lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] {n : mem_Iic, true_and, Subtype.exists] exact ⟨argmax u, by grind, fun i hi ↦ le_argmax u ⟨i, mem_Iic.mpr hi⟩⟩ -variable [MeasurableSpace α] [TopologicalSpace α] [BorelSpace α] [OpensMeasurableSpace α] - [SecondCountableTopology α] +variable [MeasurableSpace α] + +variable [TopologicalSpace α] [BorelSpace α] [OpensMeasurableSpace α] [SecondCountableTopology α] @[fun_prop] -lemma measurable_max [ContinuousSup α] : Measurable (fun (t : Iic n → α) => Tuple.max t) := by +lemma measurable_max [ContinuousSup α] : Measurable (fun (t : Iic n → α) => max t) := by fun_prop @[fun_prop] -lemma measurable_min [ContinuousInf α] : Measurable (fun (t : Iic n → α) => Tuple.min t) := by +lemma measurable_min [ContinuousInf α] : Measurable (fun (t : Iic n → α) => min t) := by fun_prop +@[fun_prop] +lemma measurable_argmax [MeasurableEq α] [ContinuousSup α] : + Measurable fun (u : Iic n → α) ↦ argmax u := by + refine measurable_to_countable' fun i ↦ ?_ + simp only [Set.preimage, Set.mem_singleton_iff] + let Maximizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {i | u i = max u} + have : {u : Iic n → α | argmax u = i} = ⋃ (S) + (hS : ∀ x, Maximizers x = S → argmax x = i), {u | Maximizers u = S} := by + ext u + simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] + constructor + · intro hu x hx + rw [← hu] + exact Classical.choose.congr_simp hx (exists_argmax x) + · intro h + exact h u rfl + rw [this] + refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) + exact measurableSet_eq_fun (by fun_prop) measurable_const + +@[fun_prop] +lemma measurable_argmin [MeasurableEq α] [ContinuousInf α] : + Measurable fun (u : Iic n → α) ↦ argmin u := by + refine measurable_to_countable' fun i ↦ ?_ + simp only [Set.preimage, Set.mem_singleton_iff] + let Minimizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {i | u i = Tuple.min u} + have : {u : Iic n → α | argmin u = i} = ⋃ (S) + (hS : ∀ x, Minimizers x = S → argmin x = i), {u | Minimizers u = S} := by + ext u + simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] + constructor + · intro hu x hx + rw [← hu] + exact Classical.choose.congr_simp hx (exists_argmin x) + · intro h + exact h u rfl + rw [this] + refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) + exact measurableSet_eq_fun (by fun_prop) measurable_const end Tuple From 7fc66cd54aaccf65e9660aee9732eb7eb1106ee5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 15:17:15 +0200 Subject: [PATCH 34/49] notation --- LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index 0fc79df7..1da55300 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -87,7 +87,7 @@ lemma measurable_argmax [MeasurableEq α] [ContinuousSup α] : Measurable fun (u : Iic n → α) ↦ argmax u := by refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] - let Maximizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {i | u i = max u} + let Maximizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {j | u j = max u} have : {u : Iic n → α | argmax u = i} = ⋃ (S) (hS : ∀ x, Maximizers x = S → argmax x = i), {u | Maximizers u = S} := by ext u @@ -107,7 +107,7 @@ lemma measurable_argmin [MeasurableEq α] [ContinuousInf α] : Measurable fun (u : Iic n → α) ↦ argmin u := by refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] - let Minimizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {i | u i = Tuple.min u} + let Minimizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {j | u j = Tuple.min u} have : {u : Iic n → α | argmin u = i} = ⋃ (S) (hS : ∀ x, Minimizers x = S → argmin x = i), {u | Minimizers u = S} := by ext u From da439a40dc68c025dc04a91dfe342c2d7bf39d19 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 20 Apr 2026 15:18:57 +0200 Subject: [PATCH 35/49] generalize --- .../Optimization/Algorithms/Utils/Tuple.lean | 46 +++++++++---------- 1 file changed, 21 insertions(+), 25 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index 0fc79df7..d7b7aabd 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -28,7 +28,7 @@ lemma min_le (x : ι) : min f ≤ f x := inf'_le _ (by simp) instance {n : ℕ} : Nonempty (Iic n) := ⟨0, insert_eq_self.mp rfl⟩ -variable {n : ℕ} (u : Iic n → α) +variable {n : ℕ} (u : ι → α) lemma exists_argmax : ∃ i, u i = max u := by obtain ⟨i, _, hi⟩ := Finset.exists_mem_eq_sup' (by simp : Finset.univ.Nonempty) u @@ -40,7 +40,7 @@ noncomputable def argmax := (exists_argmax u).choose lemma argmax_spec : u (argmax u) = max u := (exists_argmax u).choose_spec -lemma le_argmax (x : Iic n) : u x ≤ u (argmax u) := by +lemma le_argmax (x : ι) : u x ≤ u (argmax u) := by rw [argmax_spec u] exact le_max u x @@ -54,41 +54,37 @@ noncomputable def argmin := (exists_argmin u).choose lemma argmin_spec : u (argmin u) = min u := (exists_argmin u).choose_spec -lemma argmin_le (x : Iic n) : u (argmin u) ≤ u x := by +lemma argmin_le (x : ι) : u (argmin u) ≤ u x := by rw [argmin_spec u] exact min_le u x -lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] {n : ℕ} (u : Iic n → α) : +lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] (u : ι → α) : -(max u) = min (-u) := by - simp only [max, univ_eq_attach, min, Pi.neg_apply] + simp only [max, min, Pi.neg_apply] refine le_antisymm ?_ ?_ - · simp only [le_inf'_iff, mem_attach, neg_le_neg_iff, le_sup'_iff, true_and, Subtype.exists, - mem_Iic, forall_const, Subtype.forall] - intro i hi - exact ⟨i, hi, le_rfl⟩ - · simp only [inf'_le_iff, mem_attach, neg_le_neg_iff, sup'_le_iff, forall_const, Subtype.forall, - mem_Iic, true_and, Subtype.exists] - exact ⟨argmax u, by grind, fun i hi ↦ le_argmax u ⟨i, mem_Iic.mpr hi⟩⟩ + · simp only [le_inf'_iff, mem_univ, neg_le_neg_iff, le_sup'_iff, true_and, forall_const] + intro i + exact ⟨i, le_rfl⟩ + · simp only [inf'_le_iff, mem_univ, neg_le_neg_iff, sup'_le_iff, forall_const, true_and] + exact ⟨argmax u, le_argmax u⟩ -variable [MeasurableSpace α] - -variable [TopologicalSpace α] [BorelSpace α] [OpensMeasurableSpace α] [SecondCountableTopology α] +variable [MeasurableSpace α] [TopologicalSpace α] [BorelSpace α] [SecondCountableTopology α] @[fun_prop] -lemma measurable_max [ContinuousSup α] : Measurable (fun (t : Iic n → α) => max t) := by +lemma measurable_max [ContinuousSup α] : Measurable (fun (t : ι → α) => max t) := by fun_prop @[fun_prop] -lemma measurable_min [ContinuousInf α] : Measurable (fun (t : Iic n → α) => min t) := by +lemma measurable_min [ContinuousInf α] : Measurable (fun (t : ι → α) => min t) := by fun_prop @[fun_prop] -lemma measurable_argmax [MeasurableEq α] [ContinuousSup α] : - Measurable fun (u : Iic n → α) ↦ argmax u := by +lemma measurable_argmax [MeasurableSpace ι] [MeasurableEq α] [ContinuousSup α] : + Measurable fun (u : ι → α) ↦ argmax u := by refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] - let Maximizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {i | u i = max u} - have : {u : Iic n → α | argmax u = i} = ⋃ (S) + let Maximizers (u : ι → α) : Set ι := {i | u i = max u} + have : {u : ι → α | argmax u = i} = ⋃ (S) (hS : ∀ x, Maximizers x = S → argmax x = i), {u | Maximizers u = S} := by ext u simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] @@ -103,12 +99,12 @@ lemma measurable_argmax [MeasurableEq α] [ContinuousSup α] : exact measurableSet_eq_fun (by fun_prop) measurable_const @[fun_prop] -lemma measurable_argmin [MeasurableEq α] [ContinuousInf α] : - Measurable fun (u : Iic n → α) ↦ argmin u := by +lemma measurable_argmin [MeasurableSpace ι] [MeasurableEq α] [ContinuousInf α] : + Measurable fun (u : ι → α) ↦ argmin u := by refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] - let Minimizers {n : ℕ} (u : Iic n → α) : Set (Iic n) := {i | u i = Tuple.min u} - have : {u : Iic n → α | argmin u = i} = ⋃ (S) + let Minimizers (u : ι → α) : Set ι := {i | u i = Tuple.min u} + have : {u : ι → α | argmin u = i} = ⋃ (S) (hS : ∀ x, Minimizers x = S → argmin x = i), {u | Minimizers u = S} := by ext u simp only [Set.mem_setOf_eq, Set.mem_iUnion, exists_prop, exists_eq_right'] From a401eb9afa6fd4d1002e9bcdbe4e6cf399d492e9 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Mon, 20 Apr 2026 15:25:49 +0200 Subject: [PATCH 36/49] golf --- .../Optimization/Algorithms/Utils/Tuple.lean | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index d7b7aabd..b9fb2de2 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -60,12 +60,10 @@ lemma argmin_le (x : ι) : u (argmin u) ≤ u x := by lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] (u : ι → α) : -(max u) = min (-u) := by - simp only [max, min, Pi.neg_apply] refine le_antisymm ?_ ?_ - · simp only [le_inf'_iff, mem_univ, neg_le_neg_iff, le_sup'_iff, true_and, forall_const] - intro i - exact ⟨i, le_rfl⟩ - · simp only [inf'_le_iff, mem_univ, neg_le_neg_iff, sup'_le_iff, forall_const, true_and] + · simp; grind + · simp only [inf'_le_iff, mem_univ, Pi.neg_apply, neg_le_neg_iff, sup'_le_iff, forall_const, + true_and] exact ⟨argmax u, le_argmax u⟩ variable [MeasurableSpace α] [TopologicalSpace α] [BorelSpace α] [SecondCountableTopology α] From 55827ff60311de48b33ef01f88617499beefafb0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 19:13:18 +0200 Subject: [PATCH 37/49] generalize `measurable_argmax` --- .../MeasureTheory/Order/Lattice.lean | 23 +++++++++++++++ .../Optimization/Algorithms/Utils/Tuple.lean | 28 ++++++++++++------- 2 files changed, 41 insertions(+), 10 deletions(-) create mode 100644 LeanMachineLearning/MeasureTheory/Order/Lattice.lean diff --git a/LeanMachineLearning/MeasureTheory/Order/Lattice.lean b/LeanMachineLearning/MeasureTheory/Order/Lattice.lean new file mode 100644 index 00000000..f4a5b703 --- /dev/null +++ b/LeanMachineLearning/MeasureTheory/Order/Lattice.lean @@ -0,0 +1,23 @@ +/- +Copyright (c) 2026 Gaëtan Serré. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Gaëtan Serré +-/ +module + +public import Mathlib.MeasureTheory.Order.Lattice + +/-! # Measurable inf of a finite set + +-/ + +@[expose] public section + +open Finset + +variable {δ α : Type*} [MeasurableSpace δ] [SemilatticeInf α] [MeasurableSpace α] [MeasurableInf₂ α] + +@[measurability] +theorem Finset.measurable_inf' {ι : Type*} {s : Finset ι} (hs : s.Nonempty) {f : ι → δ → α} + (hf : ∀ n ∈ s, Measurable (f n)) : Measurable (s.inf' hs f) := + Finset.inf'_induction hs _ (fun _f hf _g hg => hf.inf hg) fun n hn => hf n hn diff --git a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean index b9fb2de2..5cc5810d 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Utils/Tuple.lean @@ -7,6 +7,8 @@ module public import Mathlib.Analysis.Normed.Order.Lattice public import Mathlib.MeasureTheory.Constructions.BorelSpace.Basic +public import Mathlib.MeasureTheory.Order.Lattice +public import LeanMachineLearning.MeasureTheory.Order.Lattice @[expose] public section @@ -66,18 +68,17 @@ lemma neg_max_eq_min_neg [AddGroup α] [AddLeftMono α] [AddRightMono α] (u : true_and] exact ⟨argmax u, le_argmax u⟩ -variable [MeasurableSpace α] [TopologicalSpace α] [BorelSpace α] [SecondCountableTopology α] +variable [MeasurableSpace α] @[fun_prop] -lemma measurable_max [ContinuousSup α] : Measurable (fun (t : ι → α) => max t) := by - fun_prop +lemma measurable_max [MeasurableSup₂ α] : Measurable (fun (t : ι → α) => max t) := by + suffices (fun (t : ι → α) => max t) = (univ.sup' univ_nonempty fun i t => t i) by + rw [this] + exact measurable_sup' univ_nonempty (fun i _ => measurable_pi_apply i) + ext t; simp [max] @[fun_prop] -lemma measurable_min [ContinuousInf α] : Measurable (fun (t : ι → α) => min t) := by - fun_prop - -@[fun_prop] -lemma measurable_argmax [MeasurableSpace ι] [MeasurableEq α] [ContinuousSup α] : +lemma measurable_argmax [MeasurableSpace ι] [MeasurableEq α] [MeasurableSup₂ α] : Measurable fun (u : ι → α) ↦ argmax u := by refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] @@ -94,10 +95,17 @@ lemma measurable_argmax [MeasurableSpace ι] [MeasurableEq α] [ContinuousSup α exact h u rfl rw [this] refine MeasurableSet.iUnion fun S ↦ (.iUnion fun hS ↦ ?_) - exact measurableSet_eq_fun (by fun_prop) measurable_const + refine measurableSet_eq_fun (by fun_prop) measurable_const + +@[fun_prop] +lemma measurable_min [MeasurableInf₂ α] : Measurable (fun (t : ι → α) => min t) := by + suffices (fun (t : ι → α) => min t) = (univ.inf' univ_nonempty fun i t => t i) by + rw [this] + exact measurable_inf' univ_nonempty (fun i _ => measurable_pi_apply i) + ext t; simp [min] @[fun_prop] -lemma measurable_argmin [MeasurableSpace ι] [MeasurableEq α] [ContinuousInf α] : +lemma measurable_argmin [MeasurableSpace ι] [MeasurableEq α] [MeasurableInf₂ α] : Measurable fun (u : ι → α) ↦ argmin u := by refine measurable_to_countable' fun i ↦ ?_ simp only [Set.preimage, Set.mem_singleton_iff] From 90b3be2107871ae594d8bfb2e3f6accc1adeda57 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Mon, 20 Apr 2026 19:14:38 +0200 Subject: [PATCH 38/49] `LeanMachineLearning.lean` --- LeanMachineLearning.lean | 1 + 1 file changed, 1 insertion(+) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index c1f21981..d6da693d 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,6 +1,7 @@ import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel import LeanMachineLearning.MeasureTheory.Measurable +import LeanMachineLearning.MeasureTheory.Order.Lattice import LeanMachineLearning.Online.Bandit.Algorithms.ETC import LeanMachineLearning.Online.Bandit.Algorithms.UCB import LeanMachineLearning.Online.Bandit.ArrayProbSpace From 64c6e87f046d281c70cf74cdbaa0adb2dab3c1dc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Wed, 22 Apr 2026 10:35:25 +0200 Subject: [PATCH 39/49] generalization --- .../Algorithms/RandomSampling.lean | 28 +++++++++---------- .../SequentialLearning/EvaluationEnv.lean | 8 +++--- 2 files changed, 18 insertions(+), 18 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index e66be459..8e8a30b0 100644 --- a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -60,8 +60,8 @@ namespace randomSampling variable {A : ℕ → Ω → α} {R : ℕ → Ω → β} {f : α → β} (hf : Measurable f) /-- Each action follows the distribution μ. -/ -lemma hasLaw_actions (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n : ℕ) : - HasLaw (A n) μ P := by +lemma hasLaw_actions {env : Environment α β} + (h : IsAlgEnvSeq A R (randomSampling μ) env P) (n : ℕ) : HasLaw (A n) μ P := by by_cases hn : n = 0 · rw [hn] exact h.hasLaw_action_zero @@ -72,20 +72,20 @@ lemma hasLaw_actions (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n /-- Each reward follows the distribution μ.map f. -/ lemma hasLaw_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n : ℕ) : HasLaw (R n) (μ.map f) P := by - refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_ae_eq_eval_action hf h n) + refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_ae_eq_eval_action h n) have hA := h.measurable_A n refine ⟨by fun_prop, ?_⟩ - rw [← Measure.map_map hf hA, (hasLaw_actions hf h n).map_eq] + rw [← Measure.map_map hf hA, (hasLaw_actions h n).map_eq] /-- Actions are mutually independent. -/ -lemma iIndep_actions (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : - iIndepFun A P := by +lemma iIndep_actions {env : Environment α β} + (h : IsAlgEnvSeq A R (randomSampling μ) env P) : iIndepFun A P := by have hA := h.measurable_A rw [iIndepFun_nat_iff_forall_indepFun (by fun_prop)] intro n have condDistrib_eq := (h.hasCondDistrib_action n).condDistrib_eq simp only [randomSampling_policy] at condDistrib_eq - have law_eq := (hasLaw_actions hf h (n + 1)).map_eq + have law_eq := (hasLaw_actions h (n + 1)).map_eq rw [← law_eq, ← indepFun_iff_condDistrib_eq_const ?_ (by fun_prop)] at condDistrib_eq · have meas_fst : Measurable (fun (f : Iic n → α × β) ↦ (fun i ↦ (f i).1)) := by fun_prop @@ -96,8 +96,8 @@ lemma iIndep_actions (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : lemma iIndep_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : iIndepFun R P := have (n : ℕ) : f ∘ A n =ᵐ[P] R n := - (IsAlgEnvSeq.reward_ae_eq_eval_action hf h n).symm - iIndepFun.congr this <| (iIndep_actions hf h).comp _ (fun _ ↦ hf) + (IsAlgEnvSeq.reward_ae_eq_eval_action h n).symm + iIndepFun.congr this <| (iIndep_actions h).comp _ (fun _ ↦ hf) variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] [μ.IsOpenPosMeasure] @@ -114,7 +114,7 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf refine iIndepSet.meas_biInter ?_ _ rw [iIndepSet_iff_meas_biInter fun i ↦ ?_] · intro s - have iIndep_actions := randomSampling.iIndep_actions hf h + have iIndep_actions := randomSampling.iIndep_actions h rw [iIndepFun_iff_measure_inter_preimage_eq_mul] at iIndep_actions have meas_dist : ∀ i ∈ s, MeasurableSet {x | ε ≤ dist x a} := by intro i hs @@ -127,7 +127,7 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf have prod_law (n : ℕ) : ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} = ∏ j ∈ Iic n, μ {x | ε ≤ dist x a} := by refine prod_congr rfl fun j hj ↦ ?_ - have hlaw (n : ℕ) : HasLaw (A n) μ P := randomSampling.hasLaw_actions hf h n + have hlaw (n : ℕ) : HasLaw (A n) μ P := randomSampling.hasLaw_actions h n rw [← (hlaw j).map_eq, P.map_apply] · simp · exact h.measurable_A j @@ -177,7 +177,7 @@ lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc. convert image_actions_tendsto_any hfc h a ε hε using 2 with n refine measure_congr ?_ let g : ((Iic n) → β) → ℝ := fun r ↦ Tuple.min (fun i ↦ dist (r i) (f a)) - filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h g] with ω hω + filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp h g] with ω hω simp only [eq_iff_iff] change ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (R j ω) (f a)) ↔ ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (f (A j ω)) (f a)) @@ -212,7 +212,7 @@ lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurab (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_min₀ hfc h hf_min - filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h Tuple.min] with ω hω + filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp h Tuple.min] with ω hω rw [← hω] /-- The maximum function value converges to the global maximum. -/ @@ -242,7 +242,7 @@ lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurab (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_max₀ hfc h hf_max - filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp hfc.measurable h Tuple.max] with ω hω + filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp h Tuple.max] with ω hω rw [← hω] end randomSampling diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index b9b40dc0..b5356279 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -41,7 +41,7 @@ noncomputable def evalEnv {f : α → R} (hf : Measurable f) := namespace IsAlgEnvSeq variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] - {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {f : α → R} (hf : Measurable f) + {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {f : α → R} {hf : Measurable f} {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} lemma hascondDistrib_reward_evalEnv (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : @@ -53,18 +53,18 @@ lemma hascondDistrib_reward_evalEnv (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n lemma reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : R' n =ᵐ[P] f ∘ A n := ae_eq_of_condDistrib_eq_deterministic hf (h.measurable_A n).aemeasurable - (h.measurable_R n).aemeasurable (hascondDistrib_reward_evalEnv hf h n).condDistrib_eq + (h.measurable_R n).aemeasurable (hascondDistrib_reward_evalEnv h n).condDistrib_eq lemma reward_ae_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : ∀ᵐ ω ∂P, ∀ n, R' n ω = f (A n ω) := by rw [ae_all_iff] intro n - exact reward_ae_eq_eval_action hf h n + exact reward_ae_eq_eval_action h n open Finset in lemma reward_ae_eq_evals_actions_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} (g : (Iic n → R) → β) : ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by - filter_upwards [reward_ae_eq_evals_actions hf h] with ω hω + filter_upwards [reward_ae_eq_evals_actions h] with ω hω simp_rw [hω] end IsAlgEnvSeq From 8a1da6ce2d8520b4f9a5495a9de378f1fd976598 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 24 Apr 2026 17:22:59 +0200 Subject: [PATCH 40/49] `EvalEnv` namespace --- .../SequentialLearning/Algorithms/RandomSampling.lean | 10 +++++----- .../SequentialLearning/EvaluationEnv.lean | 4 ++-- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index 8e8a30b0..b3477e62 100644 --- a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -72,7 +72,7 @@ lemma hasLaw_actions {env : Environment α β} /-- Each reward follows the distribution μ.map f. -/ lemma hasLaw_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n : ℕ) : HasLaw (R n) (μ.map f) P := by - refine HasLaw.congr ?_ (IsAlgEnvSeq.reward_ae_eq_eval_action h n) + refine HasLaw.congr ?_ (EvalEnv.reward_ae_eq_eval_action h n) have hA := h.measurable_A n refine ⟨by fun_prop, ?_⟩ rw [← Measure.map_map hf hA, (hasLaw_actions h n).map_eq] @@ -96,7 +96,7 @@ lemma iIndep_actions {env : Environment α β} lemma iIndep_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : iIndepFun R P := have (n : ℕ) : f ∘ A n =ᵐ[P] R n := - (IsAlgEnvSeq.reward_ae_eq_eval_action h n).symm + (EvalEnv.reward_ae_eq_eval_action h n).symm iIndepFun.congr this <| (iIndep_actions h).comp _ (fun _ ↦ hf) variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] @@ -177,7 +177,7 @@ lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc. convert image_actions_tendsto_any hfc h a ε hε using 2 with n refine measure_congr ?_ let g : ((Iic n) → β) → ℝ := fun r ↦ Tuple.min (fun i ↦ dist (r i) (f a)) - filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp h g] with ω hω + filter_upwards [EvalEnv.reward_ae_eq_evals_actions_comp h g] with ω hω simp only [eq_iff_iff] change ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (R j ω) (f a)) ↔ ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (f (A j ω)) (f a)) @@ -212,7 +212,7 @@ lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurab (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_min₀ hfc h hf_min - filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp h Tuple.min] with ω hω + filter_upwards [EvalEnv.reward_ae_eq_evals_actions_comp h Tuple.min] with ω hω rw [← hω] /-- The maximum function value converges to the global maximum. -/ @@ -242,7 +242,7 @@ lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurab (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_max₀ hfc h hf_max - filter_upwards [IsAlgEnvSeq.reward_ae_eq_evals_actions_comp h Tuple.max] with ω hω + filter_upwards [EvalEnv.reward_ae_eq_evals_actions_comp h Tuple.max] with ω hω rw [← hω] end randomSampling diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index b5356279..9d2a8c93 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -38,7 +38,7 @@ variable {α R : Type*} [MeasurableSpace α] [MeasurableSpace R] noncomputable def evalEnv {f : α → R} (hf : Measurable f) := stationaryEnv <| Kernel.deterministic f hf -namespace IsAlgEnvSeq +namespace EvalEnv variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty R] {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {f : α → R} {hf : Measurable f} @@ -67,6 +67,6 @@ lemma reward_ae_eq_evals_actions_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (ev filter_upwards [reward_ae_eq_evals_actions h] with ω hω simp_rw [hω] -end IsAlgEnvSeq +end EvalEnv end Learning From b0f3338f6e55a4c509b1d9ce39668aa11136183c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 24 Apr 2026 17:23:17 +0200 Subject: [PATCH 41/49] `hascondDistrib_reward` --- LeanMachineLearning/SequentialLearning/EvaluationEnv.lean | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index 9d2a8c93..1e162406 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -44,7 +44,7 @@ variable [StandardBorelSpace α] [Nonempty α] [StandardBorelSpace R] [Nonempty {Ω : Type*} {mΩ : MeasurableSpace Ω} {alg : Algorithm α R} {f : α → R} {hf : Measurable f} {P : Measure Ω} [IsProbabilityMeasure P] {A : ℕ → Ω → α} {R' : ℕ → Ω → R} -lemma hascondDistrib_reward_evalEnv (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : +lemma hascondDistrib_reward (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : HasCondDistrib (R' n) (A n) (Kernel.deterministic f hf) P := have hRn := h.measurable_R n have hAn := h.measurable_A n @@ -53,7 +53,7 @@ lemma hascondDistrib_reward_evalEnv (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n lemma reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ℕ) : R' n =ᵐ[P] f ∘ A n := ae_eq_of_condDistrib_eq_deterministic hf (h.measurable_A n).aemeasurable - (h.measurable_R n).aemeasurable (hascondDistrib_reward_evalEnv h n).condDistrib_eq + (h.measurable_R n).aemeasurable (hascondDistrib_reward h n).condDistrib_eq lemma reward_ae_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : ∀ᵐ ω ∂P, ∀ n, R' n ω = f (A n ω) := by From f514d67a3e38991510506b8dbef99d17d0ca0ed0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 24 Apr 2026 17:23:48 +0200 Subject: [PATCH 42/49] `forall_reward_ae_eq_eval_action` --- LeanMachineLearning/SequentialLearning/EvaluationEnv.lean | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index 1e162406..34fd0dc6 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -55,7 +55,7 @@ lemma reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) (n : ae_eq_of_condDistrib_eq_deterministic hf (h.measurable_A n).aemeasurable (h.measurable_R n).aemeasurable (hascondDistrib_reward h n).condDistrib_eq -lemma reward_ae_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : +lemma forall_reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : ∀ᵐ ω ∂P, ∀ n, R' n ω = f (A n ω) := by rw [ae_all_iff] intro n @@ -64,7 +64,7 @@ lemma reward_ae_eq_evals_actions (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) : open Finset in lemma reward_ae_eq_evals_actions_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} (g : (Iic n → R) → β) : ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by - filter_upwards [reward_ae_eq_evals_actions h] with ω hω + filter_upwards [forall_reward_ae_eq_eval_action h] with ω hω simp_rw [hω] end EvalEnv From 8c12e7aa21b33f0191f24d42d35d25ec5e748c2a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 24 Apr 2026 17:24:08 +0200 Subject: [PATCH 43/49] `reward_ae_eq_eval_action_comp` --- LeanMachineLearning/SequentialLearning/EvaluationEnv.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean index 34fd0dc6..5feb7db3 100644 --- a/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean +++ b/LeanMachineLearning/SequentialLearning/EvaluationEnv.lean @@ -62,7 +62,7 @@ lemma forall_reward_ae_eq_eval_action (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) exact reward_ae_eq_eval_action h n open Finset in -lemma reward_ae_eq_evals_actions_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} +lemma reward_ae_eq_eval_action_comp {β : Type*} (h : IsAlgEnvSeq A R' alg (evalEnv hf) P) {n : ℕ} (g : (Iic n → R) → β) : ∀ᵐ ω ∂P, g (fun i ↦ R' i ω) = g (fun i ↦ f (A i ω)) := by filter_upwards [forall_reward_ae_eq_eval_action h] with ω hω simp_rw [hω] From 9d55bdba3ce2c112a4560f39d32e49c804dcc905 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 24 Apr 2026 17:24:40 +0200 Subject: [PATCH 44/49] refactor --- .../SequentialLearning/Algorithms/RandomSampling.lean | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index b3477e62..714c7f49 100644 --- a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -177,7 +177,7 @@ lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc. convert image_actions_tendsto_any hfc h a ε hε using 2 with n refine measure_congr ?_ let g : ((Iic n) → β) → ℝ := fun r ↦ Tuple.min (fun i ↦ dist (r i) (f a)) - filter_upwards [EvalEnv.reward_ae_eq_evals_actions_comp h g] with ω hω + filter_upwards [EvalEnv.reward_ae_eq_eval_action_comp h g] with ω hω simp only [eq_iff_iff] change ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (R j ω) (f a)) ↔ ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (f (A j ω)) (f a)) @@ -212,7 +212,7 @@ lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurab (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_min₀ hfc h hf_min - filter_upwards [EvalEnv.reward_ae_eq_evals_actions_comp h Tuple.min] with ω hω + filter_upwards [EvalEnv.reward_ae_eq_eval_action_comp h Tuple.min] with ω hω rw [← hω] /-- The maximum function value converges to the global maximum. -/ @@ -242,7 +242,7 @@ lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurab (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_max₀ hfc h hf_max - filter_upwards [EvalEnv.reward_ae_eq_evals_actions_comp h Tuple.max] with ω hω + filter_upwards [EvalEnv.reward_ae_eq_eval_action_comp h Tuple.max] with ω hω rw [← hω] end randomSampling From 085cf30fc4530a8bd0b8fdf1be2e0a7a59287509 Mon Sep 17 00:00:00 2001 From: Remy Degenne Date: Sat, 9 May 2026 15:52:24 +0200 Subject: [PATCH 45/49] fix --- LeanMachineLearning.lean | 6 ++ .../Algorithms/RandomSampling.lean | 72 ++++++++++--------- 2 files changed, 44 insertions(+), 34 deletions(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 7d539fc3..d7965792 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -3,12 +3,18 @@ module -- shake: keep-all public import LeanMachineLearning.MeasureTheory.Constructions.BorelSpace.MeasurableArgMax public import LeanMachineLearning.MeasureTheory.Constructions.Polish.StandardBorel public import LeanMachineLearning.MeasureTheory.Measurable +public import LeanMachineLearning.MeasureTheory.Order.Lattice public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.UCB public import LeanMachineLearning.Online.Bandit.ArrayProbSpace public import LeanMachineLearning.Online.Bandit.Regret public import LeanMachineLearning.Online.Bandit.RewardByCountMeasure public import LeanMachineLearning.Online.Bandit.SumRewards +public import LeanMachineLearning.Optimization.Algorithms.Decision +public import LeanMachineLearning.Optimization.Algorithms.LIPO +public import LeanMachineLearning.Optimization.Algorithms.RankOpt +public import LeanMachineLearning.Optimization.Algorithms.Utils.Tuple +public import LeanMachineLearning.Optimization.ENNReal public import LeanMachineLearning.Probability.HasCondDistrib public import LeanMachineLearning.Probability.Independence.CondDistrib public import LeanMachineLearning.Probability.Independence.CondIndepFun diff --git a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean index 78d4df41..2f1b6a83 100644 --- a/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean +++ b/LeanMachineLearning/SequentialLearning/Algorithms/RandomSampling.lean @@ -61,6 +61,7 @@ noncomputable def randomSampling (μ : Measure 𝓐) [IsProbabilityMeasure μ] : namespace randomSampling variable {A : ℕ → Ω → 𝓐} {Y : ℕ → Ω → 𝓨} {env : Environment 𝓐 𝓨} + {f : 𝓐 → 𝓨} {hf : Measurable f} /-- Each action follows the distribution μ. -/ lemma hasLaw_action (h : IsAlgEnvSeq A Y (randomSampling μ) env P) (n : ℕ) : @@ -73,12 +74,12 @@ lemma hasLaw_action (h : IsAlgEnvSeq A Y (randomSampling μ) env P) (n : ℕ) : exact hasLaw_of_hasCondDistrib_const <| h.hasCondDistrib_action k /-- Each reward follows the distribution μ.map f. -/ -lemma hasLaw_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (n : ℕ) : - HasLaw (R n) (μ.map f) P := by - refine HasLaw.congr ?_ (EvalEnv.reward_ae_eq_eval_action h n) - have hA := h.measurable_A n +lemma hasLaw_rewards (h : IsAlgEnvSeq A Y (randomSampling μ) (evalEnv f hf) P) (n : ℕ) : + HasLaw (Y n) (μ.map f) P := by + refine HasLaw.congr ?_ (feedback_evalEnv_ae_eq_eval_action h n) + have hA := h.measurable_action n refine ⟨by fun_prop, ?_⟩ - rw [← Measure.map_map hf hA, (hasLaw_actions h n).map_eq] + rw [← Measure.map_map hf hA, (hasLaw_action h n).map_eq] /-- Actions are mutually independent. -/ lemma iIndep_action (h : IsAlgEnvSeq A Y (randomSampling μ) env P) : @@ -96,20 +97,20 @@ lemma iIndep_action (h : IsAlgEnvSeq A Y (randomSampling μ) env P) : · exact (IsAlgEnvSeq.measurable_hist (h.measurable_action) (h.measurable_feedback) n).aemeasurable /-- Rewards are mutually independent. -/ -lemma iIndep_rewards (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) : - iIndepFun R P := - have (n : ℕ) : f ∘ A n =ᵐ[P] R n := - (EvalEnv.reward_ae_eq_eval_action h n).symm - iIndepFun.congr this <| (iIndep_actions h).comp _ (fun _ ↦ hf) +lemma iIndep_rewards (h : IsAlgEnvSeq A Y (randomSampling μ) (evalEnv f hf) P) : + iIndepFun Y P := + have (n : ℕ) : f ∘ A n =ᵐ[P] Y n := + (feedback_evalEnv_ae_eq_eval_action h n).symm + iIndepFun.congr this <| (iIndep_action h).comp _ (fun _ ↦ hf) -variable [PseudoMetricSpace α] [SecondCountableTopology α] [OpensMeasurableSpace α] +variable [PseudoMetricSpace 𝓐] [SecondCountableTopology 𝓐] [OpensMeasurableSpace 𝓐] [μ.IsOpenPosMeasure] /-- The minimum distance from sampled actions to any point tends to zero. -/ -theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf) P) (a : α) : +theorem actions_tendsto_any (h : IsAlgEnvSeq A Y (randomSampling μ) (evalEnv f hf) P) (a : 𝓐) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (A j.1 x) a)}) atTop (𝓝 0) := by - set randomSampling_alg := randomSampling (β := β) μ + set randomSampling_alg := randomSampling (𝓨 := 𝓨) μ intro ε hε refine tendsto_zero_le (g := fun n ↦ P (⋂ i ∈ Iic n, {x | ε ≤ dist (A i x) a})) ?_ ?_ · have inter_prod (n : ℕ) : P (⋂ j ∈ Iic n, {x | ε ≤ dist (A j x) a}) = @@ -117,23 +118,23 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf refine iIndepSet.meas_biInter ?_ _ rw [iIndepSet_iff_meas_biInter fun i ↦ ?_] · intro s - have iIndep_actions := randomSampling.iIndep_actions h + have iIndep_actions := randomSampling.iIndep_action h rw [iIndepFun_iff_measure_inter_preimage_eq_mul] at iIndep_actions have meas_dist : ∀ i ∈ s, MeasurableSet {x | ε ≤ dist x a} := by intro i hs measurability specialize iIndep_actions s meas_dist simpa only [Set.preimage] using iIndep_actions - · have hAi := h.measurable_A i + · have hAi := h.measurable_action i measurability simp_rw [inter_prod] have prod_law (n : ℕ) : ∏ j ∈ Iic n, P {x | ε ≤ dist (A j x) a} = ∏ j ∈ Iic n, μ {x | ε ≤ dist x a} := by refine prod_congr rfl fun j hj ↦ ?_ - have hlaw (n : ℕ) : HasLaw (A n) μ P := randomSampling.hasLaw_actions h n + have hlaw (n : ℕ) : HasLaw (A n) μ P := randomSampling.hasLaw_action h n rw [← (hlaw j).map_eq, P.map_apply] · simp - · exact h.measurable_A j + · exact h.measurable_action j · measurability simp_rw [prod_law] simp only [prod_const, Nat.card_Iic] @@ -150,17 +151,18 @@ theorem actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hf intro i hi ω (hω : ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (A j.1 ω) a)) simp_all only [univ_eq_attach, le_inf'_iff, mem_attach, forall_const, Subtype.forall, mem_Iic] -variable [PseudoMetricSpace β] [BorelSpace β] (hfc : Continuous f) +variable [PseudoMetricSpace 𝓨] [BorelSpace 𝓨] (hfc : Continuous f) /-- The minimum distance from image of actions to any value tends to zero. -/ -lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) - (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P +lemma image_actions_tendsto_any + (h : IsAlgEnvSeq A Y (randomSampling μ) (evalEnv f hfc.measurable) P) + (a : 𝓐) : ∀ ε, 0 < ε → Tendsto (fun i => P {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (f (A j.1 x)) (f a))}) atTop (𝓝 0) := by intro ε hε have hf := hfc.measurable rw [Metric.continuous_iff] at hfc obtain ⟨δ, hδ, hfc⟩ := hfc a ε hε - refine actions_tendsto_any hf h a δ hδ |> tendsto_zero_le <| ?_ + refine actions_tendsto_any h a δ hδ |> tendsto_zero_le <| ?_ intro n refine measure_mono ?_ simp only [Set.setOf_subset_setOf] @@ -173,23 +175,23 @@ lemma image_actions_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEn linarith /-- The minimum distance from rewards to any value tends to zero. -/ -lemma rewards_tendsto_any (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) - (a : α) : ∀ ε, 0 < ε → Tendsto (fun i => P - {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (R j.1 x) (f a))}) atTop (𝓝 0) := by +lemma rewards_tendsto_any (h : IsAlgEnvSeq A Y (randomSampling μ) (evalEnv f hfc.measurable) P) + (a : 𝓐) : ∀ ε, 0 < ε → Tendsto (fun i => P + {x | ε ≤ Tuple.min (fun (j : Iic i) ↦ dist (Y j.1 x) (f a))}) atTop (𝓝 0) := by intro ε hε convert image_actions_tendsto_any hfc h a ε hε using 2 with n refine measure_congr ?_ - let g : ((Iic n) → β) → ℝ := fun r ↦ Tuple.min (fun i ↦ dist (r i) (f a)) - filter_upwards [EvalEnv.reward_ae_eq_eval_action_comp h g] with ω hω + let g : ((Iic n) → 𝓨) → ℝ := fun r ↦ Tuple.min (fun i ↦ dist (r i) (f a)) + filter_upwards [feedback_evalEnv_ae_eq_eval_action_comp h g] with ω hω simp only [eq_iff_iff] - change ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (R j ω) (f a)) ↔ + change ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (Y j ω) (f a)) ↔ ε ≤ Tuple.min (fun (j : Iic n) ↦ dist (f (A j ω)) (f a)) simp [g, hω] -variable {R : ℕ → Ω → ℝ} {f : α → ℝ} (hfc : Continuous f) {a : α} +variable {R : ℕ → Ω → ℝ} {f : 𝓐 → ℝ} (hfc : Continuous f) {a : 𝓐} /-- The minimum function value converges to the global minimum. -/ -lemma tendsto_min₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) +lemma tendsto_min₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv f hfc.measurable) P) (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ f (A i.1 ω))) atTop (fun _ ↦ f a) := by rw [tendstoInMeasure_iff_dist] @@ -211,15 +213,15 @@ lemma tendsto_min₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measu grind /-- The minimum reward converges to the global minimum value. -/ -lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) +lemma tendsto_min (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv f hfc.measurable) P) (hf_min : ∀ x, f a ≤ f x) : TendstoInMeasure P (fun n ω ↦ Tuple.min (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_min₀ hfc h hf_min - filter_upwards [EvalEnv.reward_ae_eq_eval_action_comp h Tuple.min] with ω hω + filter_upwards [feedback_evalEnv_ae_eq_eval_action_comp h Tuple.min] with ω hω rw [← hω] /-- The maximum function value converges to the global maximum. -/ -lemma tendsto_max₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) +lemma tendsto_max₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv f hfc.measurable) P) (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ f (A i.1 ω))) atTop (fun _ ↦ f a) := by rw [tendstoInMeasure_iff_dist] @@ -241,11 +243,13 @@ lemma tendsto_max₀ (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measu grind /-- The maximum reward converges to the global maximum value. -/ -lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv hfc.measurable) P) +lemma tendsto_max (h : IsAlgEnvSeq A R (randomSampling μ) (evalEnv f hfc.measurable) P) (hf_max : ∀ x, f x ≤ f a) : TendstoInMeasure P (fun n ω ↦ Tuple.max (fun (i : Iic n) ↦ R i.1 ω)) atTop (fun _ ↦ f a) := by refine TendstoInMeasure.congr_left (fun n ↦ ?_) <| tendsto_max₀ hfc h hf_max - filter_upwards [EvalEnv.reward_ae_eq_eval_action_comp h Tuple.max] with ω hω + filter_upwards [feedback_evalEnv_ae_eq_eval_action_comp h Tuple.max] with ω hω rw [← hω] end randomSampling + +end Learning From 511261fd131ade9bd3dc44e728b654ea3ee2ef26 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 29 May 2026 18:29:47 +0200 Subject: [PATCH 46/49] docstring --- .../Optimization/Algorithms/Decision.lean | 47 ++++++++++--------- 1 file changed, 24 insertions(+), 23 deletions(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Decision.lean b/LeanMachineLearning/Optimization/Algorithms/Decision.lean index b923a779..4c01aa9b 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Decision.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Decision.lean @@ -11,17 +11,18 @@ public import LeanMachineLearning.SequentialLearning.Algorithm /-! # Decision-based Optimization Algorithms -An interface for decision-based optimization algorithms, which sample from a set of potential -maximizers using a fixed probability measure at each iteration. This module defines the `Decision` -algorithm, which relies on a user-defined set of potential maximizers and a probability measure to -sample from it. +An interface for decision-based optimization algorithms, which sample points satisfying a +user-defined decision rule at each iteration. These algorithms are defined by a sequence of +decision rules that determine from which set to sample at each iteration, based on the observed +data. The `Decision` algorithm is a special case of the `Algorithm` structure, where the Markov +kernel is defined through the decision rules. ## Main definitions -* `potential_max_kernel`: The Markov kernel that samples from the set of potential maximizers -according to a given measure `μ`. +* `decision_kernel`: The Markov kernel that samples from the decision set rules according to a +given measure `μ`. * `Decision`: The Decision algorithm that starts by sampling from the initial measure `μ` and then -samples from the set of potential maximizers at each iteration using the defined kernel. +samples points satisfying the decision rules at each iteration using the defined kernel. -/ @[expose] public section @@ -29,40 +30,40 @@ samples from the set of potential maximizers at each iteration using the defined open MeasureTheory ProbabilityTheory Finset Learning variable {α β : Type*} [MeasurableSpace α] [MeasurableSpace β] - (μ : Measure α) [IsProbabilityMeasure μ] {potential_max : (n : ℕ) → ((Iic n) → α × β) → Set α} - (measurableSet_potential_max_prod : - ∀ n, MeasurableSet {p : (Iic n → α × β) × α | p.2 ∈ potential_max n p.1}) {n : ℕ} + (μ : Measure α) [IsProbabilityMeasure μ] {decision : (n : ℕ) → ((Iic n) → α × β) → Set α} + (measurableSet_decision_prod : + ∀ n, MeasurableSet {p : (Iic n → α × β) × α | p.2 ∈ decision n p.1}) {n : ℕ} -include measurableSet_potential_max_prod in -lemma measurable_potential_max_inter {s : Set α} (hs : MeasurableSet s) : - Measurable (fun data : Iic n → α × β ↦ μ (potential_max n data ∩ s)) := by - set E := {p : (Iic n → α × β) × α | p.2 ∈ potential_max n p.1 ∩ s} +include measurableSet_decision_prod in +lemma measurable_decision_inter {s : Set α} (hs : MeasurableSet s) : + Measurable (fun data : Iic n → α × β ↦ μ (decision n data ∩ s)) := by + set E := {p : (Iic n → α × β) × α | p.2 ∈ decision n p.1 ∩ s} have hE_meas : MeasurableSet E := - (measurableSet_potential_max_prod n).inter + (measurableSet_decision_prod n).inter <| measurableSet_preimage measurable_snd hs exact measurable_measure_prodMk_left hE_meas -/-- The Markov kernel that samples from the set of potential maximizers according to a given +/-- The Markov kernel that samples from the decision set according to a given measure `μ`. -/ -noncomputable def potential_max_kernel : Kernel (Iic n → α × β) α := by - refine ⟨fun data ↦ cond μ <| potential_max n data, ?_⟩ +noncomputable def decision_kernel : Kernel (Iic n → α × β) α := by + refine ⟨fun data ↦ cond μ <| decision n data, ?_⟩ rw [Measure.measurable_measure] intro s hs simp only [ProbabilityTheory.cond, Measure.smul_apply, smul_eq_mul] refine Measurable.mul ?_ ?_ · refine Measurable.inv ?_ - convert measurable_potential_max_inter μ measurableSet_potential_max_prod (MeasurableSet.univ) + convert measurable_decision_inter μ measurableSet_decision_prod (MeasurableSet.univ) simp [Set.inter_univ] · simp_rw [μ.restrict_apply hs] - convert measurable_potential_max_inter μ measurableSet_potential_max_prod hs using 1 + convert measurable_decision_inter μ measurableSet_decision_prod hs using 1 simp [Set.inter_comm] -/- We need that the set of potential maximizers has non-zero measure at each iteration, +/- We need that the decisions has non-zero measure at each iteration, ensuring that the algorithm can sample from it. -/ -variable (h : ∀ n (data : Iic n → α × β), μ (potential_max n data) ≠ 0) +variable (h : ∀ n (data : Iic n → α × β), μ (decision n data) ≠ 0) /-- The interface for decision-based optimization algorithms. -/ noncomputable def Decision : Algorithm α β where - policy _ := potential_max_kernel μ measurableSet_potential_max_prod + policy _ := decision_kernel μ measurableSet_decision_prod p0 := μ h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ From 0cc458004025a0a29b13c05a6d540c0049a1d095 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 29 May 2026 18:41:08 +0200 Subject: [PATCH 47/49] Remove unused variable --- LeanMachineLearning/Optimization/Algorithms/LIPO.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean index 38e4f5d3..a5a3e5f0 100644 --- a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean +++ b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean @@ -64,4 +64,4 @@ constant `κ`. It starts with an arbitrary probability measure `μ` as initial d iteratively samples from the set of potential maximizers, ensuring consistency and convergence to the global optimum [(Malherbe et al., 2017)](https://arxiv.org/abs/1703.02628). -/ noncomputable def LIPO : Algorithm α ℝ := - Decision μ (fun n ↦ measurableSet_potential_max_prod (n := n) κ) h + Decision μ (fun _ ↦ measurableSet_potential_max_prod κ) h From 8ce013a6779ea1d86fc3527ed74b50a2096eba0b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 29 May 2026 18:48:21 +0200 Subject: [PATCH 48/49] Typo --- LeanMachineLearning/Optimization/Algorithms/Decision.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/Decision.lean b/LeanMachineLearning/Optimization/Algorithms/Decision.lean index 4c01aa9b..f7a60041 100644 --- a/LeanMachineLearning/Optimization/Algorithms/Decision.lean +++ b/LeanMachineLearning/Optimization/Algorithms/Decision.lean @@ -66,4 +66,4 @@ variable (h : ∀ n (data : Iic n → α × β), μ (decision n data) ≠ 0) noncomputable def Decision : Algorithm α β where policy _ := decision_kernel μ measurableSet_decision_prod p0 := μ - h_policy n := ⟨fun data => cond_isProbabilityMeasure (h n data)⟩ + h_policy n := ⟨fun data ↦ cond_isProbabilityMeasure (h n data)⟩ From dd0d1f7c5b4c6d46d92600f944c7b29b8a955d56 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Serr=C3=A9?= Date: Fri, 29 May 2026 19:26:01 +0200 Subject: [PATCH 49/49] remove useless type annotation --- LeanMachineLearning/Optimization/Algorithms/LIPO.lean | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean index a5a3e5f0..ff827d17 100644 --- a/LeanMachineLearning/Optimization/Algorithms/LIPO.lean +++ b/LeanMachineLearning/Optimization/Algorithms/LIPO.lean @@ -39,7 +39,7 @@ Given observed data points and function values, this set contains all points `x` the maximum observed value is at most the minimum Lipschitz upper bound across all observations. The upper bound at `x` from observation `i` is `f(xᵢ) + κ · d(xᵢ, x)`, where `κ` is the Lipschitz constant. -/ -def potential_max : Set α := +def potential_max := {x | Tuple.max (fun i ↦ (data i).2) ≤ Tuple.min (fun i ↦ (data i).2 + κ * dist (data i).1 x)} lemma measurableSet_potential_max_prod :