From 51d6faab4e5dc9f29856ab8e743c44f8ea34d529 Mon Sep 17 00:00:00 2001 From: Yi Yuan Date: Sat, 12 Sep 2026 11:16:28 +0800 Subject: [PATCH 1/8] Formalize the Leshno universal approximation theorem --- LeanMachineLearning.lean | 24 ++ .../ForMathlib/Algebra/Polynomial/Affine.lean | 65 +++ .../Algebra/Polynomial/Function.lean | 94 +++++ .../Calculus/ContinuousMapComposition.lean | 99 +++++ .../Analysis/Distribution/Polynomial.lean | 393 ++++++++++++++++++ .../PolynomialCharacterization.lean | 150 +++++++ .../Analysis/Distribution/TestFunction.lean | 31 ++ .../Distribution/TestFunction/Normalize.lean | 91 ++++ .../Analysis/LocallyConvex/Annihilator.lean | 95 +++++ .../Multilinear/Polarization.lean | 248 +++++++++++ .../Integral/ClosedSubmodule.lean | 95 +++++ .../Algebra/Module/FiniteDimension.lean | 45 ++ .../Topology/ContinuousMap/Dense.lean | 40 ++ .../Topology/ContinuousMap/Discrete.lean | 42 ++ .../Topology/ContinuousMap/InnerProduct.lean | 68 +++ .../Topology/ContinuousMap/Moments.lean | 192 +++++++++ .../NeuralNetwork/Shallow/Basic.lean | 125 ++++++ .../UniversalApproximation/Convolution.lean | 212 ++++++++++ .../ConvolutionSmooth.lean | 95 +++++ .../Discriminatory.lean | 60 +++ .../UniversalApproximation/Main.lean | 68 +++ .../UniversalApproximation/Nonpolynomial.lean | 45 ++ .../NonpolynomialWitness.lean | 128 ++++++ .../PolynomialObstruction.lean | 135 ++++++ .../SmoothActivation.lean | 164 ++++++++ 25 files changed, 2804 insertions(+) create mode 100644 LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean create mode 100644 LeanMachineLearning/ForMathlib/Algebra/Polynomial/Function.lean create mode 100644 LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean create mode 100644 LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean create mode 100644 LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean create mode 100644 LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction.lean create mode 100644 LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean create mode 100644 LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean create mode 100644 LeanMachineLearning/ForMathlib/LinearAlgebra/Multilinear/Polarization.lean create mode 100644 LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean create mode 100644 LeanMachineLearning/ForMathlib/Topology/Algebra/Module/FiniteDimension.lean create mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean create mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean create mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean create mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean create mode 100644 LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean create mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 4f59f7d8..84f05987 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,4 +1,13 @@ module -- shake: keep-all --deprecated_module: ignore +public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Affine +public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function +public import LeanMachineLearning.ForMathlib.Analysis.Calculus.ContinuousMapComposition +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.Polynomial +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.PolynomialCharacterization +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction.Normalize +public import LeanMachineLearning.ForMathlib.Analysis.LocallyConvex.Annihilator +public import LeanMachineLearning.ForMathlib.LinearAlgebra.Multilinear.Polarization public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.ChainRule public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.CompProd @@ -8,6 +17,7 @@ public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.M public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.Restrict public import LeanMachineLearning.ForMathlib.MeasureTheory.Measurable public import LeanMachineLearning.ForMathlib.MeasureTheory.MeasurableSpace.Embedding +public import LeanMachineLearning.ForMathlib.MeasureTheory.Integral.ClosedSubmodule public import LeanMachineLearning.ForMathlib.MeasureTheory.Measure.AbsolutelyContinuous public import LeanMachineLearning.ForMathlib.MeasureTheory.Order.Lattice public import LeanMachineLearning.ForMathlib.MeasureTheory.Order.MeasurableArg @@ -32,7 +42,21 @@ public import LeanMachineLearning.ForMathlib.Probability.Kernel.MeasurableSpace public import LeanMachineLearning.ForMathlib.Probability.Moments.SubExponential public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity +public import LeanMachineLearning.ForMathlib.Topology.Algebra.Module.FiniteDimension +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Dense +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Discrete +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.InnerProduct +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Moments public import LeanMachineLearning.ForMathlib.Topology.Instances.ENNReal.Lemmas +public import LeanMachineLearning.NeuralNetwork.Shallow.Basic +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Convolution +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.ConvolutionSmooth +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Discriminatory +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Main +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Nonpolynomial +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.NonpolynomialWitness +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.PolynomialObstruction +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.SmoothActivation public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS diff --git a/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean b/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean new file mode 100644 index 00000000..0726760d --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean @@ -0,0 +1,65 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.RingTheory.Polynomial.Basic + +/-! +# Affine changes of variables in polynomials + +This file records that composition with a polynomial of degree at most one preserves the +submodule `Polynomial.degreeLT`. In particular, this applies to affine changes of variables. +It also provides simp lemmas for evaluating the resulting polynomials. +-/ + +@[expose] public section + +namespace Polynomial + +universe u + +variable {R : Type u} + +/-- Composing with a polynomial of degree at most one preserves a strict degree bound. -/ +theorem comp_mem_degreeLT_of_natDegree_le_one [Semiring R] + {p q : R[X]} {n : ℕ} (hp : p ∈ degreeLT R n) (hq : q.natDegree ≤ 1) : + p.comp q ∈ degreeLT R n := by + rw [mem_degreeLT] at hp ⊢ + by_cases hcomp : p.comp q = 0 + · simp [hcomp] + rw [← natDegree_lt_iff_degree_lt hcomp] + calc + (p.comp q).natDegree ≤ p.natDegree * q.natDegree := natDegree_comp_le + _ ≤ p.natDegree * 1 := Nat.mul_le_mul_left _ hq + _ = p.natDegree := Nat.mul_one _ + _ < n := by + by_cases hp0 : p = 0 + · simp [hp0] at hcomp + · exact (natDegree_lt_iff_degree_lt hp0).2 hp + +/-- Composition with the affine polynomial `C a * X + C b` preserves a strict degree bound. -/ +theorem compAffineDegreeLT [Semiring R] {p : R[X]} {n : ℕ} + (hp : p ∈ degreeLT R n) (a b : R) : + p.comp (C a * X + C b) ∈ degreeLT R n := by + apply comp_mem_degreeLT_of_natDegree_le_one hp + calc + (C a * X + C b).natDegree ≤ max (C a * X).natDegree (C b).natDegree := + natDegree_add_le _ _ + _ ≤ 1 := max_le (by simpa using natDegree_C_mul_X_pow_le a 1) (by simp) + +/-- Evaluation of the polynomial representing the affine function `x ↦ a * x + b`. -/ +@[simp] +theorem eval_C_mul_X_add_C [CommSemiring R] (a b x : R) : + eval x (C a * X + C b) = a * x + b := by + simp + +/-- Evaluation after composition with the affine polynomial `C a * X + C b`. -/ +@[simp] +theorem eval_comp_C_mul_X_add_C [CommSemiring R] (p : R[X]) (a b x : R) : + eval x (p.comp (C a * X + C b)) = eval (a * x + b) p := by + rw [eval_comp, eval_C_mul_X_add_C] + +end Polynomial diff --git a/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Function.lean b/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Function.lean new file mode 100644 index 00000000..a238272b --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Function.lean @@ -0,0 +1,94 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Topology.ContinuousMap.Polynomial + +/-! +# Functions represented by polynomials + +This file defines the predicate that a function is globally represented by a univariate +polynomial. This is different from the local analytic predicate `CPolynomialOn`. +-/ + +@[expose] public section + +open Polynomial + +namespace Function + +/-- A function from a semiring to itself is polynomial if it agrees everywhere with the +evaluation of a univariate polynomial. -/ +def IsPolynomial {R : Type*} [Semiring R] (f : R → R) : Prop := + ∃ p : R[X], ∀ x, p.eval x = f x + +namespace IsPolynomial + +section CommSemiring + +variable {R : Type*} [CommSemiring R] {f g : R → R} + +protected theorem const (c : R) : IsPolynomial (fun _ : R ↦ c) := + ⟨C c, by simp⟩ + +protected theorem id : IsPolynomial (id : R → R) := + ⟨X, by simp⟩ + +protected theorem add (hf : IsPolynomial f) (hg : IsPolynomial g) : + IsPolynomial (f + g) := by + obtain ⟨p, hp⟩ := hf + obtain ⟨q, hq⟩ := hg + exact ⟨p + q, fun x ↦ by simp only [eval_add, Pi.add_apply, hp x, hq x]⟩ + +protected theorem mul (hf : IsPolynomial f) (hg : IsPolynomial g) : + IsPolynomial (f * g) := by + obtain ⟨p, hp⟩ := hf + obtain ⟨q, hq⟩ := hg + exact ⟨p * q, fun x ↦ by simp only [eval_mul, Pi.mul_apply, hp x, hq x]⟩ + +protected theorem comp (hf : IsPolynomial f) (hg : IsPolynomial g) : + IsPolynomial (f ∘ g) := by + obtain ⟨p, hp⟩ := hf + obtain ⟨q, hq⟩ := hg + exact ⟨p.comp q, fun x ↦ by rw [eval_comp, hp, hq]; rfl⟩ + +/-- A bundled continuous map is polynomial exactly when it is in the range of +`Polynomial.toContinuousMap`. -/ +theorem iff_exists_toContinuousMap [TopologicalSpace R] [IsTopologicalSemiring R] + (f : C(R, R)) : + IsPolynomial f ↔ ∃ p : R[X], p.toContinuousMap = f := by + constructor + · rintro ⟨p, hp⟩ + exact ⟨p, ContinuousMap.ext hp⟩ + · rintro ⟨p, rfl⟩ + exact ⟨p, by simp⟩ + +protected theorem continuous [TopologicalSpace R] [IsTopologicalSemiring R] + (hf : IsPolynomial f) : Continuous f := by + obtain ⟨p, hp⟩ := hf + rw [← funext hp] + exact p.continuous + +end CommSemiring + +section CommRing + +variable {R : Type*} [CommRing R] {f g : R → R} + +protected theorem neg (hf : IsPolynomial f) : IsPolynomial (-f) := by + obtain ⟨p, hp⟩ := hf + exact ⟨-p, fun x ↦ by simp only [eval_neg, Pi.neg_apply, hp x]⟩ + +protected theorem sub (hf : IsPolynomial f) (hg : IsPolynomial g) : + IsPolynomial (f - g) := by + rw [sub_eq_add_neg] + exact hf.add hg.neg + +end CommRing + +end IsPolynomial + +end Function diff --git a/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean b/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean new file mode 100644 index 00000000..6378008d --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean @@ -0,0 +1,99 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Calculus.Deriv.Comp +public import Mathlib.Analysis.Calculus.Deriv.Mul +public import Mathlib.MeasureTheory.Integral.IntervalIntegral.FundThmCalculus +public import Mathlib.Topology.ContinuousMap.Algebra +public import Mathlib.Topology.ContinuousMap.Compact + +/-! +# Differentiating curves of continuous maps + +This file supplies a general criterion for differentiating a curve with values in `C(X, E)` +when `X` is compact: pointwise differentiability and continuity of the proposed derivative as a +`C(X, E)`-valued curve suffice. As an application, it differentiates a continuously +differentiable function after composition with a continuously varying affine argument. + +The results are stated for maps into an arbitrary real Banach space rather than only for +real-valued maps. +-/ + +open MeasureTheory + +universe u v + +@[expose] public section + +namespace HasDerivAt + +variable {X : Type u} {E : Type v} [TopologicalSpace X] [CompactSpace X] + [NormedAddCommGroup E] [NormedSpace ℝ E] [CompleteSpace E] + +/-- A curve of continuous maps has a derivative if all of its evaluations have the proposed +derivative and the proposed derivative is continuous in the uniform norm. + +The compactness of `X` equips `C(X, E)` with its supremum norm. The proof uses the fundamental +theorem of calculus after applying each continuous evaluation map. -/ +theorem continuousMap_of_continuous + {f f' : ℝ → C(X, E)} + (hf : ∀ x t, HasDerivAt (fun s ↦ f s x) (f' t x) t) + (hf' : Continuous f') (t : ℝ) : + HasDerivAt f (f' t) t := by + have hfi : ∀ a b, IntervalIntegrable f' volume a b := + fun a b ↦ hf'.intervalIntegrable a b + let q : ℝ → C(X, E) := + (fun _ ↦ f 0) + fun s ↦ ∫ r in 0..s, f' r + have hEq : q = f := by + funext s + apply ContinuousMap.ext + intro x + have hFTC : + ∫ r in 0..s, (ContinuousMap.evalCLM ℝ x) (f' r) = f s x - f 0 x := + intervalIntegral.integral_eq_sub_of_hasDerivAt + (fun r _ ↦ hf x r) + (((ContinuousMap.evalCLM ℝ x).continuous.comp hf').intervalIntegrable 0 s) + change f 0 x + (ContinuousMap.evalCLM ℝ x) (∫ r in 0..s, f' r) = f s x + rw [← ContinuousLinearMap.intervalIntegral_comp_comm + (ContinuousMap.evalCLM ℝ x) (hfi 0 s), hFTC] + simp + have hder : + HasDerivAt q (0 + f' t) t := + (hasDerivAt_const t (f 0)).add (hf'.integral_hasStrictDerivAt 0 t).hasDerivAt + rw [hEq] at hder + simpa using hder + +/-- Compose a differentiable Banach-valued function with the family of affine arguments +`x ↦ u x + t * v x`. Differentiation in `t` may be performed in the uniform norm on +`C(X, E)`. + +Bundling `g` and `dg` as continuous maps records exactly the continuity needed to upgrade the +pointwise derivatives `hg` to a derivative in the function space. -/ +theorem continuousMap_comp_affine + {g dg : C(ℝ, E)} + (hg : ∀ y, HasDerivAt g (dg y) y) + (u v : C(X, ℝ)) (t : ℝ) : + HasDerivAt + (fun s ↦ g.comp (u + ContinuousMap.const X s * v)) + ⟨fun x ↦ v x • dg (u x + t * v x), + v.continuous.smul + (dg.continuous.comp + (u.continuous.add (continuous_const.mul v.continuous)))⟩ + t := by + apply continuousMap_of_continuous (t := t) + · intro x s + convert + (hg (u x + s * v x)).scomp s + ((hasDerivAt_const s (u x)).add (hasDerivAt_mul_const (v x))) using 1 <;> + simp [Function.comp_def] + · apply ContinuousMap.continuous_of_continuous_uncurry + exact (v.continuous.comp continuous_snd).smul <| + dg.continuous.comp <| + (u.continuous.comp continuous_snd).add <| + continuous_fst.mul (v.continuous.comp continuous_snd) + +end HasDerivAt diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean new file mode 100644 index 00000000..cbb2ad64 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean @@ -0,0 +1,393 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Calculus.Deriv.Polynomial +public import Mathlib.Analysis.Distribution.Distribution +public import Mathlib.Analysis.Distribution.SchwartzSpace.Deriv +public import Mathlib.Analysis.Normed.Group.Bounded +public import Mathlib.Algebra.Exact.Basic +public import Mathlib.MeasureTheory.Integral.IntervalIntegral.FundThmCalculus +public import Mathlib.Topology.Algebra.Polynomial +public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function + +/-! +# Regular distributions induced by polynomials + +This file proves that distributional differentiation agrees with classical differentiation for +regular distributions on the real line. The result is first established for functions with a +classical derivative and then iterated for arbitrary derivative towers. Polynomials are obtained as +a special case. + +The converse statement, that a distribution whose sufficiently high derivative vanishes is a +regular polynomial distribution, needs an additional exactness result for the derivative on the +space of test functions. In particular, one needs that every compactly supported smooth function +with integral zero has a compactly supported smooth primitive. That result is not currently +available in Mathlib. +-/ + +@[expose] public section + +open MeasureTheory +open scoped Distributions + +namespace Polynomial + +/-- Over a field of characteristic zero, formal differentiation of univariate polynomials is +surjective. + +This is the algebraic input for recovering a polynomial from its derivative. The construction +divides the coefficient of `X ^ n` by `n + 1`. -/ +theorem derivative_surjective_of_charZero {K : Type*} [Field K] [CharZero K] : + Function.Surjective (derivative : Polynomial K → Polynomial K) := by + intro p + refine ⟨p.sum fun n a ↦ C (a / (n + 1 : ℕ)) * X ^ (n + 1), ?_⟩ + rw [sum_def, derivative_sum] + simp only [derivative_C_mul_X_pow, Nat.add_sub_cancel] + calc + (∑ n ∈ p.support, C (p.coeff n / (n + 1 : ℕ) * (n + 1 : ℕ)) * X ^ n) = + ∑ n ∈ p.support, C (p.coeff n) * X ^ n := by + apply Finset.sum_congr rfl + intro n hn + congr 2 + rw [div_mul_cancel₀] + exact_mod_cast Nat.succ_ne_zero n + _ = p := p.as_sum_support_C_mul_X_pow.symm + +end Polynomial + +namespace MeasureTheory + +/-- The integral of an integrable continuous function over `Set.Iic y`, regarded as a function of +the upper endpoint `y`, has derivative equal to the integrand. + +This vector-valued version is obtained by comparing two lower-ray integrals and applying the +fundamental theorem of calculus to the resulting interval integral. -/ +theorem hasDerivAt_integral_Iic + {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] [CompleteSpace F] + {f : ℝ → F} (hf : Continuous f) (hfi : Integrable f) (b : ℝ) : + HasDerivAt (fun y ↦ ∫ x in Set.Iic y, f x) (f b) b := by + have hEq : (fun y ↦ ∫ x in Set.Iic y, f x) = + fun y ↦ (∫ x in b..y, f x) + ∫ x in Set.Iic b, f x := by + funext y + exact sub_eq_iff_eq_add.mp <| intervalIntegral.integral_Iic_sub_Iic + (hfi.integrableOn : IntegrableOn f (Set.Iic b)) + (hfi.integrableOn : IntegrableOn f (Set.Iic y)) + rw [hEq] + exact (hf.integral_hasStrictDerivAt b b).hasDerivAt.add_const _ + +/-- If a compactly supported integrable function on the real line has integral zero, then its +indefinite integral over lower rays is compactly supported. + +This statement is independent of differentiability and works for functions with values in any +complete real normed space. -/ +theorem hasCompactSupport_integral_Iic_of_integral_eq_zero + {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] [CompleteSpace F] + {f : ℝ → F} (hfc : HasCompactSupport f) (hfi : Integrable f) + (hzero : ∫ x, f x = 0) : + HasCompactSupport (fun b ↦ ∫ x in Set.Iic b, f x) := by + obtain ⟨R, hR, hout⟩ := hfc.exists_pos_le_norm + apply HasCompactSupport.intro (isCompact_Icc : IsCompact (Set.Icc (-R) R)) + intro x hx + simp only [Set.mem_Icc, not_and_or, not_le] at hx + rcases hx with hx | hx + · apply setIntegral_eq_zero_of_forall_eq_zero + intro y hy + change y ≤ x at hy + apply hout y + rw [Real.norm_eq_abs, abs_of_nonpos] + · linarith + · linarith + · have hIoi : ∫ y in Set.Ioi x, f y = 0 := by + apply setIntegral_eq_zero_of_forall_eq_zero + intro y hy + change x < y at hy + apply hout y + rw [Real.norm_eq_abs, abs_of_nonneg] + · linarith + · linarith + have hsplit := intervalIntegral.integral_Iic_add_Ioi + (hfi.integrableOn : IntegrableOn f (Set.Iic x)) + (hfi.integrableOn : IntegrableOn f (Set.Ioi x)) + rw [hIoi, hzero, add_zero] at hsplit + exact hsplit + +/-- Every smooth compactly supported function on the real line with integral zero has a smooth +compactly supported primitive. + +The codomain is allowed to be any complete real normed space. This is the vector-valued analytic +core of exactness of differentiation and integration on real test functions. -/ +theorem exists_contDiff_primitive_hasCompactSupport + {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] [CompleteSpace F] + {f : ℝ → F} (hf : ContDiff ℝ (↑(⊤ : ℕ∞)) f) (hfc : HasCompactSupport f) + (hzero : ∫ x, f x = 0) : + ∃ g : ℝ → F, ContDiff ℝ (↑(⊤ : ℕ∞)) g ∧ HasCompactSupport g ∧ + ∀ x, HasDerivAt g (f x) x := by + have hfcont : Continuous f := hf.continuous + have hfi : Integrable f := hfcont.integrable_of_hasCompactSupport hfc + let g : ℝ → F := fun b ↦ ∫ x in Set.Iic b, f x + have hgDeriv (x : ℝ) : HasDerivAt g (f x) x := + hasDerivAt_integral_Iic hfcont hfi x + have hgDiff : Differentiable ℝ g := fun x ↦ (hgDeriv x).differentiableAt + have hgderiv : deriv g = f := by + funext x + exact (hgDeriv x).deriv + have hgSmooth : ContDiff ℝ (↑(⊤ : ℕ∞)) g := by + rw [contDiff_infty_iff_deriv] + exact ⟨hgDiff, hgderiv ▸ hf⟩ + have hgCompact : HasCompactSupport g := + hasCompactSupport_integral_Iic_of_integral_eq_zero hfc hfi hzero + exact ⟨g, hgSmooth, hgCompact, hgDeriv⟩ + +end MeasureTheory + +namespace LinearMap + +/-- A linear map annihilating the first map in an exact pair is determined by its value on any +element sent to `1` by the second map. + +This is the algebraic factorization argument underlying the fact that a distribution with zero +derivative is constant. -/ +theorem apply_eq_smul_apply_of_exact {R X Y : Type*} [CommRing R] + [AddCommGroup X] [Module R X] [AddCommGroup Y] [Module R Y] + (D : X →ₗ[R] X) (I : X →ₗ[R] R) (T : X →ₗ[R] Y) + (hExact : Function.Exact D I) {ρ : X} (hρ : I ρ = 1) + (hTD : ∀ x, T (D x) = 0) (x : X) : + T x = I x • T ρ := by + let ψ := x - I x • ρ + have hIψ : I ψ = 0 := by simp [ψ, hρ] + obtain ⟨θ, hθ⟩ := (hExact ψ).mp hIψ + have hTψ : T ψ = 0 := by + rw [← hθ] + exact hTD θ + calc + T x = T (ψ + I x • ρ) := by simp [ψ] + _ = I x • T ρ := by simp [hTψ] + +end LinearMap + +namespace TestFunction + +/-- On real-valued test functions on the real line, the directional derivative in direction `1` +is the usual one-dimensional derivative. -/ +lemma lineDerivCLM_one_apply {Ω : TopologicalSpace.Opens ℝ} (φ : 𝓓(Ω, ℝ)) (x : ℝ) : + (lineDerivCLM ℝ (1 : ℝ) φ : 𝓓(Ω, ℝ)) x = deriv φ x := by + rw [lineDerivCLM_apply_of_le] + · calc + lineDeriv ℝ (φ : ℝ → ℝ) x 1 = (fderiv ℝ (φ : ℝ → ℝ) x) 1 := + (φ.contDiff.differentiable (by simp)).differentiableAt.lineDeriv_eq_fderiv + _ = deriv (φ : ℝ → ℝ) x := fderiv_apply_one_eq_deriv + · simp + +/-- The analytic exactness property needed to recognize constant distributions on an open subset +of the real line: every test function of integral zero is the derivative of another test +function. + +This is a class so downstream distribution theory can be stated independently of the particular +construction of compactly supported primitives. -/ +class HasCompactSupportPrimitive (Ω : TopologicalSpace.Opens ℝ) : Prop where + exists_eq_lineDerivCLM (φ : 𝓓(Ω, ℝ)) (hφ : ∫ x, φ x = 0) : + ∃ ψ : 𝓓(Ω, ℝ), lineDerivCLM ℝ (1 : ℝ) ψ = φ + +/-- On the whole real line, smooth compactly supported primitives exist. -/ +instance instHasCompactSupportPrimitiveTop : + HasCompactSupportPrimitive (⊤ : TopologicalSpace.Opens ℝ) where + exists_eq_lineDerivCLM φ hφ := by + obtain ⟨g, hgSmooth, hgCompact, hgDeriv⟩ := + MeasureTheory.exists_contDiff_primitive_hasCompactSupport + φ.contDiff φ.hasCompactSupport hφ + let ψ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ) := + ⟨g, hgSmooth, hgCompact, by simp⟩ + refine ⟨ψ, ?_⟩ + ext x + rw [lineDerivCLM_one_apply] + simpa [ψ] using (hgDeriv x).deriv + +end TestFunction + +namespace Distribution + +open LineDeriv + +variable {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] + +/-- The distributional derivative of the regular distribution induced by `f` is the regular +distribution induced by the classical derivative of `f`. + +This vector-valued version only assumes the local integrability needed to define the two regular +distributions. -/ +theorem lineDerivCLM_ofFun_eq_of_hasDerivAt {Ω : TopologicalSpace.Opens ℝ} + {f f' : ℝ → F} (hf : ∀ x, HasDerivAt f (f' x) x) + (hfloc : LocallyIntegrableOn f Ω volume) + (hf'loc : LocallyIntegrableOn f' Ω volume) : + lineDerivCLM (1 : ℝ) (ofFun Ω f volume ⊤) = ofFun Ω f' volume ⊤ := by + ext φ + rw [lineDerivCLM_apply, ofFun_apply hfloc, ofFun_apply hf'loc] + let dφ : 𝓓(Ω, ℝ) := TestFunction.lineDerivCLM ℝ (1 : ℝ) φ + have hdφ (x : ℝ) : dφ x = deriv (φ : ℝ → ℝ) x := + TestFunction.lineDerivCLM_one_apply φ x + have hibp := MeasureTheory.integral_bilinear_hasDerivAt_right_eq_neg_left_of_integrable + (L := ContinuousLinearMap.lsmul ℝ ℝ) (u := (φ : ℝ → ℝ)) (v := f) + (u' := fun x => deriv (φ : ℝ → ℝ) x) (v' := f') + (fun _ _ => (φ.contDiff.differentiable (by simp)).differentiableAt.hasDerivAt) + (fun x _ => hf x) + (φ.integrable_smul hf'loc) + (by simpa only [hdφ, ContinuousLinearMap.lsmul_apply] using dφ.integrable_smul hfloc) + (φ.integrable_smul hfloc) + simp only [ContinuousLinearMap.lsmul_apply] at hibp + rw [show (TestFunction.lineDerivCLM ℝ (1 : ℝ) φ : 𝓓(Ω, ℝ)) = dφ from rfl] + simp_rw [hdφ] + exact hibp.symm + +/-- The distributional derivative of a regular constant distribution vanishes. -/ +theorem lineDerivCLM_ofFun_const_eq_zero {Ω : TopologicalSpace.Opens ℝ} (c : F) : + (lineDerivCLM (1 : ℝ) (ofFun Ω (fun _ : ℝ => c) volume ⊤) : 𝓓'(Ω, F)) = 0 := by + have hc : LocallyIntegrableOn (fun _ : ℝ => c) Ω volume := + (continuous_const : Continuous (fun _ : ℝ => c)).locallyIntegrable.locallyIntegrableOn Ω + rw [lineDerivCLM_ofFun_eq_of_hasDerivAt + (fun x => hasDerivAt_const x c) hc locallyIntegrableOn_zero] + exact ofFun_zero + +/-- Integration annihilates derivatives of test functions. This is the easy half of the +derivative--integral exactness statement. -/ +theorem ofFun_one_comp_testFunction_lineDerivCLM_eq_zero + {Ω : TopologicalSpace.Opens ℝ} : + (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤) ∘ + (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) = 0 := by + funext φ + have h := congrArg (fun T : 𝓓'(Ω, ℝ) => T φ) + (lineDerivCLM_ofFun_const_eq_zero (Ω := Ω) (1 : ℝ)) + rw [lineDerivCLM_apply] at h + exact neg_eq_zero.mp h + +/-- Compactly supported primitives identify the range of differentiation with the kernel of +integration on test functions. -/ +theorem exact_testFunction_lineDerivCLM_of_hasCompactSupportPrimitive + {Ω : TopologicalSpace.Opens ℝ} [TestFunction.HasCompactSupportPrimitive Ω] : + Function.Exact + (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) + (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤) := by + apply Function.Exact.of_comp_of_mem_range + ofFun_one_comp_testFunction_lineDerivCLM_eq_zero + intro φ hφ + have hOneLoc : LocallyIntegrableOn (fun _ : ℝ => (1 : ℝ)) Ω volume := + (continuous_const : Continuous (fun _ : ℝ => (1 : ℝ))).locallyIntegrable + |>.locallyIntegrableOn Ω + have hIntegral : ∫ x, φ x = 0 := by + rw [ofFun_apply hOneLoc] at hφ + simpa using hφ + exact TestFunction.HasCompactSupportPrimitive.exists_eq_lineDerivCLM φ hIntegral + +/-- A distribution with zero derivative is a regular constant distribution, provided the +derivative--integral pair on test functions is exact and a normalized test function is given. + +The exactness hypothesis precisely isolates the missing analytic input: every test function with +integral zero must have a compactly supported smooth primitive. -/ +theorem eq_ofFun_const_of_lineDerivCLM_eq_zero [CompleteSpace F] + {Ω : TopologicalSpace.Opens ℝ} (ρ : 𝓓(Ω, ℝ)) + (hρ : ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤ ρ = 1) + (hExact : Function.Exact + (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) + (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤)) + (T : 𝓓'(Ω, F)) (hT : (lineDerivCLM (1 : ℝ) T : 𝓓'(Ω, F)) = 0) : + T = ofFun Ω (fun _ => T ρ) volume ⊤ := by + have hOneLoc : LocallyIntegrableOn (fun _ : ℝ => (1 : ℝ)) Ω volume := + (continuous_const : Continuous (fun _ : ℝ => (1 : ℝ))).locallyIntegrable + |>.locallyIntegrableOn Ω + have hConstLoc : LocallyIntegrableOn (fun _ : ℝ => T ρ) Ω volume := + (continuous_const : Continuous (fun _ : ℝ => T ρ)).locallyIntegrable + |>.locallyIntegrableOn Ω + have hTD (φ : 𝓓(Ω, ℝ)) : + T (TestFunction.lineDerivCLM ℝ (1 : ℝ) φ) = 0 := by + have h := congrArg (fun S : 𝓓'(Ω, F) => S φ) hT + rw [lineDerivCLM_apply] at h + exact neg_eq_zero.mp h + ext φ + rw [ofFun_apply hConstLoc] + calc + T φ = (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤) φ • T ρ := + LinearMap.apply_eq_smul_apply_of_exact + (TestFunction.lineDerivCLM ℝ (1 : ℝ)).toLinearMap + (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤).toLinearMap + T.toLinearMap hExact hρ hTD φ + _ = (∫ x, φ x) • T ρ := by + rw [ofFun_apply hOneLoc] + congr 1 + simp + _ = ∫ x, φ x • T ρ := (integral_smul_const (φ : ℝ → ℝ) (T ρ)).symm + +/-- A distribution with zero derivative is constant whenever compactly supported primitives exist. +The test function `ρ` fixes the normalization of the resulting constant. -/ +theorem eq_ofFun_const_of_lineDerivCLM_eq_zero_of_hasCompactSupportPrimitive [CompleteSpace F] + {Ω : TopologicalSpace.Opens ℝ} [TestFunction.HasCompactSupportPrimitive Ω] + (ρ : 𝓓(Ω, ℝ)) (hρ : ∫ x, ρ x = 1) (T : 𝓓'(Ω, F)) + (hT : (lineDerivCLM (1 : ℝ) T : 𝓓'(Ω, F)) = 0) : + T = ofFun Ω (fun _ => T ρ) volume ⊤ := by + have hOneLoc : LocallyIntegrableOn (fun _ : ℝ => (1 : ℝ)) Ω volume := + (continuous_const : Continuous (fun _ : ℝ => (1 : ℝ))).locallyIntegrable + |>.locallyIntegrableOn Ω + apply eq_ofFun_const_of_lineDerivCLM_eq_zero ρ + · rw [ofFun_apply hOneLoc] + simpa using hρ + · exact exact_testFunction_lineDerivCLM_of_hasCompactSupportPrimitive + · exact hT + +/-- Iterated distributional differentiation commutes with a tower of classical derivatives. + +Using a sequence rather than iterating `deriv` makes the statement applicable when the classical +derivatives are supplied together with proofs of their values. -/ +theorem iteratedLineDerivOp_ofFun_eq_of_hasDerivAt {Ω : TopologicalSpace.Opens ℝ} + (f : ℕ → ℝ → F) (hf : ∀ n x, HasDerivAt (f n) (f (n + 1) x) x) + (hfloc : ∀ n, LocallyIntegrableOn (f n) Ω volume) (k : ℕ) : + iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) (ofFun Ω (f 0) volume ⊤) = + ofFun Ω (f k) volume ⊤ := by + rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] + induction k with + | zero => simp + | succ k ih => + rw [Function.iterate_succ_apply', ih] + exact lineDerivCLM_ofFun_eq_of_hasDerivAt (hf k) (hfloc k) (hfloc (k + 1)) + +/-- Iterated distributional differentiation of the regular distribution induced by a polynomial +agrees with iterated formal differentiation of that polynomial. -/ +theorem iteratedLineDerivOp_ofFun_polynomial {Ω : TopologicalSpace.Opens ℝ} + (p : Polynomial ℝ) (k : ℕ) : + iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) + (ofFun Ω (fun x => p.eval x) volume ⊤) = + ofFun Ω + (fun x => (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[k]) p).eval x) + volume ⊤ := by + apply iteratedLineDerivOp_ofFun_eq_of_hasDerivAt + (f := fun n x => (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[n]) p).eval x) + · intro n x + simpa only [Function.iterate_succ_apply'] using + (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[n]) p).hasDerivAt x + · intro n + exact (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[n]) p).continuous + |>.locallyIntegrable.locallyIntegrableOn Ω + +/-- Every distributional derivative of order strictly larger than the degree of a regular +polynomial distribution vanishes. -/ +theorem iteratedLineDerivOp_ofFun_polynomial_eq_zero_of_natDegree_lt + {Ω : TopologicalSpace.Opens ℝ} (p : Polynomial ℝ) (k : ℕ) (hpk : p.natDegree < k) : + iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) + (ofFun Ω (fun x => p.eval x) volume ⊤) = 0 := by + rw [iteratedLineDerivOp_ofFun_polynomial, Polynomial.iterate_derivative_eq_zero hpk] + have hz : (fun x : ℝ => Polynomial.eval x (0 : Polynomial ℝ)) = 0 := by + funext x + simp + rw [hz] + exact ofFun_zero + +/-- The derivative of order one above the natural degree of a regular polynomial distribution +vanishes. -/ +theorem iteratedLineDerivOp_ofFun_polynomial_natDegree_add_one_eq_zero + {Ω : TopologicalSpace.Opens ℝ} (p : Polynomial ℝ) : + iteratedLineDerivOp (fun _ : Fin (p.natDegree + 1) => (1 : ℝ)) + (ofFun Ω (fun x => p.eval x) volume ⊤) = 0 := + iteratedLineDerivOp_ofFun_polynomial_eq_zero_of_natDegree_lt p _ (Nat.lt_succ_self _) + +end Distribution diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean new file mode 100644 index 00000000..f8c84ccc --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean @@ -0,0 +1,150 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.Polynomial + +/-! +# Characterizing polynomial functions by distributional derivatives + +This file proves the converse to the elementary fact that a sufficiently high distributional +derivative of a polynomial vanishes. The analytic input is isolated in +`TestFunction.HasCompactSupportPrimitive`: every test function of integral zero must be the +derivative of a compactly supported test function. + +The main results are: + +* `Distribution.exists_polynomial_of_iteratedLineDerivOp_eq_zero`, for arbitrary distributions; +* `Distribution.ofFun_eq_iff_eq_of_continuous`, a general injectivity result for regular + distributions induced by continuous functions; +* `Distribution.isPolynomial_iff_exists_iteratedLineDerivOp_ofFun_eq_zero`, the desired + characterization for continuous real functions. +-/ + +@[expose] public section + +open MeasureTheory +open scoped Distributions + +namespace Distribution + +open LineDeriv + +/-- A distribution on an open subset of the real line whose `k`th derivative vanishes is induced +by a polynomial, provided compactly supported primitives exist. + +No smoothness or regularity assumption is made on the distribution. The proof repeatedly uses +that a distribution with zero derivative is constant and the surjectivity of polynomial +differentiation over `ℝ`. -/ +theorem exists_polynomial_of_iteratedLineDerivOp_eq_zero + {Ω : TopologicalSpace.Opens ℝ} [TestFunction.HasCompactSupportPrimitive Ω] + (ρ : 𝓓(Ω, ℝ)) (hρ : ∫ x, ρ x = 1) + (T : 𝓓'(Ω, ℝ)) (k : ℕ) + (hT : iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) T = 0) : + ∃ p : Polynomial ℝ, T = ofFun Ω (fun x => p.eval x) volume ⊤ := by + induction k generalizing T with + | zero => + simp only [iteratedLineDerivOp_fin_zero] at hT + refine ⟨0, ?_⟩ + rw [show (fun x : ℝ => (0 : Polynomial ℝ).eval x) = 0 by funext x; simp, + ofFun_zero] + exact hT + | succ k ih => + let DT : 𝓓'(Ω, ℝ) := ∂_{(1 : ℝ)} T + have hDT : iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) DT = 0 := by + rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] at hT ⊢ + exact hT + obtain ⟨p, hp⟩ := ih DT hDT + obtain ⟨q, hq⟩ := Polynomial.derivative_surjective_of_charZero p + let Q : 𝓓'(Ω, ℝ) := ofFun Ω (fun x => q.eval x) volume ⊤ + have hqLoc : LocallyIntegrableOn (fun x => q.eval x) Ω volume := + q.continuous.locallyIntegrable.locallyIntegrableOn _ + have hDQ : lineDerivCLM (1 : ℝ) Q = + ofFun Ω (fun x => p.eval x) volume ⊤ := by + dsimp only [Q] + rw [lineDerivCLM_ofFun_eq_of_hasDerivAt + (fun x => q.hasDerivAt x) hqLoc + ((q.derivative.continuous).locallyIntegrable.locallyIntegrableOn _)] + rw [hq] + have hDsub : (lineDerivCLM (1 : ℝ) (T - Q) : 𝓓'(Ω, ℝ)) = 0 := by + rw [map_sub, hDQ, ← hp] + change DT - DT = 0 + exact sub_self _ + have hconst := eq_ofFun_const_of_lineDerivCLM_eq_zero_of_hasCompactSupportPrimitive + ρ hρ (T - Q) hDsub + let c : ℝ := (T - Q) ρ + have hcLoc : LocallyIntegrableOn (fun _ : ℝ => c) Ω volume := + (continuous_const : Continuous (fun _ : ℝ => c)).locallyIntegrable + |>.locallyIntegrableOn _ + refine ⟨q + Polynomial.C c, ?_⟩ + have heval : (fun x => (q + Polynomial.C c).eval x) = + (fun x => q.eval x) + (fun _ : ℝ => c) := by + funext x + simp + rw [heval, ofFun_add hqLoc hcLoc] + exact sub_eq_iff_eq_add'.mp hconst + +/-- Two locally integrable continuous functions on a finite-dimensional real normed space induce +the same regular distribution if and only if they are equal. + +This packages `Distribution.ofFun_injective`, whose conclusion is only almost-everywhere equality, +with the fact that a full-support measure detects equality of continuous functions. -/ +theorem ofFun_eq_iff_eq_of_continuous + {E F : Type*} [NormedAddCommGroup E] [NormedSpace ℝ E] + [MeasurableSpace E] [BorelSpace E] [FiniteDimensional ℝ E] + [NormedAddCommGroup F] [NormedSpace ℝ F] [CompleteSpace F] + {μ : Measure E} [μ.IsOpenPosMeasure] {n : ℕ∞} {f g : E → F} + (hf : Continuous f) (hg : Continuous g) + (hfloc : LocallyIntegrableOn f (Set.univ : Set E) μ) + (hgloc : LocallyIntegrableOn g (Set.univ : Set E) μ) : + ofFun (⊤ : TopologicalSpace.Opens E) f μ n = + ofFun (⊤ : TopologicalSpace.Opens E) g μ n ↔ f = g := by + constructor + · intro h + have hae : f =ᵐ[μ.restrict (Set.univ : Set E)] g := + ofFun_injective hfloc hgloc h + exact (Continuous.ae_eq_iff_eq μ hf hg).mp <| by simpa using hae + · rintro rfl + rfl + +/-- A continuous real function is polynomial if one of its distributional derivatives vanishes, +assuming the compactly supported primitive property on the real line. -/ +theorem isPolynomial_of_iteratedLineDerivOp_ofFun_eq_zero + [TestFunction.HasCompactSupportPrimitive (⊤ : TopologicalSpace.Opens ℝ)] + (ρ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (hρ : ∫ x, ρ x = 1) + {f : ℝ → ℝ} (hf : Continuous f) (k : ℕ) + (h : iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) + (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) = 0) : + Function.IsPolynomial f := by + obtain ⟨p, hp⟩ := exists_polynomial_of_iteratedLineDerivOp_eq_zero ρ hρ + (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) k h + have hfloc : LocallyIntegrableOn f (Set.univ : Set ℝ) volume := + hf.locallyIntegrable.locallyIntegrableOn _ + have hploc : LocallyIntegrableOn (fun x => p.eval x) (Set.univ : Set ℝ) volume := + p.continuous.locallyIntegrable.locallyIntegrableOn _ + have heq : f = fun x => p.eval x := + (ofFun_eq_iff_eq_of_continuous hf p.continuous hfloc hploc).mp hp + exact ⟨p, fun x => (congrFun heq x).symm⟩ + +/-- A continuous real function is polynomial exactly when one of its distributional derivatives +vanishes, assuming the compactly supported primitive property on the real line. -/ +theorem isPolynomial_iff_exists_iteratedLineDerivOp_ofFun_eq_zero + [TestFunction.HasCompactSupportPrimitive (⊤ : TopologicalSpace.Opens ℝ)] + (ρ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (hρ : ∫ x, ρ x = 1) + {f : ℝ → ℝ} (hf : Continuous f) : + Function.IsPolynomial f ↔ + ∃ k : ℕ, iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) + (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) = 0 := by + constructor + · rintro ⟨p, hp⟩ + refine ⟨p.natDegree + 1, ?_⟩ + have heval : f = fun x => p.eval x := funext fun x => (hp x).symm + rw [heval] + exact iteratedLineDerivOp_ofFun_polynomial_natDegree_add_one_eq_zero p + · rintro ⟨k, hk⟩ + exact isPolynomial_of_iteratedLineDerivOp_ofFun_eq_zero ρ hρ hf k hk + +end Distribution diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction.lean new file mode 100644 index 00000000..55d1e83c --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction.lean @@ -0,0 +1,31 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Distribution.TestFunction +public import Mathlib.Topology.ContinuousMap.CompactlySupported + +/-! +# Test functions as compactly supported continuous maps + +Every member of a `TestFunctionClass` is continuous and has compact support. This file packages +those two existing facts in the standard `CompactlySupportedContinuousMapClass` interface. +-/ + +@[expose] public section + +namespace TestFunctionClass + +variable {B E F : Type*} [NormedAddCommGroup E] [NormedSpace ℝ E] + {Ω : TopologicalSpace.Opens E} [NormedAddCommGroup F] [NormedSpace ℝ F] + {n : ℕ∞} [TestFunctionClass B Ω F n] + +/-- A test-function class is, in particular, a class of compactly supported continuous maps. -/ +instance instCompactlySupportedContinuousMapClass : + CompactlySupportedContinuousMapClass B E F := + CompactlySupportedContinuousMapClass.mk map_hasCompactSupport + +end TestFunctionClass diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean new file mode 100644 index 00000000..d2fa4eb5 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean @@ -0,0 +1,91 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Calculus.BumpFunction.FiniteDimension +public import Mathlib.Analysis.Calculus.BumpFunction.Normed +public import Mathlib.Analysis.Distribution.TestFunction + +/-! +# Test functions normalized by their integral + +This file constructs real-valued test functions of integral one from normalized smooth bump +functions, both inside an arbitrary nonempty open subset of a finite-dimensional real normed +space and as a fixed test function on the real line. +-/ + +@[expose] public section + +noncomputable section + +open Function Set TopologicalSpace MeasureTheory +open scoped Distributions + +namespace ContDiffBump + +variable {E : Type*} [NormedAddCommGroup E] [NormedSpace ℝ E] [HasContDiffBump E] + [MeasurableSpace E] [BorelSpace E] [FiniteDimensional ℝ E] + {Ω : Opens E} {c : E} (f : ContDiffBump c) (μ : Measure E) + [IsLocallyFiniteMeasure μ] [μ.IsOpenPosMeasure] + +/-- Regard a normalized smooth bump as a test function on an open set containing its support. -/ +def toTestFunctionNormed + (h : Metric.closedBall c f.rOut ⊆ Ω) : 𝓓(Ω, ℝ) where + toFun := f.normed μ + contDiff' := f.contDiff_normed + hasCompactSupport' := f.hasCompactSupport_normed + tsupport_subset' := by simpa only [f.tsupport_normed_eq] using h + +@[simp] +theorem toTestFunctionNormed_apply + (h : Metric.closedBall c f.rOut ⊆ Ω) (x : E) : + f.toTestFunctionNormed μ h x = f.normed μ x := + rfl + +/-- A normalized smooth bump, regarded as a test function, still has integral one. -/ +@[simp] +theorem integral_toTestFunctionNormed + (h : Metric.closedBall c f.rOut ⊆ Ω) : + ∫ x, f.toTestFunctionNormed μ h x ∂μ = 1 := by + simpa only [toTestFunctionNormed_apply] using f.integral_normed (μ := μ) + +end ContDiffBump + +namespace TestFunction + +variable {E : Type*} [NormedAddCommGroup E] [NormedSpace ℝ E] + [FiniteDimensional ℝ E] [MeasurableSpace E] [BorelSpace E] + {Ω : Opens E} (μ : Measure E) [IsLocallyFiniteMeasure μ] [μ.IsOpenPosMeasure] + +/-- Every nonempty open subset of a finite-dimensional real normed space supports a smooth test +function of integral one. -/ +theorem exists_integral_eq_one (hΩ : (Ω : Set E).Nonempty) : + ∃ ρ : 𝓓(Ω, ℝ), ∫ x, ρ x ∂μ = 1 := by + obtain ⟨c, hc⟩ := hΩ + obtain ⟨ε, hε, hball⟩ := Metric.mem_nhds_iff.mp (Ω.isOpen.mem_nhds hc) + let f : ContDiffBump c := + ContDiffBump.mk (ε / 4) (ε / 2) (by positivity) (by linarith) + have hf : Metric.closedBall c f.rOut ⊆ Ω := by + refine (Metric.closedBall_subset_ball ?_).trans hball + change ε / 2 < ε + exact half_lt_self hε + exact ⟨f.toTestFunctionNormed μ hf, f.integral_toTestFunctionNormed μ hf⟩ + +/-- A fixed smooth compactly supported function on `ℝ` whose Lebesgue integral is one. -/ +def normalizedBumpReal : 𝓓((⊤ : Opens ℝ), ℝ) := + let f : ContDiffBump (0 : ℝ) := + ContDiffBump.mk 1 2 zero_lt_one one_lt_two + f.toTestFunctionNormed volume (subset_univ _) + +@[simp] +theorem integral_normalizedBumpReal : + ∫ x : ℝ, normalizedBumpReal x = 1 := by + let f : ContDiffBump (0 : ℝ) := + ContDiffBump.mk 1 2 zero_lt_one one_lt_two + simpa only [normalizedBumpReal] using + f.integral_toTestFunctionNormed volume (subset_univ _) + +end TestFunction diff --git a/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean b/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean new file mode 100644 index 00000000..711a3293 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean @@ -0,0 +1,95 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.LocallyConvex.Polar +public import Mathlib.Analysis.LocallyConvex.Separation + +/-! +# Density and annihilators + +This file characterizes dense real submodules of locally convex spaces in terms of their +continuous dual annihilators. The reverse implication is an application of geometric +Hahn--Banach separation. +-/ + +@[expose] public section + +open Set + +namespace Submodule + +/-- A real submodule of a locally convex topological vector space is dense exactly when every +continuous linear functional vanishing on it is zero. -/ +theorem dense_iff_forall_dual_eq_zero + {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] + [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] + (s : Submodule ℝ E) : + Dense (s : Set E) ↔ + ∀ f : StrongDual ℝ E, (∀ x ∈ s, f x = 0) → f = 0 := by + constructor + · intro hs f hf + ext x + have hfun : (f : E → ℝ) = (0 : E → ℝ) := + Continuous.ext_on hs f.continuous continuous_zero (by + intro y hy + simpa using hf y hy) + exact congrFun hfun x + · intro h + rw [Submodule.dense_iff_topologicalClosure_eq_top] + apply top_unique + intro x hx + by_contra hxc + obtain ⟨f, u, hfc, hfx⟩ := + geometric_hahn_banach_closed_point + s.topologicalClosure.convex + s.isClosed_topologicalClosure hxc + have hfzero : ∀ y ∈ s.topologicalClosure, f y = 0 := by + intro y hy + by_contra hfy + have hlt := hfc ((u / f y) • y) + (s.topologicalClosure.smul_mem (u / f y) hy) + rw [map_smul, smul_eq_mul, div_mul_cancel₀ u hfy] at hlt + exact (lt_irrefl u) hlt + have hf : f = 0 := + h f fun y hy ↦ hfzero y (s.le_topologicalClosure hy) + have hu0 : 0 < u := by + simpa using hfc 0 s.topologicalClosure.zero_mem + have hux0 : u < 0 := by + simpa [hf] using hfx + exact (not_lt_of_ge hu0.le) hux0 + +/-- If a real submodule of a locally convex space is not dense, a nonzero continuous linear +functional annihilates it. -/ +theorem exists_dual_annihilator_of_not_dense + {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] + [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] + (s : Submodule ℝ E) (hs : ¬ Dense (s : Set E)) : + ∃ f : StrongDual ℝ E, f ≠ 0 ∧ ∀ x ∈ s, f x = 0 := by + grind [Submodule.dense_iff_forall_dual_eq_zero] + +/-- A real submodule is dense exactly when its polar submodule is trivial. -/ +theorem dense_iff_polarSubmodule_eq_bot + {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] + [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] + (s : Submodule ℝ E) : + Dense (s : Set E) ↔ StrongDual.polarSubmodule ℝ s = ⊥ := by + rw [dense_iff_forall_dual_eq_zero] + constructor + · intro h + ext f + rw [StrongDual.mem_polarSubmodule, Submodule.mem_bot] + constructor + · exact h f + · rintro rfl x hx + rfl + · intro h f hf + have hmem : f ∈ StrongDual.polarSubmodule ℝ s := + (StrongDual.mem_polarSubmodule ℝ s f).2 hf + rw [h] at hmem + exact hmem + +end Submodule diff --git a/LeanMachineLearning/ForMathlib/LinearAlgebra/Multilinear/Polarization.lean b/LeanMachineLearning/ForMathlib/LinearAlgebra/Multilinear/Polarization.lean new file mode 100644 index 00000000..30be1a96 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/LinearAlgebra/Multilinear/Polarization.lean @@ -0,0 +1,248 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Analytic.IteratedFDeriv + +/-! +# Polarization of symmetric multilinear maps + +This file proves that a symmetric continuous multilinear map is determined by its values on the +diagonal. The main input is the existing formula +`ContinuousMultilinearMap.iteratedFDeriv_comp_diagonal`: the `n`-th derivative of the diagonal +of an `n`-linear map is the sum of the map over all permutations of its arguments. For a +symmetric map this sum is `n!` times the original value. + +The most general statements only assume that `n!` is nonzero in the scalar field. Characteristic +zero versions are provided as convenient corollaries. +-/ + +@[expose] public section + +open scoped BigOperators + +namespace MultilinearMap + +universe uR uM uN uι + +variable {R : Type uR} [Semiring R] +variable {M : Type uM} [AddCommMonoid M] [Module R M] +variable {N : Type uN} [AddCommMonoid N] [Module R N] +variable {ι : Type uι} + +/-- A multilinear map is symmetric if reindexing its arguments by a permutation leaves it +unchanged. -/ +class IsSymm (f : MultilinearMap R (fun _ : ι => M) N) : Prop where + domDomCongr_eq (σ : Equiv.Perm ι) : f.domDomCongr σ = f + +namespace IsSymm + +variable {f g : MultilinearMap R (fun _ : ι => M) N} + +/-- Pointwise characterization of symmetry. -/ +theorem isSymm_iff : f.IsSymm ↔ ∀ (v : ι → M) (σ : Equiv.Perm ι), f (v ∘ σ) = f v where + mp hf v σ := by + have h := congrArg (fun p : MultilinearMap R (fun _ : ι => M) N => p v) + (hf.domDomCongr_eq σ) + change f (fun i => v (σ i)) = f v at h + exact h + mpr h := by + constructor + intro σ + ext v + change f (v ∘ σ) = f v + exact h v σ + +/-- A symmetric multilinear map has the same value after permuting its arguments. -/ +@[simp] +lemma map_perm [hf : f.IsSymm] (v : ι → M) (σ : Equiv.Perm ι) : + f (v ∘ σ) = f v := + isSymm_iff.mp hf v σ + +instance zero : IsSymm (0 : MultilinearMap R (fun _ : ι => M) N) where + domDomCongr_eq _ := by ext; simp + +instance add [f.IsSymm] [g.IsSymm] : IsSymm (f + g) where + domDomCongr_eq σ := by + ext v + change f (v ∘ σ) + g (v ∘ σ) = f v + g v + rw [map_perm, map_perm] + +end IsSymm + +section Ring + +variable {R : Type uR} [Ring R] +variable {M : Type uM} [AddCommGroup M] [Module R M] +variable {N : Type uN} [AddCommGroup N] [Module R N] +variable {ι : Type uι} + +namespace IsSymm + +variable {f g : MultilinearMap R (fun _ : ι => M) N} + +instance neg [f.IsSymm] : IsSymm (-f) where + domDomCongr_eq σ := by + ext v + change -f (v ∘ σ) = -f v + rw [map_perm] + +instance sub [f.IsSymm] [g.IsSymm] : IsSymm (f - g) where + domDomCongr_eq σ := by + ext v + change f (v ∘ σ) - g (v ∘ σ) = f v - g v + rw [map_perm, map_perm] + +end IsSymm + +end Ring + +end MultilinearMap + +namespace ContinuousMultilinearMap + +universe u𝕜 uE uF + +variable {𝕜 : Type u𝕜} [NontriviallyNormedField 𝕜] +variable {E : Type uE} [NormedAddCommGroup E] [NormedSpace 𝕜 E] +variable {F : Type uF} [NormedAddCommGroup F] [NormedSpace 𝕜 F] +variable {n : ℕ} + +/-- Symmetry of a continuous multilinear map, inherited from its underlying multilinear map. -/ +class IsSymm (f : E [×n]→L[𝕜] F) : Prop extends f.toMultilinearMap.IsSymm + +namespace IsSymm + +variable {f g : E [×n]→L[𝕜] F} + +/-- Pointwise characterization of symmetry for continuous multilinear maps. -/ +theorem isSymm_iff : f.IsSymm ↔ + ∀ (v : Fin n → E) (σ : Equiv.Perm (Fin n)), f (v ∘ σ) = f v where + mp hf := MultilinearMap.IsSymm.isSymm_iff.mp hf.toIsSymm + mpr h := by + have hf : f.toMultilinearMap.IsSymm := MultilinearMap.IsSymm.isSymm_iff.mpr h + exact { domDomCongr_eq := hf.domDomCongr_eq } + +/-- A symmetric continuous multilinear map has the same value after permuting its arguments. -/ +@[simp] +lemma map_perm [f.IsSymm] (v : Fin n → E) (σ : Equiv.Perm (Fin n)) : + f (v ∘ σ) = f v := + isSymm_iff.mp ‹f.IsSymm› v σ + +instance zero : IsSymm (0 : E [×n]→L[𝕜] F) where + domDomCongr_eq _ := by ext; simp + +instance add [f.IsSymm] [g.IsSymm] : IsSymm (f + g) where + domDomCongr_eq σ := by + ext v + change f (v ∘ σ) + g (v ∘ σ) = f v + g v + rw [map_perm, map_perm] + +instance neg [f.IsSymm] : IsSymm (-f) where + domDomCongr_eq σ := by + ext v + change -f (v ∘ σ) = -f v + rw [map_perm] + +instance sub [f.IsSymm] [g.IsSymm] : IsSymm (f - g) where + domDomCongr_eq σ := by + ext v + change f (v ∘ σ) - g (v ∘ σ) = f v - g v + rw [map_perm, map_perm] + +/-- Differential form of the polarization identity: for a symmetric `n`-linear map, the `n`-th +derivative of its restriction to the diagonal is `n!` times the original map. -/ +theorem factorial_smul_eq_iteratedFDeriv_comp_diagonal [f.IsSymm] + (x : E) (v : Fin n → E) : + (n.factorial : 𝕜) • f v = + iteratedFDeriv 𝕜 n (fun y => f (fun _ => y)) x v := by + rw [f.iteratedFDeriv_comp_diagonal] + have hperm : ∀ σ : Equiv.Perm (Fin n), f (fun i => v (σ i)) = f v := by + intro σ + change f (v ∘ σ) = f v + exact map_perm v σ + simp_rw [hperm] + simp only [Finset.sum_const, Finset.card_univ, Fintype.card_perm, Fintype.card_fin, + Nat.cast_smul_eq_nsmul] + +end IsSymm + +variable {f g : E [×n]→L[𝕜] F} + +/-- A symmetric continuous `n`-linear map is zero if it vanishes on the diagonal and `n!` is +nonzero in the scalar field. -/ +theorem eq_zero_of_diagonal_eq_zero_of_factorial_ne_zero [f.IsSymm] + (hn : (n.factorial : 𝕜) ≠ 0) (hdiag : ∀ x : E, f (fun _ => x) = 0) : f = 0 := by + ext v + have h := IsSymm.factorial_smul_eq_iteratedFDeriv_comp_diagonal (f := f) (0 : E) v + have hfun : (fun x : E => f (fun _ => x)) = (fun _ : E => (0 : F)) := + funext hdiag + rw [hfun, iteratedFDeriv_fun_zero] at h + exact (smul_eq_zero.mp h).resolve_left hn + +/-- Over a characteristic-zero nontrivially normed field, a symmetric continuous multilinear map +is zero if it vanishes on the diagonal. -/ +theorem eq_zero_of_diagonal_eq_zero [CharZero 𝕜] [f.IsSymm] + (hdiag : ∀ x : E, f (fun _ => x) = 0) : f = 0 := + f.eq_zero_of_diagonal_eq_zero_of_factorial_ne_zero + (Nat.cast_ne_zero.mpr n.factorial_ne_zero) hdiag + +/-- Two symmetric continuous `n`-linear maps agree if their diagonal values agree and `n!` is +nonzero in the scalar field. -/ +theorem ext_of_diagonal_of_factorial_ne_zero [f.IsSymm] [g.IsSymm] + (hn : (n.factorial : 𝕜) ≠ 0) + (hdiag : ∀ x : E, f (fun _ => x) = g (fun _ => x)) : f = g := by + apply sub_eq_zero.mp + apply eq_zero_of_diagonal_eq_zero_of_factorial_ne_zero hn + intro x + simp only [sub_apply, hdiag x, sub_self] + +/-- Over a characteristic-zero nontrivially normed field, symmetric continuous multilinear maps +are determined by their diagonal values. -/ +theorem ext_of_diagonal [CharZero 𝕜] [f.IsSymm] [g.IsSymm] + (hdiag : ∀ x : E, f (fun _ => x) = g (fun _ => x)) : f = g := + ext_of_diagonal_of_factorial_ne_zero (Nat.cast_ne_zero.mpr n.factorial_ne_zero) hdiag + +/-- Equality of symmetric continuous multilinear maps is equivalent to equality on the diagonal. -/ +theorem ext_iff_of_isSymm [CharZero 𝕜] [f.IsSymm] [g.IsSymm] : + f = g ↔ ∀ x : E, f (fun _ => x) = g (fun _ => x) where + mp h := by simp [h] + mpr := ext_of_diagonal + +end ContinuousMultilinearMap + +namespace MultilinearMap + +universe u𝕜 uE uF + +variable {𝕜 : Type u𝕜} [NontriviallyNormedField 𝕜] +variable {E : Type uE} [NormedAddCommGroup E] [NormedSpace 𝕜 E] +variable {F : Type uF} [NormedAddCommGroup F] [NormedSpace 𝕜 F] +variable {n : ℕ} +variable {f : MultilinearMap 𝕜 (fun _ : Fin n => E) F} + +/-- An unbundled symmetric multilinear map which is continuous is zero if it vanishes on the +diagonal and `n!` is nonzero in the scalar field. -/ +theorem eq_zero_of_diagonal_eq_zero_of_continuous_of_factorial_ne_zero [f.IsSymm] + (hn : (n.factorial : 𝕜) ≠ 0) (hcont : Continuous f) + (hdiag : ∀ x : E, f (fun _ => x) = 0) : f = 0 := by + let fc : E [×n]→L[𝕜] F := ⟨f, hcont⟩ + have : fc.IsSymm := { domDomCongr_eq := IsSymm.domDomCongr_eq } + have hc : fc = 0 := + fc.eq_zero_of_diagonal_eq_zero_of_factorial_ne_zero hn hdiag + ext v + change fc v = 0 + rw [hc] + rfl + +/-- A continuous symmetric multilinear map over a characteristic-zero nontrivially normed field +is zero if it vanishes on the diagonal. -/ +theorem eq_zero_of_diagonal_eq_zero_of_continuous [CharZero 𝕜] [f.IsSymm] + (hcont : Continuous f) (hdiag : ∀ x : E, f (fun _ => x) = 0) : f = 0 := + f.eq_zero_of_diagonal_eq_zero_of_continuous_of_factorial_ne_zero + (Nat.cast_ne_zero.mpr n.factorial_ne_zero) hcont hdiag + +end MultilinearMap diff --git a/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean b/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean new file mode 100644 index 00000000..208b67cd --- /dev/null +++ b/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean @@ -0,0 +1,95 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Normed.Group.Quotient +public import Mathlib.MeasureTheory.Integral.Bochner.ContinuousLinearMap +public import Mathlib.Topology.Algebra.Module.ClosedSubmodule +public import Mathlib.Topology.Algebra.Module.ContinuousLinearMap.Quotient + +/-! +# Bochner integrals valued in closed submodules + +This file proves that a Bochner integral of a function taking values almost everywhere in a +closed submodule still belongs to that submodule. The main ingredient is the continuous linear +quotient map: after passing to the quotient, the integrand is almost everywhere zero. +-/ + +@[expose] public section + +open MeasureTheory + +namespace ContinuousLinearMap + +variable {α E F 𝕜 : Type*} [MeasurableSpace α] {μ : Measure α} + [RCLike 𝕜] [NormedAddCommGroup E] [NormedSpace 𝕜 E] [NormedSpace ℝ E] + [NormedAddCommGroup F] [NormedSpace 𝕜 F] [NormedSpace ℝ F] + [CompleteSpace F] + +/-- The Bochner integral of a function valued almost everywhere in the kernel of a continuous +linear map still belongs to its kernel. -/ +theorem integral_mem_ker (L : E →L[𝕜] F) {f : α → E} + (hf : ∀ᵐ x ∂μ, f x ∈ L.ker) : + (∫ x, f x ∂μ) ∈ L.ker := by + by_cases hE : CompleteSpace E + · let _ := hE + by_cases hfi : Integrable f μ + · apply LinearMap.mem_ker.mpr + change L (∫ x, f x ∂μ) = 0 + rw [← L.integral_comp_comm hfi] + exact integral_eq_zero_of_ae (hf.mono fun x hx ↦ by + change L (f x) = 0 + exact LinearMap.mem_ker.mp hx) + · simp [integral_undef hfi] + · simp [integral, hE] + +end ContinuousLinearMap + +namespace Submodule + +variable {α E 𝕜 : Type*} [MeasurableSpace α] {μ : Measure α} + [RCLike 𝕜] [NormedAddCommGroup E] [NormedSpace 𝕜 E] [NormedSpace ℝ E] + +/-- The Bochner integral of a function valued almost everywhere in a closed submodule belongs to +that submodule. No integrability or completeness assumption is needed: in the remaining cases, +the Bochner integral is defined to be zero. -/ +theorem integral_mem (S : Submodule 𝕜 E) (hS : IsClosed (S : Set E)) + {f : α → E} (hf : ∀ᵐ x ∂μ, f x ∈ S) : + (∫ x, f x ∂μ) ∈ S := by + by_cases hE : CompleteSpace E + · let _ := hE + let _ : IsClosed (S : Set E) := hS + rw [← S.ker_mkQ] + exact S.mkQL.integral_mem_ker (by + filter_upwards [hf] with x hx + apply LinearMap.mem_ker.mpr + change S.mkQL (f x) = 0 + exact (Submodule.Quotient.mk_eq_zero S).2 hx) + · simp [integral, hE] + +/-- The Bochner integral of a function valued almost everywhere in a submodule belongs to the +topological closure of that submodule. -/ +theorem integral_mem_topologicalClosure (S : Submodule 𝕜 E) {f : α → E} + (hf : ∀ᵐ x ∂μ, f x ∈ S) : + (∫ x, f x ∂μ) ∈ S.topologicalClosure := + S.topologicalClosure.integral_mem S.isClosed_topologicalClosure + (hf.mono fun _ hx ↦ S.le_topologicalClosure hx) + +end Submodule + +namespace ClosedSubmodule + +variable {α E 𝕜 : Type*} [MeasurableSpace α] {μ : Measure α} + [RCLike 𝕜] [NormedAddCommGroup E] [NormedSpace 𝕜 E] [NormedSpace ℝ E] + +/-- The Bochner integral of a function valued almost everywhere in a closed submodule belongs to +that closed submodule. -/ +theorem integral_mem (S : ClosedSubmodule 𝕜 E) {f : α → E} + (hf : ∀ᵐ x ∂μ, f x ∈ S) : + (∫ x, f x ∂μ) ∈ S := + S.toSubmodule.integral_mem S.isClosed (by simpa using hf) + +end ClosedSubmodule diff --git a/LeanMachineLearning/ForMathlib/Topology/Algebra/Module/FiniteDimension.lean b/LeanMachineLearning/ForMathlib/Topology/Algebra/Module/FiniteDimension.lean new file mode 100644 index 00000000..4a022eff --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Topology/Algebra/Module/FiniteDimension.lean @@ -0,0 +1,45 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Topology.Algebra.Module.FiniteDimension + +/-! +# Density and finite-dimensional submodules + +This file records that no subset of a proper finite-dimensional submodule can be dense. +-/ + +@[expose] public section + +open Set + +namespace Submodule + +variable {𝕜 E : Type*} [NontriviallyNormedField 𝕜] [CompleteSpace 𝕜] +variable [AddCommGroup E] [TopologicalSpace E] [IsTopologicalAddGroup E] +variable [Module 𝕜 E] [ContinuousSMul 𝕜 E] [T2Space E] + +/-- A subset of a proper finite-dimensional submodule is not dense in the ambient space. -/ +theorem not_dense_of_subset_of_finiteDimensional (s : Submodule 𝕜 E) + [FiniteDimensional 𝕜 s] (hs : s ≠ ⊤) {t : Set E} (ht : t ⊆ s) : + ¬ Dense t := by + intro ht_dense + apply hs + apply top_unique + intro x _ + have hx : x ∈ closure (s : Set E) := by + rw [(ht_dense.mono ht).closure_eq] + exact Set.mem_univ x + rwa [s.closed_of_finiteDimensional.closure_eq] at hx + +/-- A proper finite-dimensional submodule is not dense in the ambient space. -/ +theorem not_dense_of_finiteDimensional (s : Submodule 𝕜 E) + [FiniteDimensional 𝕜 s] (hs : s ≠ ⊤) : + ¬ Dense (s : Set E) := + s.not_dense_of_subset_of_finiteDimensional hs fun _ ↦ id + +end Submodule diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean new file mode 100644 index 00000000..d2bbb381 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean @@ -0,0 +1,40 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Topology.ContinuousMap.Compact + +/-! +# Dense families of continuous maps on compact spaces + +This file gives the uniform epsilon formulation of density in a continuous-map space with +compact domain. +-/ + +@[expose] public section + +namespace ContinuousMap + +variable {X Y : Type*} [TopologicalSpace X] [CompactSpace X] [PseudoMetricSpace Y] + +/-- A family of continuous maps on a compact space is dense exactly when every continuous map can +be approximated pointwise with one uniform positive error bound. -/ +theorem dense_iff_forall_exists_forall_dist_lt {S : Set C(X, Y)} : + Dense S ↔ ∀ (f : C(X, Y)) (ε : ℝ), 0 < ε → + ∃ g ∈ S, ∀ x, dist (g x) (f x) < ε := by + rw [Metric.dense_iff] + constructor + · intro h f ε hε + obtain ⟨g, hgBall, hgS⟩ := h f ε hε + refine ⟨g, hgS, ?_⟩ + rwa [Metric.mem_ball, ContinuousMap.dist_lt_iff hε] at hgBall + · intro h f ε hε + obtain ⟨g, hgS, hg⟩ := h f ε hε + refine ⟨g, ?_, hgS⟩ + rw [Metric.mem_ball, ContinuousMap.dist_lt_iff hε] + exact hg + +end ContinuousMap diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean new file mode 100644 index 00000000..835738cb --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean @@ -0,0 +1,42 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Topology.ContinuousMap.Algebra + +/-! +# Continuous maps from a discrete space + +This file upgrades the equivalence between continuous maps from a discrete space and arbitrary +functions to a linear equivalence. +-/ + +universe u v w + +@[expose] public section + +namespace ContinuousMap + +variable (R : Type u) {X : Type v} {M : Type w} +variable [Semiring R] [TopologicalSpace X] [DiscreteTopology X] +variable [TopologicalSpace M] [AddCommMonoid M] [ContinuousAdd M] +variable [Module R M] [ContinuousConstSMul R M] + +/-- Continuous maps from a discrete space are linearly equivalent to arbitrary functions. -/ +def linearEquivFnOfDiscrete : C(X, M) ≃ₗ[R] (X → M) where + __ := equivFnOfDiscrete + map_add' _ _ := rfl + map_smul' _ _ := rfl + +@[simp] +theorem linearEquivFnOfDiscrete_apply (f : C(X, M)) (x : X) : + linearEquivFnOfDiscrete R f x = f x := rfl + +@[simp] +theorem linearEquivFnOfDiscrete_symm_apply_apply (f : X → M) (x : X) : + (linearEquivFnOfDiscrete R).symm f x = f x := rfl + +end ContinuousMap diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean new file mode 100644 index 00000000..bca8e520 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean @@ -0,0 +1,68 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.InnerProductSpace.Continuous +public import Mathlib.Topology.ContinuousMap.StoneWeierstrass + +/-! +# Density of polynomials in inner-product coordinates + +The linear coordinate functions given by an inner product separate points. Consequently, on a +compact subtype, the algebra that they generate is dense in the continuous real-valued functions. +-/ + +@[expose] public section + +namespace ContinuousMap + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- The algebra generated by the real inner-product coordinates on a subtype separates points. + +No compactness or finite-dimensionality assumption is needed for this fact. +-/ +theorem innerProduct_adjoin_separatesPoints (K : Set E) : + (Algebra.adjoin ℝ + (Set.range fun w : E => + (⟨fun x : K => inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ : + C(K, ℝ)))).SeparatesPoints := by + intro x y hxy + let w : E := x.1 - y.1 + let f : C(K, ℝ) := + ⟨fun z => inner ℝ w z.1, continuous_const.inner continuous_subtype_val⟩ + refine ⟨f, ?_, ?_⟩ + · exact ⟨f, ⟨Algebra.subset_adjoin ⟨w, rfl⟩, rfl⟩⟩ + · intro h + apply hxy + apply Subtype.ext + apply sub_eq_zero.mp + apply (inner_self_eq_zero.mp : inner ℝ w w = 0 → w = 0) + rw [inner_sub_right] + exact sub_eq_zero.mpr h + +/-- On a compact subtype of a real inner-product space, the closure of the algebra generated by +the inner-product coordinates is the full algebra of continuous real-valued functions. -/ +theorem innerProduct_adjoin_topologicalClosure_eq_top (K : Set E) [CompactSpace K] : + (Algebra.adjoin ℝ + (Set.range fun w : E => + (⟨fun x : K => inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ : + C(K, ℝ)))).topologicalClosure = ⊤ := + subalgebra_topologicalClosure_eq_top_of_separatesPoints _ + (innerProduct_adjoin_separatesPoints K) + +/-- On a compact subtype of a real inner-product space, the algebra generated by inner-product +coordinates is dense in the continuous real-valued functions. -/ +theorem dense_innerProduct_adjoin (K : Set E) [CompactSpace K] : + Dense (Algebra.adjoin ℝ + (Set.range fun w : E => + (⟨fun x : K => inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ : + C(K, ℝ))) : Set C(K, ℝ)) := by + rw [dense_iff_closure_eq, ← Subalgebra.topologicalClosure_coe, + innerProduct_adjoin_topologicalClosure_eq_top] + rfl + +end ContinuousMap diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean new file mode 100644 index 00000000..d2e76ce5 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean @@ -0,0 +1,192 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.ForMathlib.LinearAlgebra.Multilinear.Polarization +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.InnerProduct +public import Mathlib.Algebra.Group.Submonoid.Membership + +/-! +# Determinacy by polynomial moments + +This file proves that a continuous linear functional on continuous functions over a compact space +vanishes if all of its moments along a linearly parametrized, algebraically generating family +vanish. Polarization first recovers mixed moments from pure powers; the algebraic span description +of `Algebra.adjoin` and density then finish the proof. + +The general theorem works over any characteristic-zero nontrivially normed field. We also provide +the specialization to the inner-product coordinates of a compact subset of a real inner-product +space. No finite-dimensionality assumption on the ambient inner-product space is needed. +-/ + +@[expose] public section + +open scoped BigOperators + +namespace StrongDual + +section Coordinate + +variable {𝕜 X V : Type*} [NontriviallyNormedField 𝕜] + [TopologicalSpace X] [CompactSpace X] + [NormedAddCommGroup V] [NormedSpace 𝕜 V] + +/-- The multilinear moment obtained by applying `Λ` to a product of coordinate functions. -/ +noncomputable def coordinateMoment (coordinate : V →L[𝕜] C(X, 𝕜)) + (Λ : StrongDual 𝕜 C(X, 𝕜)) (n : ℕ) : V [×n]→L[𝕜] 𝕜 := + Λ.compContinuousMultilinearMap <| + (ContinuousMultilinearMap.mkPiAlgebra 𝕜 (Fin n) C(X, 𝕜)).compContinuousLinearMap + fun _ ↦ coordinate + +@[simp] +theorem coordinateMoment_apply (coordinate : V →L[𝕜] C(X, 𝕜)) + (Λ : StrongDual 𝕜 C(X, 𝕜)) (n : ℕ) (v : Fin n → V) : + coordinateMoment coordinate Λ n v = Λ (∏ i, coordinate (v i)) := by + simp [coordinateMoment] + +instance coordinateMoment_isSymm (coordinate : V →L[𝕜] C(X, 𝕜)) + (Λ : StrongDual 𝕜 C(X, 𝕜)) (n : ℕ) : (coordinateMoment coordinate Λ n).IsSymm := by + rw [ContinuousMultilinearMap.IsSymm.isSymm_iff] + intro v e + simp only [coordinateMoment_apply, Function.comp_apply] + congr 1 + exact Equiv.prod_comp e (fun i ↦ coordinate (v i)) + +end Coordinate + +section CharZero + +variable {𝕜 X V : Type*} [NontriviallyNormedField 𝕜] [CharZero 𝕜] + [TopologicalSpace X] [CompactSpace X] + [NormedAddCommGroup V] [NormedSpace 𝕜 V] + +/-- A continuous linear functional is zero if it annihilates every pure coordinate power and the +coordinates algebraically generate a dense subalgebra. + +The proof first uses polarization to show that all mixed coordinate products vanish. Such products +span the generated algebra, so continuity and density imply that the functional is zero everywhere. +-/ +theorem eq_zero_of_coordinate_powers + (coordinate : V →L[𝕜] C(X, 𝕜)) + (hdense : Dense (Algebra.adjoin 𝕜 (Set.range coordinate) : Set C(X, 𝕜))) + (Λ : StrongDual 𝕜 C(X, 𝕜)) + (hpow : ∀ (n : ℕ) (v : V), Λ ((coordinate v) ^ n) = 0) : + Λ = 0 := by + have hproduct : ∀ (n : ℕ) (v : Fin n → V), Λ (∏ i, coordinate (v i)) = 0 := by + intro n v + have hdiag : ∀ w : V, coordinateMoment coordinate Λ n (fun _ ↦ w) = 0 := by + intro w + rw [coordinateMoment_apply] + simpa using hpow n w + have hm : coordinateMoment coordinate Λ n = 0 := + (coordinateMoment coordinate Λ n).eq_zero_of_diagonal_eq_zero hdiag + have hv := congrArg (fun p : V [×n]→L[𝕜] 𝕜 ↦ p v) hm + simpa using hv + have hmonoid : ∀ p ∈ Submonoid.closure (Set.range coordinate), Λ p = 0 := by + intro p hp + obtain ⟨l, hl, rfl⟩ := Submonoid.exists_list_of_mem_closure hp + have hexi : ∀ i : Fin l.length, ∃ v : V, coordinate v = l[i] := by + intro i + exact hl l[i] (List.getElem_mem ..) + choose v hv using hexi + rw [← Fin.prod_univ_getElem l] + convert hproduct l.length v using 1 + congr 1 + apply Finset.prod_congr rfl + intro i _ + exact (hv i).symm + have hadjoin : ∀ p ∈ Algebra.adjoin 𝕜 (Set.range coordinate), Λ p = 0 := by + intro p hp + change p ∈ (Algebra.adjoin 𝕜 (Set.range coordinate)).toSubmodule at hp + rw [Algebra.adjoin_eq_span] at hp + have hle : Submodule.span 𝕜 + (Submonoid.closure (Set.range coordinate) : Set C(X, 𝕜)) ≤ Λ.ker := by + rw [Submodule.span_le] + intro q hq + exact hmonoid q hq + exact hle hp + apply ContinuousLinearMap.ext_on + (hdense.mono (Submodule.subset_span (R := 𝕜))) + intro p hp + simpa using hadjoin p hp + +end CharZero + +end StrongDual + +namespace ContinuousMap + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- The real inner-product coordinate `x ↦ ⟪w, x⟫` on a subtype. -/ +def innerProductCoordinate (K : Set E) (w : E) : C(K, ℝ) := + ⟨fun x ↦ inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ + +@[simp] +theorem innerProductCoordinate_apply (K : Set E) (w : E) (x : K) : + innerProductCoordinate K w x = inner ℝ w x.1 := rfl + +/-- The continuous linear map sending a vector to its inner-product coordinate on a compact +subtype. Compactness bounds the subtype, so finite-dimensionality is not required. -/ +noncomputable def innerProductCoordinateCLM (K : Set E) [CompactSpace K] : + E →L[ℝ] C(K, ℝ) := by + let coordinate : E →ₗ[ℝ] C(K, ℝ) := + { toFun := innerProductCoordinate K + map_add' := by + intro x y + ext z + simp [innerProductCoordinate, inner_add_left] + map_smul' := by + intro c x + ext z + simp [innerProductCoordinate, real_inner_smul_left] } + apply coordinate.mkContinuousOfExistsBound + have hcompact : IsCompact (Set.range fun x : K ↦ x.1) := + isCompact_range continuous_subtype_val + obtain ⟨R, hRpos, hR⟩ := hcompact.isBounded.exists_pos_norm_le + refine ⟨R, fun w ↦ + (ContinuousMap.norm_le (innerProductCoordinate K w) ?_).2 (fun x ↦ ?_)⟩ + · positivity + · change ‖inner ℝ w x.1‖ ≤ R * ‖w‖ + calc + ‖inner ℝ w x.1‖ ≤ ‖w‖ * ‖x.1‖ := norm_inner_le_norm _ _ + _ ≤ ‖w‖ * R := mul_le_mul_of_nonneg_left (hR x.1 ⟨x, rfl⟩) (norm_nonneg _) + _ = R * ‖w‖ := mul_comm _ _ + +@[simp] +theorem innerProductCoordinateCLM_apply (K : Set E) [CompactSpace K] (w : E) : + innerProductCoordinateCLM K w = innerProductCoordinate K w := by + rfl + +end ContinuousMap + +namespace StrongDual + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- A continuous linear functional on a compact subtype of a real inner-product space is zero if +it annihilates every power of every inner-product coordinate. -/ +theorem eq_zero_of_innerProductCoordinate_powers (K : Set E) [CompactSpace K] + (Λ : StrongDual ℝ C(K, ℝ)) + (hpow : ∀ (n : ℕ) (w : E), + Λ ((ContinuousMap.innerProductCoordinate K w) ^ n) = 0) : + Λ = 0 := by + apply eq_zero_of_coordinate_powers (ContinuousMap.innerProductCoordinateCLM K) + (Λ := Λ) + · have hrange : Set.range (ContinuousMap.innerProductCoordinateCLM K) = + Set.range (ContinuousMap.innerProductCoordinate K) := by + ext f + constructor + · rintro ⟨w, rfl⟩ + exact ⟨w, (ContinuousMap.innerProductCoordinateCLM_apply K w).symm⟩ + · rintro ⟨w, rfl⟩ + exact ⟨w, ContinuousMap.innerProductCoordinateCLM_apply K w⟩ + rw [hrange] + exact ContinuousMap.dense_innerProduct_adjoin K + · intro n w + simpa only [ContinuousMap.innerProductCoordinateCLM_apply] using hpow n w + +end StrongDual diff --git a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean new file mode 100644 index 00000000..d7cd640a --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean @@ -0,0 +1,125 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.InnerProductSpace.Continuous +public import Mathlib.LinearAlgebra.Finsupp.LinearCombination +public import Mathlib.Topology.ContinuousMap.Algebra +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Dense + +/-! +# Single-hidden-layer neural networks + +We define a neuron on a real inner product space and the vector space of finite-width shallow +networks generated by an activation function. Using `Submodule.span` means that finite linear +combinations and all their algebraic laws come from Mathlib's existing linear-algebra API. +-/ + +@[expose] public section + +namespace Learning.ShallowNetwork + +variable {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- A single neuron with weight `w`, bias `b`, and activation `σ`. -/ +def neuron (σ : C(ℝ, ℝ)) (w : E) (b : ℝ) : C(E, ℝ) := + σ.comp ⟨fun x ↦ inner ℝ w x + b, + (continuous_const.inner continuous_id).add continuous_const⟩ + +@[simp] +theorem neuron_apply (σ : C(ℝ, ℝ)) (w : E) (b : ℝ) (x : E) : + neuron σ w b x = σ (inner ℝ w x + b) := rfl + +/-- The real vector space of finite-width, single-hidden-layer networks with activation `σ`. -/ +def space (σ : C(ℝ, ℝ)) : Submodule ℝ C(E, ℝ) := + Submodule.span ℝ (Set.range fun p : E × ℝ ↦ neuron σ p.1 p.2) + +/-- The restrictions to `K` of finite-width, single-hidden-layer networks with activation `σ`. -/ +def spaceOn (σ : C(ℝ, ℝ)) (K : Set E) : Submodule ℝ C(K, ℝ) := + Submodule.span ℝ (Set.range fun p : E × ℝ ↦ (neuron σ p.1 p.2).restrict K) + +/-- Membership in `space` is exactly representability by a finite-width shallow network. -/ +theorem mem_space_iff (σ : C(ℝ, ℝ)) (f : C(E, ℝ)) : + f ∈ space σ ↔ + ∃ (m : ℕ) (a : Fin m → ℝ) (w : Fin m → E) (b : Fin m → ℝ), + ∑ j, a j • neuron σ (w j) (b j) = f := by + constructor + · intro hf + rw [space, Submodule.mem_span_set'] at hf + obtain ⟨m, a, g, h⟩ := hf + have hg : ∀ j, ∃ w b, neuron σ w b = (g j : C(E, ℝ)) := by + intro j + obtain ⟨p, hp⟩ := (g j).property + exact ⟨p.1, p.2, hp⟩ + choose w b hb using hg + exact ⟨m, a, w, b, by simpa only [← hb] using h⟩ + · rintro ⟨m, a, w, b, rfl⟩ + apply Submodule.sum_mem + intro j hj + apply Submodule.smul_mem + exact Submodule.subset_span ⟨(w j, b j), rfl⟩ + +/-- Membership in `spaceOn` is exactly representability on `K` by a finite-width shallow +network. -/ +theorem mem_spaceOn_iff (σ : C(ℝ, ℝ)) (K : Set E) (f : C(K, ℝ)) : + f ∈ spaceOn σ K ↔ + ∃ (m : ℕ) (a : Fin m → ℝ) (w : Fin m → E) (b : Fin m → ℝ), + ∑ j, a j • (neuron σ (w j) (b j)).restrict K = f := by + constructor + · intro hf + rw [spaceOn, Submodule.mem_span_set'] at hf + obtain ⟨m, a, g, h⟩ := hf + have hg : ∀ j, ∃ w b, (neuron σ w b).restrict K = (g j : C(K, ℝ)) := by + intro j + obtain ⟨p, hp⟩ := (g j).property + exact ⟨p.1, p.2, hp⟩ + choose w b hb using hg + exact ⟨m, a, w, b, by simpa only [← hb] using h⟩ + · rintro ⟨m, a, w, b, rfl⟩ + apply Submodule.sum_mem + intro j hj + apply Submodule.smul_mem + exact Submodule.subset_span ⟨(w j, b j), rfl⟩ + +/-- `spaceOn` is the image of the global network space under restriction. -/ +theorem spaceOn_eq_map (σ : C(ℝ, ℝ)) (K : Set E) : + spaceOn σ K = + (space σ).map + (ContinuousMap.compCLM ℝ ℝ + ⟨((↑) : K → E), continuous_subtype_val⟩).toLinearMap := by + rw [spaceOn, space, Submodule.map_span] + congr 1 + ext f + simp only [Set.mem_range, Set.mem_image] + aesop + +/-- An activation is universal on `E` if its shallow networks are dense on every compact +subset. This class packages the property for downstream approximation theorems. -/ +class IsUniversal (σ : C(ℝ, ℝ)) : Prop where + dense_on_compact : ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) + +/-- The typeclass formulation of universality unfolds to density on every compact subset. -/ +theorem isUniversal_iff (σ : C(ℝ, ℝ)) : + IsUniversal (E := E) σ ↔ + ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := by + grind [IsUniversal] + +/-- The usual uniform epsilon formulation of universal approximation on every compact set. -/ +theorem isUniversal_iff_uniform_approximation (σ : C(ℝ, ℝ)) : + IsUniversal (E := E) σ ↔ + ∀ (K : Set E), IsCompact K → ∀ (f : C(K, ℝ)) (ε : ℝ), 0 < ε → + ∃ g ∈ spaceOn σ K, ∀ x, dist (g x) (f x) < ε := by + constructor + · rintro ⟨h⟩ K hK + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + exact ContinuousMap.dense_iff_forall_exists_forall_dist_lt.mp (h K hK) + · intro h + constructor + intro K hK + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + exact ContinuousMap.dense_iff_forall_exists_forall_dist_lt.mpr (h K hK) + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean new file mode 100644 index 00000000..f48b1766 --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean @@ -0,0 +1,212 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.Calculus.ContDiff.Convolution +public import Mathlib.MeasureTheory.Measure.Haar.OfBasis +public import Mathlib.MeasureTheory.Measure.Haar.Unique +public import Mathlib.Topology.ContinuousMap.CompactlySupported +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction +public import LeanMachineLearning.ForMathlib.MeasureTheory.Integral.ClosedSubmodule +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Discriminatory + +/-! +# Convolution smoothing of activation functions + +Convolution against a compactly supported continuous kernel turns an activation into another +continuous activation. On a compact domain, a ridge function for the convolved activation is a +Bochner integral of ridge functions for the original activation. Consequently it belongs to the +closure of their span, and every functional annihilating the original neurons also annihilates the +convolved neurons. + +The kernel is abstracted by `CompactlySupportedContinuousMapClass`, so this file applies directly +both to compactly supported continuous maps and to smooth test functions. +-/ + +@[expose] public section + +open MeasureTheory + +namespace Learning.ShallowNetwork + +/-- Convolution of a continuous activation with a compactly supported continuous kernel. + +The convention is +`convolutionActivation φ σ t = ∫ s, φ s * σ (t - s)`. -/ +noncomputable def convolutionActivation {B : Type*} [FunLike B ℝ ℝ] + [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) : C(ℝ, ℝ) := by + letI : (volume : Measure ℝ).IsNegInvariant := + Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume + exact + ⟨MeasureTheory.convolution φ σ (ContinuousLinearMap.mul ℝ ℝ) volume, + (CompactlySupportedContinuousMapClass.hasCompactSupport φ).continuous_convolution_left + (ContinuousLinearMap.mul ℝ ℝ) + (ContinuousMapClass.map_continuous φ) σ.continuous.locallyIntegrable⟩ + +@[simp] +theorem convolutionActivation_apply {B : Type*} [FunLike B ℝ ℝ] + [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (t : ℝ) : + convolutionActivation φ σ t = ∫ s, φ s * σ (t - s) := by + exact MeasureTheory.convolution_mul + +/-- Convolution with a compactly supported `C^n` kernel makes a continuous activation `C^n`. -/ +theorem convolutionActivation_contDiff {B : Type*} [FunLike B ℝ ℝ] + [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) {n : ℕ∞} (hφ : ContDiff ℝ n φ) : + ContDiff ℝ n (convolutionActivation φ σ) := by + let _ : (volume : Measure ℝ).IsNegInvariant := + Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume + exact (CompactlySupportedContinuousMapClass.hasCompactSupport φ).contDiff_convolution_left + (ContinuousLinearMap.mul ℝ ℝ) hφ σ.continuous.locallyIntegrable + +/-- The activation `σ` applied to a scalar-valued continuous feature `u`, with bias `b`. -/ +def activationAlong {X : Type*} [TopologicalSpace X] (σ : C(ℝ, ℝ)) + (u : C(X, ℝ)) (b : ℝ) : C(X, ℝ) := + σ.comp (u + ContinuousMap.const X b) + +@[simp] +theorem activationAlong_apply {X : Type*} [TopologicalSpace X] + (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (b : ℝ) (x : X) : + activationAlong σ u b x = σ (u x + b) := rfl + +/-- The `C(X, ℝ)`-valued integrand expressing a ridge function of a convolved activation is +Bochner integrable. -/ +theorem integrable_smul_activationAlong_sub + {X : Type*} [TopologicalSpace X] [CompactSpace X] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (b : ℝ) : + Integrable (fun s : ℝ => φ s • activationAlong σ u (b - s)) := by + apply Continuous.integrable_of_hasCompactSupport + · exact (ContinuousMapClass.map_continuous φ).smul + (ContinuousMap.continuous_of_continuous_uncurry _ <| + σ.continuous.comp + ((u.continuous.comp continuous_snd).add + (continuous_const.sub continuous_fst))) + · exact (CompactlySupportedContinuousMapClass.hasCompactSupport φ).smul_right + +/-- A ridge function of a convolved activation is the Bochner integral of shifted ridge functions +of the original activation. -/ +theorem activationAlong_convolutionActivation + {X : Type*} [TopologicalSpace X] [CompactSpace X] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (b : ℝ) : + activationAlong (convolutionActivation φ σ) u b = + ∫ s : ℝ, φ s • activationAlong σ u (b - s) := by + apply ContinuousMap.ext + intro x + rw [ContinuousMap.integral_apply (integrable_smul_activationAlong_sub φ σ u b)] + simp only [activationAlong_apply, convolutionActivation_apply] + congr 1 + funext s + change φ s * σ (u x + b - s) = φ s * σ (u x + (b - s)) + simp only [sub_eq_add_neg, add_assoc] + +/-- A ridge function of a convolved activation belongs to the closure of any submodule containing +all bias translates of the corresponding ridge function for the original activation. -/ +theorem activationAlong_convolutionActivation_mem_topologicalClosure + {X : Type*} [TopologicalSpace X] [CompactSpace X] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (b : ℝ) + (S : Submodule ℝ C(X, ℝ)) (hS : ∀ c, activationAlong σ u c ∈ S) : + activationAlong (convolutionActivation φ σ) u b ∈ S.topologicalClosure := by + rw [activationAlong_convolutionActivation] + apply S.integral_mem_topologicalClosure + filter_upwards with s + exact S.smul_mem (φ s) (hS (b - s)) + +/-- Every neuron for a convolved activation lies in the closure of the shallow-network space for +the original activation. -/ +theorem convolved_neuron_mem_spaceOn_topologicalClosure + {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (K : Set E) (hK : IsCompact K) (w : E) (b : ℝ) : + (neuron (convolutionActivation φ σ) w b).restrict K ∈ + (spaceOn σ K).topologicalClosure := by + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + let u : C(K, ℝ) := + ⟨fun x => inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ + rw [show (neuron (convolutionActivation φ σ) w b).restrict K = + activationAlong (convolutionActivation φ σ) u b by ext; rfl] + apply activationAlong_convolutionActivation_mem_topologicalClosure φ σ u b + intro c + rw [show activationAlong σ u c = (neuron σ w c).restrict K by ext; rfl] + exact Submodule.subset_span ⟨(w, c), rfl⟩ + +/-- On every compact set, the network space of a convolved activation is contained in the closure +of the network space of the original activation. -/ +theorem convolved_spaceOn_le_topologicalClosure + {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (K : Set E) (hK : IsCompact K) : + spaceOn (convolutionActivation φ σ) K ≤ (spaceOn σ K).topologicalClosure := by + rw [spaceOn] + apply Submodule.span_le.2 + rintro f ⟨p, rfl⟩ + exact convolved_neuron_mem_spaceOn_topologicalClosure φ σ K hK p.1 p.2 + +/-- A continuous functional annihilating all translates of a ridge function also annihilates the +corresponding ridge function for every compactly supported convolution smoothing. -/ +theorem annihilates_convolutionActivation + {X : Type*} [TopologicalSpace X] [CompactSpace X] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (Λ : StrongDual ℝ C(X, ℝ)) + (hΛ : ∀ b, Λ (activationAlong σ u b) = 0) (b : ℝ) : + Λ (activationAlong (convolutionActivation φ σ) u b) = 0 := by + rw [activationAlong_convolutionActivation, + ← Λ.integral_comp_comm (integrable_smul_activationAlong_sub φ σ u b)] + simp [hΛ] + +/-- A continuous functional annihilating all neurons for `σ` also annihilates all neurons for a +compactly supported convolution smoothing of `σ`. -/ +theorem annihilates_convolutionActivation_neurons + {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) (K : Set E) (hK : IsCompact K) + (Λ : StrongDual ℝ C(K, ℝ)) + (hΛ : ∀ w b, Λ ((neuron σ w b).restrict K) = 0) : + ∀ w b, Λ ((neuron (convolutionActivation φ σ) w b).restrict K) = 0 := by + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + intro w b + let u : C(K, ℝ) := + ⟨fun x => inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ + have htrans : ∀ c, Λ (activationAlong σ u c) = 0 := by + intro c + rw [show activationAlong σ u c = (neuron σ w c).restrict K by ext; rfl] + exact hΛ w c + rw [show (neuron (convolutionActivation φ σ) w b).restrict K = + activationAlong (convolutionActivation φ σ) u b by ext; rfl] + exact annihilates_convolutionActivation φ σ u Λ htrans b + +/-- If one convolution smoothing of `σ` is discriminatory, then `σ` itself is discriminatory. -/ +theorem isDiscriminatory_of_convolutionActivation + {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) + [IsDiscriminatory (E := E) (convolutionActivation φ σ)] : + IsDiscriminatory (E := E) σ := by + constructor + intro K hK Λ hΛ + apply IsDiscriminatory.annihilator_eq_zero + (σ := convolutionActivation φ σ) (E := E) K hK Λ + exact annihilates_convolutionActivation_neurons φ σ K hK Λ hΛ + +/-- Universality of one compactly supported convolution smoothing implies universality of the +original activation. -/ +theorem isUniversal_of_convolutionActivation + {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] + (φ : B) (σ : C(ℝ, ℝ)) + [IsUniversal (E := E) (convolutionActivation φ σ)] : + IsUniversal (E := E) σ := by + apply (isUniversal_iff_isDiscriminatory σ).mpr + let _ : IsDiscriminatory (E := E) (convolutionActivation φ σ) := + (isUniversal_iff_isDiscriminatory (convolutionActivation φ σ)).mp + (inferInstance : IsUniversal (E := E) (convolutionActivation φ σ)) + exact isDiscriminatory_of_convolutionActivation φ σ + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean new file mode 100644 index 00000000..cb6b776d --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean @@ -0,0 +1,95 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Convolution +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.NonpolynomialWitness +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.SmoothActivation + +/-! +# Derivative towers for convolution-smoothed activations + +Convolving a continuous real activation with a smooth compactly supported test function produces +a smooth activation. More precisely, its `n`-th derivative is the convolution whose kernel is +obtained by applying `TestFunction.lineDerivCLM` `n` times. + +The resulting derivative tower is registered as an instance of +`HasContinuousDerivativeTower`. The last theorem combines its value at the origin with the +distributional witness for a nonpolynomial activation. +-/ + +@[expose] public section + +open MeasureTheory +open scoped Distributions + +namespace Learning.ShallowNetwork + +/-- Successive derivatives of the left kernel give successive derivatives of its convolution +with a continuous activation. -/ +theorem hasDerivAt_convolutionActivation_iterate_lineDerivCLM + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) (x : ℝ) : + HasDerivAt + (convolutionActivation + (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ) + (convolutionActivation + (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) σ x) x := by + let _ : (volume : Measure ℝ).IsNegInvariant := + Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume + have h := + (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ).hasCompactSupport + |>.hasDerivAt_convolution_left + (μ := volume) (ContinuousLinearMap.mul ℝ ℝ) + ((TestFunction.contDiff + (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ)).of_le (by simp)) + σ.continuous.locallyIntegrable x + convert h using 1 + · rfl + · rw [Function.iterate_succ_apply'] + congr 1 + +/-- The canonical continuous derivative tower on the convolution of a test function with a +continuous activation. -/ +noncomputable instance instHasContinuousDerivativeTowerConvolutionActivation + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) : + HasContinuousDerivativeTower (convolutionActivation φ σ) where + derivative n := + convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ + derivative_zero := by simp + hasDerivAt_derivative := + hasDerivAt_convolutionActivation_iterate_lineDerivCLM φ σ + +/-- The `n`-th member of the canonical derivative tower is convolution with the `n`-fold +derivative of the test-function kernel. -/ +@[simp] +theorem derivative_convolutionActivation_testFunction + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : + HasContinuousDerivativeTower.derivative (convolutionActivation φ σ) n = + convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ := + rfl + +/-- At the origin, the `n`-th derivative of a test-function convolution is the pairing of the +`n`-fold derivative of its kernel with the reflected activation. -/ +theorem derivative_convolutionActivation_testFunction_apply_zero + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : + HasContinuousDerivativeTower.derivative (convolutionActivation φ σ) n 0 = + ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by + rw [derivative_convolutionActivation_testFunction, convolutionActivation_apply] + simp only [zero_sub] + +/-- For every order, a nonpolynomial continuous activation has a test-function convolution whose +canonical derivative tower is nonzero at the origin in that order. -/ +theorem exists_testFunction_derivative_convolutionActivation_ne_zero + (σ : C(ℝ, ℝ)) (hσ : ¬ Function.IsPolynomial σ) (n : ℕ) : + ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), + HasContinuousDerivativeTower.derivative (convolutionActivation φ σ) n 0 ≠ 0 := by + obtain ⟨φ, hφ⟩ := + exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero σ hσ n + refine ⟨φ, ?_⟩ + rw [derivative_convolutionActivation_testFunction_apply_zero] + exact hφ + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean new file mode 100644 index 00000000..d5fd3dee --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean @@ -0,0 +1,60 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.ForMathlib.Analysis.LocallyConvex.Annihilator +public import LeanMachineLearning.NeuralNetwork.Shallow.Basic + +/-! +# Discriminatory activation functions + +An activation is discriminatory on an input space if the only continuous linear functional on +`C(K, ℝ)` that annihilates every neuron is zero, for every compact `K`. Hahn--Banach makes this +property equivalent to universal approximation. +-/ + +@[expose] public section + +namespace Learning.ShallowNetwork + +variable {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- An activation is discriminatory on `E` if no nonzero continuous linear functional annihilates +all of its neurons on a compact subset of `E`. -/ +class IsDiscriminatory (σ : C(ℝ, ℝ)) : Prop where + annihilator_eq_zero : + ∀ (K : Set E), IsCompact K → ∀ Λ : StrongDual ℝ C(K, ℝ), + (∀ w b, Λ ((neuron σ w b).restrict K) = 0) → Λ = 0 + +/-- For shallow networks, the discriminatory-functional criterion is equivalent to universal +approximation. -/ +theorem isUniversal_iff_isDiscriminatory (σ : C(ℝ, ℝ)) : + IsUniversal (E := E) σ ↔ IsDiscriminatory (E := E) σ := by + constructor + · rintro ⟨h_dense⟩ + constructor + intro K hK + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + intro Λ hΛ + refine (spaceOn σ K).dense_iff_forall_dual_eq_zero.mp (h_dense K hK) Λ ?_ + intro f hf + have hle : spaceOn σ K ≤ Λ.ker := by + rw [spaceOn] + apply Submodule.span_le.2 + rintro g ⟨p, rfl⟩ + exact hΛ p.1 p.2 + exact hle hf + · rintro ⟨h_disc⟩ + constructor + intro K hK + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + rw [Submodule.dense_iff_forall_dual_eq_zero] + intro Λ hΛ + apply h_disc K hK Λ + intro w b + exact hΛ _ (Submodule.subset_span ⟨(w, b), rfl⟩) + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean new file mode 100644 index 00000000..069cb0fa --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean @@ -0,0 +1,68 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Analysis.InnerProductSpace.PiL2 +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Nonpolynomial +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.PolynomialObstruction + +/-! +# The Leshno--Lin--Pinkus--Schocken theorem + +This file states the precise universal approximation theorem. The input space is required to be +nontrivial: in dimension zero, every shallow-network function is constant, and a polynomial +activation can still be universal. + +The sufficient direction follows from the distributional, convolution-smoothing, and +Stone--Weierstrass arguments, while the reverse implication follows from the polynomial +obstruction. +-/ + +@[expose] public section + +namespace Learning.ShallowNetwork + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- Every continuous nonpolynomial activation is universal on compact subsets of a real +inner-product space. -/ +theorem isUniversal_of_not_isPolynomial + (σ : C(ℝ, ℝ)) (hσ : ¬ Function.IsPolynomial σ) : IsUniversal (E := E) σ := + (isUniversal_iff_isDiscriminatory σ).2 + (isDiscriminatory_of_not_isPolynomial σ hσ) + +/-- Abstract form of the Leshno--Lin--Pinkus--Schocken equivalence on an arbitrary nontrivial real +inner-product space. -/ +theorem not_isPolynomial_iff_isUniversal [Nontrivial E] (σ : C(ℝ, ℝ)) : + ¬ Function.IsPolynomial σ ↔ IsUniversal (E := E) σ := + ⟨isUniversal_of_not_isPolynomial σ, + fun hUniversal hPolynomial ↦ + not_isUniversal_of_isPolynomial σ hPolynomial hUniversal⟩ + +/-- Precise compact-set form of the Leshno--Lin--Pinkus--Schocken equivalence. + +The approximating subspace is `spaceOn σ K`, whose generators are exactly the restrictions to +`K` of biased ridge functions `x ↦ σ (⟪w, x⟫ + b)`. +-/ +theorem not_isPolynomial_iff_dense_on_compact [Nontrivial E] (σ : C(ℝ, ℝ)) : + ¬ Function.IsPolynomial σ ↔ + ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := + (not_isPolynomial_iff_isUniversal σ).trans (isUniversal_iff σ) + +/-- The classical theorem on `ℝ^d`, represented as `EuclideanSpace ℝ (Fin d)`. + +The hypothesis `0 < d` is essential: the claimed equivalence is false for the zero-dimensional +input space. +-/ +theorem leshno_lin_pinkus_schocken {d : ℕ} (hd : 0 < d) + (σ : C(ℝ, ℝ)) : + ¬ Function.IsPolynomial σ ↔ + ∀ (K : Set (EuclideanSpace ℝ (Fin d))), IsCompact K → + Dense (spaceOn σ K : Set C(K, ℝ)) := by + let _ : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp hd + exact not_isPolynomial_iff_dense_on_compact σ + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean new file mode 100644 index 00000000..d38f80b2 --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean @@ -0,0 +1,45 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.ConvolutionSmooth + +/-! +# Nonpolynomial activations are discriminatory + +This file assembles the analytic ingredients of the sufficient direction of the +Leshno--Lin--Pinkus--Schocken theorem. Distributional differentiation supplies, at every order, +a compactly supported smooth convolution whose derivative is nonzero. Annihilator transfer for +convolution and the smooth-ridge argument then imply that the original activation is +discriminatory. + +The result applies to an arbitrary real inner-product space. Finite dimensionality is not needed: +the approximation problem is posed on compact subsets, where the algebra generated by inner-product +coordinates still separates points. +-/ + +@[expose] public section + +namespace Learning.ShallowNetwork + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- Every continuous nonpolynomial activation is discriminatory on compact subsets of an arbitrary +real inner-product space. -/ +theorem isDiscriminatory_of_not_isPolynomial (σ : C(ℝ, ℝ)) + (hσ : ¬ Function.IsPolynomial σ) : IsDiscriminatory (E := E) σ := by + apply HasContinuousDerivativeTower.isDiscriminatory_of_smooth_ridges σ + intro n + obtain ⟨φ, hφ⟩ := + exists_testFunction_derivative_convolutionActivation_ne_zero σ hσ n + let g : C(ℝ, ℝ) := convolutionActivation φ σ + let hg : HasContinuousDerivativeTower g := inferInstance + refine ⟨g, hg, 0, ?_, ?_⟩ + · exact hφ + · intro K hK Λ hΛ + exact annihilates_convolutionActivation_neurons φ σ K hK Λ hΛ + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean new file mode 100644 index 00000000..0f36e720 --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean @@ -0,0 +1,128 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.PolynomialCharacterization +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction.Normalize + +/-! +# Test-function witnesses for nonpolynomial activations + +This file extracts the analytic witness needed by convolution-based universal-approximation +arguments. A continuous nonpolynomial function defines a regular distribution whose derivative +of every order is nonzero. Evaluating such a derivative on a suitable test function gives a +nonzero integral involving the corresponding iterated derivative of the test function. + +The final theorem uses the reflected activation `x ↦ σ (-x)`. This is exactly the orientation +that occurs when evaluating the left convolution `φ ⋆ σ` at zero. +-/ + +@[expose] public section + +open MeasureTheory +open scoped Distributions + +namespace Learning.ShallowNetwork + +/-- Reflect a continuous real function through the origin. -/ +noncomputable def reflectedActivation (σ : C(ℝ, ℝ)) : C(ℝ, ℝ) := + σ.comp (-ContinuousMap.id ℝ) + +@[simp] +theorem reflectedActivation_apply (σ : C(ℝ, ℝ)) (x : ℝ) : + reflectedActivation σ x = σ (-x) := rfl + +/-- Reflection preserves the property of being a polynomial function. -/ +theorem isPolynomial_reflectedActivation_iff (σ : C(ℝ, ℝ)) : + Function.IsPolynomial (reflectedActivation σ) ↔ Function.IsPolynomial σ := by + constructor + · rintro ⟨p, hp⟩ + refine ⟨p.comp (-Polynomial.X), fun x ↦ ?_⟩ + simp only [Polynomial.eval_comp, Polynomial.eval_neg, Polynomial.eval_X, + hp, reflectedActivation_apply, neg_neg] + · rintro ⟨p, hp⟩ + refine ⟨p.comp (-Polynomial.X), fun x ↦ ?_⟩ + simp only [Polynomial.eval_comp, Polynomial.eval_neg, Polynomial.eval_X, + hp, reflectedActivation_apply] + +end Learning.ShallowNetwork + +namespace Distribution + +open LineDeriv + +/-- Evaluate an iterated distributional derivative by moving all derivatives onto the test +function. The statement is vector-valued and valid on every open subset of the real line. -/ +theorem iteratedLineDerivOp_apply_iterated_testFunction + {F : Type*} [AddCommGroup F] [Module ℝ F] [TopologicalSpace F] + [IsTopologicalAddGroup F] [ContinuousSMul ℝ F] + {Ω : TopologicalSpace.Opens ℝ} (T : 𝓓'(Ω, F)) (n : ℕ) (φ : 𝓓(Ω, ℝ)) : + iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) T φ = + (-1 : ℝ) ^ n • T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) := by + rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] + induction n generalizing T φ with + | zero => simp + | succ n ih => + rw [Function.iterate_succ_apply'] + change Distribution.lineDerivCLM (1 : ℝ) ((∂_{(1 : ℝ)})^[n] T) φ = + (-1 : ℝ) ^ (n + 1) • + T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) + rw [Distribution.lineDerivCLM_apply, ih] + rw [Function.iterate_succ_apply] + simp only [pow_succ] + module + +/-- Every distributional derivative of a continuous nonpolynomial function is nonzero. + +The compactly-supported-primitive class is the exactness input used by the converse +characterization of polynomial regular distributions. -/ +theorem iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial + (f : C(ℝ, ℝ)) (hf : ¬ Function.IsPolynomial f) (n : ℕ) : + iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) + (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) ≠ 0 := by + intro hzero + apply hf + exact isPolynomial_of_iteratedLineDerivOp_ofFun_eq_zero + TestFunction.normalizedBumpReal TestFunction.integral_normalizedBumpReal + f.continuous n hzero + +end Distribution + +namespace Learning.ShallowNetwork + +/-- For each order, a continuous nonpolynomial activation admits a test function whose iterated +derivative has nonzero pairing with the reflected activation. + +Equivalently, this is the nonzero value at the origin of the corresponding derivative of the +left convolution of the test function with `σ`. -/ +theorem exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero + (σ : C(ℝ, ℝ)) (hσ : ¬ Function.IsPolynomial σ) (n : ℕ) : + ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), + ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) ≠ 0 := by + let f : C(ℝ, ℝ) := reflectedActivation σ + let T : 𝓓'((⊤ : TopologicalSpace.Opens ℝ), ℝ) := + LineDeriv.iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) + (Distribution.ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) + have hf : ¬ Function.IsPolynomial f := by + simpa only [f, isPolynomial_reflectedActivation_iff] using hσ + have hT : T ≠ 0 := by + dsimp only [T] + exact Distribution.iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial f hf n + obtain ⟨φ, hφ⟩ := T.exists_ne_zero hT + refine ⟨φ, ?_⟩ + have hfloc : LocallyIntegrableOn f (Set.univ : Set ℝ) volume := + f.continuous.locallyIntegrable.locallyIntegrableOn _ + have heval : T φ = (-1 : ℝ) ^ n * + ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by + dsimp only [T] + rw [Distribution.iteratedLineDerivOp_apply_iterated_testFunction] + rw [Distribution.ofFun_apply hfloc] + simp only [smul_eq_mul, f, reflectedActivation_apply] + intro hzero + apply hφ + exact heval.trans (by rw [hzero, mul_zero]) + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean new file mode 100644 index 00000000..b69a5cb5 --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean @@ -0,0 +1,135 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function +public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Affine +public import LeanMachineLearning.ForMathlib.Topology.Algebra.Module.FiniteDimension +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Discrete +public import LeanMachineLearning.NeuralNetwork.Shallow.Basic +public import Mathlib.RingTheory.Polynomial.DegreeLT +public import Mathlib.Topology.Separation.Basic + +/-! +# Polynomial obstructions to universal approximation + +On sufficiently many collinear sample points, every ridge function obtained from a polynomial +activation belongs to a fixed finite-dimensional space of univariate polynomials. This gives the +necessary direction of the Leshno--Lin--Pinkus--Schocken theorem. +-/ + +@[expose] public section + +open Polynomial + +namespace Learning.ShallowNetwork + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +private theorem normalized_inner_smul_self (e : E) (he : e ≠ 0) (r : ℝ) : + inner ℝ e (r • e) / inner ℝ e e = r := by + rw [real_inner_smul_right, div_eq_iff (inner_self_ne_zero.mpr he)] + +/-- The space of continuous real-valued functions on `n` distinct scalar multiples of a nonzero +vector has dimension `n`. + +The statement is phrased using a range, so it applies without choosing a finite set enumeration. +-/ +theorem finrank_continuousMap_range_fin_smul (e : E) (he : e ≠ 0) (n : ℕ) : + Module.finrank ℝ C(Set.range (fun i : Fin n ↦ (i : ℝ) • e), ℝ) = n := by + classical + let emb : Fin n ↪ E := + ⟨fun i ↦ (i : ℝ) • e, fun i j hij ↦ by + apply Fin.ext + exact_mod_cast (smul_left_injective ℝ he hij)⟩ + let K : Set E := Set.range emb + let _ : Fintype K := (Set.finite_range emb).fintype + let _ : DiscreteTopology K := Finite.instDiscreteTopology + have hcard : Fintype.card K = n := + (Fintype.card_congr emb.toEquivRange).symm.trans (Fintype.card_fin n) + change Module.finrank ℝ C(K, ℝ) = n + rw [(ContinuousMap.linearEquivFnOfDiscrete ℝ).finrank_eq, + Module.finrank_pi, hcard] + +private theorem aeval_degreeLT_range_ne_top {X : Type*} [TopologicalSpace X] + (coordinate : C(X, ℝ)) (n : ℕ) + (hfinrank : Module.finrank ℝ C(X, ℝ) = n + 1) : + ((Polynomial.aeval coordinate).toLinearMap.domRestrict + (degreeLT ℝ n)).range ≠ ⊤ := by + intro htop + have hrank := LinearMap.finrank_range_le + ((Polynomial.aeval coordinate).toLinearMap.domRestrict (degreeLT ℝ n)) + rw [htop, finrank_top, hfinrank, + Module.finrank_eq_card_basis (degreeLT.basis ℝ n), Fintype.card_fin] at hrank + omega + +/-- If the activation is polynomial, then its shallow-network space fails to be dense on some +finite (hence compact) set. No finite-dimensionality assumption on the input space is needed. -/ +theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] + (σ : C(ℝ, ℝ)) (hσ : Function.IsPolynomial σ) : + ∃ K : Set E, IsCompact K ∧ ¬ Dense (spaceOn σ K : Set C(K, ℝ)) := by + classical + obtain ⟨p, hp⟩ := hσ + obtain ⟨e, he⟩ : ∃ e : E, e ≠ 0 := exists_ne 0 + let emb : Fin (p.natDegree + 2) ↪ E := + ⟨fun i ↦ (i : ℝ) • e, fun i j hij ↦ by + apply Fin.ext + exact_mod_cast (smul_left_injective ℝ he hij)⟩ + let K : Set E := Set.range emb + let coordinate : C(K, ℝ) := + ⟨fun x ↦ inner ℝ e x / inner ℝ e e, + (continuous_const.inner continuous_subtype_val).div_const _⟩ + let evalDegree : ↥(degreeLT ℝ (p.natDegree + 1)) →ₗ[ℝ] C(K, ℝ) := + (Polynomial.aeval coordinate).toLinearMap.domRestrict + (degreeLT ℝ (p.natDegree + 1)) + have hpDegree : p ∈ degreeLT ℝ (p.natDegree + 1) := by + rw [degreeLT_succ_eq_degreeLE, mem_degreeLE] + exact degree_le_natDegree + have hfinrank : Module.finrank ℝ C(K, ℝ) = p.natDegree + 2 := by + change Module.finrank ℝ + C(Set.range (fun i : Fin (p.natDegree + 2) ↦ (i : ℝ) • e), ℝ) = _ + exact finrank_continuousMap_range_fin_smul e he _ + have hproper : evalDegree.range ≠ ⊤ := + aeval_degreeLT_range_ne_top coordinate (p.natDegree + 1) (by simpa using hfinrank) + have hspace : (spaceOn σ K : Set C(K, ℝ)) ⊆ evalDegree.range := by + rw [SetLike.coe_subset_coe, spaceOn] + apply Submodule.span_le.2 + rintro _ ⟨⟨w, b⟩, rfl⟩ + let q := p.comp (C (inner ℝ w e) * X + C b) + have hq : q ∈ degreeLT ℝ (p.natDegree + 1) := + Polynomial.compAffineDegreeLT hpDegree _ _ + change (neuron σ w b).restrict K ∈ evalDegree.range + refine ⟨⟨q, hq⟩, ?_⟩ + ext x + obtain ⟨i, hi⟩ := x.property + simp only [evalDegree, LinearMap.domRestrict_apply, AlgHom.toLinearMap_apply, + Polynomial.aeval_continuousMap_apply, coordinate, ContinuousMap.coe_mk, + ContinuousMap.restrict_apply, neuron_apply] + rw [← hp] + rw [← hi] + have hemb : emb i = (i : ℝ) • e := rfl + rw [hemb, normalized_inner_smul_self e he] + simp only [q, Polynomial.eval_comp_C_mul_X_add_C, real_inner_smul_right] + rw [mul_comm (inner ℝ w e)] + exact ⟨K, (Set.finite_range emb).isCompact, + evalDegree.range.not_dense_of_subset_of_finiteDimensional hproper hspace⟩ + +/-- A polynomial activation is not universal on a nontrivial real inner product space. -/ +theorem not_isUniversal_of_isPolynomial [Nontrivial E] + (σ : C(ℝ, ℝ)) (hσ : Function.IsPolynomial σ) : + ¬ IsUniversal (E := E) σ := by + intro hUniversal + obtain ⟨K, hK, hnotDense⟩ := + exists_compact_not_dense_of_isPolynomial (E := E) σ hσ + exact hnotDense (hUniversal.dense_on_compact K hK) + +/-- Universality forces the activation not to be a polynomial. -/ +theorem not_isPolynomial_of_isUniversal [Nontrivial E] + (σ : C(ℝ, ℝ)) [hσ : IsUniversal (E := E) σ] : + ¬ Function.IsPolynomial σ := + fun hPolynomial ↦ not_isUniversal_of_isPolynomial (E := E) σ hPolynomial hσ + +end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean new file mode 100644 index 00000000..4888704b --- /dev/null +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean @@ -0,0 +1,164 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import LeanMachineLearning.ForMathlib.Analysis.Calculus.ContinuousMapComposition +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Moments +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Discriminatory + +/-! +# Universal approximation for smooth activations + +This file isolates the smooth part of the discriminatory-function argument. A continuous +derivative tower is packaged as a class, so later convolution arguments may supply a different +smooth ridge function at each required degree. The core lemma says that annihilating all ridges +of a function whose `n`-th derivative is nonzero forces a functional to annihilate every `n`-th +power of a linear coordinate. +-/ + +@[expose] public section + +namespace Learning.ShallowNetwork + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- A choice of all successive continuous derivatives of a continuous real function. -/ +class HasContinuousDerivativeTower (g : C(ℝ, ℝ)) : Type where + /-- The `n`-th derivative, bundled as a continuous map. -/ + derivative : ℕ → C(ℝ, ℝ) + derivative_zero : derivative 0 = g + hasDerivAt_derivative : + ∀ (n : ℕ) (x : ℝ), HasDerivAt (derivative n) (derivative (n + 1) x) x + +/-- A smooth function whose derivative of every order is not identically zero. -/ +class HasNonzeroContinuousDerivativeTower (g : C(ℝ, ℝ)) : Type + extends HasContinuousDerivativeTower g where + exists_derivative_ne_zero : ∀ n, ∃ x, derivative n x ≠ 0 + +namespace HasContinuousDerivativeTower + +variable {g : C(ℝ, ℝ)} [hg : HasContinuousDerivativeTower g] + +@[simp] +theorem derivative_zero_eq : hg.derivative 0 = g := + hg.derivative_zero + +/-- A functional annihilating every ridge of `g` also annihilates every power of a linear +coordinate for which the corresponding derivative of `g` is nonzero somewhere. -/ +theorem annihilates_coordinate_pow_of_derivative_ne_zero + (K : Set E) (hK : IsCompact K) (Λ : StrongDual ℝ C(K, ℝ)) + (hΛ : ∀ w b, Λ ((neuron g w b).restrict K) = 0) + (n : ℕ) {b : ℝ} (hb : hg.derivative n b ≠ 0) (w : E) : + Λ ((ContinuousMap.innerProductCoordinate K w) ^ n) = 0 := by + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + let u : C(K, ℝ) := ContinuousMap.innerProductCoordinate K w + have hstep : ∀ m t, + Λ (u ^ m * (hg.derivative m).comp + (ContinuousMap.const K b + ContinuousMap.const K t * u)) = 0 := by + intro m + induction m with + | zero => + intro t + rw [hg.derivative_zero] + have heq : g.comp + (ContinuousMap.const K b + ContinuousMap.const K t * u) = + (neuron g (t • w) b).restrict K := by + ext x + simp only [ContinuousMap.comp_apply, ContinuousMap.add_apply, + ContinuousMap.const_apply, ContinuousMap.mul_apply, u, + ContinuousMap.innerProductCoordinate_apply, + neuron_apply, ContinuousMap.restrict_apply, real_inner_smul_left] + congr 1 + ring + rw [heq] + simpa using hΛ (t • w) b + | succ m ihm => + intro t + have hcurve := HasDerivAt.continuousMap_comp_affine + (fun y ↦ hg.hasDerivAt_derivative m y) + (ContinuousMap.const K b) u t + have hmul := hcurve.const_mul (u ^ m) + have hmul' : HasDerivAt + (fun s ↦ u ^ m * (hg.derivative m).comp + (ContinuousMap.const K b + ContinuousMap.const K s * u)) + (u ^ m * (u * (hg.derivative (m + 1)).comp + (ContinuousMap.const K b + ContinuousMap.const K t * u))) t := by + convert hmul using 1 + ext x + simp [smul_eq_mul] + have happly : HasDerivAt + (fun s ↦ Λ (u ^ m * (hg.derivative m).comp + (ContinuousMap.const K b + ContinuousMap.const K s * u))) + (Λ (u ^ m * (u * (hg.derivative (m + 1)).comp + (ContinuousMap.const K b + ContinuousMap.const K t * u)))) t := by + simpa [Function.comp_def] using + Λ.hasFDerivAt.comp_hasDerivAt_of_eq t hmul' rfl + have hzero : HasDerivAt + (fun s ↦ Λ (u ^ m * (hg.derivative m).comp + (ContinuousMap.const K b + ContinuousMap.const K s * u))) 0 t := by + convert hasDerivAt_const t (0 : ℝ) using 1 + funext s + exact ihm s + have hz := happly.unique hzero + simpa [pow_succ, mul_assoc] using hz + have h := hstep n 0 + have harg : (hg.derivative n).comp + (ContinuousMap.const K b + ContinuousMap.const K 0 * u) = + ContinuousMap.const K (hg.derivative n b) := by + ext x + simp + rw [harg] at h + have heq : u ^ n * ContinuousMap.const K (hg.derivative n b) = + hg.derivative n b • u ^ n := by + ext x + simp [mul_comm] + rw [heq, map_smul] at h + exact (mul_eq_zero.mp h).resolve_left hb + +/-- A degree-by-degree smooth ridge family is enough for the discriminatory property. The +smooth function may depend on the degree; this is the form needed after mollification. -/ +theorem isDiscriminatory_of_smooth_ridges + (σ : C(ℝ, ℝ)) + (hsmooth : ∀ n : ℕ, + ∃ (g : C(ℝ, ℝ)) (hg : HasContinuousDerivativeTower g) (b : ℝ), + hg.derivative n b ≠ 0 ∧ + ∀ (K : Set E) (_hK : IsCompact K) (Λ : StrongDual ℝ C(K, ℝ)), + (∀ w c, Λ ((neuron σ w c).restrict K) = 0) → + ∀ w c, Λ ((neuron g w c).restrict K) = 0) : + IsDiscriminatory (E := E) σ := by + constructor + intro K hK Λ hΛ + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + apply StrongDual.eq_zero_of_innerProductCoordinate_powers K Λ + intro n w + obtain ⟨g, hg, b, hb, htransfer⟩ := hsmooth n + let _ : HasContinuousDerivativeTower g := hg + exact annihilates_coordinate_pow_of_derivative_ne_zero K hK Λ + (htransfer K hK Λ hΛ) n hb w + +/-- A smooth activation with no identically-zero derivative is discriminatory on every real +inner-product space. -/ +theorem isDiscriminatory_of_hasNonzeroContinuousDerivativeTower + (g : C(ℝ, ℝ)) [hg : HasNonzeroContinuousDerivativeTower g] : + IsDiscriminatory (E := E) g := by + apply isDiscriminatory_of_smooth_ridges g + intro n + obtain ⟨b, hb⟩ := hg.exists_derivative_ne_zero n + exact ⟨g, hg.toHasContinuousDerivativeTower, b, hb, by + intro K hK Λ hΛ + exact hΛ⟩ + +/-- A smooth activation with no identically-zero derivative has the universal approximation +property on every real inner-product space. -/ +theorem isUniversal_of_hasNonzeroContinuousDerivativeTower + (g : C(ℝ, ℝ)) [HasNonzeroContinuousDerivativeTower g] : + IsUniversal (E := E) g := + (isUniversal_iff_isDiscriminatory g).2 + (isDiscriminatory_of_hasNonzeroContinuousDerivativeTower g) + +end HasContinuousDerivativeTower + +end Learning.ShallowNetwork From 34981bc3084edd43952a364182cdf44d28c41394 Mon Sep 17 00:00:00 2001 From: yuanyi-350 Date: Sat, 12 Sep 2026 13:52:01 +0800 Subject: [PATCH 2/8] Refactor universal approximation proofs and APIs --- LeanMachineLearning.lean | 7 +- .../ForMathlib/Algebra/Polynomial/Affine.lean | 65 ------- .../Distribution/TestFunction/Normalize.lean | 17 +- .../Topology/ContinuousMap/Dense.lean | 40 ----- .../Topology/ContinuousMap/Discrete.lean | 16 +- .../Topology/ContinuousMap/InnerProduct.lean | 26 ++- .../Topology/ContinuousMap/Moments.lean | 8 - .../NeuralNetwork/Shallow/Basic.lean | 27 +-- .../UniversalApproximation/Convolution.lean | 128 +++++++++----- .../ConvolutionSmooth.lean | 95 ---------- .../Discriminatory.lean | 137 ++++++++++++++- .../{Main.lean => Leshno.lean} | 32 ++-- .../UniversalApproximation/Nonpolynomial.lean | 155 ++++++++++++++--- .../NonpolynomialWitness.lean | 128 -------------- .../PolynomialObstruction.lean | 65 ++++--- .../SmoothActivation.lean | 164 ------------------ 16 files changed, 421 insertions(+), 689 deletions(-) delete mode 100644 LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean delete mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean delete mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean rename LeanMachineLearning/NeuralNetwork/UniversalApproximation/{Main.lean => Leshno.lean} (58%) delete mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean delete mode 100644 LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 84f05987..cd6c808a 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,5 +1,4 @@ module -- shake: keep-all --deprecated_module: ignore -public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Affine public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function public import LeanMachineLearning.ForMathlib.Analysis.Calculus.ContinuousMapComposition public import LeanMachineLearning.ForMathlib.Analysis.Distribution.Polynomial @@ -43,20 +42,16 @@ public import LeanMachineLearning.ForMathlib.Probability.Moments.SubExponential public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.ForMathlib.Topology.Algebra.Module.FiniteDimension -public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Dense public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Discrete public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.InnerProduct public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Moments public import LeanMachineLearning.ForMathlib.Topology.Instances.ENNReal.Lemmas public import LeanMachineLearning.NeuralNetwork.Shallow.Basic public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Convolution -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.ConvolutionSmooth public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Discriminatory -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Main +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Leshno public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Nonpolynomial -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.NonpolynomialWitness public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.PolynomialObstruction -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.SmoothActivation public import LeanMachineLearning.Online.Bandit.Algorithms.ETC public import LeanMachineLearning.Online.Bandit.Algorithms.Regret.BayesRegretTS public import LeanMachineLearning.Online.Bandit.Algorithms.TS diff --git a/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean b/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean deleted file mode 100644 index 0726760d..00000000 --- a/LeanMachineLearning/ForMathlib/Algebra/Polynomial/Affine.lean +++ /dev/null @@ -1,65 +0,0 @@ -/- -Copyright (c) 2026 Yi Yuan. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Yi Yuan --/ -module - -public import Mathlib.RingTheory.Polynomial.Basic - -/-! -# Affine changes of variables in polynomials - -This file records that composition with a polynomial of degree at most one preserves the -submodule `Polynomial.degreeLT`. In particular, this applies to affine changes of variables. -It also provides simp lemmas for evaluating the resulting polynomials. --/ - -@[expose] public section - -namespace Polynomial - -universe u - -variable {R : Type u} - -/-- Composing with a polynomial of degree at most one preserves a strict degree bound. -/ -theorem comp_mem_degreeLT_of_natDegree_le_one [Semiring R] - {p q : R[X]} {n : ℕ} (hp : p ∈ degreeLT R n) (hq : q.natDegree ≤ 1) : - p.comp q ∈ degreeLT R n := by - rw [mem_degreeLT] at hp ⊢ - by_cases hcomp : p.comp q = 0 - · simp [hcomp] - rw [← natDegree_lt_iff_degree_lt hcomp] - calc - (p.comp q).natDegree ≤ p.natDegree * q.natDegree := natDegree_comp_le - _ ≤ p.natDegree * 1 := Nat.mul_le_mul_left _ hq - _ = p.natDegree := Nat.mul_one _ - _ < n := by - by_cases hp0 : p = 0 - · simp [hp0] at hcomp - · exact (natDegree_lt_iff_degree_lt hp0).2 hp - -/-- Composition with the affine polynomial `C a * X + C b` preserves a strict degree bound. -/ -theorem compAffineDegreeLT [Semiring R] {p : R[X]} {n : ℕ} - (hp : p ∈ degreeLT R n) (a b : R) : - p.comp (C a * X + C b) ∈ degreeLT R n := by - apply comp_mem_degreeLT_of_natDegree_le_one hp - calc - (C a * X + C b).natDegree ≤ max (C a * X).natDegree (C b).natDegree := - natDegree_add_le _ _ - _ ≤ 1 := max_le (by simpa using natDegree_C_mul_X_pow_le a 1) (by simp) - -/-- Evaluation of the polynomial representing the affine function `x ↦ a * x + b`. -/ -@[simp] -theorem eval_C_mul_X_add_C [CommSemiring R] (a b x : R) : - eval x (C a * X + C b) = a * x + b := by - simp - -/-- Evaluation after composition with the affine polynomial `C a * X + C b`. -/ -@[simp] -theorem eval_comp_C_mul_X_add_C [CommSemiring R] (p : R[X]) (a b x : R) : - eval x (p.comp (C a * X + C b)) = eval (a * x + b) p := by - rw [eval_comp, eval_C_mul_X_add_C] - -end Polynomial diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean index d2fa4eb5..f09c64a8 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean @@ -13,8 +13,7 @@ public import Mathlib.Analysis.Distribution.TestFunction # Test functions normalized by their integral This file constructs real-valued test functions of integral one from normalized smooth bump -functions, both inside an arbitrary nonempty open subset of a finite-dimensional real normed -space and as a fixed test function on the real line. +functions inside an arbitrary nonempty open subset of a finite-dimensional real normed space. -/ @[expose] public section @@ -74,18 +73,4 @@ theorem exists_integral_eq_one (hΩ : (Ω : Set E).Nonempty) : exact half_lt_self hε exact ⟨f.toTestFunctionNormed μ hf, f.integral_toTestFunctionNormed μ hf⟩ -/-- A fixed smooth compactly supported function on `ℝ` whose Lebesgue integral is one. -/ -def normalizedBumpReal : 𝓓((⊤ : Opens ℝ), ℝ) := - let f : ContDiffBump (0 : ℝ) := - ContDiffBump.mk 1 2 zero_lt_one one_lt_two - f.toTestFunctionNormed volume (subset_univ _) - -@[simp] -theorem integral_normalizedBumpReal : - ∫ x : ℝ, normalizedBumpReal x = 1 := by - let f : ContDiffBump (0 : ℝ) := - ContDiffBump.mk 1 2 zero_lt_one one_lt_two - simpa only [normalizedBumpReal] using - f.integral_toTestFunctionNormed volume (subset_univ _) - end TestFunction diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean deleted file mode 100644 index d2bbb381..00000000 --- a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Dense.lean +++ /dev/null @@ -1,40 +0,0 @@ -/- -Copyright (c) 2026 Yi Yuan. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Yi Yuan --/ -module - -public import Mathlib.Topology.ContinuousMap.Compact - -/-! -# Dense families of continuous maps on compact spaces - -This file gives the uniform epsilon formulation of density in a continuous-map space with -compact domain. --/ - -@[expose] public section - -namespace ContinuousMap - -variable {X Y : Type*} [TopologicalSpace X] [CompactSpace X] [PseudoMetricSpace Y] - -/-- A family of continuous maps on a compact space is dense exactly when every continuous map can -be approximated pointwise with one uniform positive error bound. -/ -theorem dense_iff_forall_exists_forall_dist_lt {S : Set C(X, Y)} : - Dense S ↔ ∀ (f : C(X, Y)) (ε : ℝ), 0 < ε → - ∃ g ∈ S, ∀ x, dist (g x) (f x) < ε := by - rw [Metric.dense_iff] - constructor - · intro h f ε hε - obtain ⟨g, hgBall, hgS⟩ := h f ε hε - refine ⟨g, hgS, ?_⟩ - rwa [Metric.mem_ball, ContinuousMap.dist_lt_iff hε] at hgBall - · intro h f ε hε - obtain ⟨g, hgS, hg⟩ := h f ε hε - refine ⟨g, ?_, hgS⟩ - rw [Metric.mem_ball, ContinuousMap.dist_lt_iff hε] - exact hg - -end ContinuousMap diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean index 835738cb..ca80c777 100644 --- a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean @@ -26,17 +26,9 @@ variable [TopologicalSpace M] [AddCommMonoid M] [ContinuousAdd M] variable [Module R M] [ContinuousConstSMul R M] /-- Continuous maps from a discrete space are linearly equivalent to arbitrary functions. -/ -def linearEquivFnOfDiscrete : C(X, M) ≃ₗ[R] (X → M) where - __ := equivFnOfDiscrete - map_add' _ _ := rfl - map_smul' _ _ := rfl - -@[simp] -theorem linearEquivFnOfDiscrete_apply (f : C(X, M)) (x : X) : - linearEquivFnOfDiscrete R f x = f x := rfl - -@[simp] -theorem linearEquivFnOfDiscrete_symm_apply_apply (f : X → M) (x : X) : - (linearEquivFnOfDiscrete R).symm f x = f x := rfl +def linearEquivFnOfDiscrete : C(X, M) ≃ₗ[R] (X → M) := + equivFnOfDiscrete.toLinearEquiv + { map_add := fun _ _ ↦ rfl + map_smul := fun _ _ ↦ rfl } end ContinuousMap diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean index bca8e520..a564703f 100644 --- a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/InnerProduct.lean @@ -21,19 +21,23 @@ namespace ContinuousMap variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] +/-- The real inner-product coordinate `x ↦ ⟪w, x⟫` on a subtype. -/ +def innerProductCoordinate (K : Set E) (w : E) : C(K, ℝ) := + ⟨fun x ↦ inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ + +@[simp] +theorem innerProductCoordinate_apply (K : Set E) (w : E) (x : K) : + innerProductCoordinate K w x = inner ℝ w x.1 := rfl + /-- The algebra generated by the real inner-product coordinates on a subtype separates points. No compactness or finite-dimensionality assumption is needed for this fact. -/ theorem innerProduct_adjoin_separatesPoints (K : Set E) : - (Algebra.adjoin ℝ - (Set.range fun w : E => - (⟨fun x : K => inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ : - C(K, ℝ)))).SeparatesPoints := by + (Algebra.adjoin ℝ (Set.range (innerProductCoordinate K))).SeparatesPoints := by intro x y hxy let w : E := x.1 - y.1 - let f : C(K, ℝ) := - ⟨fun z => inner ℝ w z.1, continuous_const.inner continuous_subtype_val⟩ + let f : C(K, ℝ) := innerProductCoordinate K w refine ⟨f, ?_, ?_⟩ · exact ⟨f, ⟨Algebra.subset_adjoin ⟨w, rfl⟩, rfl⟩⟩ · intro h @@ -47,20 +51,14 @@ theorem innerProduct_adjoin_separatesPoints (K : Set E) : /-- On a compact subtype of a real inner-product space, the closure of the algebra generated by the inner-product coordinates is the full algebra of continuous real-valued functions. -/ theorem innerProduct_adjoin_topologicalClosure_eq_top (K : Set E) [CompactSpace K] : - (Algebra.adjoin ℝ - (Set.range fun w : E => - (⟨fun x : K => inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ : - C(K, ℝ)))).topologicalClosure = ⊤ := + (Algebra.adjoin ℝ (Set.range (innerProductCoordinate K))).topologicalClosure = ⊤ := subalgebra_topologicalClosure_eq_top_of_separatesPoints _ (innerProduct_adjoin_separatesPoints K) /-- On a compact subtype of a real inner-product space, the algebra generated by inner-product coordinates is dense in the continuous real-valued functions. -/ theorem dense_innerProduct_adjoin (K : Set E) [CompactSpace K] : - Dense (Algebra.adjoin ℝ - (Set.range fun w : E => - (⟨fun x : K => inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ : - C(K, ℝ))) : Set C(K, ℝ)) := by + Dense (Algebra.adjoin ℝ (Set.range (innerProductCoordinate K)) : Set C(K, ℝ)) := by rw [dense_iff_closure_eq, ← Subalgebra.topologicalClosure_coe, innerProduct_adjoin_topologicalClosure_eq_top] rfl diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean index d2e76ce5..0e5799f2 100644 --- a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Moments.lean @@ -121,14 +121,6 @@ namespace ContinuousMap variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] -/-- The real inner-product coordinate `x ↦ ⟪w, x⟫` on a subtype. -/ -def innerProductCoordinate (K : Set E) (w : E) : C(K, ℝ) := - ⟨fun x ↦ inner ℝ w x.1, continuous_const.inner continuous_subtype_val⟩ - -@[simp] -theorem innerProductCoordinate_apply (K : Set E) (w : E) (x : K) : - innerProductCoordinate K w x = inner ℝ w x.1 := rfl - /-- The continuous linear map sending a vector to its inner-product coordinate on a compact subtype. Compactness bounds the subtype, so finite-dimensionality is not required. -/ noncomputable def innerProductCoordinateCLM (K : Set E) [CompactSpace K] : diff --git a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean index d7cd640a..dc60c9ae 100644 --- a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean +++ b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean @@ -8,7 +8,7 @@ module public import Mathlib.Analysis.InnerProductSpace.Continuous public import Mathlib.LinearAlgebra.Finsupp.LinearCombination public import Mathlib.Topology.ContinuousMap.Algebra -public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Dense +public import Mathlib.Topology.ContinuousMap.Compact /-! # Single-hidden-layer neural networks @@ -26,8 +26,7 @@ variable {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] /-- A single neuron with weight `w`, bias `b`, and activation `σ`. -/ def neuron (σ : C(ℝ, ℝ)) (w : E) (b : ℝ) : C(E, ℝ) := - σ.comp ⟨fun x ↦ inner ℝ w x + b, - (continuous_const.inner continuous_id).add continuous_const⟩ + σ.comp ⟨fun x ↦ inner ℝ w x + b, by fun_prop⟩ @[simp] theorem neuron_apply (σ : C(ℝ, ℝ)) (w : E) (b : ℝ) (x : E) : @@ -86,16 +85,14 @@ theorem mem_spaceOn_iff (σ : C(ℝ, ℝ)) (K : Set E) (f : C(K, ℝ)) : /-- `spaceOn` is the image of the global network space under restriction. -/ theorem spaceOn_eq_map (σ : C(ℝ, ℝ)) (K : Set E) : - spaceOn σ K = - (space σ).map - (ContinuousMap.compCLM ℝ ℝ - ⟨((↑) : K → E), continuous_subtype_val⟩).toLinearMap := by + spaceOn σ K = (space σ).map + (ContinuousMap.compCLM ℝ ℝ ⟨((↑) : K → E), continuous_subtype_val⟩).toLinearMap := by rw [spaceOn, space, Submodule.map_span] congr 1 ext f - simp only [Set.mem_range, Set.mem_image] aesop +variable (E) in /-- An activation is universal on `E` if its shallow networks are dense on every compact subset. This class packages the property for downstream approximation theorems. -/ class IsUniversal (σ : C(ℝ, ℝ)) : Prop where @@ -103,23 +100,27 @@ class IsUniversal (σ : C(ℝ, ℝ)) : Prop where /-- The typeclass formulation of universality unfolds to density on every compact subset. -/ theorem isUniversal_iff (σ : C(ℝ, ℝ)) : - IsUniversal (E := E) σ ↔ - ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := by + IsUniversal E σ ↔ ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := by grind [IsUniversal] /-- The usual uniform epsilon formulation of universal approximation on every compact set. -/ theorem isUniversal_iff_uniform_approximation (σ : C(ℝ, ℝ)) : - IsUniversal (E := E) σ ↔ + IsUniversal E σ ↔ ∀ (K : Set E), IsCompact K → ∀ (f : C(K, ℝ)) (ε : ℝ), 0 < ε → ∃ g ∈ spaceOn σ K, ∀ x, dist (g x) (f x) < ε := by constructor · rintro ⟨h⟩ K hK let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - exact ContinuousMap.dense_iff_forall_exists_forall_dist_lt.mp (h K hK) + intro f ε hε + obtain ⟨g, hgBall, hgSpace⟩ := Metric.dense_iff.mp (h K hK) f ε hε + exact ⟨g, hgSpace, (ContinuousMap.dist_lt_iff hε).mp hgBall⟩ · intro h constructor intro K hK let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - exact ContinuousMap.dense_iff_forall_exists_forall_dist_lt.mpr (h K hK) + rw [Metric.dense_iff] + intro f ε hε + obtain ⟨g, hgSpace, hgDist⟩ := h K hK f ε hε + exact ⟨g, (ContinuousMap.dist_lt_iff hε).mpr hgDist, hgSpace⟩ end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean index f48b1766..0ffc8817 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean @@ -17,52 +17,54 @@ public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Discrimin # Convolution smoothing of activation functions Convolution against a compactly supported continuous kernel turns an activation into another -continuous activation. On a compact domain, a ridge function for the convolved activation is a -Bochner integral of ridge functions for the original activation. Consequently it belongs to the +continuous activation. On a compact domain, a ridge function for the convolved activation is a +Bochner integral of ridge functions for the original activation. Consequently it belongs to the closure of their span, and every functional annihilating the original neurons also annihilates the convolved neurons. -The kernel is abstracted by `CompactlySupportedContinuousMapClass`, so this file applies directly -both to compactly supported continuous maps and to smooth test functions. +The kernel is abstracted by `CompactlySupportedContinuousMapClass`, so the closure and annihilator +results apply both to compactly supported continuous maps and to smooth test functions. + +For a smooth test-function kernel, successive derivatives of the kernel give successive +derivatives of the convolution. The resulting `iteratedDeriv` formula also identifies their +values at the origin. -/ @[expose] public section open MeasureTheory +open scoped Distributions namespace Learning.ShallowNetwork +/-! ## Compactly supported convolution kernels -/ + /-- Convolution of a continuous activation with a compactly supported continuous kernel. The convention is `convolutionActivation φ σ t = ∫ s, φ s * σ (t - s)`. -/ noncomputable def convolutionActivation {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) : C(ℝ, ℝ) := by - letI : (volume : Measure ℝ).IsNegInvariant := - Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume - exact - ⟨MeasureTheory.convolution φ σ (ContinuousLinearMap.mul ℝ ℝ) volume, - (CompactlySupportedContinuousMapClass.hasCompactSupport φ).continuous_convolution_left - (ContinuousLinearMap.mul ℝ ℝ) - (ContinuousMapClass.map_continuous φ) σ.continuous.locallyIntegrable⟩ + (φ : B) (σ : C(ℝ, ℝ)) : C(ℝ, ℝ) := + ⟨MeasureTheory.convolution φ σ (ContinuousLinearMap.mul ℝ ℝ) volume, + (CompactlySupportedContinuousMapClass.hasCompactSupport φ).continuous_convolution_left + _ (ContinuousMapClass.map_continuous φ) σ.continuous.locallyIntegrable⟩ @[simp] theorem convolutionActivation_apply {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] (φ : B) (σ : C(ℝ, ℝ)) (t : ℝ) : - convolutionActivation φ σ t = ∫ s, φ s * σ (t - s) := by - exact MeasureTheory.convolution_mul + convolutionActivation φ σ t = ∫ s, φ s * σ (t - s) := MeasureTheory.convolution_mul /-- Convolution with a compactly supported `C^n` kernel makes a continuous activation `C^n`. -/ theorem convolutionActivation_contDiff {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] (φ : B) (σ : C(ℝ, ℝ)) {n : ℕ∞} (hφ : ContDiff ℝ n φ) : - ContDiff ℝ n (convolutionActivation φ σ) := by - let _ : (volume : Measure ℝ).IsNegInvariant := - Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume - exact (CompactlySupportedContinuousMapClass.hasCompactSupport φ).contDiff_convolution_left - (ContinuousLinearMap.mul ℝ ℝ) hφ σ.continuous.locallyIntegrable + ContDiff ℝ n (convolutionActivation φ σ) := + (CompactlySupportedContinuousMapClass.hasCompactSupport φ).contDiff_convolution_left + _ hφ σ.continuous.locallyIntegrable + +/-! ## Network spaces and annihilator transfer -/ /-- The activation `σ` applied to a scalar-valued continuous feature `u`, with bias `b`. -/ def activationAlong {X : Type*} [TopologicalSpace X] (σ : C(ℝ, ℝ)) @@ -84,9 +86,8 @@ theorem integrable_smul_activationAlong_sub apply Continuous.integrable_of_hasCompactSupport · exact (ContinuousMapClass.map_continuous φ).smul (ContinuousMap.continuous_of_continuous_uncurry _ <| - σ.continuous.comp - ((u.continuous.comp continuous_snd).add - (continuous_const.sub continuous_fst))) + (σ.continuous.comp + ((u.continuous.comp continuous_snd).add (continuous_const.sub continuous_fst)))) · exact (CompactlySupportedContinuousMapClass.hasCompactSupport φ).smul_right /-- A ridge function of a convolved activation is the Bochner integral of shifted ridge functions @@ -103,8 +104,7 @@ theorem activationAlong_convolutionActivation simp only [activationAlong_apply, convolutionActivation_apply] congr 1 funext s - change φ s * σ (u x + b - s) = φ s * σ (u x + (b - s)) - simp only [sub_eq_add_neg, add_assoc] + simp [sub_eq_add_neg, add_assoc] /-- A ridge function of a convolved activation belongs to the closure of any submodule containing all bias translates of the corresponding ridge function for the original activation. -/ @@ -124,17 +124,17 @@ the original activation. -/ theorem convolved_neuron_mem_spaceOn_topologicalClosure {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) (K : Set E) (hK : IsCompact K) (w : E) (b : ℝ) : + (φ : B) (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) {w : E} {b : ℝ} : (neuron (convolutionActivation φ σ) w b).restrict K ∈ (spaceOn σ K).topologicalClosure := by let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK let u : C(K, ℝ) := ⟨fun x => inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ rw [show (neuron (convolutionActivation φ σ) w b).restrict K = - activationAlong (convolutionActivation φ σ) u b by ext; rfl] + activationAlong (convolutionActivation φ σ) u b by rfl] apply activationAlong_convolutionActivation_mem_topologicalClosure φ σ u b intro c - rw [show activationAlong σ u c = (neuron σ w c).restrict K by ext; rfl] + rw [show activationAlong σ u c = (neuron σ w c).restrict K by rfl] exact Submodule.subset_span ⟨(w, c), rfl⟩ /-- On every compact set, the network space of a convolved activation is contained in the closure @@ -142,20 +142,20 @@ of the network space of the original activation. -/ theorem convolved_spaceOn_le_topologicalClosure {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) (K : Set E) (hK : IsCompact K) : + (φ : B) (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) : spaceOn (convolutionActivation φ σ) K ≤ (spaceOn σ K).topologicalClosure := by rw [spaceOn] apply Submodule.span_le.2 rintro f ⟨p, rfl⟩ - exact convolved_neuron_mem_spaceOn_topologicalClosure φ σ K hK p.1 p.2 + exact convolved_neuron_mem_spaceOn_topologicalClosure φ σ hK /-- A continuous functional annihilating all translates of a ridge function also annihilates the corresponding ridge function for every compactly supported convolution smoothing. -/ theorem annihilates_convolutionActivation {X : Type*} [TopologicalSpace X] [CompactSpace X] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (Λ : StrongDual ℝ C(X, ℝ)) - (hΛ : ∀ b, Λ (activationAlong σ u b) = 0) (b : ℝ) : + (φ : B) {σ : C(ℝ, ℝ)} {u : C(X, ℝ)} {Λ : StrongDual ℝ C(X, ℝ)} {b : ℝ} + (hΛ : ∀ b, Λ (activationAlong σ u b) = 0) : Λ (activationAlong (convolutionActivation φ σ) u b) = 0 := by rw [activationAlong_convolutionActivation, ← Λ.integral_comp_comm (integrable_smul_activationAlong_sub φ σ u b)] @@ -166,9 +166,8 @@ compactly supported convolution smoothing of `σ`. -/ theorem annihilates_convolutionActivation_neurons {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) (K : Set E) (hK : IsCompact K) - (Λ : StrongDual ℝ C(K, ℝ)) - (hΛ : ∀ w b, Λ ((neuron σ w b).restrict K) = 0) : + (φ : B) {σ : C(ℝ, ℝ)} {K : Set E} {Λ : StrongDual ℝ C(K, ℝ)} + (hK : IsCompact K) (hΛ : ∀ w b, Λ ((neuron σ w b).restrict K) = 0) : ∀ w b, Λ ((neuron (convolutionActivation φ σ) w b).restrict K) = 0 := by let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK intro w b @@ -176,24 +175,24 @@ theorem annihilates_convolutionActivation_neurons ⟨fun x => inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ have htrans : ∀ c, Λ (activationAlong σ u c) = 0 := by intro c - rw [show activationAlong σ u c = (neuron σ w c).restrict K by ext; rfl] + rw [show activationAlong σ u c = (neuron σ w c).restrict K by rfl] exact hΛ w c rw [show (neuron (convolutionActivation φ σ) w b).restrict K = - activationAlong (convolutionActivation φ σ) u b by ext; rfl] - exact annihilates_convolutionActivation φ σ u Λ htrans b + activationAlong (convolutionActivation φ σ) u b by rfl] + exact annihilates_convolutionActivation φ htrans /-- If one convolution smoothing of `σ` is discriminatory, then `σ` itself is discriminatory. -/ theorem isDiscriminatory_of_convolutionActivation {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] (φ : B) (σ : C(ℝ, ℝ)) - [IsDiscriminatory (E := E) (convolutionActivation φ σ)] : - IsDiscriminatory (E := E) σ := by + [IsDiscriminatory E (convolutionActivation φ σ)] : + IsDiscriminatory E σ := by constructor intro K hK Λ hΛ apply IsDiscriminatory.annihilator_eq_zero (σ := convolutionActivation φ σ) (E := E) K hK Λ - exact annihilates_convolutionActivation_neurons φ σ K hK Λ hΛ + exact annihilates_convolutionActivation_neurons φ hK hΛ /-- Universality of one compactly supported convolution smoothing implies universality of the original activation. -/ @@ -201,12 +200,53 @@ theorem isUniversal_of_convolutionActivation {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] (φ : B) (σ : C(ℝ, ℝ)) - [IsUniversal (E := E) (convolutionActivation φ σ)] : - IsUniversal (E := E) σ := by + [IsUniversal E (convolutionActivation φ σ)] : + IsUniversal E σ := by apply (isUniversal_iff_isDiscriminatory σ).mpr - let _ : IsDiscriminatory (E := E) (convolutionActivation φ σ) := + let _ : IsDiscriminatory E (convolutionActivation φ σ) := (isUniversal_iff_isDiscriminatory (convolutionActivation φ σ)).mp - (inferInstance : IsUniversal (E := E) (convolutionActivation φ σ)) + (inferInstance : IsUniversal E (convolutionActivation φ σ)) exact isDiscriminatory_of_convolutionActivation φ σ +/-! ## Iterated derivatives for smooth kernels -/ + +/-- Successive derivatives of the left kernel give successive derivatives of its convolution +with a continuous activation. -/ +theorem hasDerivAt_convolutionActivation_iterate_lineDerivCLM + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) (x : ℝ) : + HasDerivAt (convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ) + (convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) σ x) x := by + let _ : (volume : Measure ℝ).IsNegInvariant := + Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume + have h := (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ).hasCompactSupport + |>.hasDerivAt_convolution_left (μ := volume) (ContinuousLinearMap.mul ℝ ℝ) + ((TestFunction.contDiff (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ)).of_le (by simp)) + σ.continuous.locallyIntegrable x + convert h using 1 + · rfl + · rw [Function.iterate_succ_apply'] + congr 1 + +/-- The `n`-th derivative of a test-function convolution is the convolution with the +`n`-fold derivative of its kernel. -/ +@[simp] +theorem iteratedDeriv_convolutionActivation_testFunction + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : + iteratedDeriv n (convolutionActivation φ σ) = + convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ := by + induction n with + | zero => simp + | succ n ih => + rw [iteratedDeriv_succ, ih] + funext x + exact (hasDerivAt_convolutionActivation_iterate_lineDerivCLM φ σ n x).deriv + +/-- At the origin, the `n`-th derivative of a test-function convolution is the pairing of the +`n`-fold derivative of its kernel with the reflected activation. -/ +theorem iteratedDeriv_convolutionActivation_testFunction_apply_zero + (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : + iteratedDeriv n (convolutionActivation φ σ) 0 = + ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by + simp + end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean deleted file mode 100644 index cb6b776d..00000000 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/ConvolutionSmooth.lean +++ /dev/null @@ -1,95 +0,0 @@ -/- -Copyright (c) 2026 Yi Yuan. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Yi Yuan --/ -module - -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Convolution -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.NonpolynomialWitness -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.SmoothActivation - -/-! -# Derivative towers for convolution-smoothed activations - -Convolving a continuous real activation with a smooth compactly supported test function produces -a smooth activation. More precisely, its `n`-th derivative is the convolution whose kernel is -obtained by applying `TestFunction.lineDerivCLM` `n` times. - -The resulting derivative tower is registered as an instance of -`HasContinuousDerivativeTower`. The last theorem combines its value at the origin with the -distributional witness for a nonpolynomial activation. --/ - -@[expose] public section - -open MeasureTheory -open scoped Distributions - -namespace Learning.ShallowNetwork - -/-- Successive derivatives of the left kernel give successive derivatives of its convolution -with a continuous activation. -/ -theorem hasDerivAt_convolutionActivation_iterate_lineDerivCLM - (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) (x : ℝ) : - HasDerivAt - (convolutionActivation - (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ) - (convolutionActivation - (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) σ x) x := by - let _ : (volume : Measure ℝ).IsNegInvariant := - Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume - have h := - (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ).hasCompactSupport - |>.hasDerivAt_convolution_left - (μ := volume) (ContinuousLinearMap.mul ℝ ℝ) - ((TestFunction.contDiff - (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ)).of_le (by simp)) - σ.continuous.locallyIntegrable x - convert h using 1 - · rfl - · rw [Function.iterate_succ_apply'] - congr 1 - -/-- The canonical continuous derivative tower on the convolution of a test function with a -continuous activation. -/ -noncomputable instance instHasContinuousDerivativeTowerConvolutionActivation - (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) : - HasContinuousDerivativeTower (convolutionActivation φ σ) where - derivative n := - convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ - derivative_zero := by simp - hasDerivAt_derivative := - hasDerivAt_convolutionActivation_iterate_lineDerivCLM φ σ - -/-- The `n`-th member of the canonical derivative tower is convolution with the `n`-fold -derivative of the test-function kernel. -/ -@[simp] -theorem derivative_convolutionActivation_testFunction - (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : - HasContinuousDerivativeTower.derivative (convolutionActivation φ σ) n = - convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ := - rfl - -/-- At the origin, the `n`-th derivative of a test-function convolution is the pairing of the -`n`-fold derivative of its kernel with the reflected activation. -/ -theorem derivative_convolutionActivation_testFunction_apply_zero - (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : - HasContinuousDerivativeTower.derivative (convolutionActivation φ σ) n 0 = - ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by - rw [derivative_convolutionActivation_testFunction, convolutionActivation_apply] - simp only [zero_sub] - -/-- For every order, a nonpolynomial continuous activation has a test-function convolution whose -canonical derivative tower is nonzero at the origin in that order. -/ -theorem exists_testFunction_derivative_convolutionActivation_ne_zero - (σ : C(ℝ, ℝ)) (hσ : ¬ Function.IsPolynomial σ) (n : ℕ) : - ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), - HasContinuousDerivativeTower.derivative (convolutionActivation φ σ) n 0 ≠ 0 := by - obtain ⟨φ, hφ⟩ := - exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero σ hσ n - refine ⟨φ, ?_⟩ - rw [derivative_convolutionActivation_testFunction_apply_zero] - exact hφ - -end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean index d5fd3dee..30e69a44 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean @@ -5,34 +5,48 @@ Authors: Yi Yuan -/ module +public import LeanMachineLearning.ForMathlib.Analysis.Calculus.ContinuousMapComposition public import LeanMachineLearning.ForMathlib.Analysis.LocallyConvex.Annihilator +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Moments public import LeanMachineLearning.NeuralNetwork.Shallow.Basic +public import Mathlib.Analysis.Calculus.IteratedDeriv.Defs /-! -# Discriminatory activation functions +# Discriminatory criteria for universal approximation An activation is discriminatory on an input space if the only continuous linear functional on `C(K, ℝ)` that annihilates every neuron is zero, for every compact `K`. Hahn--Banach makes this property equivalent to universal approximation. + +For smooth activations, annihilating all ridges whose `n`-th derivative is nonzero somewhere +forces a functional to annihilate every `n`-th power of a linear coordinate. The resulting +criterion permits a different smooth ridge function at each degree. Smoothness and successive +derivatives are expressed using mathlib's `ContDiff` and `iteratedDeriv`. -/ @[expose] public section +open scoped ContDiff + namespace Learning.ShallowNetwork -variable {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] +/-! ## The dual criterion -/ + +section Discriminatory /-- An activation is discriminatory on `E` if no nonzero continuous linear functional annihilates all of its neurons on a compact subset of `E`. -/ -class IsDiscriminatory (σ : C(ℝ, ℝ)) : Prop where - annihilator_eq_zero : - ∀ (K : Set E), IsCompact K → ∀ Λ : StrongDual ℝ C(K, ℝ), +class IsDiscriminatory (E : Type*) [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + (σ : C(ℝ, ℝ)) : Prop where + annihilator_eq_zero : ∀ (K : Set E), IsCompact K → ∀ Λ : StrongDual ℝ C(K, ℝ), (∀ w b, Λ ((neuron σ w b).restrict K) = 0) → Λ = 0 +variable {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] + /-- For shallow networks, the discriminatory-functional criterion is equivalent to universal approximation. -/ theorem isUniversal_iff_isDiscriminatory (σ : C(ℝ, ℝ)) : - IsUniversal (E := E) σ ↔ IsDiscriminatory (E := E) σ := by + IsUniversal E σ ↔ IsDiscriminatory E σ := by constructor · rintro ⟨h_dense⟩ constructor @@ -57,4 +71,115 @@ theorem isUniversal_iff_isDiscriminatory (σ : C(ℝ, ℝ)) : intro w b exact hΛ _ (Submodule.subset_span ⟨(w, b), rfl⟩) +end Discriminatory + +/-! ## Smooth ridge criteria -/ + +section SmoothActivation + +variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] + +/-- A functional annihilating every ridge of `g` also annihilates every power of a linear +coordinate for which the corresponding derivative of `g` is nonzero somewhere. -/ +theorem annihilates_coordinate_pow_of_iteratedDeriv_ne_zero + {g : C(ℝ, ℝ)} {K : Set E} {Λ : StrongDual ℝ C(K, ℝ)} {n : ℕ} {b : ℝ} {w : E} + (hg : ContDiff ℝ ∞ g) (hK : IsCompact K) + (hΛ : ∀ w b, Λ ((neuron g w b).restrict K) = 0) (hb : iteratedDeriv n g b ≠ 0) : + Λ ((ContinuousMap.innerProductCoordinate K w) ^ n) = 0 := by + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + let u : C(K, ℝ) := ContinuousMap.innerProductCoordinate K w + let arg (t : ℝ) : C(K, ℝ) := ContinuousMap.const K b + ContinuousMap.const K t * u + let d (m : ℕ) : C(ℝ, ℝ) := ⟨iteratedDeriv m g, hg.continuous_iteratedDeriv m (by simp)⟩ + have hstep : ∀ m t, Λ (u ^ m * (d m).comp (arg t)) = 0 := by + intro m + induction m with + | zero => + intro t + have hd0 : d 0 = g := by + ext x + simp [d] + rw [hd0] + have heq : g.comp (arg t) = (neuron g (t • w) b).restrict K := by + ext x + simp only [arg, ContinuousMap.comp_apply, ContinuousMap.add_apply, + ContinuousMap.const_apply, ContinuousMap.mul_apply, u, + ContinuousMap.innerProductCoordinate_apply, + neuron_apply, ContinuousMap.restrict_apply, real_inner_smul_left] + congr 1 + ring + rw [heq] + simpa using hΛ (t • w) b + | succ m ihm => + intro t + have hd : ∀ y, HasDerivAt (d m) (d (m + 1) y) y := by + intro y + simpa only [d, ContinuousMap.coe_mk, iteratedDeriv_succ] using + (hg.differentiable_iteratedDeriv m + (by exact_mod_cast ENat.natCast_lt_top m) y).hasDerivAt + have hcurve := HasDerivAt.continuousMap_comp_affine hd (ContinuousMap.const K b) u t + have hmul := hcurve.const_mul (u ^ m) + have hmul' : HasDerivAt (fun s ↦ u ^ m * (d m).comp (arg s)) + (u ^ m * (u * (d (m + 1)).comp (arg t))) t := by + convert hmul using 1 + ext x + simp [arg, smul_eq_mul] + have happly : HasDerivAt (fun s ↦ Λ (u ^ m * (d m).comp (arg s))) + (Λ (u ^ m * (u * (d (m + 1)).comp (arg t)))) t := by + simpa [Function.comp_def] using Λ.hasFDerivAt.comp_hasDerivAt_of_eq t hmul' rfl + have hzero : HasDerivAt (fun s ↦ Λ (u ^ m * (d m).comp (arg s))) 0 t := by + convert hasDerivAt_const t (0 : ℝ) using 1 + funext s + exact ihm s + have hz := happly.unique hzero + simpa [pow_succ, mul_assoc] using hz + have h := hstep n 0 + have harg : (d n).comp (arg 0) = ContinuousMap.const K (d n b) := by + ext x + simp [arg] + rw [harg] at h + have heq : u ^ n * ContinuousMap.const K (d n b) = d n b • u ^ n := by + ext x + simp [mul_comm] + rw [heq, map_smul] at h + exact (mul_eq_zero.mp h).resolve_left hb + +/-- A degree-by-degree smooth ridge family is enough for the discriminatory property. The +smooth function may depend on the degree, as needed after mollification. -/ +theorem isDiscriminatory_of_smooth_ridges {σ : C(ℝ, ℝ)} (hsmooth : ∀ n : ℕ, ∃ (g : C(ℝ, ℝ)) (b : ℝ), + ContDiff ℝ ∞ g ∧ iteratedDeriv n g b ≠ 0 ∧ + ∀ (K : Set E) (_hK : IsCompact K) (Λ : StrongDual ℝ C(K, ℝ)), + (∀ w c, Λ ((neuron σ w c).restrict K) = 0) → + ∀ w c, Λ ((neuron g w c).restrict K) = 0) : + IsDiscriminatory E σ := by + constructor + intro K hK Λ hΛ + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK + apply StrongDual.eq_zero_of_innerProductCoordinate_powers K Λ + intro n w + obtain ⟨g, b, hg, hb, htransfer⟩ := hsmooth n + exact annihilates_coordinate_pow_of_iteratedDeriv_ne_zero hg hK (htransfer K hK Λ hΛ) hb + +/-- A smooth activation with no identically-zero derivative is discriminatory on every real +inner-product space. -/ +theorem isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero + {g : C(ℝ, ℝ)} (hg : ContDiff ℝ ∞ g) (hne : ∀ n : ℕ, ∃ b : ℝ, iteratedDeriv n g b ≠ 0) : + IsDiscriminatory E g := by + apply isDiscriminatory_of_smooth_ridges + intro n + obtain ⟨b, hb⟩ := hne n + exact ⟨g, b, hg, hb, by + intro K hK Λ hΛ + exact hΛ⟩ + +/-- A smooth activation with no identically-zero derivative has the universal approximation +property on every real inner-product space. -/ +theorem isUniversal_of_contDiff_of_iteratedDeriv_ne_zero + {g : C(ℝ, ℝ)} (hg : ContDiff ℝ ∞ g) + (hne : ∀ n : ℕ, ∃ b : ℝ, iteratedDeriv n g b ≠ 0) : + IsUniversal E g := + (isUniversal_iff_isDiscriminatory g).2 + (isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero hg hne) + +end SmoothActivation + end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean similarity index 58% rename from LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean rename to LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean index 069cb0fa..7af93a15 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Main.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean @@ -12,13 +12,13 @@ public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Polynomia /-! # The Leshno--Lin--Pinkus--Schocken theorem -This file states the precise universal approximation theorem. The input space is required to be -nontrivial: in dimension zero, every shallow-network function is constant, and a polynomial -activation can still be universal. +This file combines the sufficient direction from `Nonpolynomial` with the necessary direction +from `PolynomialObstruction` to characterize continuous universal activations. -The sufficient direction follows from the distributional, convolution-smoothing, and -Stone--Weierstrass arguments, while the reverse implication follows from the polynomial -obstruction. +The input space is required to be nontrivial: in dimension zero, every shallow-network function +is constant, and a polynomial activation can still be universal. The results include the +equivalence on arbitrary nontrivial real inner-product spaces, its compact-set formulation, +and the classical Euclidean-space theorem. -/ @[expose] public section @@ -27,20 +27,12 @@ namespace Learning.ShallowNetwork variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] -/-- Every continuous nonpolynomial activation is universal on compact subsets of a real -inner-product space. -/ -theorem isUniversal_of_not_isPolynomial - (σ : C(ℝ, ℝ)) (hσ : ¬ Function.IsPolynomial σ) : IsUniversal (E := E) σ := - (isUniversal_iff_isDiscriminatory σ).2 - (isDiscriminatory_of_not_isPolynomial σ hσ) - /-- Abstract form of the Leshno--Lin--Pinkus--Schocken equivalence on an arbitrary nontrivial real inner-product space. -/ theorem not_isPolynomial_iff_isUniversal [Nontrivial E] (σ : C(ℝ, ℝ)) : - ¬ Function.IsPolynomial σ ↔ IsUniversal (E := E) σ := - ⟨isUniversal_of_not_isPolynomial σ, - fun hUniversal hPolynomial ↦ - not_isUniversal_of_isPolynomial σ hPolynomial hUniversal⟩ + ¬ Function.IsPolynomial σ ↔ IsUniversal E σ := + ⟨isUniversal_of_not_isPolynomial, fun hUniversal hPolynomial ↦ + not_isUniversal_of_isPolynomial hPolynomial hUniversal⟩ /-- Precise compact-set form of the Leshno--Lin--Pinkus--Schocken equivalence. @@ -48,8 +40,7 @@ The approximating subspace is `spaceOn σ K`, whose generators are exactly the r `K` of biased ridge functions `x ↦ σ (⟪w, x⟫ + b)`. -/ theorem not_isPolynomial_iff_dense_on_compact [Nontrivial E] (σ : C(ℝ, ℝ)) : - ¬ Function.IsPolynomial σ ↔ - ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := + ¬ Function.IsPolynomial σ ↔ ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := (not_isPolynomial_iff_isUniversal σ).trans (isUniversal_iff σ) /-- The classical theorem on `ℝ^d`, represented as `EuclideanSpace ℝ (Fin d)`. @@ -57,8 +48,7 @@ theorem not_isPolynomial_iff_dense_on_compact [Nontrivial E] (σ : C(ℝ, ℝ)) The hypothesis `0 < d` is essential: the claimed equivalence is false for the zero-dimensional input space. -/ -theorem leshno_lin_pinkus_schocken {d : ℕ} (hd : 0 < d) - (σ : C(ℝ, ℝ)) : +theorem leshno_lin_pinkus_schocken {d : ℕ} (hd : 0 < d) (σ : C(ℝ, ℝ)) : ¬ Function.IsPolynomial σ ↔ ∀ (K : Set (EuclideanSpace ℝ (Fin d))), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := by diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean index d38f80b2..e64ac8db 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean @@ -5,41 +5,154 @@ Authors: Yi Yuan -/ module -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.ConvolutionSmooth +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.PolynomialCharacterization +public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction.Normalize +public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Convolution /-! -# Nonpolynomial activations are discriminatory +# Universal approximation for nonpolynomial activations -This file assembles the analytic ingredients of the sufficient direction of the -Leshno--Lin--Pinkus--Schocken theorem. Distributional differentiation supplies, at every order, -a compactly supported smooth convolution whose derivative is nonzero. Annihilator transfer for -convolution and the smooth-ridge argument then imply that the original activation is -discriminatory. +A continuous nonpolynomial function defines a regular distribution whose derivative of every +order is nonzero. Evaluating the derivative on a suitable test function gives a nonzero integral +against the reflected activation `x ↦ σ (-x)`. -The result applies to an arbitrary real inner-product space. Finite dimensionality is not needed: -the approximation problem is posed on compact subsets, where the algebra generated by inner-product -coordinates still separates points. +For each order `n`, this constructs a smooth test-function convolution whose `n`-th derivative +at the origin is nonzero. The kernel may depend on `n`. Annihilator transfer for convolution +and the degree-by-degree smooth ridge criterion then prove that the original activation is +discriminatory and universal. + +The result applies to arbitrary real inner-product spaces. Neither nontriviality nor finite +dimensionality is needed for this sufficient direction. -/ @[expose] public section +open MeasureTheory +open scoped Distributions + namespace Learning.ShallowNetwork +/-! ## Reflection of the activation -/ + +/-- Reflect a continuous real function through the origin. -/ +noncomputable def reflectedActivation (σ : C(ℝ, ℝ)) : C(ℝ, ℝ) := + σ.comp (-ContinuousMap.id ℝ) + +@[simp] +theorem reflectedActivation_apply (σ : C(ℝ, ℝ)) (x : ℝ) : + reflectedActivation σ x = σ (-x) := rfl + +/-- Reflection preserves the property of being a polynomial function. -/ +theorem isPolynomial_reflectedActivation_iff (σ : C(ℝ, ℝ)) : + Function.IsPolynomial (reflectedActivation σ) ↔ Function.IsPolynomial σ := by + constructor + · rintro ⟨p, hp⟩ + exact ⟨p.comp (-Polynomial.X), by simp [hp]⟩ + · rintro ⟨p, hp⟩ + exact ⟨p.comp (-Polynomial.X), by simp [hp]⟩ + +end Learning.ShallowNetwork + +namespace Distribution + +/-! ## Nonzero distributional derivatives -/ + +open LineDeriv + +/-- Evaluate an iterated distributional derivative by moving all derivatives onto the test +function. The statement is vector-valued and valid on every open subset of the real line. -/ +theorem iteratedLineDerivOp_apply_iterated_testFunction + {F : Type*} [AddCommGroup F] [Module ℝ F] [TopologicalSpace F] + [IsTopologicalAddGroup F] [ContinuousSMul ℝ F] + {Ω : TopologicalSpace.Opens ℝ} (T : 𝓓'(Ω, F)) (n : ℕ) (φ : 𝓓(Ω, ℝ)) : + iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) T φ = + (-1 : ℝ) ^ n • T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) := by + rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] + induction n generalizing T φ with + | zero => simp + | succ n ih => + rw [Function.iterate_succ_apply'] + change Distribution.lineDerivCLM (1 : ℝ) ((∂_{(1 : ℝ)})^[n] T) φ = + (-1 : ℝ) ^ (n + 1) • T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) + simp [Distribution.lineDerivCLM_apply, ih, Function.iterate_succ_apply, pow_succ] + +/-- Every distributional derivative of a continuous nonpolynomial function is nonzero. + +The compactly-supported-primitive class is the exactness input used by the converse +characterization of polynomial regular distributions. -/ +theorem iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial + {f : C(ℝ, ℝ)} (hf : ¬ Function.IsPolynomial f) (n : ℕ) : + iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) + (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) ≠ 0 := by + intro hzero + apply hf + obtain ⟨ρ, hρ⟩ := TestFunction.exists_integral_eq_one + (Ω := (⊤ : TopologicalSpace.Opens ℝ)) volume Set.univ_nonempty + exact isPolynomial_of_iteratedLineDerivOp_ofFun_eq_zero ρ hρ f.continuous n hzero + +end Distribution + +namespace Learning.ShallowNetwork + +/-! ## Test-function and convolution witnesses -/ + +/-- For each order, a continuous nonpolynomial activation admits a test function whose iterated +derivative has nonzero pairing with the reflected activation. + +Equivalently, this is the nonzero value at the origin of the corresponding derivative of the +left convolution of the test function with `σ`. -/ +theorem exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero + {σ : C(ℝ, ℝ)} (hσ : ¬ Function.IsPolynomial σ) (n : ℕ) : + ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), + ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) ≠ 0 := by + let f : C(ℝ, ℝ) := reflectedActivation σ + let T : 𝓓'((⊤ : TopologicalSpace.Opens ℝ), ℝ) := + LineDeriv.iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) + (Distribution.ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) + have hf : ¬ Function.IsPolynomial f := by + simpa only [f, isPolynomial_reflectedActivation_iff] using hσ + have hT : T ≠ 0 := Distribution.iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial hf n + obtain ⟨φ, hφ⟩ := T.exists_ne_zero hT + refine ⟨φ, ?_⟩ + have hfloc : LocallyIntegrableOn f (Set.univ : Set ℝ) volume := + f.continuous.locallyIntegrable.locallyIntegrableOn _ + have heval : T φ = (-1 : ℝ) ^ n * + ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by + rw [Distribution.iteratedLineDerivOp_apply_iterated_testFunction, + Distribution.ofFun_apply hfloc] + simp [smul_eq_mul, f, reflectedActivation_apply] + intro hzero + apply hφ + exact heval.trans (by rw [hzero, mul_zero]) + +/-- For every order, a nonpolynomial continuous activation has a test-function convolution whose +derivative is nonzero at the origin in that order. -/ +theorem exists_testFunction_iteratedDeriv_convolutionActivation_ne_zero + {σ : C(ℝ, ℝ)} (hσ : ¬ Function.IsPolynomial σ) (n : ℕ) : + ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), + iteratedDeriv n (convolutionActivation φ σ) 0 ≠ 0 := by + obtain ⟨φ, hφ⟩ := exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero hσ n + exact ⟨φ, by rwa [iteratedDeriv_convolutionActivation_testFunction_apply_zero]⟩ + +/-! ## Universality of nonpolynomial activations -/ + variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] /-- Every continuous nonpolynomial activation is discriminatory on compact subsets of an arbitrary real inner-product space. -/ -theorem isDiscriminatory_of_not_isPolynomial (σ : C(ℝ, ℝ)) - (hσ : ¬ Function.IsPolynomial σ) : IsDiscriminatory (E := E) σ := by - apply HasContinuousDerivativeTower.isDiscriminatory_of_smooth_ridges σ +theorem isDiscriminatory_of_not_isPolynomial {σ : C(ℝ, ℝ)} + (hσ : ¬ Function.IsPolynomial σ) : IsDiscriminatory E σ := by + apply isDiscriminatory_of_smooth_ridges intro n - obtain ⟨φ, hφ⟩ := - exists_testFunction_derivative_convolutionActivation_ne_zero σ hσ n - let g : C(ℝ, ℝ) := convolutionActivation φ σ - let hg : HasContinuousDerivativeTower g := inferInstance - refine ⟨g, hg, 0, ?_, ?_⟩ - · exact hφ - · intro K hK Λ hΛ - exact annihilates_convolutionActivation_neurons φ σ K hK Λ hΛ + obtain ⟨φ, hφ⟩ := exists_testFunction_iteratedDeriv_convolutionActivation_ne_zero hσ n + refine ⟨convolutionActivation φ σ, 0, convolutionActivation_contDiff φ σ φ.contDiff, hφ, ?_⟩ + intro K hK Λ hΛ + exact annihilates_convolutionActivation_neurons φ hK hΛ + +/-- Every continuous nonpolynomial activation is universal on compact subsets of a real +inner-product space. -/ +theorem isUniversal_of_not_isPolynomial + {σ : C(ℝ, ℝ)} (hσ : ¬ Function.IsPolynomial σ) : IsUniversal E σ := + (isUniversal_iff_isDiscriminatory σ).2 (isDiscriminatory_of_not_isPolynomial hσ) end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean deleted file mode 100644 index 0f36e720..00000000 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/NonpolynomialWitness.lean +++ /dev/null @@ -1,128 +0,0 @@ -/- -Copyright (c) 2026 Yi Yuan. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Yi Yuan --/ -module - -public import LeanMachineLearning.ForMathlib.Analysis.Distribution.PolynomialCharacterization -public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction.Normalize - -/-! -# Test-function witnesses for nonpolynomial activations - -This file extracts the analytic witness needed by convolution-based universal-approximation -arguments. A continuous nonpolynomial function defines a regular distribution whose derivative -of every order is nonzero. Evaluating such a derivative on a suitable test function gives a -nonzero integral involving the corresponding iterated derivative of the test function. - -The final theorem uses the reflected activation `x ↦ σ (-x)`. This is exactly the orientation -that occurs when evaluating the left convolution `φ ⋆ σ` at zero. --/ - -@[expose] public section - -open MeasureTheory -open scoped Distributions - -namespace Learning.ShallowNetwork - -/-- Reflect a continuous real function through the origin. -/ -noncomputable def reflectedActivation (σ : C(ℝ, ℝ)) : C(ℝ, ℝ) := - σ.comp (-ContinuousMap.id ℝ) - -@[simp] -theorem reflectedActivation_apply (σ : C(ℝ, ℝ)) (x : ℝ) : - reflectedActivation σ x = σ (-x) := rfl - -/-- Reflection preserves the property of being a polynomial function. -/ -theorem isPolynomial_reflectedActivation_iff (σ : C(ℝ, ℝ)) : - Function.IsPolynomial (reflectedActivation σ) ↔ Function.IsPolynomial σ := by - constructor - · rintro ⟨p, hp⟩ - refine ⟨p.comp (-Polynomial.X), fun x ↦ ?_⟩ - simp only [Polynomial.eval_comp, Polynomial.eval_neg, Polynomial.eval_X, - hp, reflectedActivation_apply, neg_neg] - · rintro ⟨p, hp⟩ - refine ⟨p.comp (-Polynomial.X), fun x ↦ ?_⟩ - simp only [Polynomial.eval_comp, Polynomial.eval_neg, Polynomial.eval_X, - hp, reflectedActivation_apply] - -end Learning.ShallowNetwork - -namespace Distribution - -open LineDeriv - -/-- Evaluate an iterated distributional derivative by moving all derivatives onto the test -function. The statement is vector-valued and valid on every open subset of the real line. -/ -theorem iteratedLineDerivOp_apply_iterated_testFunction - {F : Type*} [AddCommGroup F] [Module ℝ F] [TopologicalSpace F] - [IsTopologicalAddGroup F] [ContinuousSMul ℝ F] - {Ω : TopologicalSpace.Opens ℝ} (T : 𝓓'(Ω, F)) (n : ℕ) (φ : 𝓓(Ω, ℝ)) : - iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) T φ = - (-1 : ℝ) ^ n • T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) := by - rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] - induction n generalizing T φ with - | zero => simp - | succ n ih => - rw [Function.iterate_succ_apply'] - change Distribution.lineDerivCLM (1 : ℝ) ((∂_{(1 : ℝ)})^[n] T) φ = - (-1 : ℝ) ^ (n + 1) • - T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) - rw [Distribution.lineDerivCLM_apply, ih] - rw [Function.iterate_succ_apply] - simp only [pow_succ] - module - -/-- Every distributional derivative of a continuous nonpolynomial function is nonzero. - -The compactly-supported-primitive class is the exactness input used by the converse -characterization of polynomial regular distributions. -/ -theorem iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial - (f : C(ℝ, ℝ)) (hf : ¬ Function.IsPolynomial f) (n : ℕ) : - iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) - (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) ≠ 0 := by - intro hzero - apply hf - exact isPolynomial_of_iteratedLineDerivOp_ofFun_eq_zero - TestFunction.normalizedBumpReal TestFunction.integral_normalizedBumpReal - f.continuous n hzero - -end Distribution - -namespace Learning.ShallowNetwork - -/-- For each order, a continuous nonpolynomial activation admits a test function whose iterated -derivative has nonzero pairing with the reflected activation. - -Equivalently, this is the nonzero value at the origin of the corresponding derivative of the -left convolution of the test function with `σ`. -/ -theorem exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero - (σ : C(ℝ, ℝ)) (hσ : ¬ Function.IsPolynomial σ) (n : ℕ) : - ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), - ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) ≠ 0 := by - let f : C(ℝ, ℝ) := reflectedActivation σ - let T : 𝓓'((⊤ : TopologicalSpace.Opens ℝ), ℝ) := - LineDeriv.iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) - (Distribution.ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) - have hf : ¬ Function.IsPolynomial f := by - simpa only [f, isPolynomial_reflectedActivation_iff] using hσ - have hT : T ≠ 0 := by - dsimp only [T] - exact Distribution.iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial f hf n - obtain ⟨φ, hφ⟩ := T.exists_ne_zero hT - refine ⟨φ, ?_⟩ - have hfloc : LocallyIntegrableOn f (Set.univ : Set ℝ) volume := - f.continuous.locallyIntegrable.locallyIntegrableOn _ - have heval : T φ = (-1 : ℝ) ^ n * - ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by - dsimp only [T] - rw [Distribution.iteratedLineDerivOp_apply_iterated_testFunction] - rw [Distribution.ofFun_apply hfloc] - simp only [smul_eq_mul, f, reflectedActivation_apply] - intro hzero - apply hφ - exact heval.trans (by rw [hzero, mul_zero]) - -end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean index b69a5cb5..b833e8e6 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean @@ -6,7 +6,6 @@ Authors: Yi Yuan module public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function -public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Affine public import LeanMachineLearning.ForMathlib.Topology.Algebra.Module.FiniteDimension public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Discrete public import LeanMachineLearning.NeuralNetwork.Shallow.Basic @@ -29,16 +28,12 @@ namespace Learning.ShallowNetwork variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] -private theorem normalized_inner_smul_self (e : E) (he : e ≠ 0) (r : ℝ) : - inner ℝ e (r • e) / inner ℝ e e = r := by - rw [real_inner_smul_right, div_eq_iff (inner_self_ne_zero.mpr he)] - /-- The space of continuous real-valued functions on `n` distinct scalar multiples of a nonzero vector has dimension `n`. The statement is phrased using a range, so it applies without choosing a finite set enumeration. -/ -theorem finrank_continuousMap_range_fin_smul (e : E) (he : e ≠ 0) (n : ℕ) : +theorem finrank_continuousMap_range_fin_smul {e : E} (he : e ≠ 0) (n : ℕ) : Module.finrank ℝ C(Set.range (fun i : Fin n ↦ (i : ℝ) • e), ℝ) = n := by classical let emb : Fin n ↪ E := @@ -51,25 +46,23 @@ theorem finrank_continuousMap_range_fin_smul (e : E) (he : e ≠ 0) (n : ℕ) : have hcard : Fintype.card K = n := (Fintype.card_congr emb.toEquivRange).symm.trans (Fintype.card_fin n) change Module.finrank ℝ C(K, ℝ) = n - rw [(ContinuousMap.linearEquivFnOfDiscrete ℝ).finrank_eq, - Module.finrank_pi, hcard] + rw [(ContinuousMap.linearEquivFnOfDiscrete ℝ).finrank_eq, Module.finrank_pi, hcard] private theorem aeval_degreeLT_range_ne_top {X : Type*} [TopologicalSpace X] - (coordinate : C(X, ℝ)) (n : ℕ) + {coordinate : C(X, ℝ)} {n : ℕ} (hfinrank : Module.finrank ℝ C(X, ℝ) = n + 1) : - ((Polynomial.aeval coordinate).toLinearMap.domRestrict - (degreeLT ℝ n)).range ≠ ⊤ := by + ((Polynomial.aeval coordinate).toLinearMap.domRestrict (degreeLT ℝ n)).range ≠ ⊤ := by intro htop have hrank := LinearMap.finrank_range_le ((Polynomial.aeval coordinate).toLinearMap.domRestrict (degreeLT ℝ n)) rw [htop, finrank_top, hfinrank, Module.finrank_eq_card_basis (degreeLT.basis ℝ n), Fintype.card_fin] at hrank - omega + lia /-- If the activation is polynomial, then its shallow-network space fails to be dense on some finite (hence compact) set. No finite-dimensionality assumption on the input space is needed. -/ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] - (σ : C(ℝ, ℝ)) (hσ : Function.IsPolynomial σ) : + {σ : C(ℝ, ℝ)} (hσ : Function.IsPolynomial σ) : ∃ K : Set E, IsCompact K ∧ ¬ Dense (spaceOn σ K : Set C(K, ℝ)) := by classical obtain ⟨p, hp⟩ := hσ @@ -83,53 +76,53 @@ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] ⟨fun x ↦ inner ℝ e x / inner ℝ e e, (continuous_const.inner continuous_subtype_val).div_const _⟩ let evalDegree : ↥(degreeLT ℝ (p.natDegree + 1)) →ₗ[ℝ] C(K, ℝ) := - (Polynomial.aeval coordinate).toLinearMap.domRestrict - (degreeLT ℝ (p.natDegree + 1)) - have hpDegree : p ∈ degreeLT ℝ (p.natDegree + 1) := by - rw [degreeLT_succ_eq_degreeLE, mem_degreeLE] - exact degree_le_natDegree - have hfinrank : Module.finrank ℝ C(K, ℝ) = p.natDegree + 2 := by - change Module.finrank ℝ - C(Set.range (fun i : Fin (p.natDegree + 2) ↦ (i : ℝ) • e), ℝ) = _ - exact finrank_continuousMap_range_fin_smul e he _ + (Polynomial.aeval coordinate).toLinearMap.domRestrict (degreeLT ℝ (p.natDegree + 1)) + have hfinrank : Module.finrank ℝ C(K, ℝ) = p.natDegree + 2 := + finrank_continuousMap_range_fin_smul he _ have hproper : evalDegree.range ≠ ⊤ := - aeval_degreeLT_range_ne_top coordinate (p.natDegree + 1) (by simpa using hfinrank) + aeval_degreeLT_range_ne_top (by simpa using hfinrank) have hspace : (spaceOn σ K : Set C(K, ℝ)) ⊆ evalDegree.range := by rw [SetLike.coe_subset_coe, spaceOn] apply Submodule.span_le.2 rintro _ ⟨⟨w, b⟩, rfl⟩ let q := p.comp (C (inner ℝ w e) * X + C b) - have hq : q ∈ degreeLT ℝ (p.natDegree + 1) := - Polynomial.compAffineDegreeLT hpDegree _ _ - change (neuron σ w b).restrict K ∈ evalDegree.range + have hq : q ∈ degreeLT ℝ (p.natDegree + 1) := by + rw [degreeLT_succ_eq_degreeLE, mem_degreeLE, ← natDegree_le_iff_degree_le] + calc + q.natDegree ≤ p.natDegree * (C (inner ℝ w e) * X + C b).natDegree := natDegree_comp_le + _ ≤ p.natDegree * 1 := Nat.mul_le_mul_left _ <| + natDegree_add_le_of_degree_le + (by simpa using natDegree_C_mul_X_pow_le (inner ℝ w e) 1) (by simp) + _ = p.natDegree := Nat.mul_one _ refine ⟨⟨q, hq⟩, ?_⟩ ext x obtain ⟨i, hi⟩ := x.property simp only [evalDegree, LinearMap.domRestrict_apply, AlgHom.toLinearMap_apply, Polynomial.aeval_continuousMap_apply, coordinate, ContinuousMap.coe_mk, ContinuousMap.restrict_apply, neuron_apply] - rw [← hp] - rw [← hi] + rw [← hp, ← hi] have hemb : emb i = (i : ℝ) • e := rfl - rw [hemb, normalized_inner_smul_self e he] - simp only [q, Polynomial.eval_comp_C_mul_X_add_C, real_inner_smul_right] + have : inner ℝ e ((i : ℝ) • e) / inner ℝ e e = (i : ℝ) := by + rw [real_inner_smul_right, div_eq_iff (inner_self_ne_zero.mpr he)] + rw [hemb, this] + simp only [q, Polynomial.eval_comp, Polynomial.eval_add, Polynomial.eval_C_mul, + Polynomial.eval_X, Polynomial.eval_C, real_inner_smul_right] rw [mul_comm (inner ℝ w e)] exact ⟨K, (Set.finite_range emb).isCompact, evalDegree.range.not_dense_of_subset_of_finiteDimensional hproper hspace⟩ /-- A polynomial activation is not universal on a nontrivial real inner product space. -/ theorem not_isUniversal_of_isPolynomial [Nontrivial E] - (σ : C(ℝ, ℝ)) (hσ : Function.IsPolynomial σ) : - ¬ IsUniversal (E := E) σ := by + {σ : C(ℝ, ℝ)} (hσ : Function.IsPolynomial σ) : + ¬ IsUniversal E σ := by intro hUniversal - obtain ⟨K, hK, hnotDense⟩ := - exists_compact_not_dense_of_isPolynomial (E := E) σ hσ + obtain ⟨K, hK, hnotDense⟩ := exists_compact_not_dense_of_isPolynomial (E := E) hσ exact hnotDense (hUniversal.dense_on_compact K hK) /-- Universality forces the activation not to be a polynomial. -/ theorem not_isPolynomial_of_isUniversal [Nontrivial E] - (σ : C(ℝ, ℝ)) [hσ : IsUniversal (E := E) σ] : + (σ : C(ℝ, ℝ)) [hσ : IsUniversal E σ] : ¬ Function.IsPolynomial σ := - fun hPolynomial ↦ not_isUniversal_of_isPolynomial (E := E) σ hPolynomial hσ + fun hPolynomial ↦ not_isUniversal_of_isPolynomial (E := E) hPolynomial hσ end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean deleted file mode 100644 index 4888704b..00000000 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/SmoothActivation.lean +++ /dev/null @@ -1,164 +0,0 @@ -/- -Copyright (c) 2026 Yi Yuan. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Yi Yuan --/ -module - -public import LeanMachineLearning.ForMathlib.Analysis.Calculus.ContinuousMapComposition -public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Moments -public import LeanMachineLearning.NeuralNetwork.UniversalApproximation.Discriminatory - -/-! -# Universal approximation for smooth activations - -This file isolates the smooth part of the discriminatory-function argument. A continuous -derivative tower is packaged as a class, so later convolution arguments may supply a different -smooth ridge function at each required degree. The core lemma says that annihilating all ridges -of a function whose `n`-th derivative is nonzero forces a functional to annihilate every `n`-th -power of a linear coordinate. --/ - -@[expose] public section - -namespace Learning.ShallowNetwork - -variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] - -/-- A choice of all successive continuous derivatives of a continuous real function. -/ -class HasContinuousDerivativeTower (g : C(ℝ, ℝ)) : Type where - /-- The `n`-th derivative, bundled as a continuous map. -/ - derivative : ℕ → C(ℝ, ℝ) - derivative_zero : derivative 0 = g - hasDerivAt_derivative : - ∀ (n : ℕ) (x : ℝ), HasDerivAt (derivative n) (derivative (n + 1) x) x - -/-- A smooth function whose derivative of every order is not identically zero. -/ -class HasNonzeroContinuousDerivativeTower (g : C(ℝ, ℝ)) : Type - extends HasContinuousDerivativeTower g where - exists_derivative_ne_zero : ∀ n, ∃ x, derivative n x ≠ 0 - -namespace HasContinuousDerivativeTower - -variable {g : C(ℝ, ℝ)} [hg : HasContinuousDerivativeTower g] - -@[simp] -theorem derivative_zero_eq : hg.derivative 0 = g := - hg.derivative_zero - -/-- A functional annihilating every ridge of `g` also annihilates every power of a linear -coordinate for which the corresponding derivative of `g` is nonzero somewhere. -/ -theorem annihilates_coordinate_pow_of_derivative_ne_zero - (K : Set E) (hK : IsCompact K) (Λ : StrongDual ℝ C(K, ℝ)) - (hΛ : ∀ w b, Λ ((neuron g w b).restrict K) = 0) - (n : ℕ) {b : ℝ} (hb : hg.derivative n b ≠ 0) (w : E) : - Λ ((ContinuousMap.innerProductCoordinate K w) ^ n) = 0 := by - let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - let u : C(K, ℝ) := ContinuousMap.innerProductCoordinate K w - have hstep : ∀ m t, - Λ (u ^ m * (hg.derivative m).comp - (ContinuousMap.const K b + ContinuousMap.const K t * u)) = 0 := by - intro m - induction m with - | zero => - intro t - rw [hg.derivative_zero] - have heq : g.comp - (ContinuousMap.const K b + ContinuousMap.const K t * u) = - (neuron g (t • w) b).restrict K := by - ext x - simp only [ContinuousMap.comp_apply, ContinuousMap.add_apply, - ContinuousMap.const_apply, ContinuousMap.mul_apply, u, - ContinuousMap.innerProductCoordinate_apply, - neuron_apply, ContinuousMap.restrict_apply, real_inner_smul_left] - congr 1 - ring - rw [heq] - simpa using hΛ (t • w) b - | succ m ihm => - intro t - have hcurve := HasDerivAt.continuousMap_comp_affine - (fun y ↦ hg.hasDerivAt_derivative m y) - (ContinuousMap.const K b) u t - have hmul := hcurve.const_mul (u ^ m) - have hmul' : HasDerivAt - (fun s ↦ u ^ m * (hg.derivative m).comp - (ContinuousMap.const K b + ContinuousMap.const K s * u)) - (u ^ m * (u * (hg.derivative (m + 1)).comp - (ContinuousMap.const K b + ContinuousMap.const K t * u))) t := by - convert hmul using 1 - ext x - simp [smul_eq_mul] - have happly : HasDerivAt - (fun s ↦ Λ (u ^ m * (hg.derivative m).comp - (ContinuousMap.const K b + ContinuousMap.const K s * u))) - (Λ (u ^ m * (u * (hg.derivative (m + 1)).comp - (ContinuousMap.const K b + ContinuousMap.const K t * u)))) t := by - simpa [Function.comp_def] using - Λ.hasFDerivAt.comp_hasDerivAt_of_eq t hmul' rfl - have hzero : HasDerivAt - (fun s ↦ Λ (u ^ m * (hg.derivative m).comp - (ContinuousMap.const K b + ContinuousMap.const K s * u))) 0 t := by - convert hasDerivAt_const t (0 : ℝ) using 1 - funext s - exact ihm s - have hz := happly.unique hzero - simpa [pow_succ, mul_assoc] using hz - have h := hstep n 0 - have harg : (hg.derivative n).comp - (ContinuousMap.const K b + ContinuousMap.const K 0 * u) = - ContinuousMap.const K (hg.derivative n b) := by - ext x - simp - rw [harg] at h - have heq : u ^ n * ContinuousMap.const K (hg.derivative n b) = - hg.derivative n b • u ^ n := by - ext x - simp [mul_comm] - rw [heq, map_smul] at h - exact (mul_eq_zero.mp h).resolve_left hb - -/-- A degree-by-degree smooth ridge family is enough for the discriminatory property. The -smooth function may depend on the degree; this is the form needed after mollification. -/ -theorem isDiscriminatory_of_smooth_ridges - (σ : C(ℝ, ℝ)) - (hsmooth : ∀ n : ℕ, - ∃ (g : C(ℝ, ℝ)) (hg : HasContinuousDerivativeTower g) (b : ℝ), - hg.derivative n b ≠ 0 ∧ - ∀ (K : Set E) (_hK : IsCompact K) (Λ : StrongDual ℝ C(K, ℝ)), - (∀ w c, Λ ((neuron σ w c).restrict K) = 0) → - ∀ w c, Λ ((neuron g w c).restrict K) = 0) : - IsDiscriminatory (E := E) σ := by - constructor - intro K hK Λ hΛ - let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - apply StrongDual.eq_zero_of_innerProductCoordinate_powers K Λ - intro n w - obtain ⟨g, hg, b, hb, htransfer⟩ := hsmooth n - let _ : HasContinuousDerivativeTower g := hg - exact annihilates_coordinate_pow_of_derivative_ne_zero K hK Λ - (htransfer K hK Λ hΛ) n hb w - -/-- A smooth activation with no identically-zero derivative is discriminatory on every real -inner-product space. -/ -theorem isDiscriminatory_of_hasNonzeroContinuousDerivativeTower - (g : C(ℝ, ℝ)) [hg : HasNonzeroContinuousDerivativeTower g] : - IsDiscriminatory (E := E) g := by - apply isDiscriminatory_of_smooth_ridges g - intro n - obtain ⟨b, hb⟩ := hg.exists_derivative_ne_zero n - exact ⟨g, hg.toHasContinuousDerivativeTower, b, hb, by - intro K hK Λ hΛ - exact hΛ⟩ - -/-- A smooth activation with no identically-zero derivative has the universal approximation -property on every real inner-product space. -/ -theorem isUniversal_of_hasNonzeroContinuousDerivativeTower - (g : C(ℝ, ℝ)) [HasNonzeroContinuousDerivativeTower g] : - IsUniversal (E := E) g := - (isUniversal_iff_isDiscriminatory g).2 - (isDiscriminatory_of_hasNonzeroContinuousDerivativeTower g) - -end HasContinuousDerivativeTower - -end Learning.ShallowNetwork From 34975504557ef96984a96ef2d42ccea221d09741 Mon Sep 17 00:00:00 2001 From: yuanyi-350 Date: Sat, 12 Sep 2026 14:51:13 +0800 Subject: [PATCH 3/8] Refactor annihilator proofs and continuous map coercion --- LeanMachineLearning.lean | 2 +- .../Analysis/LocallyConvex/Annihilator.lean | 84 +++++-------------- .../Integral/ClosedSubmodule.lean | 12 +-- .../Topology/ContinuousMap/Algebra.lean | 33 ++++++++ .../Topology/ContinuousMap/Discrete.lean | 34 -------- .../UniversalApproximation/Convolution.lean | 2 +- .../Discriminatory.lean | 28 +++---- .../UniversalApproximation/Nonpolynomial.lean | 6 +- .../PolynomialObstruction.lean | 13 ++- 9 files changed, 80 insertions(+), 134 deletions(-) create mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Algebra.lean delete mode 100644 LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index cd6c808a..0c04a1ca 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -42,7 +42,7 @@ public import LeanMachineLearning.ForMathlib.Probability.Moments.SubExponential public import LeanMachineLearning.ForMathlib.Probability.Moments.SubGaussian public import LeanMachineLearning.ForMathlib.Probability.WithDensity public import LeanMachineLearning.ForMathlib.Topology.Algebra.Module.FiniteDimension -public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Discrete +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Algebra public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.InnerProduct public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Moments public import LeanMachineLearning.ForMathlib.Topology.Instances.ENNReal.Lemmas diff --git a/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean b/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean index 711a3293..8f746974 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/LocallyConvex/Annihilator.lean @@ -12,8 +12,8 @@ public import Mathlib.Analysis.LocallyConvex.Separation # Density and annihilators This file characterizes dense real submodules of locally convex spaces in terms of their -continuous dual annihilators. The reverse implication is an application of geometric -Hahn--Banach separation. +continuous dual annihilators. Geometric Hahn--Banach separation bounds a functional on a +nondense submodule. Its restriction must vanish, since a nonzero linear functional is surjective. -/ @[expose] public section @@ -22,74 +22,36 @@ open Set namespace Submodule -/-- A real submodule of a locally convex topological vector space is dense exactly when every -continuous linear functional vanishing on it is zero. -/ -theorem dense_iff_forall_dual_eq_zero - {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] - [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] - (s : Submodule ℝ E) : - Dense (s : Set E) ↔ - ∀ f : StrongDual ℝ E, (∀ x ∈ s, f x = 0) → f = 0 := by +variable {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] + [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] + (s : Submodule ℝ E) + +theorem dense_iff_forall_dual_eq_zero : + Dense (s : Set E) ↔ ∀ f : StrongDual ℝ E, (∀ x ∈ s, f x = 0) → f = 0 := by constructor · intro hs f hf - ext x - have hfun : (f : E → ℝ) = (0 : E → ℝ) := - Continuous.ext_on hs f.continuous continuous_zero (by - intro y hy - simpa using hf y hy) - exact congrFun hfun x + exact ContinuousLinearMap.ext_on (by simpa using hs) hf · intro h rw [Submodule.dense_iff_topologicalClosure_eq_top] apply top_unique intro x hx by_contra hxc - obtain ⟨f, u, hfc, hfx⟩ := - geometric_hahn_banach_closed_point - s.topologicalClosure.convex - s.isClosed_topologicalClosure hxc - have hfzero : ∀ y ∈ s.topologicalClosure, f y = 0 := by - intro y hy - by_contra hfy - have hlt := hfc ((u / f y) • y) - (s.topologicalClosure.smul_mem (u / f y) hy) - rw [map_smul, smul_eq_mul, div_mul_cancel₀ u hfy] at hlt - exact (lt_irrefl u) hlt - have hf : f = 0 := - h f fun y hy ↦ hfzero y (s.le_topologicalClosure hy) - have hu0 : 0 < u := by - simpa using hfc 0 s.topologicalClosure.zero_mem - have hux0 : u < 0 := by - simpa [hf] using hfx - exact (not_lt_of_ge hu0.le) hux0 - -/-- If a real submodule of a locally convex space is not dense, a nonzero continuous linear -functional annihilates it. -/ -theorem exists_dual_annihilator_of_not_dense - {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] - [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] - (s : Submodule ℝ E) (hs : ¬ Dense (s : Set E)) : + obtain ⟨f, u, hfc, hfx⟩ := geometric_hahn_banach_closed_point s.topologicalClosure.convex + s.isClosed_topologicalClosure hxc + have hrestr : f.toLinearMap.comp s.subtype = 0 := by + by_contra hf + obtain ⟨y, hy⟩ := (f.toLinearMap.comp s.subtype).surjective hf u + exact (hfc y (s.le_topologicalClosure y.property)).ne hy + have hf : f = 0 := h f fun y hy ↦ DFunLike.congr_fun hrestr ⟨y, hy⟩ + simpa [hf] using (hfc 0 s.topologicalClosure.zero_mem).trans hfx + +theorem exists_dual_annihilator_of_not_dense (hs : ¬ Dense (s : Set E)) : ∃ f : StrongDual ℝ E, f ≠ 0 ∧ ∀ x ∈ s, f x = 0 := by - grind [Submodule.dense_iff_forall_dual_eq_zero] + simpa only [dense_iff_forall_dual_eq_zero, not_forall, exists_prop, and_comm] using hs -/-- A real submodule is dense exactly when its polar submodule is trivial. -/ -theorem dense_iff_polarSubmodule_eq_bot - {E : Type*} [TopologicalSpace E] [AddCommGroup E] [Module ℝ E] - [IsTopologicalAddGroup E] [ContinuousSMul ℝ E] [LocallyConvexSpace ℝ E] - (s : Submodule ℝ E) : +theorem dense_iff_polarSubmodule_eq_bot : Dense (s : Set E) ↔ StrongDual.polarSubmodule ℝ s = ⊥ := by - rw [dense_iff_forall_dual_eq_zero] - constructor - · intro h - ext f - rw [StrongDual.mem_polarSubmodule, Submodule.mem_bot] - constructor - · exact h f - · rintro rfl x hx - rfl - · intro h f hf - have hmem : f ∈ StrongDual.polarSubmodule ℝ s := - (StrongDual.mem_polarSubmodule ℝ s f).2 hf - rw [h] at hmem - exact hmem + simp only [dense_iff_forall_dual_eq_zero, Submodule.eq_bot_iff, + StrongDual.mem_polarSubmodule] end Submodule diff --git a/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean b/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean index 208b67cd..febe5fe3 100644 --- a/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean +++ b/LeanMachineLearning/ForMathlib/MeasureTheory/Integral/ClosedSubmodule.lean @@ -37,12 +37,8 @@ theorem integral_mem_ker (L : E →L[𝕜] F) {f : α → E} by_cases hE : CompleteSpace E · let _ := hE by_cases hfi : Integrable f μ - · apply LinearMap.mem_ker.mpr - change L (∫ x, f x ∂μ) = 0 - rw [← L.integral_comp_comm hfi] - exact integral_eq_zero_of_ae (hf.mono fun x hx ↦ by - change L (f x) = 0 - exact LinearMap.mem_ker.mp hx) + · simp only [LinearMap.mem_ker, coe_coe, ← L.integral_comp_comm hfi] + exact integral_eq_zero_of_ae (hf.mono fun x hx ↦ LinearMap.mem_ker.mp hx) · simp [integral_undef hfi] · simp [integral, hE] @@ -65,9 +61,7 @@ theorem integral_mem (S : Submodule 𝕜 E) (hS : IsClosed (S : Set E)) rw [← S.ker_mkQ] exact S.mkQL.integral_mem_ker (by filter_upwards [hf] with x hx - apply LinearMap.mem_ker.mpr - change S.mkQL (f x) = 0 - exact (Submodule.Quotient.mk_eq_zero S).2 hx) + exact LinearMap.mem_ker.mpr ((Submodule.Quotient.mk_eq_zero S).2 hx)) · simp [integral, hE] /-- The Bochner integral of a function valued almost everywhere in a submodule belongs to the diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Algebra.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Algebra.lean new file mode 100644 index 00000000..8394ec64 --- /dev/null +++ b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Algebra.lean @@ -0,0 +1,33 @@ +/- +Copyright (c) 2026 Yi Yuan. All rights reserved. +Released under Apache 2.0 license as described in the file LICENSE. +Authors: Yi Yuan +-/ +module + +public import Mathlib.Topology.ContinuousMap.Algebra + +/-! +# Continuous linear maps on continuous function spaces + +This file bundles coercion from continuous maps to arbitrary functions as a continuous linear map. +-/ + +universe u v w + +@[expose] public section + +namespace ContinuousMap + +variable (R : Type u) {X : Type v} {M : Type w} +variable [Semiring R] [TopologicalSpace X] +variable [TopologicalSpace M] [AddCommMonoid M] [ContinuousAdd M] +variable [Module R M] [ContinuousConstSMul R M] + +/-- Coercion to a function as a continuous linear map. -/ +@[simps! apply] +def coeFnCLM : C(X, M) →L[R] (X → M) where + __ := coeFnLinearMap R + cont := continuous_coeFun + +end ContinuousMap diff --git a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean b/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean deleted file mode 100644 index ca80c777..00000000 --- a/LeanMachineLearning/ForMathlib/Topology/ContinuousMap/Discrete.lean +++ /dev/null @@ -1,34 +0,0 @@ -/- -Copyright (c) 2026 Yi Yuan. All rights reserved. -Released under Apache 2.0 license as described in the file LICENSE. -Authors: Yi Yuan --/ -module - -public import Mathlib.Topology.ContinuousMap.Algebra - -/-! -# Continuous maps from a discrete space - -This file upgrades the equivalence between continuous maps from a discrete space and arbitrary -functions to a linear equivalence. --/ - -universe u v w - -@[expose] public section - -namespace ContinuousMap - -variable (R : Type u) {X : Type v} {M : Type w} -variable [Semiring R] [TopologicalSpace X] [DiscreteTopology X] -variable [TopologicalSpace M] [AddCommMonoid M] [ContinuousAdd M] -variable [Module R M] [ContinuousConstSMul R M] - -/-- Continuous maps from a discrete space are linearly equivalent to arbitrary functions. -/ -def linearEquivFnOfDiscrete : C(X, M) ≃ₗ[R] (X → M) := - equivFnOfDiscrete.toLinearEquiv - { map_add := fun _ _ ↦ rfl - map_smul := fun _ _ ↦ rfl } - -end ContinuousMap diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean index 0ffc8817..a06251df 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean @@ -172,7 +172,7 @@ theorem annihilates_convolutionActivation_neurons let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK intro w b let u : C(K, ℝ) := - ⟨fun x => inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ + ⟨fun x ↦ inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ have htrans : ∀ c, Λ (activationAlong σ u c) = 0 := by intro c rw [show activationAlong σ u c = (neuron σ w c).restrict K by rfl] diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean index 30e69a44..b73bd05b 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean @@ -113,14 +113,13 @@ theorem annihilates_coordinate_pow_of_iteratedDeriv_ne_zero intro t have hd : ∀ y, HasDerivAt (d m) (d (m + 1) y) y := by intro y - simpa only [d, ContinuousMap.coe_mk, iteratedDeriv_succ] using + simpa [d, ContinuousMap.coe_mk, iteratedDeriv_succ] using (hg.differentiable_iteratedDeriv m (by exact_mod_cast ENat.natCast_lt_top m) y).hasDerivAt have hcurve := HasDerivAt.continuousMap_comp_affine hd (ContinuousMap.const K b) u t - have hmul := hcurve.const_mul (u ^ m) have hmul' : HasDerivAt (fun s ↦ u ^ m * (d m).comp (arg s)) (u ^ m * (u * (d (m + 1)).comp (arg t))) t := by - convert hmul using 1 + convert hcurve.const_mul (u ^ m) using 1 ext x simp [arg, smul_eq_mul] have happly : HasDerivAt (fun s ↦ Λ (u ^ m * (d m).comp (arg s))) @@ -130,17 +129,14 @@ theorem annihilates_coordinate_pow_of_iteratedDeriv_ne_zero convert hasDerivAt_const t (0 : ℝ) using 1 funext s exact ihm s - have hz := happly.unique hzero - simpa [pow_succ, mul_assoc] using hz - have h := hstep n 0 - have harg : (d n).comp (arg 0) = ContinuousMap.const K (d n b) := by - ext x - simp [arg] - rw [harg] at h - have heq : u ^ n * ContinuousMap.const K (d n b) = d n b • u ^ n := by - ext x - simp [mul_comm] - rw [heq, map_smul] at h + simpa [pow_succ, mul_assoc] using happly.unique hzero + have h : d n b * Λ (u ^ n) = 0 := calc + d n b * Λ (u ^ n) = Λ (d n b • u ^ n) := by simp + _ = Λ (u ^ n * (d n).comp (arg 0)) := by + congr 1 + ext x + simp [arg, mul_comm] + _ = 0 := hstep n 0 exact (mul_eq_zero.mp h).resolve_left hb /-- A degree-by-degree smooth ridge family is enough for the discriminatory property. The @@ -167,9 +163,7 @@ theorem isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero apply isDiscriminatory_of_smooth_ridges intro n obtain ⟨b, hb⟩ := hne n - exact ⟨g, b, hg, hb, by - intro K hK Λ hΛ - exact hΛ⟩ + exact ⟨g, b, hg, hb, by simp⟩ /-- A smooth activation with no identically-zero derivative has the universal approximation property on every real inner-product space. -/ diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean index e64ac8db..b303ea87 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean @@ -110,14 +110,14 @@ theorem exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero LineDeriv.iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) (Distribution.ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) have hf : ¬ Function.IsPolynomial f := by - simpa only [f, isPolynomial_reflectedActivation_iff] using hσ + simpa [f, isPolynomial_reflectedActivation_iff] using hσ have hT : T ≠ 0 := Distribution.iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial hf n obtain ⟨φ, hφ⟩ := T.exists_ne_zero hT refine ⟨φ, ?_⟩ - have hfloc : LocallyIntegrableOn f (Set.univ : Set ℝ) volume := + have hfloc : LocallyIntegrableOn f ⊤ volume := f.continuous.locallyIntegrable.locallyIntegrableOn _ have heval : T φ = (-1 : ℝ) ^ n * - ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by + ∫ s : ℝ, ((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ s * σ (-s) := by rw [Distribution.iteratedLineDerivOp_apply_iterated_testFunction, Distribution.ofFun_apply hfloc] simp [smul_eq_mul, f, reflectedActivation_apply] diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean index b833e8e6..00d744b2 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean @@ -7,7 +7,7 @@ module public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function public import LeanMachineLearning.ForMathlib.Topology.Algebra.Module.FiniteDimension -public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Discrete +public import LeanMachineLearning.ForMathlib.Topology.ContinuousMap.Algebra public import LeanMachineLearning.NeuralNetwork.Shallow.Basic public import Mathlib.RingTheory.Polynomial.DegreeLT public import Mathlib.Topology.Separation.Basic @@ -46,7 +46,8 @@ theorem finrank_continuousMap_range_fin_smul {e : E} (he : e ≠ 0) (n : ℕ) : have hcard : Fintype.card K = n := (Fintype.card_congr emb.toEquivRange).symm.trans (Fintype.card_fin n) change Module.finrank ℝ C(K, ℝ) = n - rw [(ContinuousMap.linearEquivFnOfDiscrete ℝ).finrank_eq, Module.finrank_pi, hcard] + rw [(LinearEquiv.ofBijective (ContinuousMap.coeFnCLM ℝ).toLinearMap + ContinuousMap.equivFnOfDiscrete.bijective).finrank_eq, Module.finrank_pi, hcard] private theorem aeval_degreeLT_range_ne_top {X : Type*} [TopologicalSpace X] {coordinate : C(X, ℝ)} {n : ℕ} @@ -64,7 +65,6 @@ finite (hence compact) set. No finite-dimensionality assumption on the input sp theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] {σ : C(ℝ, ℝ)} (hσ : Function.IsPolynomial σ) : ∃ K : Set E, IsCompact K ∧ ¬ Dense (spaceOn σ K : Set C(K, ℝ)) := by - classical obtain ⟨p, hp⟩ := hσ obtain ⟨e, he⟩ : ∃ e : E, e ≠ 0 := exists_ne 0 let emb : Fin (p.natDegree + 2) ↪ E := @@ -73,8 +73,7 @@ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] exact_mod_cast (smul_left_injective ℝ he hij)⟩ let K : Set E := Set.range emb let coordinate : C(K, ℝ) := - ⟨fun x ↦ inner ℝ e x / inner ℝ e e, - (continuous_const.inner continuous_subtype_val).div_const _⟩ + ⟨fun x ↦ inner ℝ e x / inner ℝ e e, (continuous_const.inner continuous_subtype_val).div_const _⟩ let evalDegree : ↥(degreeLT ℝ (p.natDegree + 1)) →ₗ[ℝ] C(K, ℝ) := (Polynomial.aeval coordinate).toLinearMap.domRestrict (degreeLT ℝ (p.natDegree + 1)) have hfinrank : Module.finrank ℝ C(K, ℝ) = p.natDegree + 2 := @@ -105,9 +104,7 @@ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] have : inner ℝ e ((i : ℝ) • e) / inner ℝ e e = (i : ℝ) := by rw [real_inner_smul_right, div_eq_iff (inner_self_ne_zero.mpr he)] rw [hemb, this] - simp only [q, Polynomial.eval_comp, Polynomial.eval_add, Polynomial.eval_C_mul, - Polynomial.eval_X, Polynomial.eval_C, real_inner_smul_right] - rw [mul_comm (inner ℝ w e)] + simp [q, real_inner_smul_right, mul_comm (inner ℝ w e)] exact ⟨K, (Set.finite_range emb).isCompact, evalDegree.range.not_dense_of_subset_of_finiteDimensional hproper hspace⟩ From b0d8af1bd8103e807f24c03ef00772a9c3b4499a Mon Sep 17 00:00:00 2001 From: yuanyi-350 Date: Sat, 12 Sep 2026 14:54:12 +0800 Subject: [PATCH 4/8] run lake exe mk_all --- LeanMachineLearning.lean | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/LeanMachineLearning.lean b/LeanMachineLearning.lean index 0c04a1ca..8ab0ef02 100644 --- a/LeanMachineLearning.lean +++ b/LeanMachineLearning.lean @@ -1,4 +1,5 @@ module -- shake: keep-all --deprecated_module: ignore + public import LeanMachineLearning.ForMathlib.Algebra.Polynomial.Function public import LeanMachineLearning.ForMathlib.Analysis.Calculus.ContinuousMapComposition public import LeanMachineLearning.ForMathlib.Analysis.Distribution.Polynomial @@ -6,17 +7,16 @@ public import LeanMachineLearning.ForMathlib.Analysis.Distribution.PolynomialCha public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction public import LeanMachineLearning.ForMathlib.Analysis.Distribution.TestFunction.Normalize public import LeanMachineLearning.ForMathlib.Analysis.LocallyConvex.Annihilator -public import LeanMachineLearning.ForMathlib.LinearAlgebra.Multilinear.Polarization - public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.ChainRule public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.CompProd public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.Convex public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.DataProcessing public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.MapSequence public import LeanMachineLearning.ForMathlib.InformationTheory.KullbackLeibler.Restrict +public import LeanMachineLearning.ForMathlib.LinearAlgebra.Multilinear.Polarization +public import LeanMachineLearning.ForMathlib.MeasureTheory.Integral.ClosedSubmodule public import LeanMachineLearning.ForMathlib.MeasureTheory.Measurable public import LeanMachineLearning.ForMathlib.MeasureTheory.MeasurableSpace.Embedding -public import LeanMachineLearning.ForMathlib.MeasureTheory.Integral.ClosedSubmodule public import LeanMachineLearning.ForMathlib.MeasureTheory.Measure.AbsolutelyContinuous public import LeanMachineLearning.ForMathlib.MeasureTheory.Order.Lattice public import LeanMachineLearning.ForMathlib.MeasureTheory.Order.MeasurableArg From b7289414a0e7eaeb496f4a0b7b403e014dbf8e58 Mon Sep 17 00:00:00 2001 From: yuanyi-350 Date: Sun, 13 Sep 2026 00:17:38 +0800 Subject: [PATCH 5/8] fix ci --- .../Analysis/Distribution/Polynomial.lean | 2 +- .../PolynomialCharacterization.lean | 33 +++++-------------- .../Distribution/TestFunction/Normalize.lean | 19 ++++------- .../UniversalApproximation/Convolution.lean | 2 -- .../Discriminatory.lean | 10 ++---- .../UniversalApproximation/Nonpolynomial.lean | 8 ++--- .../PolynomialObstruction.lean | 13 +++----- 7 files changed, 25 insertions(+), 62 deletions(-) diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean index cbb2ad64..cde2dd0b 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean @@ -85,7 +85,7 @@ indefinite integral over lower rays is compactly supported. This statement is independent of differentiability and works for functions with values in any complete real normed space. -/ theorem hasCompactSupport_integral_Iic_of_integral_eq_zero - {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] [CompleteSpace F] + {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] {f : ℝ → F} (hfc : HasCompactSupport f) (hfi : Integrable f) (hzero : ∫ x, f x = 0) : HasCompactSupport (fun b ↦ ∫ x in Set.Iic b, f x) := by diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean index f8c84ccc..0a578183 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean @@ -47,38 +47,26 @@ theorem exists_polynomial_of_iteratedLineDerivOp_eq_zero ∃ p : Polynomial ℝ, T = ofFun Ω (fun x => p.eval x) volume ⊤ := by induction k generalizing T with | zero => - simp only [iteratedLineDerivOp_fin_zero] at hT refine ⟨0, ?_⟩ - rw [show (fun x : ℝ => (0 : Polynomial ℝ).eval x) = 0 by funext x; simp, - ofFun_zero] - exact hT + rwa [show (fun x : ℝ => (0 : Polynomial ℝ).eval x) = 0 by funext x; simp, ofFun_zero] | succ k ih => let DT : 𝓓'(Ω, ℝ) := ∂_{(1 : ℝ)} T have hDT : iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) DT = 0 := by - rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] at hT ⊢ - exact hT + rwa [iteratedLineDerivOp_const_eq_iter_lineDerivOp] at hT ⊢ obtain ⟨p, hp⟩ := ih DT hDT obtain ⟨q, hq⟩ := Polynomial.derivative_surjective_of_charZero p let Q : 𝓓'(Ω, ℝ) := ofFun Ω (fun x => q.eval x) volume ⊤ have hqLoc : LocallyIntegrableOn (fun x => q.eval x) Ω volume := q.continuous.locallyIntegrable.locallyIntegrableOn _ - have hDQ : lineDerivCLM (1 : ℝ) Q = - ofFun Ω (fun x => p.eval x) volume ⊤ := by - dsimp only [Q] - rw [lineDerivCLM_ofFun_eq_of_hasDerivAt - (fun x => q.hasDerivAt x) hqLoc - ((q.derivative.continuous).locallyIntegrable.locallyIntegrableOn _)] - rw [hq] have hDsub : (lineDerivCLM (1 : ℝ) (T - Q) : 𝓓'(Ω, ℝ)) = 0 := by - rw [map_sub, hDQ, ← hp] - change DT - DT = 0 + rw [map_sub, lineDerivCLM_ofFun_eq_of_hasDerivAt (fun x => q.hasDerivAt x) hqLoc + ((q.derivative.continuous).locallyIntegrable.locallyIntegrableOn _), hq, ← hp] exact sub_self _ have hconst := eq_ofFun_const_of_lineDerivCLM_eq_zero_of_hasCompactSupportPrimitive ρ hρ (T - Q) hDsub let c : ℝ := (T - Q) ρ have hcLoc : LocallyIntegrableOn (fun _ : ℝ => c) Ω volume := - (continuous_const : Continuous (fun _ : ℝ => c)).locallyIntegrable - |>.locallyIntegrableOn _ + continuous_const.locallyIntegrable.locallyIntegrableOn _ refine ⟨q + Polynomial.C c, ?_⟩ have heval : (fun x => (q + Polynomial.C c).eval x) = (fun x => q.eval x) + (fun _ : ℝ => c) := by @@ -104,8 +92,7 @@ theorem ofFun_eq_iff_eq_of_continuous ofFun (⊤ : TopologicalSpace.Opens E) g μ n ↔ f = g := by constructor · intro h - have hae : f =ᵐ[μ.restrict (Set.univ : Set E)] g := - ofFun_injective hfloc hgloc h + have hae : f =ᵐ[μ.restrict (Set.univ : Set E)] g := ofFun_injective hfloc hgloc h exact (Continuous.ae_eq_iff_eq μ hf hg).mp <| by simpa using hae · rintro rfl rfl @@ -133,11 +120,9 @@ theorem isPolynomial_of_iteratedLineDerivOp_ofFun_eq_zero vanishes, assuming the compactly supported primitive property on the real line. -/ theorem isPolynomial_iff_exists_iteratedLineDerivOp_ofFun_eq_zero [TestFunction.HasCompactSupportPrimitive (⊤ : TopologicalSpace.Opens ℝ)] - (ρ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (hρ : ∫ x, ρ x = 1) - {f : ℝ → ℝ} (hf : Continuous f) : - Function.IsPolynomial f ↔ - ∃ k : ℕ, iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) - (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) = 0 := by + (ρ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (hρ : ∫ x, ρ x = 1) {f : ℝ → ℝ} (hf : Continuous f) : + Function.IsPolynomial f ↔ ∃ k : ℕ, iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) + (ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) = 0 := by constructor · rintro ⟨p, hp⟩ refine ⟨p.natDegree + 1, ?_⟩ diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean index f09c64a8..a849a8b9 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/TestFunction/Normalize.lean @@ -31,8 +31,7 @@ variable {E : Type*} [NormedAddCommGroup E] [NormedSpace ℝ E] [HasContDiffBump [IsLocallyFiniteMeasure μ] [μ.IsOpenPosMeasure] /-- Regard a normalized smooth bump as a test function on an open set containing its support. -/ -def toTestFunctionNormed - (h : Metric.closedBall c f.rOut ⊆ Ω) : 𝓓(Ω, ℝ) where +def toTestFunctionNormed (h : Metric.closedBall c f.rOut ⊆ Ω) : 𝓓(Ω, ℝ) where toFun := f.normed μ contDiff' := f.contDiff_normed hasCompactSupport' := f.hasCompactSupport_normed @@ -40,16 +39,12 @@ def toTestFunctionNormed @[simp] theorem toTestFunctionNormed_apply - (h : Metric.closedBall c f.rOut ⊆ Ω) (x : E) : - f.toTestFunctionNormed μ h x = f.normed μ x := + (h : Metric.closedBall c f.rOut ⊆ Ω) (x : E) : f.toTestFunctionNormed μ h x = f.normed μ x := rfl /-- A normalized smooth bump, regarded as a test function, still has integral one. -/ @[simp] -theorem integral_toTestFunctionNormed - (h : Metric.closedBall c f.rOut ⊆ Ω) : - ∫ x, f.toTestFunctionNormed μ h x ∂μ = 1 := by - simpa only [toTestFunctionNormed_apply] using f.integral_normed (μ := μ) +theorem integral_toTestFunctionNormed : ∫ (x : E), f.normed μ x ∂μ = 1 := f.integral_normed end ContDiffBump @@ -67,10 +62,8 @@ theorem exists_integral_eq_one (hΩ : (Ω : Set E).Nonempty) : obtain ⟨ε, hε, hball⟩ := Metric.mem_nhds_iff.mp (Ω.isOpen.mem_nhds hc) let f : ContDiffBump c := ContDiffBump.mk (ε / 4) (ε / 2) (by positivity) (by linarith) - have hf : Metric.closedBall c f.rOut ⊆ Ω := by - refine (Metric.closedBall_subset_ball ?_).trans hball - change ε / 2 < ε - exact half_lt_self hε - exact ⟨f.toTestFunctionNormed μ hf, f.integral_toTestFunctionNormed μ hf⟩ + have hf : Metric.closedBall c f.rOut ⊆ Ω := + (Metric.closedBall_subset_ball (half_lt_self hε)).trans hball + exact ⟨f.toTestFunctionNormed μ hf, by simp⟩ end TestFunction diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean index a06251df..cc303075 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean @@ -216,8 +216,6 @@ theorem hasDerivAt_convolutionActivation_iterate_lineDerivCLM (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) (x : ℝ) : HasDerivAt (convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) σ) (convolutionActivation (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) σ x) x := by - let _ : (volume : Measure ℝ).IsNegInvariant := - Measure.IsAddHaarMeasure.isNegInvariant_of_regular volume have h := (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ).hasCompactSupport |>.hasDerivAt_convolution_left (μ := volume) (ContinuousLinearMap.mul ℝ ℝ) ((TestFunction.contDiff (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ)).of_le (by simp)) diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean index b73bd05b..bdc8d26e 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean @@ -101,12 +101,7 @@ theorem annihilates_coordinate_pow_of_iteratedDeriv_ne_zero rw [hd0] have heq : g.comp (arg t) = (neuron g (t • w) b).restrict K := by ext x - simp only [arg, ContinuousMap.comp_apply, ContinuousMap.add_apply, - ContinuousMap.const_apply, ContinuousMap.mul_apply, u, - ContinuousMap.innerProductCoordinate_apply, - neuron_apply, ContinuousMap.restrict_apply, real_inner_smul_left] - congr 1 - ring + simp [arg, u, neuron_apply, real_inner_smul_left, add_comm] rw [heq] simpa using hΛ (t • w) b | succ m ihm => @@ -144,8 +139,7 @@ smooth function may depend on the degree, as needed after mollification. -/ theorem isDiscriminatory_of_smooth_ridges {σ : C(ℝ, ℝ)} (hsmooth : ∀ n : ℕ, ∃ (g : C(ℝ, ℝ)) (b : ℝ), ContDiff ℝ ∞ g ∧ iteratedDeriv n g b ≠ 0 ∧ ∀ (K : Set E) (_hK : IsCompact K) (Λ : StrongDual ℝ C(K, ℝ)), - (∀ w c, Λ ((neuron σ w c).restrict K) = 0) → - ∀ w c, Λ ((neuron g w c).restrict K) = 0) : + (∀ w c, Λ ((neuron σ w c).restrict K) = 0) → ∀ w c, Λ ((neuron g w c).restrict K) = 0) : IsDiscriminatory E σ := by constructor intro K hK Λ hΛ diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean index b303ea87..d8ff1a0e 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean @@ -72,8 +72,7 @@ theorem iteratedLineDerivOp_apply_iterated_testFunction | zero => simp | succ n ih => rw [Function.iterate_succ_apply'] - change Distribution.lineDerivCLM (1 : ℝ) ((∂_{(1 : ℝ)})^[n] T) φ = - (-1 : ℝ) ^ (n + 1) • T (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n + 1]) φ) + change Distribution.lineDerivCLM (1 : ℝ) _ φ = _ simp [Distribution.lineDerivCLM_apply, ih, Function.iterate_succ_apply, pow_succ] /-- Every distributional derivative of a continuous nonpolynomial function is nonzero. @@ -109,8 +108,7 @@ theorem exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero let T : 𝓓'((⊤ : TopologicalSpace.Opens ℝ), ℝ) := LineDeriv.iteratedLineDerivOp (fun _ : Fin n ↦ (1 : ℝ)) (Distribution.ofFun (⊤ : TopologicalSpace.Opens ℝ) f volume ⊤) - have hf : ¬ Function.IsPolynomial f := by - simpa [f, isPolynomial_reflectedActivation_iff] using hσ + have hf : ¬ Function.IsPolynomial f := by simpa [f, isPolynomial_reflectedActivation_iff] have hT : T ≠ 0 := Distribution.iteratedLineDerivOp_ofFun_ne_zero_of_not_isPolynomial hf n obtain ⟨φ, hφ⟩ := T.exists_ne_zero hT refine ⟨φ, ?_⟩ @@ -123,7 +121,7 @@ theorem exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero simp [smul_eq_mul, f, reflectedActivation_apply] intro hzero apply hφ - exact heval.trans (by rw [hzero, mul_zero]) + exact heval.trans (by simpa) /-- For every order, a nonpolynomial continuous activation has a test-function convolution whose derivative is nonzero at the origin in that order. -/ diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean index 00d744b2..667a4bc1 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean @@ -41,8 +41,6 @@ theorem finrank_continuousMap_range_fin_smul {e : E} (he : e ≠ 0) (n : ℕ) : apply Fin.ext exact_mod_cast (smul_left_injective ℝ he hij)⟩ let K : Set E := Set.range emb - let _ : Fintype K := (Set.finite_range emb).fintype - let _ : DiscreteTopology K := Finite.instDiscreteTopology have hcard : Fintype.card K = n := (Fintype.card_congr emb.toEquivRange).symm.trans (Fintype.card_fin n) change Module.finrank ℝ C(K, ℝ) = n @@ -72,14 +70,12 @@ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] apply Fin.ext exact_mod_cast (smul_left_injective ℝ he hij)⟩ let K : Set E := Set.range emb - let coordinate : C(K, ℝ) := - ⟨fun x ↦ inner ℝ e x / inner ℝ e e, (continuous_const.inner continuous_subtype_val).div_const _⟩ - let evalDegree : ↥(degreeLT ℝ (p.natDegree + 1)) →ₗ[ℝ] C(K, ℝ) := + let coordinate : C(K, ℝ) := ⟨fun x ↦ inner ℝ e x / inner ℝ e e, by fun_prop⟩ + let evalDegree : degreeLT ℝ (p.natDegree + 1) →ₗ[ℝ] C(K, ℝ) := (Polynomial.aeval coordinate).toLinearMap.domRestrict (degreeLT ℝ (p.natDegree + 1)) have hfinrank : Module.finrank ℝ C(K, ℝ) = p.natDegree + 2 := finrank_continuousMap_range_fin_smul he _ - have hproper : evalDegree.range ≠ ⊤ := - aeval_degreeLT_range_ne_top (by simpa using hfinrank) + have hproper : evalDegree.range ≠ ⊤ := aeval_degreeLT_range_ne_top hfinrank have hspace : (spaceOn σ K : Set C(K, ℝ)) ⊆ evalDegree.range := by rw [SetLike.coe_subset_coe, spaceOn] apply Submodule.span_le.2 @@ -89,8 +85,7 @@ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] rw [degreeLT_succ_eq_degreeLE, mem_degreeLE, ← natDegree_le_iff_degree_le] calc q.natDegree ≤ p.natDegree * (C (inner ℝ w e) * X + C b).natDegree := natDegree_comp_le - _ ≤ p.natDegree * 1 := Nat.mul_le_mul_left _ <| - natDegree_add_le_of_degree_le + _ ≤ p.natDegree * 1 := Nat.mul_le_mul_left _ <| natDegree_add_le_of_degree_le (by simpa using natDegree_C_mul_X_pow_le (inner ℝ w e) 1) (by simp) _ = p.natDegree := Nat.mul_one _ refine ⟨⟨q, hq⟩, ?_⟩ From 7196327fc076287da3aad0bb25a6c3ff8cfed6a0 Mon Sep 17 00:00:00 2001 From: Yi Yuan Date: Sun, 13 Sep 2026 18:04:33 +0800 Subject: [PATCH 6/8] Refactor universal approximation statements and simplify proofs --- .../Calculus/ContinuousMapComposition.lean | 40 ++---- .../Analysis/Distribution/Polynomial.lean | 128 +++++++----------- .../PolynomialCharacterization.lean | 15 +- .../NeuralNetwork/Shallow/Basic.lean | 31 ++--- .../UniversalApproximation/Convolution.lean | 18 +-- .../Discriminatory.lean | 26 ++-- .../UniversalApproximation/Leshno.lean | 27 ++-- .../UniversalApproximation/Nonpolynomial.lean | 11 +- .../PolynomialObstruction.lean | 21 ++- 9 files changed, 124 insertions(+), 193 deletions(-) diff --git a/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean b/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean index 6378008d..ad9e65d6 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Calculus/ContinuousMapComposition.lean @@ -39,10 +39,8 @@ derivative and the proposed derivative is continuous in the uniform norm. The compactness of `X` equips `C(X, E)` with its supremum norm. The proof uses the fundamental theorem of calculus after applying each continuous evaluation map. -/ -theorem continuousMap_of_continuous - {f f' : ℝ → C(X, E)} - (hf : ∀ x t, HasDerivAt (fun s ↦ f s x) (f' t x) t) - (hf' : Continuous f') (t : ℝ) : +theorem continuousMap_of_continuous {f f' : ℝ → C(X, E)} + (hf : ∀ x t, HasDerivAt (fun s ↦ f s x) (f' t x) t) (hf' : Continuous f') (t : ℝ) : HasDerivAt f (f' t) t := by have hfi : ∀ a b, IntervalIntegrable f' volume a b := fun a b ↦ hf'.intervalIntegrable a b @@ -52,20 +50,16 @@ theorem continuousMap_of_continuous funext s apply ContinuousMap.ext intro x - have hFTC : - ∫ r in 0..s, (ContinuousMap.evalCLM ℝ x) (f' r) = f s x - f 0 x := - intervalIntegral.integral_eq_sub_of_hasDerivAt - (fun r _ ↦ hf x r) + have hFTC : ∫ r in 0..s, (ContinuousMap.evalCLM ℝ x) (f' r) = f s x - f 0 x := + intervalIntegral.integral_eq_sub_of_hasDerivAt (fun r _ ↦ hf x r) (((ContinuousMap.evalCLM ℝ x).continuous.comp hf').intervalIntegrable 0 s) change f 0 x + (ContinuousMap.evalCLM ℝ x) (∫ r in 0..s, f' r) = f s x rw [← ContinuousLinearMap.intervalIntegral_comp_comm (ContinuousMap.evalCLM ℝ x) (hfi 0 s), hFTC] simp - have hder : - HasDerivAt q (0 + f' t) t := + have hder : HasDerivAt q (0 + f' t) t := (hasDerivAt_const t (f 0)).add (hf'.integral_hasStrictDerivAt 0 t).hasDerivAt - rw [hEq] at hder - simpa using hder + simpa [hEq] using hder /-- Compose a differentiable Banach-valued function with the family of affine arguments `x ↦ u x + t * v x`. Differentiation in `t` may be performed in the uniform norm on @@ -73,27 +67,15 @@ theorem continuousMap_of_continuous Bundling `g` and `dg` as continuous maps records exactly the continuity needed to upgrade the pointwise derivatives `hg` to a derivative in the function space. -/ -theorem continuousMap_comp_affine - {g dg : C(ℝ, E)} - (hg : ∀ y, HasDerivAt g (dg y) y) - (u v : C(X, ℝ)) (t : ℝ) : - HasDerivAt - (fun s ↦ g.comp (u + ContinuousMap.const X s * v)) - ⟨fun x ↦ v x • dg (u x + t * v x), - v.continuous.smul - (dg.continuous.comp - (u.continuous.add (continuous_const.mul v.continuous)))⟩ - t := by +theorem continuousMap_comp_affine {g dg : C(ℝ, E)} (hg : ∀ y, HasDerivAt g (dg y) y) + (u v : C(X, ℝ)) (t : ℝ) : HasDerivAt (fun s ↦ g.comp (u + ContinuousMap.const X s * v)) + ⟨fun x ↦ v x • dg (u x + t * v x), by fun_prop⟩ t := by apply continuousMap_of_continuous (t := t) · intro x s - convert - (hg (u x + s * v x)).scomp s + convert (hg (u x + s * v x)).scomp s ((hasDerivAt_const s (u x)).add (hasDerivAt_mul_const (v x))) using 1 <;> simp [Function.comp_def] · apply ContinuousMap.continuous_of_continuous_uncurry - exact (v.continuous.comp continuous_snd).smul <| - dg.continuous.comp <| - (u.continuous.comp continuous_snd).add <| - continuous_fst.mul (v.continuous.comp continuous_snd) + exact (v.continuous.comp continuous_snd).smul <| by fun_prop end HasDerivAt diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean index cde2dd0b..154ee143 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/Polynomial.lean @@ -74,8 +74,7 @@ theorem hasDerivAt_integral_Iic fun y ↦ (∫ x in b..y, f x) + ∫ x in Set.Iic b, f x := by funext y exact sub_eq_iff_eq_add.mp <| intervalIntegral.integral_Iic_sub_Iic - (hfi.integrableOn : IntegrableOn f (Set.Iic b)) - (hfi.integrableOn : IntegrableOn f (Set.Iic y)) + hfi.integrableOn hfi.integrableOn rw [hEq] exact (hf.integral_hasStrictDerivAt b b).hasDerivAt.add_const _ @@ -112,8 +111,7 @@ theorem hasCompactSupport_integral_Iic_of_integral_eq_zero have hsplit := intervalIntegral.integral_Iic_add_Ioi (hfi.integrableOn : IntegrableOn f (Set.Iic x)) (hfi.integrableOn : IntegrableOn f (Set.Ioi x)) - rw [hIoi, hzero, add_zero] at hsplit - exact hsplit + rwa [hIoi, hzero, add_zero] at hsplit /-- Every smooth compactly supported function on the real line with integral zero has a smooth compactly supported primitive. @@ -123,24 +121,19 @@ core of exactness of differentiation and integration on real test functions. -/ theorem exists_contDiff_primitive_hasCompactSupport {F : Type*} [NormedAddCommGroup F] [NormedSpace ℝ F] [CompleteSpace F] {f : ℝ → F} (hf : ContDiff ℝ (↑(⊤ : ℕ∞)) f) (hfc : HasCompactSupport f) - (hzero : ∫ x, f x = 0) : - ∃ g : ℝ → F, ContDiff ℝ (↑(⊤ : ℕ∞)) g ∧ HasCompactSupport g ∧ + (hzero : ∫ x, f x = 0) : ∃ g : ℝ → F, ContDiff ℝ (↑(⊤ : ℕ∞)) g ∧ HasCompactSupport g ∧ ∀ x, HasDerivAt g (f x) x := by have hfcont : Continuous f := hf.continuous have hfi : Integrable f := hfcont.integrable_of_hasCompactSupport hfc let g : ℝ → F := fun b ↦ ∫ x in Set.Iic b, f x - have hgDeriv (x : ℝ) : HasDerivAt g (f x) x := - hasDerivAt_integral_Iic hfcont hfi x - have hgDiff : Differentiable ℝ g := fun x ↦ (hgDeriv x).differentiableAt + have hgDeriv (x : ℝ) : HasDerivAt g (f x) x := hasDerivAt_integral_Iic hfcont hfi x have hgderiv : deriv g = f := by funext x exact (hgDeriv x).deriv have hgSmooth : ContDiff ℝ (↑(⊤ : ℕ∞)) g := by rw [contDiff_infty_iff_deriv] - exact ⟨hgDiff, hgderiv ▸ hf⟩ - have hgCompact : HasCompactSupport g := - hasCompactSupport_integral_Iic_of_integral_eq_zero hfc hfi hzero - exact ⟨g, hgSmooth, hgCompact, hgDeriv⟩ + exact ⟨fun x ↦ (hgDeriv x).differentiableAt, hgderiv ▸ hf⟩ + exact ⟨g, hgSmooth, hasCompactSupport_integral_Iic_of_integral_eq_zero hfc hfi hzero, hgDeriv⟩ end MeasureTheory @@ -177,7 +170,7 @@ lemma lineDerivCLM_one_apply {Ω : TopologicalSpace.Opens ℝ} (φ : 𝓓(Ω, (lineDerivCLM ℝ (1 : ℝ) φ : 𝓓(Ω, ℝ)) x = deriv φ x := by rw [lineDerivCLM_apply_of_le] · calc - lineDeriv ℝ (φ : ℝ → ℝ) x 1 = (fderiv ℝ (φ : ℝ → ℝ) x) 1 := + _ = (fderiv ℝ (φ : ℝ → ℝ) x) 1 := (φ.contDiff.differentiable (by simp)).differentiableAt.lineDeriv_eq_fderiv _ = deriv (φ : ℝ → ℝ) x := fderiv_apply_one_eq_deriv · simp @@ -199,8 +192,7 @@ instance instHasCompactSupportPrimitiveTop : obtain ⟨g, hgSmooth, hgCompact, hgDeriv⟩ := MeasureTheory.exists_contDiff_primitive_hasCompactSupport φ.contDiff φ.hasCompactSupport hφ - let ψ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ) := - ⟨g, hgSmooth, hgCompact, by simp⟩ + let ψ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ) := ⟨g, hgSmooth, hgCompact, by simp⟩ refine ⟨ψ, ?_⟩ ext x rw [lineDerivCLM_one_apply] @@ -230,36 +222,30 @@ theorem lineDerivCLM_ofFun_eq_of_hasDerivAt {Ω : TopologicalSpace.Opens ℝ} have hdφ (x : ℝ) : dφ x = deriv (φ : ℝ → ℝ) x := TestFunction.lineDerivCLM_one_apply φ x have hibp := MeasureTheory.integral_bilinear_hasDerivAt_right_eq_neg_left_of_integrable - (L := ContinuousLinearMap.lsmul ℝ ℝ) (u := (φ : ℝ → ℝ)) (v := f) - (u' := fun x => deriv (φ : ℝ → ℝ) x) (v' := f') - (fun _ _ => (φ.contDiff.differentiable (by simp)).differentiableAt.hasDerivAt) - (fun x _ => hf x) - (φ.integrable_smul hf'loc) - (by simpa only [hdφ, ContinuousLinearMap.lsmul_apply] using dφ.integrable_smul hfloc) - (φ.integrable_smul hfloc) + (L := ContinuousLinearMap.lsmul ℝ ℝ) + (fun _ _ ↦ (φ.contDiff.differentiable (by simp)).differentiableAt.hasDerivAt) + (fun x _ ↦ hf x) (φ.integrable_smul hf'loc) + (by simpa [hdφ] using dφ.integrable_smul hfloc) (φ.integrable_smul hfloc) simp only [ContinuousLinearMap.lsmul_apply] at hibp - rw [show (TestFunction.lineDerivCLM ℝ (1 : ℝ) φ : 𝓓(Ω, ℝ)) = dφ from rfl] - simp_rw [hdφ] + rw [DFunLike.congr_fun rfl φ] exact hibp.symm /-- The distributional derivative of a regular constant distribution vanishes. -/ theorem lineDerivCLM_ofFun_const_eq_zero {Ω : TopologicalSpace.Opens ℝ} (c : F) : - (lineDerivCLM (1 : ℝ) (ofFun Ω (fun _ : ℝ => c) volume ⊤) : 𝓓'(Ω, F)) = 0 := by - have hc : LocallyIntegrableOn (fun _ : ℝ => c) Ω volume := - (continuous_const : Continuous (fun _ : ℝ => c)).locallyIntegrable.locallyIntegrableOn Ω + (lineDerivCLM (1 : ℝ) (ofFun Ω (fun _ : ℝ ↦ c) volume ⊤) : 𝓓'(Ω, F)) = 0 := by + have hc : LocallyIntegrableOn (fun _ : ℝ ↦ c) Ω volume := + (continuous_const : Continuous (fun _ : ℝ ↦ c)).locallyIntegrable.locallyIntegrableOn Ω rw [lineDerivCLM_ofFun_eq_of_hasDerivAt - (fun x => hasDerivAt_const x c) hc locallyIntegrableOn_zero] + (fun x ↦ hasDerivAt_const x c) hc locallyIntegrableOn_zero] exact ofFun_zero /-- Integration annihilates derivatives of test functions. This is the easy half of the derivative--integral exactness statement. -/ -theorem ofFun_one_comp_testFunction_lineDerivCLM_eq_zero - {Ω : TopologicalSpace.Opens ℝ} : - (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤) ∘ - (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) = 0 := by +theorem ofFun_one_comp_testFunction_lineDerivCLM_eq_zero {Ω : TopologicalSpace.Opens ℝ} : + (ofFun Ω (fun _ : ℝ ↦ (1 : ℝ)) volume ⊤) ∘ + (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) = 0 := by funext φ - have h := congrArg (fun T : 𝓓'(Ω, ℝ) => T φ) - (lineDerivCLM_ofFun_const_eq_zero (Ω := Ω) (1 : ℝ)) + have h := congrArg (fun T : 𝓓'(Ω, ℝ) ↦ T φ) (lineDerivCLM_ofFun_const_eq_zero (Ω := Ω) (1 : ℝ)) rw [lineDerivCLM_apply] at h exact neg_eq_zero.mp h @@ -267,18 +253,14 @@ theorem ofFun_one_comp_testFunction_lineDerivCLM_eq_zero integration on test functions. -/ theorem exact_testFunction_lineDerivCLM_of_hasCompactSupportPrimitive {Ω : TopologicalSpace.Opens ℝ} [TestFunction.HasCompactSupportPrimitive Ω] : - Function.Exact - (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) - (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤) := by - apply Function.Exact.of_comp_of_mem_range - ofFun_one_comp_testFunction_lineDerivCLM_eq_zero + Function.Exact (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) + (ofFun Ω (fun _ : ℝ ↦ (1 : ℝ)) volume ⊤) := by + apply Function.Exact.of_comp_of_mem_range ofFun_one_comp_testFunction_lineDerivCLM_eq_zero intro φ hφ - have hOneLoc : LocallyIntegrableOn (fun _ : ℝ => (1 : ℝ)) Ω volume := - (continuous_const : Continuous (fun _ : ℝ => (1 : ℝ))).locallyIntegrable + have hOneLoc : LocallyIntegrableOn (fun _ : ℝ ↦ (1 : ℝ)) Ω volume := + (continuous_const : Continuous (fun _ : ℝ ↦ (1 : ℝ))).locallyIntegrable |>.locallyIntegrableOn Ω - have hIntegral : ∫ x, φ x = 0 := by - rw [ofFun_apply hOneLoc] at hφ - simpa using hφ + have hIntegral : ∫ x, φ x = 0 := by simpa [ofFun_apply hOneLoc] using hφ exact TestFunction.HasCompactSupportPrimitive.exists_eq_lineDerivCLM φ hIntegral /-- A distribution with zero derivative is a regular constant distribution, provided the @@ -288,30 +270,25 @@ The exactness hypothesis precisely isolates the missing analytic input: every te integral zero must have a compactly supported smooth primitive. -/ theorem eq_ofFun_const_of_lineDerivCLM_eq_zero [CompleteSpace F] {Ω : TopologicalSpace.Opens ℝ} (ρ : 𝓓(Ω, ℝ)) - (hρ : ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤ ρ = 1) - (hExact : Function.Exact - (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) - (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤)) - (T : 𝓓'(Ω, F)) (hT : (lineDerivCLM (1 : ℝ) T : 𝓓'(Ω, F)) = 0) : - T = ofFun Ω (fun _ => T ρ) volume ⊤ := by - have hOneLoc : LocallyIntegrableOn (fun _ : ℝ => (1 : ℝ)) Ω volume := - (continuous_const : Continuous (fun _ : ℝ => (1 : ℝ))).locallyIntegrable - |>.locallyIntegrableOn Ω - have hConstLoc : LocallyIntegrableOn (fun _ : ℝ => T ρ) Ω volume := - (continuous_const : Continuous (fun _ : ℝ => T ρ)).locallyIntegrable - |>.locallyIntegrableOn Ω - have hTD (φ : 𝓓(Ω, ℝ)) : - T (TestFunction.lineDerivCLM ℝ (1 : ℝ) φ) = 0 := by - have h := congrArg (fun S : 𝓓'(Ω, F) => S φ) hT + (hρ : ofFun Ω (fun _ : ℝ ↦ (1 : ℝ)) volume ⊤ ρ = 1) + (hExact : Function.Exact (TestFunction.lineDerivCLM ℝ (1 : ℝ) : 𝓓(Ω, ℝ) → 𝓓(Ω, ℝ)) + (ofFun Ω (fun _ : ℝ ↦ (1 : ℝ)) volume ⊤)) (T : 𝓓'(Ω, F)) + (hT : (lineDerivCLM (1 : ℝ) T : 𝓓'(Ω, F)) = 0) : T = ofFun Ω (fun _ ↦ T ρ) volume ⊤ := by + have hOneLoc : LocallyIntegrableOn (fun _ : ℝ ↦ (1 : ℝ)) Ω volume := + (continuous_const : Continuous (fun _ : ℝ ↦ (1 : ℝ))).locallyIntegrable |>.locallyIntegrableOn Ω + have hConstLoc : LocallyIntegrableOn (fun _ : ℝ ↦ T ρ) Ω volume := + (continuous_const : Continuous (fun _ : ℝ ↦ T ρ)).locallyIntegrable |>.locallyIntegrableOn Ω + have hTD (φ : 𝓓(Ω, ℝ)) : T (TestFunction.lineDerivCLM ℝ (1 : ℝ) φ) = 0 := by + have h := congrArg (fun S : 𝓓'(Ω, F) ↦ S φ) hT rw [lineDerivCLM_apply] at h exact neg_eq_zero.mp h ext φ rw [ofFun_apply hConstLoc] calc - T φ = (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤) φ • T ρ := + T φ = (ofFun Ω (fun _ : ℝ ↦ (1 : ℝ)) volume ⊤) φ • T ρ := LinearMap.apply_eq_smul_apply_of_exact (TestFunction.lineDerivCLM ℝ (1 : ℝ)).toLinearMap - (ofFun Ω (fun _ : ℝ => (1 : ℝ)) volume ⊤).toLinearMap + (ofFun Ω (fun _ : ℝ ↦ (1 : ℝ)) volume ⊤).toLinearMap T.toLinearMap hExact hρ hTD φ _ = (∫ x, φ x) • T ρ := by rw [ofFun_apply hOneLoc] @@ -325,10 +302,9 @@ theorem eq_ofFun_const_of_lineDerivCLM_eq_zero_of_hasCompactSupportPrimitive [Co {Ω : TopologicalSpace.Opens ℝ} [TestFunction.HasCompactSupportPrimitive Ω] (ρ : 𝓓(Ω, ℝ)) (hρ : ∫ x, ρ x = 1) (T : 𝓓'(Ω, F)) (hT : (lineDerivCLM (1 : ℝ) T : 𝓓'(Ω, F)) = 0) : - T = ofFun Ω (fun _ => T ρ) volume ⊤ := by - have hOneLoc : LocallyIntegrableOn (fun _ : ℝ => (1 : ℝ)) Ω volume := - (continuous_const : Continuous (fun _ : ℝ => (1 : ℝ))).locallyIntegrable - |>.locallyIntegrableOn Ω + T = ofFun Ω (fun _ ↦ T ρ) volume ⊤ := by + have hOneLoc : LocallyIntegrableOn (fun _ : ℝ ↦ (1 : ℝ)) Ω volume := + (continuous_const : Continuous (fun _ : ℝ ↦ (1 : ℝ))).locallyIntegrable |>.locallyIntegrableOn Ω apply eq_ofFun_const_of_lineDerivCLM_eq_zero ρ · rw [ofFun_apply hOneLoc] simpa using hρ @@ -342,7 +318,7 @@ derivatives are supplied together with proofs of their values. -/ theorem iteratedLineDerivOp_ofFun_eq_of_hasDerivAt {Ω : TopologicalSpace.Opens ℝ} (f : ℕ → ℝ → F) (hf : ∀ n x, HasDerivAt (f n) (f (n + 1) x) x) (hfloc : ∀ n, LocallyIntegrableOn (f n) Ω volume) (k : ℕ) : - iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) (ofFun Ω (f 0) volume ⊤) = + iteratedLineDerivOp (fun _ : Fin k ↦ (1 : ℝ)) (ofFun Ω (f 0) volume ⊤) = ofFun Ω (f k) volume ⊤ := by rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] induction k with @@ -355,13 +331,11 @@ theorem iteratedLineDerivOp_ofFun_eq_of_hasDerivAt {Ω : TopologicalSpace.Opens agrees with iterated formal differentiation of that polynomial. -/ theorem iteratedLineDerivOp_ofFun_polynomial {Ω : TopologicalSpace.Opens ℝ} (p : Polynomial ℝ) (k : ℕ) : - iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) - (ofFun Ω (fun x => p.eval x) volume ⊤) = - ofFun Ω - (fun x => (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[k]) p).eval x) + iteratedLineDerivOp (fun _ : Fin k ↦ (1 : ℝ)) (ofFun Ω (fun x ↦ p.eval x) volume ⊤) = + ofFun Ω (fun x ↦ (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[k]) p).eval x) volume ⊤ := by apply iteratedLineDerivOp_ofFun_eq_of_hasDerivAt - (f := fun n x => (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[n]) p).eval x) + (fun n x ↦ ((Polynomial.derivative^[n]) p).eval x) · intro n x simpa only [Function.iterate_succ_apply'] using (((Polynomial.derivative : Polynomial ℝ → Polynomial ℝ)^[n]) p).hasDerivAt x @@ -373,10 +347,10 @@ theorem iteratedLineDerivOp_ofFun_polynomial {Ω : TopologicalSpace.Opens ℝ} polynomial distribution vanishes. -/ theorem iteratedLineDerivOp_ofFun_polynomial_eq_zero_of_natDegree_lt {Ω : TopologicalSpace.Opens ℝ} (p : Polynomial ℝ) (k : ℕ) (hpk : p.natDegree < k) : - iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) - (ofFun Ω (fun x => p.eval x) volume ⊤) = 0 := by + iteratedLineDerivOp (fun _ : Fin k ↦ (1 : ℝ)) + (ofFun Ω (fun x ↦ p.eval x) volume ⊤) = 0 := by rw [iteratedLineDerivOp_ofFun_polynomial, Polynomial.iterate_derivative_eq_zero hpk] - have hz : (fun x : ℝ => Polynomial.eval x (0 : Polynomial ℝ)) = 0 := by + have hz : (fun x : ℝ ↦ Polynomial.eval x (0 : Polynomial ℝ)) = 0 := by funext x simp rw [hz] @@ -386,8 +360,8 @@ theorem iteratedLineDerivOp_ofFun_polynomial_eq_zero_of_natDegree_lt vanishes. -/ theorem iteratedLineDerivOp_ofFun_polynomial_natDegree_add_one_eq_zero {Ω : TopologicalSpace.Opens ℝ} (p : Polynomial ℝ) : - iteratedLineDerivOp (fun _ : Fin (p.natDegree + 1) => (1 : ℝ)) - (ofFun Ω (fun x => p.eval x) volume ⊤) = 0 := + iteratedLineDerivOp (fun _ : Fin (p.natDegree + 1) ↦ (1 : ℝ)) + (ofFun Ω (fun x ↦ p.eval x) volume ⊤) = 0 := iteratedLineDerivOp_ofFun_polynomial_eq_zero_of_natDegree_lt p _ (Nat.lt_succ_self _) end Distribution diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean index 0a578183..da257f9d 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean @@ -62,18 +62,15 @@ theorem exists_polynomial_of_iteratedLineDerivOp_eq_zero rw [map_sub, lineDerivCLM_ofFun_eq_of_hasDerivAt (fun x => q.hasDerivAt x) hqLoc ((q.derivative.continuous).locallyIntegrable.locallyIntegrableOn _), hq, ← hp] exact sub_self _ - have hconst := eq_ofFun_const_of_lineDerivCLM_eq_zero_of_hasCompactSupportPrimitive - ρ hρ (T - Q) hDsub - let c : ℝ := (T - Q) ρ - have hcLoc : LocallyIntegrableOn (fun _ : ℝ => c) Ω volume := + have hcLoc : LocallyIntegrableOn (fun _ : ℝ => (T - Q) ρ) Ω volume := continuous_const.locallyIntegrable.locallyIntegrableOn _ - refine ⟨q + Polynomial.C c, ?_⟩ - have heval : (fun x => (q + Polynomial.C c).eval x) = - (fun x => q.eval x) + (fun _ : ℝ => c) := by + refine ⟨q + Polynomial.C ((T - Q) ρ), ?_⟩ + have heval : (fun x => (q + Polynomial.C ((T - Q) ρ)).eval x) = + (fun x => q.eval x) + (fun _ : ℝ => (T - Q) ρ) := by funext x simp - rw [heval, ofFun_add hqLoc hcLoc] - exact sub_eq_iff_eq_add'.mp hconst + rw [heval, ofFun_add hqLoc hcLoc, ← sub_eq_iff_eq_add'] + exact eq_ofFun_const_of_lineDerivCLM_eq_zero_of_hasCompactSupportPrimitive ρ hρ (T - Q) hDsub /-- Two locally integrable continuous functions on a finite-dimensional real normed space induce the same regular distribution if and only if they are equal. diff --git a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean index dc60c9ae..dcd27219 100644 --- a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean +++ b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean @@ -92,35 +92,20 @@ theorem spaceOn_eq_map (σ : C(ℝ, ℝ)) (K : Set E) : ext f aesop -variable (E) in -/-- An activation is universal on `E` if its shallow networks are dense on every compact -subset. This class packages the property for downstream approximation theorems. -/ -class IsUniversal (σ : C(ℝ, ℝ)) : Prop where - dense_on_compact : ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) - -/-- The typeclass formulation of universality unfolds to density on every compact subset. -/ -theorem isUniversal_iff (σ : C(ℝ, ℝ)) : - IsUniversal E σ ↔ ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := by - grind [IsUniversal] - -/-- The usual uniform epsilon formulation of universal approximation on every compact set. -/ -theorem isUniversal_iff_uniform_approximation (σ : C(ℝ, ℝ)) : - IsUniversal E σ ↔ - ∀ (K : Set E), IsCompact K → ∀ (f : C(K, ℝ)) (ε : ℝ), 0 < ε → +/-- Density of the network space on a compact set is equivalent to uniform approximation. -/ +theorem dense_spaceOn_iff_uniform_approximation (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) : + Dense (spaceOn σ K : Set C(K, ℝ)) ↔ + ∀ (f : C(K, ℝ)) (ε : ℝ), 0 < ε → ∃ g ∈ spaceOn σ K, ∀ x, dist (g x) (f x) < ε := by + let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK constructor - · rintro ⟨h⟩ K hK - let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - intro f ε hε - obtain ⟨g, hgBall, hgSpace⟩ := Metric.dense_iff.mp (h K hK) f ε hε + · intro h f ε hε + obtain ⟨g, hgBall, hgSpace⟩ := Metric.dense_iff.mp h f ε hε exact ⟨g, hgSpace, (ContinuousMap.dist_lt_iff hε).mp hgBall⟩ · intro h - constructor - intro K hK - let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK rw [Metric.dense_iff] intro f ε hε - obtain ⟨g, hgSpace, hgDist⟩ := h K hK f ε hε + obtain ⟨g, hgSpace, hgDist⟩ := h f ε hε exact ⟨g, (ContinuousMap.dist_lt_iff hε).mpr hgDist, hgSpace⟩ end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean index cc303075..47c4cc58 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean @@ -194,19 +194,15 @@ theorem isDiscriminatory_of_convolutionActivation (σ := convolutionActivation φ σ) (E := E) K hK Λ exact annihilates_convolutionActivation_neurons φ hK hΛ -/-- Universality of one compactly supported convolution smoothing implies universality of the -original activation. -/ -theorem isUniversal_of_convolutionActivation +/-- If the network space of one compactly supported convolution smoothing is dense on a compact +set, then the network space of the original activation is dense there. -/ +theorem dense_spaceOn_of_convolutionActivation {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) - [IsUniversal E (convolutionActivation φ σ)] : - IsUniversal E σ := by - apply (isUniversal_iff_isDiscriminatory σ).mpr - let _ : IsDiscriminatory E (convolutionActivation φ σ) := - (isUniversal_iff_isDiscriminatory (convolutionActivation φ σ)).mp - (inferInstance : IsUniversal E (convolutionActivation φ σ)) - exact isDiscriminatory_of_convolutionActivation φ σ + (φ : B) (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) + (h : Dense (spaceOn (convolutionActivation φ σ) K : Set C(K, ℝ))) : + Dense (spaceOn σ K : Set C(K, ℝ)) := + (Dense.mono (convolved_spaceOn_le_topologicalClosure φ σ hK) h).of_closure /-! ## Iterated derivatives for smooth kernels -/ diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean index bdc8d26e..17e06ce7 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean @@ -43,12 +43,13 @@ class IsDiscriminatory (E : Type*) [SeminormedAddCommGroup E] [InnerProductSpace variable {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] -/-- For shallow networks, the discriminatory-functional criterion is equivalent to universal -approximation. -/ -theorem isUniversal_iff_isDiscriminatory (σ : C(ℝ, ℝ)) : - IsUniversal E σ ↔ IsDiscriminatory E σ := by +/-- Density of the network space on every compact set is equivalent to the +discriminatory-functional criterion. -/ +theorem dense_spaceOn_iff_isDiscriminatory (σ : C(ℝ, ℝ)) : + (∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ))) ↔ + IsDiscriminatory E σ := by constructor - · rintro ⟨h_dense⟩ + · intro h_dense constructor intro K hK let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK @@ -62,7 +63,6 @@ theorem isUniversal_iff_isDiscriminatory (σ : C(ℝ, ℝ)) : exact hΛ p.1 p.2 exact hle hf · rintro ⟨h_disc⟩ - constructor intro K hK let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK rw [Submodule.dense_iff_forall_dual_eq_zero] @@ -159,14 +159,14 @@ theorem isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero obtain ⟨b, hb⟩ := hne n exact ⟨g, b, hg, hb, by simp⟩ -/-- A smooth activation with no identically-zero derivative has the universal approximation -property on every real inner-product space. -/ -theorem isUniversal_of_contDiff_of_iteratedDeriv_ne_zero +/-- A smooth activation with no identically-zero derivative has dense network space on every +compact subset of a real inner-product space. -/ +theorem dense_spaceOn_of_contDiff_of_iteratedDeriv_ne_zero {g : C(ℝ, ℝ)} (hg : ContDiff ℝ ∞ g) - (hne : ∀ n : ℕ, ∃ b : ℝ, iteratedDeriv n g b ≠ 0) : - IsUniversal E g := - (isUniversal_iff_isDiscriminatory g).2 - (isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero hg hne) + (hne : ∀ n : ℕ, ∃ b : ℝ, iteratedDeriv n g b ≠ 0) {K : Set E} (hK : IsCompact K) : + Dense (spaceOn g K : Set C(K, ℝ)) := + (dense_spaceOn_iff_isDiscriminatory g).2 + (isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero hg hne) K hK end SmoothActivation diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean index 7af93a15..ee9cbbe1 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Leshno.lean @@ -16,9 +16,9 @@ This file combines the sufficient direction from `Nonpolynomial` with the necess from `PolynomialObstruction` to characterize continuous universal activations. The input space is required to be nontrivial: in dimension zero, every shallow-network function -is constant, and a polynomial activation can still be universal. The results include the -equivalence on arbitrary nontrivial real inner-product spaces, its compact-set formulation, -and the classical Euclidean-space theorem. +is constant, and a polynomial activation can still be universal. We characterize density on +every compact subset of an arbitrary nontrivial real inner-product space, express it as +`(spaceOn σ K).topologicalClosure = ⊤`, and specialize to the classical Euclidean-space theorem. -/ @[expose] public section @@ -27,13 +27,6 @@ namespace Learning.ShallowNetwork variable {E : Type*} [NormedAddCommGroup E] [InnerProductSpace ℝ E] -/-- Abstract form of the Leshno--Lin--Pinkus--Schocken equivalence on an arbitrary nontrivial real -inner-product space. -/ -theorem not_isPolynomial_iff_isUniversal [Nontrivial E] (σ : C(ℝ, ℝ)) : - ¬ Function.IsPolynomial σ ↔ IsUniversal E σ := - ⟨isUniversal_of_not_isPolynomial, fun hUniversal hPolynomial ↦ - not_isUniversal_of_isPolynomial hPolynomial hUniversal⟩ - /-- Precise compact-set form of the Leshno--Lin--Pinkus--Schocken equivalence. The approximating subspace is `spaceOn σ K`, whose generators are exactly the restrictions to @@ -41,7 +34,15 @@ The approximating subspace is `spaceOn σ K`, whose generators are exactly the r -/ theorem not_isPolynomial_iff_dense_on_compact [Nontrivial E] (σ : C(ℝ, ℝ)) : ¬ Function.IsPolynomial σ ↔ ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ)) := - (not_isPolynomial_iff_isUniversal σ).trans (isUniversal_iff σ) + ⟨fun hσ _ hK ↦ dense_spaceOn_of_not_isPolynomial hσ hK, not_isPolynomial_of_dense_spaceOn σ⟩ + +/-- A continuous activation is nonpolynomial if and only if its network space has full +topological closure on every compact subset of a nontrivial real inner-product space. -/ +theorem not_isPolynomial_iff_spaceOn_topologicalClosure_eq_top [Nontrivial E] (σ : C(ℝ, ℝ)) : + ¬ Function.IsPolynomial σ ↔ + ∀ (K : Set E), IsCompact K → (spaceOn σ K).topologicalClosure = ⊤ := by + simpa only [Submodule.dense_iff_topologicalClosure_eq_top] using + (not_isPolynomial_iff_dense_on_compact (E := E) σ) /-- The classical theorem on `ℝ^d`, represented as `EuclideanSpace ℝ (Fin d)`. @@ -51,8 +52,8 @@ input space. theorem leshno_lin_pinkus_schocken {d : ℕ} (hd : 0 < d) (σ : C(ℝ, ℝ)) : ¬ Function.IsPolynomial σ ↔ ∀ (K : Set (EuclideanSpace ℝ (Fin d))), IsCompact K → - Dense (spaceOn σ K : Set C(K, ℝ)) := by + (spaceOn σ K).topologicalClosure = ⊤ := by let _ : Nonempty (Fin d) := Fin.pos_iff_nonempty.mp hd - exact not_isPolynomial_iff_dense_on_compact σ + exact not_isPolynomial_iff_spaceOn_topologicalClosure_eq_top σ end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean index d8ff1a0e..8c0c0816 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean @@ -147,10 +147,11 @@ theorem isDiscriminatory_of_not_isPolynomial {σ : C(ℝ, ℝ)} intro K hK Λ hΛ exact annihilates_convolutionActivation_neurons φ hK hΛ -/-- Every continuous nonpolynomial activation is universal on compact subsets of a real -inner-product space. -/ -theorem isUniversal_of_not_isPolynomial - {σ : C(ℝ, ℝ)} (hσ : ¬ Function.IsPolynomial σ) : IsUniversal E σ := - (isUniversal_iff_isDiscriminatory σ).2 (isDiscriminatory_of_not_isPolynomial hσ) +/-- Every continuous nonpolynomial activation has dense network space on every compact subset +of a real inner-product space. -/ +theorem dense_spaceOn_of_not_isPolynomial + {σ : C(ℝ, ℝ)} (hσ : ¬ Function.IsPolynomial σ) {K : Set E} (hK : IsCompact K) : + Dense (spaceOn σ K : Set C(K, ℝ)) := + (dense_spaceOn_iff_isDiscriminatory σ).2 (isDiscriminatory_of_not_isPolynomial hσ) K hK end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean index 667a4bc1..37e822a9 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/PolynomialObstruction.lean @@ -103,18 +103,13 @@ theorem exists_compact_not_dense_of_isPolynomial [Nontrivial E] exact ⟨K, (Set.finite_range emb).isCompact, evalDegree.range.not_dense_of_subset_of_finiteDimensional hproper hspace⟩ -/-- A polynomial activation is not universal on a nontrivial real inner product space. -/ -theorem not_isUniversal_of_isPolynomial [Nontrivial E] - {σ : C(ℝ, ℝ)} (hσ : Function.IsPolynomial σ) : - ¬ IsUniversal E σ := by - intro hUniversal - obtain ⟨K, hK, hnotDense⟩ := exists_compact_not_dense_of_isPolynomial (E := E) hσ - exact hnotDense (hUniversal.dense_on_compact K hK) - -/-- Universality forces the activation not to be a polynomial. -/ -theorem not_isPolynomial_of_isUniversal [Nontrivial E] - (σ : C(ℝ, ℝ)) [hσ : IsUniversal E σ] : - ¬ Function.IsPolynomial σ := - fun hPolynomial ↦ not_isUniversal_of_isPolynomial (E := E) hPolynomial hσ +/-- Density of the network space on every compact set forces the activation not to be a +polynomial, provided the input space is nontrivial. -/ +theorem not_isPolynomial_of_dense_spaceOn [Nontrivial E] (σ : C(ℝ, ℝ)) + (hσ : ∀ (K : Set E), IsCompact K → Dense (spaceOn σ K : Set C(K, ℝ))) : + ¬ Function.IsPolynomial σ := by + intro hPolynomial + obtain ⟨K, hK, hnotDense⟩ := exists_compact_not_dense_of_isPolynomial (E := E) hPolynomial + exact hnotDense (hσ K hK) end Learning.ShallowNetwork From 618825d27cf6bca9ca2a4458d9dcea73b9494b14 Mon Sep 17 00:00:00 2001 From: yuanyi-350 Date: Mon, 14 Sep 2026 19:34:39 +0800 Subject: [PATCH 7/8] remove useless lemmas --- .../NeuralNetwork/Shallow/Basic.lean | 72 ------------------ .../UniversalApproximation/Convolution.lean | 74 ------------------- .../Discriminatory.lean | 19 ----- .../UniversalApproximation/Nonpolynomial.lean | 2 +- 4 files changed, 1 insertion(+), 166 deletions(-) diff --git a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean index dcd27219..fc6aba77 100644 --- a/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean +++ b/LeanMachineLearning/NeuralNetwork/Shallow/Basic.lean @@ -32,80 +32,8 @@ def neuron (σ : C(ℝ, ℝ)) (w : E) (b : ℝ) : C(E, ℝ) := theorem neuron_apply (σ : C(ℝ, ℝ)) (w : E) (b : ℝ) (x : E) : neuron σ w b x = σ (inner ℝ w x + b) := rfl -/-- The real vector space of finite-width, single-hidden-layer networks with activation `σ`. -/ -def space (σ : C(ℝ, ℝ)) : Submodule ℝ C(E, ℝ) := - Submodule.span ℝ (Set.range fun p : E × ℝ ↦ neuron σ p.1 p.2) - /-- The restrictions to `K` of finite-width, single-hidden-layer networks with activation `σ`. -/ def spaceOn (σ : C(ℝ, ℝ)) (K : Set E) : Submodule ℝ C(K, ℝ) := Submodule.span ℝ (Set.range fun p : E × ℝ ↦ (neuron σ p.1 p.2).restrict K) -/-- Membership in `space` is exactly representability by a finite-width shallow network. -/ -theorem mem_space_iff (σ : C(ℝ, ℝ)) (f : C(E, ℝ)) : - f ∈ space σ ↔ - ∃ (m : ℕ) (a : Fin m → ℝ) (w : Fin m → E) (b : Fin m → ℝ), - ∑ j, a j • neuron σ (w j) (b j) = f := by - constructor - · intro hf - rw [space, Submodule.mem_span_set'] at hf - obtain ⟨m, a, g, h⟩ := hf - have hg : ∀ j, ∃ w b, neuron σ w b = (g j : C(E, ℝ)) := by - intro j - obtain ⟨p, hp⟩ := (g j).property - exact ⟨p.1, p.2, hp⟩ - choose w b hb using hg - exact ⟨m, a, w, b, by simpa only [← hb] using h⟩ - · rintro ⟨m, a, w, b, rfl⟩ - apply Submodule.sum_mem - intro j hj - apply Submodule.smul_mem - exact Submodule.subset_span ⟨(w j, b j), rfl⟩ - -/-- Membership in `spaceOn` is exactly representability on `K` by a finite-width shallow -network. -/ -theorem mem_spaceOn_iff (σ : C(ℝ, ℝ)) (K : Set E) (f : C(K, ℝ)) : - f ∈ spaceOn σ K ↔ - ∃ (m : ℕ) (a : Fin m → ℝ) (w : Fin m → E) (b : Fin m → ℝ), - ∑ j, a j • (neuron σ (w j) (b j)).restrict K = f := by - constructor - · intro hf - rw [spaceOn, Submodule.mem_span_set'] at hf - obtain ⟨m, a, g, h⟩ := hf - have hg : ∀ j, ∃ w b, (neuron σ w b).restrict K = (g j : C(K, ℝ)) := by - intro j - obtain ⟨p, hp⟩ := (g j).property - exact ⟨p.1, p.2, hp⟩ - choose w b hb using hg - exact ⟨m, a, w, b, by simpa only [← hb] using h⟩ - · rintro ⟨m, a, w, b, rfl⟩ - apply Submodule.sum_mem - intro j hj - apply Submodule.smul_mem - exact Submodule.subset_span ⟨(w j, b j), rfl⟩ - -/-- `spaceOn` is the image of the global network space under restriction. -/ -theorem spaceOn_eq_map (σ : C(ℝ, ℝ)) (K : Set E) : - spaceOn σ K = (space σ).map - (ContinuousMap.compCLM ℝ ℝ ⟨((↑) : K → E), continuous_subtype_val⟩).toLinearMap := by - rw [spaceOn, space, Submodule.map_span] - congr 1 - ext f - aesop - -/-- Density of the network space on a compact set is equivalent to uniform approximation. -/ -theorem dense_spaceOn_iff_uniform_approximation (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) : - Dense (spaceOn σ K : Set C(K, ℝ)) ↔ - ∀ (f : C(K, ℝ)) (ε : ℝ), 0 < ε → - ∃ g ∈ spaceOn σ K, ∀ x, dist (g x) (f x) < ε := by - let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - constructor - · intro h f ε hε - obtain ⟨g, hgBall, hgSpace⟩ := Metric.dense_iff.mp h f ε hε - exact ⟨g, hgSpace, (ContinuousMap.dist_lt_iff hε).mp hgBall⟩ - · intro h - rw [Metric.dense_iff] - intro f ε hε - obtain ⟨g, hgSpace, hgDist⟩ := h f ε hε - exact ⟨g, (ContinuousMap.dist_lt_iff hε).mpr hgDist, hgSpace⟩ - end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean index 47c4cc58..db49aab0 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Convolution.lean @@ -106,49 +106,6 @@ theorem activationAlong_convolutionActivation funext s simp [sub_eq_add_neg, add_assoc] -/-- A ridge function of a convolved activation belongs to the closure of any submodule containing -all bias translates of the corresponding ridge function for the original activation. -/ -theorem activationAlong_convolutionActivation_mem_topologicalClosure - {X : Type*} [TopologicalSpace X] [CompactSpace X] - {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) (u : C(X, ℝ)) (b : ℝ) - (S : Submodule ℝ C(X, ℝ)) (hS : ∀ c, activationAlong σ u c ∈ S) : - activationAlong (convolutionActivation φ σ) u b ∈ S.topologicalClosure := by - rw [activationAlong_convolutionActivation] - apply S.integral_mem_topologicalClosure - filter_upwards with s - exact S.smul_mem (φ s) (hS (b - s)) - -/-- Every neuron for a convolved activation lies in the closure of the shallow-network space for -the original activation. -/ -theorem convolved_neuron_mem_spaceOn_topologicalClosure - {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] - {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) {w : E} {b : ℝ} : - (neuron (convolutionActivation φ σ) w b).restrict K ∈ - (spaceOn σ K).topologicalClosure := by - let _ : CompactSpace K := isCompact_iff_compactSpace.mp hK - let u : C(K, ℝ) := - ⟨fun x => inner ℝ w (x : E), continuous_const.inner continuous_subtype_val⟩ - rw [show (neuron (convolutionActivation φ σ) w b).restrict K = - activationAlong (convolutionActivation φ σ) u b by rfl] - apply activationAlong_convolutionActivation_mem_topologicalClosure φ σ u b - intro c - rw [show activationAlong σ u c = (neuron σ w c).restrict K by rfl] - exact Submodule.subset_span ⟨(w, c), rfl⟩ - -/-- On every compact set, the network space of a convolved activation is contained in the closure -of the network space of the original activation. -/ -theorem convolved_spaceOn_le_topologicalClosure - {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] - {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) : - spaceOn (convolutionActivation φ σ) K ≤ (spaceOn σ K).topologicalClosure := by - rw [spaceOn] - apply Submodule.span_le.2 - rintro f ⟨p, rfl⟩ - exact convolved_neuron_mem_spaceOn_topologicalClosure φ σ hK - /-- A continuous functional annihilating all translates of a ridge function also annihilates the corresponding ridge function for every compactly supported convolution smoothing. -/ theorem annihilates_convolutionActivation @@ -181,29 +138,6 @@ theorem annihilates_convolutionActivation_neurons activationAlong (convolutionActivation φ σ) u b by rfl] exact annihilates_convolutionActivation φ htrans -/-- If one convolution smoothing of `σ` is discriminatory, then `σ` itself is discriminatory. -/ -theorem isDiscriminatory_of_convolutionActivation - {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] - {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) - [IsDiscriminatory E (convolutionActivation φ σ)] : - IsDiscriminatory E σ := by - constructor - intro K hK Λ hΛ - apply IsDiscriminatory.annihilator_eq_zero - (σ := convolutionActivation φ σ) (E := E) K hK Λ - exact annihilates_convolutionActivation_neurons φ hK hΛ - -/-- If the network space of one compactly supported convolution smoothing is dense on a compact -set, then the network space of the original activation is dense there. -/ -theorem dense_spaceOn_of_convolutionActivation - {E : Type*} [SeminormedAddCommGroup E] [InnerProductSpace ℝ E] - {B : Type*} [FunLike B ℝ ℝ] [CompactlySupportedContinuousMapClass B ℝ ℝ] - (φ : B) (σ : C(ℝ, ℝ)) {K : Set E} (hK : IsCompact K) - (h : Dense (spaceOn (convolutionActivation φ σ) K : Set C(K, ℝ))) : - Dense (spaceOn σ K : Set C(K, ℝ)) := - (Dense.mono (convolved_spaceOn_le_topologicalClosure φ σ hK) h).of_closure - /-! ## Iterated derivatives for smooth kernels -/ /-- Successive derivatives of the left kernel give successive derivatives of its convolution @@ -235,12 +169,4 @@ theorem iteratedDeriv_convolutionActivation_testFunction funext x exact (hasDerivAt_convolutionActivation_iterate_lineDerivCLM φ σ n x).deriv -/-- At the origin, the `n`-th derivative of a test-function convolution is the pairing of the -`n`-fold derivative of its kernel with the reflected activation. -/ -theorem iteratedDeriv_convolutionActivation_testFunction_apply_zero - (φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ)) (σ : C(ℝ, ℝ)) (n : ℕ) : - iteratedDeriv n (convolutionActivation φ σ) 0 = - ∫ s : ℝ, (((TestFunction.lineDerivCLM ℝ (1 : ℝ))^[n]) φ) s * σ (-s) := by - simp - end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean index 17e06ce7..f22919ba 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Discriminatory.lean @@ -149,25 +149,6 @@ theorem isDiscriminatory_of_smooth_ridges {σ : C(ℝ, ℝ)} (hsmooth : ∀ n : obtain ⟨g, b, hg, hb, htransfer⟩ := hsmooth n exact annihilates_coordinate_pow_of_iteratedDeriv_ne_zero hg hK (htransfer K hK Λ hΛ) hb -/-- A smooth activation with no identically-zero derivative is discriminatory on every real -inner-product space. -/ -theorem isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero - {g : C(ℝ, ℝ)} (hg : ContDiff ℝ ∞ g) (hne : ∀ n : ℕ, ∃ b : ℝ, iteratedDeriv n g b ≠ 0) : - IsDiscriminatory E g := by - apply isDiscriminatory_of_smooth_ridges - intro n - obtain ⟨b, hb⟩ := hne n - exact ⟨g, b, hg, hb, by simp⟩ - -/-- A smooth activation with no identically-zero derivative has dense network space on every -compact subset of a real inner-product space. -/ -theorem dense_spaceOn_of_contDiff_of_iteratedDeriv_ne_zero - {g : C(ℝ, ℝ)} (hg : ContDiff ℝ ∞ g) - (hne : ∀ n : ℕ, ∃ b : ℝ, iteratedDeriv n g b ≠ 0) {K : Set E} (hK : IsCompact K) : - Dense (spaceOn g K : Set C(K, ℝ)) := - (dense_spaceOn_iff_isDiscriminatory g).2 - (isDiscriminatory_of_contDiff_of_iteratedDeriv_ne_zero hg hne) K hK - end SmoothActivation end Learning.ShallowNetwork diff --git a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean index 8c0c0816..98dbbce8 100644 --- a/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean +++ b/LeanMachineLearning/NeuralNetwork/UniversalApproximation/Nonpolynomial.lean @@ -130,7 +130,7 @@ theorem exists_testFunction_iteratedDeriv_convolutionActivation_ne_zero ∃ φ : 𝓓((⊤ : TopologicalSpace.Opens ℝ), ℝ), iteratedDeriv n (convolutionActivation φ σ) 0 ≠ 0 := by obtain ⟨φ, hφ⟩ := exists_testFunction_iteratedLineDeriv_integral_mul_reflected_ne_zero hσ n - exact ⟨φ, by rwa [iteratedDeriv_convolutionActivation_testFunction_apply_zero]⟩ + exact ⟨φ, by simpa⟩ /-! ## Universality of nonpolynomial activations -/ From 6101a707a3b5ff0e8a941ae4818f2a1a735f34c0 Mon Sep 17 00:00:00 2001 From: Yi Yuan Date: Wed, 23 Sep 2026 12:10:13 +0800 Subject: [PATCH 8/8] Fix deprecated rwa syntax --- .../Analysis/Distribution/PolynomialCharacterization.lean | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean index da257f9d..d7fc10d0 100644 --- a/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean +++ b/LeanMachineLearning/ForMathlib/Analysis/Distribution/PolynomialCharacterization.lean @@ -52,7 +52,8 @@ theorem exists_polynomial_of_iteratedLineDerivOp_eq_zero | succ k ih => let DT : 𝓓'(Ω, ℝ) := ∂_{(1 : ℝ)} T have hDT : iteratedLineDerivOp (fun _ : Fin k => (1 : ℝ)) DT = 0 := by - rwa [iteratedLineDerivOp_const_eq_iter_lineDerivOp] at hT ⊢ + rw [iteratedLineDerivOp_const_eq_iter_lineDerivOp] at hT ⊢ + exact hT obtain ⟨p, hp⟩ := ih DT hDT obtain ⟨q, hq⟩ := Polynomial.derivative_surjective_of_charZero p let Q : 𝓓'(Ω, ℝ) := ofFun Ω (fun x => q.eval x) volume ⊤