From d19f874cc6fc19b85e56307751b42e817690d94f Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:14:38 -0400 Subject: [PATCH 01/18] Add TensorOperationsBenchmarks core framework Reusable benchmark suite package for TensorOperations.jl, living in benchmark/. Core pieces: pure-data contraction specs (AddSpec/TraceSpec/ ContractSpec/NetworkSpec) decoupled from an AbstractProvider interface so downstream packages can plug in their own tensor type; a category registry; analytical flop/byte cost model; and independent thread-count configuration that never pollutes a timed sample. Ships with generic pairwise/permutation/ trace categories. Co-Authored-By: Claude Sonnet 5 --- benchmark/Project.toml | 30 ++++++ benchmark/src/TensorOperationsBenchmarks.jl | 32 ++++++ benchmark/src/categories/pairwise.jl | 34 +++++++ benchmark/src/categories/permute.jl | 32 ++++++ benchmark/src/categories/trace.jl | 35 +++++++ benchmark/src/cost.jl | 98 ++++++++++++++++++ benchmark/src/lowering.jl | 102 +++++++++++++++++++ benchmark/src/provider.jl | 81 +++++++++++++++ benchmark/src/registry.jl | 52 ++++++++++ benchmark/src/report.jl | 65 ++++++++++++ benchmark/src/specs.jl | 104 ++++++++++++++++++++ benchmark/src/suite.jl | 41 ++++++++ benchmark/src/threading.jl | 66 +++++++++++++ 13 files changed, 772 insertions(+) create mode 100644 benchmark/Project.toml create mode 100644 benchmark/src/TensorOperationsBenchmarks.jl create mode 100644 benchmark/src/categories/pairwise.jl create mode 100644 benchmark/src/categories/permute.jl create mode 100644 benchmark/src/categories/trace.jl create mode 100644 benchmark/src/cost.jl create mode 100644 benchmark/src/lowering.jl create mode 100644 benchmark/src/provider.jl create mode 100644 benchmark/src/registry.jl create mode 100644 benchmark/src/report.jl create mode 100644 benchmark/src/specs.jl create mode 100644 benchmark/src/suite.jl create mode 100644 benchmark/src/threading.jl diff --git a/benchmark/Project.toml b/benchmark/Project.toml new file mode 100644 index 00000000..da297f82 --- /dev/null +++ b/benchmark/Project.toml @@ -0,0 +1,30 @@ +name = "TensorOperationsBenchmarks" +uuid = "d983fc97-4e87-46ba-abd9-b4864e69d4dd" +authors = ["Lukas Devos "] +version = "0.1.0" + +[deps] +ArgParse = "c7e460c6-2fb9-53a9-8c5b-16f535851c63" +BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" +LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" +PkgBenchmark = "32113eaa-f34f-5b0d-bd6c-c81e245fc73d" +Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" +Strided = "5e0ebb24-38b0-5f93-81fe-25c709ecae67" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" + +[compat] +ArgParse = "1" +BenchmarkTools = "1" +LinearAlgebra = "1.10" +PkgBenchmark = "0.2" +Random = "1.10" +Strided = "2.6" +TensorOperations = "5.8" +Test = "1" +julia = "1.10" + +[extras] +Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" + +[targets] +test = ["Test"] diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl new file mode 100644 index 00000000..1f6904e1 --- /dev/null +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -0,0 +1,32 @@ +module TensorOperationsBenchmarks + +using LinearAlgebra: BLAS +using Strided: Strided +using Random: Random, randn! +using BenchmarkTools +using TensorOperations +using TensorOperations: DefaultBackend, DefaultAllocator, AbstractBackend + +include("specs.jl") +include("cost.jl") +include("provider.jl") +include("threading.jl") +include("registry.jl") +include("lowering.jl") +include("suite.jl") +include("report.jl") + +include("categories/pairwise.jl") +include("categories/permute.jl") +include("categories/trace.jl") + +export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, NetworkSpec +export flops, bytes +export AbstractProvider, ArrayProvider, scalartype, randtensor, backend, allocator, label, + supports, rng +export ThreadConfig, with_threads, set_threads! +export BenchmarkCase, register_category!, REGISTRY, default_sizes +export build_suite +export resultstable + +end # module diff --git a/benchmark/src/categories/pairwise.jl b/benchmark/src/categories/pairwise.jl new file mode 100644 index 00000000..4e59138a --- /dev/null +++ b/benchmark/src/categories/pairwise.jl @@ -0,0 +1,34 @@ +# Generic pairwise contractions of varying rank/dimension, ported (as literal Julia data, +# not a regex-parsed `.dat` file) from the shapes explored in the stale `ld/benchmark` +# prototype. `sizes` is a list of leg dimensions to sweep; every dimension is used at every +# `(nopenA, ncontract, nopenB)` shape below, giving a scaling curve per shape. + +const PAIRWISE_SHAPES = ( + (1, 1, 1), # matrix-vector-like + (2, 1, 2), # single shared bond, several open legs each side + (2, 2, 2), # GEMM-like, rank 4 total + (1, 3, 1), # trace-heavy: many contracted, few open + (1, 0, 1), # pure outer product, no contraction +) + +function _pairwise_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + for (nopenA, ncontract, nopenB) in PAIRWISE_SHAPES + openA = [Symbol("a", i) for i in 1:nopenA] + contract = [Symbol("c", i) for i in 1:ncontract] + openB = [Symbol("b", i) for i in 1:nopenB] + IA = vcat(openA, contract) + IB = vcat(contract, openB) + IC = vcat(openA, openB) + dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) + spec = ContractSpec(IA, IB, IC, dims) + within_memory_budget(spec) || continue + id = "dim$(dim)_$(nopenA)_$(ncontract)_$(nopenB)" + push!(cases, BenchmarkCase(:pairwise, id, (; dim, nopenA, ncontract, nopenB), spec)) + end + end + return cases +end + +register_category!(:pairwise, _pairwise_cases; sizes = (8, 32, 64, 128, 256)) diff --git a/benchmark/src/categories/permute.jl b/benchmark/src/categories/permute.jl new file mode 100644 index 00000000..d1ede9f3 --- /dev/null +++ b/benchmark/src/categories/permute.jl @@ -0,0 +1,32 @@ +# Permutation-only benchmarks (transpose cost), a first-class category in its own right: +# TBLIS/TCL-style backends exist precisely because transpose-free vs. transpose-then-GEMM +# strategies differ, so isolating pure `tensorcopy!` cost from contraction cost matters. +# +# `sizes` is a list of leg dimensions; for each dimension we benchmark permutations of a +# rank-4 tensor across a spread of "how scrambled" the permutation is (identity-adjacent vs. +# fully reversed). + +const PERMUTE_PATTERNS = ( + [1, 2, 3, 4], # identity (still exercises the copy machinery, no real permutation) + [2, 1, 3, 4], # single adjacent swap + [4, 3, 2, 1], # full reversal + [3, 1, 4, 2], # scrambled +) + +function _permute_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + IA = [Symbol("a", i) for i in 1:4] + dims = Dict{Symbol, Int}(l => dim for l in IA) + for pattern in PERMUTE_PATTERNS + IC = IA[pattern] + spec = AddSpec(IA, IC, dims) + within_memory_budget(spec) || continue + id = "dim$(dim)_perm$(join(pattern))" + push!(cases, BenchmarkCase(:permute, id, (; dim, pattern), spec)) + end + end + return cases +end + +register_category!(:permute, _permute_cases; sizes = (8, 32, 64, 128, 256)) diff --git a/benchmark/src/categories/trace.jl b/benchmark/src/categories/trace.jl new file mode 100644 index 00000000..3dc9868d --- /dev/null +++ b/benchmark/src/categories/trace.jl @@ -0,0 +1,35 @@ +# Trace-heavy benchmarks: partial traces (some legs traced, some kept open) and full traces +# (reduce all the way to a scalar), as distinct from the pairwise-contraction category. +# +# `sizes` is a list of leg dimensions; for each dimension we benchmark a rank-6 tensor traced +# down to rank-2 (partial trace) and a rank-4 tensor traced all the way to a scalar (full +# trace). + +function _trace_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + # partial trace: rank 6 -> rank 2, trace 2 pairs, keep 2 open + IA6 = [:o1, :o2, :t1, :t1, :t2, :t2] + IC6 = [:o1, :o2] + dims6 = Dict{Symbol, Int}(l => dim for l in unique(IA6)) + partialspec = TraceSpec(IA6, IC6, dims6) + if within_memory_budget(partialspec) + push!( + cases, + BenchmarkCase(:trace, "partial_dim$(dim)", (; dim, kind = :partial), partialspec) + ) + end + + # full trace: rank 4 -> scalar + IA4 = [:t1, :t1, :t2, :t2] + IC4 = Symbol[] + dims4 = Dict{Symbol, Int}(l => dim for l in unique(IA4)) + fullspec = TraceSpec(IA4, IC4, dims4) + if within_memory_budget(fullspec) + push!(cases, BenchmarkCase(:trace, "full_dim$(dim)", (; dim, kind = :full), fullspec)) + end + end + return cases +end + +register_category!(:trace, _trace_cases; sizes = (8, 32, 64, 128, 256)) diff --git a/benchmark/src/cost.jl b/benchmark/src/cost.jl new file mode 100644 index 00000000..28e33062 --- /dev/null +++ b/benchmark/src/cost.jl @@ -0,0 +1,98 @@ +# Analytical flop/byte counts computed directly from a spec (no tensor objects needed), so +# that timings can be reported as GFLOP/s or GB/s scaling curves instead of raw wall-clock time. + +# A spec's `TA`/`TB`/`TC` (or `Ts`) are `nothing` unless the mixed-precision category set them +# explicitly -- meaning "whatever the provider's default `scalartype` turns out to be", which +# isn't known until a provider is chosen. For sizing purposes (both for reporting GB/s *before* +# a provider is picked, and for the `within_memory_budget` safety check in registry.jl, which +# runs at case-generation time) we assume the common case of `Float64`/`ComplexF64`-sized +# (8-byte) elements; this can only ever *underestimate* real memory use for a provider using a +# larger element type, never overestimate it into skipping a case that would actually fit. +_elsize(::Nothing) = sizeof(Float64) +_elsize(T::Type) = sizeof(T) + +""" + flops(spec::AbstractCaseSpec) -> Int + +Approximate number of floating point operations (multiply + add counted together) needed to +execute `spec`. +""" +function flops end + +""" + bytes(spec::AbstractCaseSpec) -> Int + +Approximate number of bytes moved (all tensors read once, output written once) to execute +`spec`. +""" +function bytes end + +# Permutation: no arithmetic, just data movement (one read + one write of every element). +flops(::AddSpec) = 0 +function bytes(spec::AddSpec) + n = prod((spec.dims[l] for l in spec.IA); init = 1) + return n * (_elsize(spec.TA) + _elsize(spec.TC)) +end + +# Trace: every element of A is read and accumulated once into the (smaller) output. +function flops(spec::TraceSpec) + return prod((spec.dims[l] for l in spec.IA); init = 1) +end +function bytes(spec::TraceSpec) + nA = prod((spec.dims[l] for l in spec.IA); init = 1) + nC = prod((spec.dims[l] for l in spec.IC); init = 1) + return nA * _elsize(spec.TA) + nC * _elsize(spec.TC) +end + +# Pairwise contraction: 2 * (open-A) * (open-B) * (contracted), the standard GEMM-equivalent +# flop count, using multiply-add pairs. +function flops(spec::ContractSpec) + contracted = intersect(spec.IA, spec.IB) + openA = setdiff(spec.IA, contracted) + openB = setdiff(spec.IB, contracted) + nopenA = prod((spec.dims[l] for l in openA); init = 1) + nopenB = prod((spec.dims[l] for l in openB); init = 1) + ncontracted = prod((spec.dims[l] for l in contracted); init = 1) + return 2 * nopenA * nopenB * ncontracted +end +function bytes(spec::ContractSpec) + nA = prod((spec.dims[l] for l in spec.IA); init = 1) + nB = prod((spec.dims[l] for l in spec.IB); init = 1) + nC = prod((spec.dims[l] for l in spec.IC); init = 1) + return nA * _elsize(spec.TA) + nB * _elsize(spec.TB) + nC * _elsize(spec.TC) +end + +# Network: walk the *actual* pairwise contraction tree that `ncon` would build for this +# network (`TensorOperations.ncontree`/`indexordertree`, the same functions `ncon` itself +# calls), rather than an arbitrary/greedy pairing -- so the reported cost matches what +# `execute(spec::NetworkSpec, ...)` actually runs, not a guess at it. +function flops(spec::NetworkSpec) + tree = spec.order === nothing ? TensorOperations.ncontree(spec.indexlists) : + TensorOperations.indexordertree(spec.indexlists, spec.order) + _, total = _tree_cost(spec, tree) do labelsA, labelsB, contracted + nopenA = prod((spec.dims[abs(l)] for l in labelsA if !(l in contracted)); init = 1) + nopenB = prod((spec.dims[abs(l)] for l in labelsB if !(l in contracted)); init = 1) + ncontracted = prod((spec.dims[abs(l)] for l in contracted); init = 1) + return 2 * nopenA * nopenB * ncontracted + end + return total +end +function bytes(spec::NetworkSpec) + Ts = something(spec.Ts, fill(nothing, length(spec.indexlists))) + return sum( + prod((spec.dims[abs(l)] for l in il); init = 1) * _elsize(T) + for (il, T) in zip(spec.indexlists, Ts) + ) +end + +# Recursively walks a `ncontree`/`indexordertree` result (leaves are `Int` indices into +# `spec.indexlists`, nodes are `Any[left, right]`), returning `(survivinglabels, totalcost)`, +# calling `f(labelsA, labelsB, contractedlabels)` at every pairwise step -- mirroring exactly +# how `ncon`'s own `contracttree` combines subtrees (`IC = symdiff(IA, IB)`). +function _tree_cost(f, spec::NetworkSpec, tree) + tree isa Int && return spec.indexlists[tree], 0 + labelsA, costA = _tree_cost(f, spec, tree[1]) + labelsB, costB = _tree_cost(f, spec, tree[2]) + contracted = intersect(labelsA, labelsB) + return symdiff(labelsA, labelsB), costA + costB + f(labelsA, labelsB, contracted) +end diff --git a/benchmark/src/lowering.jl b/benchmark/src/lowering.jl new file mode 100644 index 00000000..b2d20816 --- /dev/null +++ b/benchmark/src/lowering.jl @@ -0,0 +1,102 @@ +# Turns a (spec, provider) pair into an executable BenchmarkTools benchmark. Everything that +# isn't the operation itself -- input tensors, the *output* tensor, and the low-level +# `Index2Tuple` index-permutation computation -- is built fresh in the `setup=` block (once per +# sample, not per eval), so it doesn't count against the timed operation. In particular, `C` is +# preallocated (via `tensoralloc_add`/`tensoralloc_contract`, so it goes through the provider's +# configured allocator) and execution uses the mutating `tensorcopy!`/`tensortrace!`/ +# `tensorcontract!`, so the timed region is exactly the compute kernel, not an output +# allocation. +# +# `NetworkSpec` is the one exception: `ncon` has no public in-place variant (it always +# allocates its final and intermediate results internally), so its timed region does include +# allocation. Reimplementing `ncon`'s tree contraction manually with preallocated buffers would +# let us avoid that, but is out of scope for v1 -- `ncon`-based network cases should be read as +# "cost of ncon", allocation included, not "cost of the raw contraction kernel". + +_scalartype_or(::Nothing, provider) = TensorOperations.scalartype(provider) +_scalartype_or(T::Type, provider) = T + +function maketensors(spec::AddSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TC = _scalartype_or(spec.TC, provider) + dims = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + A = randtensor(provider, spec.IA, dims, TA) + pA = TensorOperations.add_indices(spec.IA, spec.IC) + C = TensorOperations.tensoralloc_add(TC, A, pA, spec.conjA, Val(false), allocator(provider)) + return (A, pA, C) +end + +function maketensors(spec::TraceSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TC = _scalartype_or(spec.TC, provider) + dims = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + A = randtensor(provider, spec.IA, dims, TA) + p, q = TensorOperations.trace_indices(spec.IA, spec.IC) + C = TensorOperations.tensoralloc_add(TC, A, p, spec.conjA, Val(false), allocator(provider)) + return (A, p, q, C) +end + +function maketensors(spec::ContractSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TB = _scalartype_or(spec.TB, provider) + TC = _scalartype_or(spec.TC, provider) + dimsA = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + dimsB = ntuple(i -> spec.dims[spec.IB[i]], length(spec.IB)) + A = randtensor(provider, spec.IA, dimsA, TA) + B = randtensor(provider, spec.IB, dimsB, TB) + pA, pB, pAB = TensorOperations.contract_indices(spec.IA, spec.IB, spec.IC) + C = TensorOperations.tensoralloc_contract( + TC, A, pA, spec.conjA, B, pB, spec.conjB, pAB, Val(false), allocator(provider) + ) + return (A, B, pA, pB, pAB, C) +end + +function maketensors(spec::NetworkSpec, provider) + Ts = something(spec.Ts, fill(TensorOperations.scalartype(provider), length(spec.indexlists))) + return map(spec.indexlists, Ts) do il, T + dims = ntuple(i -> spec.dims[abs(il[i])], length(il)) + randtensor(provider, il, dims, T) + end +end + +function execute(spec::AddSpec, (A, pA, C), provider) + return tensorcopy!( + C, A, pA, spec.conjA, one(eltype(C)), backend(provider), allocator(provider) + ) +end + +function execute(spec::TraceSpec, (A, p, q, C), provider) + return tensortrace!( + C, A, p, q, spec.conjA, one(eltype(C)), zero(eltype(C)), + backend(provider), allocator(provider) + ) +end + +function execute(spec::ContractSpec, (A, B, pA, pB, pAB, C), provider) + return tensorcontract!( + C, A, pA, spec.conjA, B, pB, spec.conjB, pAB, one(eltype(C)), zero(eltype(C)), + backend(provider), allocator(provider) + ) +end + +function execute(spec::NetworkSpec, tensors, provider) + return ncon( + tensors, spec.indexlists, spec.conjlist; + order = spec.order, output = spec.output, + backend = backend(provider), allocator = allocator(provider) + ) +end + +""" + make_benchmarkable(case::BenchmarkCase, provider::AbstractProvider) + +Build a `BenchmarkTools.Benchmark` for `case` run against `provider`, constructing fresh +input tensors and a fresh (preallocated, uninitialized) output tensor before every sample. +""" +function make_benchmarkable(case::BenchmarkCase, provider::AbstractProvider) + spec = case.spec + return @benchmarkable( + execute($spec, ts, $provider), + setup = (ts = maketensors($spec, $provider)) + ) +end diff --git a/benchmark/src/provider.jl b/benchmark/src/provider.jl new file mode 100644 index 00000000..0cde390c --- /dev/null +++ b/benchmark/src/provider.jl @@ -0,0 +1,81 @@ +# The downstream extension point: a `AbstractProvider` supplies the tensor type, default +# element type, backend and allocator to use when executing a (backend-agnostic) spec. This is +# deliberately a minimal interface (four required-ish methods) so that a downstream package +# (e.g. a symmetric/block-sparse tensor package) can add its own provider without touching +# anything else in this package. + +""" + AbstractProvider + +Supertype for the downstream extension point of the benchmark suite. A provider ties a spec +(pure index/dimension data) to an actual tensor type, backend, and allocator to run it with. + +Required methods: +- `scalartype(provider)`: the provider's default element type. +- `randtensor(provider, labels, dims, T=scalartype(provider))`: build a random tensor with the + given (ordered) `labels`/`dims` and element type `T`. + +Optional methods (with sensible defaults): +- `backend(provider) = DefaultBackend()` +- `allocator(provider) = DefaultAllocator()` +- `label(provider) = string(nameof(typeof(provider)))` +- `supports(provider, category::Symbol) = true` +- `rng(provider) = Random.default_rng()`: the RNG `randtensor` should draw from. Override + with a *stored, stateful* RNG seeded once at construction (as [`ArrayProvider`](@ref) does) + to make repeated suite runs reproducible -- a fresh RNG re-seeded on every call would just + make every tensor within a run identical, which isn't the same thing. +""" +abstract type AbstractProvider end + +function TensorOperations.scalartype(p::AbstractProvider) + error("`scalartype` not implemented for provider $(typeof(p))") +end + +""" + randtensor(provider, labels, dims, T=scalartype(provider)) + +Instantiate a random tensor with index order `labels` (a `Vector{Symbol}` or `Vector{Int}`, +only used for bookkeeping), size `dims` (matching `labels` elementwise), and element type `T`. +""" +function randtensor end + +backend(p::AbstractProvider) = DefaultBackend() +allocator(p::AbstractProvider) = DefaultAllocator() +label(p::AbstractProvider) = string(nameof(typeof(p))) +supports(p::AbstractProvider, category::Symbol) = true +rng(p::AbstractProvider) = Random.default_rng() + +""" + ArrayProvider{T}(; backend=DefaultBackend(), allocator=DefaultAllocator(), rng=Random.Xoshiro(0x5eed5eed5eed5eed)) + +The reference provider: plain dense `Array`s of element type `T`, exercising +TensorOperations.jl's own backends (`StridedNative`, `StridedBLAS`, `cuTENSORBackend`, ...). +This is how TensorOperations.jl dogfoods its own suite. + +Tensors are allocated via `TensorOperations.tensoralloc` (so they go through `allocator`, not +a bare `Array` constructor) and filled via `randn!(rng, ...)`. `rng` is stored (not +reconstructed per call), so it advances across successive `randtensor` calls within one suite +build -- but is seeded identically across separate `ArrayProvider()` constructions, so two runs +of the same suite see the same input data. +""" +struct ArrayProvider{T, B <: AbstractBackend, A, R <: Random.AbstractRNG} <: AbstractProvider + backend::B + allocator::A + rng::R +end +function ArrayProvider{T}(; + backend::AbstractBackend = DefaultBackend(), allocator = DefaultAllocator(), + rng::Random.AbstractRNG = Random.Xoshiro(0x5eed5eed5eed5eed) + ) where {T} + return ArrayProvider{T, typeof(backend), typeof(allocator), typeof(rng)}(backend, allocator, rng) +end + +TensorOperations.scalartype(::ArrayProvider{T}) where {T} = T +function randtensor(p::ArrayProvider, labels, dims, T = TensorOperations.scalartype(p)) + C = TensorOperations.tensoralloc(Array{T, length(dims)}, dims, Val(false), allocator(p)) + return randn!(rng(p), C) +end +backend(p::ArrayProvider) = p.backend +allocator(p::ArrayProvider) = p.allocator +label(p::ArrayProvider{T}) where {T} = "$(nameof(typeof(p.backend)))/$T" +rng(p::ArrayProvider) = p.rng diff --git a/benchmark/src/registry.jl b/benchmark/src/registry.jl new file mode 100644 index 00000000..0377384e --- /dev/null +++ b/benchmark/src/registry.jl @@ -0,0 +1,52 @@ +# A category is registered as a single generator function `sizes -> Vector{BenchmarkCase}`. +# Adding a new category to the suite is exactly: write one file defining a generator, `include` +# it, and call `register_category!` -- nothing else in the framework changes. + +""" + BenchmarkCase(category, id, params, spec) + +One concrete benchmark case: `category` groups it (e.g. `:pairwise`), `id` is a short unique +label within the category, `params` records the sweep parameters that produced it (e.g. +`(; D=64)`), and `spec` is the [`AbstractCaseSpec`](@ref) to execute. +""" +struct BenchmarkCase + category::Symbol + id::String + params::NamedTuple + spec::AbstractCaseSpec +end + +const REGISTRY = Dict{Symbol, Function}() +const DEFAULT_SIZES = Dict{Symbol, Any}() + +""" + register_category!(name::Symbol, generator::Function; sizes=(4, 8, 16, 32, 64, 128)) + +Register `generator(sizes) -> Vector{BenchmarkCase}` under category `name`, along with the +default size sweep to use when the caller doesn't supply one. Calling this again for the same +`name` overwrites the previous generator/default. +""" +function register_category!(name::Symbol, generator::Function; sizes = (4, 8, 16, 32, 64, 128)) + REGISTRY[name] = generator + DEFAULT_SIZES[name] = sizes + return nothing +end + +""" + default_sizes(category::Symbol) + +The default size sweep passed to `category`'s generator when the caller doesn't supply one. +Each category interprets `sizes` in its own way (a list of ranks, of bond dimensions, ...). +""" +default_sizes(category::Symbol) = get(DEFAULT_SIZES, category, (4, 8, 16, 32, 64, 128)) + +""" + within_memory_budget(spec::AbstractCaseSpec; maxbytes=MAX_CASE_BYTES) + +Whether executing `spec` would stay within `maxbytes` of total tensor memory, assuming (in the +absence of an explicit `TA`/`TB`/`TC`/`Ts` override) worst-case `Float64`-sized elements. +Category generators use this to silently skip dimension/shape combinations that would +otherwise OOM the benchmark process, rather than hand-tuning per-shape size ceilings. +""" +const MAX_CASE_BYTES = 2^28 # 256 MiB +within_memory_budget(spec::AbstractCaseSpec; maxbytes = MAX_CASE_BYTES) = bytes(spec) <= maxbytes diff --git a/benchmark/src/report.jl b/benchmark/src/report.jl new file mode 100644 index 00000000..666bb607 --- /dev/null +++ b/benchmark/src/report.jl @@ -0,0 +1,65 @@ +# Turns a BenchmarkTools/PkgBenchmark result `BenchmarkGroup` (as produced by running a +# `build_suite` suite) into a flat table of rows, joining the raw timings back against the +# originating specs (regenerated from `REGISTRY`) so GFLOP/s and bandwidth can be reported +# alongside wall-clock time. + +""" + ResultRow + +One row of [`resultstable`](@ref): `category`, `provider`, `id`, `params`, `mintime` (ns), +`allocs`, `memory` (bytes), `gflops` (`flops(spec) / mintime`, or `missing` if `mintime` is +zero), and `gbps` (`bytes(spec) / mintime`). +""" +struct ResultRow + category::String + provider::String + id::String + params::NamedTuple + mintime::Float64 + allocs::Int + memory::Int + gflops::Union{Float64, Missing} + gbps::Union{Float64, Missing} +end + +""" + resultstable(results::BenchmarkGroup; categories=collect(keys(REGISTRY)), sizes=nothing) + +Flatten a benchmark-run result (with the same `[category][provider][id]` nesting +`build_suite` produces) into a `Vector{ResultRow}`. `categories`/`sizes` must match what was +passed to the `build_suite` call that produced `results`, since specs (and therefore +flop/byte counts) are regenerated from `REGISTRY` rather than stored in the result itself. +""" +function resultstable( + results::BenchmarkGroup; categories = collect(keys(REGISTRY)), sizes = nothing + ) + rows = ResultRow[] + for category in categories + haskey(results, String(category)) || continue + cases = REGISTRY[category](_sizes_for(sizes, category)) + casesbyid = Dict(c.id => c for c in cases) + catgroup = results[String(category)] + for providerlabel in keys(catgroup) + provgroup = catgroup[providerlabel] + for id in keys(provgroup) + trial = provgroup[id] + case = casesbyid[id] + mintime = minimum(trial.times) + memory = trial.memory + allocs = trial.allocs + fl = flops(case.spec) + by = bytes(case.spec) + gflops = mintime > 0 ? fl / mintime : missing + gbps = mintime > 0 ? by / mintime : missing + push!( + rows, + ResultRow( + String(category), providerlabel, id, case.params, + mintime, allocs, memory, gflops, gbps + ) + ) + end + end + end + return rows +end diff --git a/benchmark/src/specs.jl b/benchmark/src/specs.jl new file mode 100644 index 00000000..287b3ce2 --- /dev/null +++ b/benchmark/src/specs.jl @@ -0,0 +1,104 @@ +# Pure-data contraction/network specifications. +# +# Specs carry index labels and leg extents only -- never actual tensor objects -- so that a +# single spec can be shared across providers (different tensor types/backends) and so that +# `cost.jl` can compute flop/byte counts analytically from the spec alone. +# +# `TA`/`TB`/`TC` (and `Ts` for `NetworkSpec`) default to `nothing`, meaning "ask the provider +# for its default `scalartype`". Only the mixed-precision category sets them explicitly. + +abstract type AbstractCaseSpec end + +""" + AddSpec(IA, IC, dims, conjA, TA=nothing, TC=nothing) + +Specification of a permutation `C = permutedims(opA(A), pA)` (executed via `tensorcopy`), +where `pA` is derived from matching labels in `IA` to `IC`. +""" +struct AddSpec <: AbstractCaseSpec + IA::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + TA::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function AddSpec(IA, IC, dims; conjA::Bool = false, TA = nothing, TC = nothing) + return AddSpec(collect(Symbol, IA), collect(Symbol, IC), dims, conjA, TA, TC) +end + +""" + TraceSpec(IA, IC, dims, conjA, TA=nothing, TC=nothing) + +Specification of a (partial) trace `C = permutedims(trace(opA(A)), p)` (executed via +`tensortrace`). Labels appearing twice in `IA` are traced; labels appearing once and also +in `IC` are kept. +""" +struct TraceSpec <: AbstractCaseSpec + IA::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + TA::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function TraceSpec(IA, IC, dims; conjA::Bool = false, TA = nothing, TC = nothing) + return TraceSpec(collect(Symbol, IA), collect(Symbol, IC), dims, conjA, TA, TC) +end + +""" + ContractSpec(IA, IB, IC, dims, conjA, conjB, TA=nothing, TB=nothing, TC=nothing) + +Specification of a pairwise contraction `C = contract(opA(A), opB(B))` (executed via +`tensorcontract`). +""" +struct ContractSpec <: AbstractCaseSpec + IA::Vector{Symbol} + IB::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + conjB::Bool + TA::Union{Nothing, Type} + TB::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function ContractSpec( + IA, IB, IC, dims; conjA::Bool = false, conjB::Bool = false, + TA = nothing, TB = nothing, TC = nothing + ) + return ContractSpec( + collect(Symbol, IA), collect(Symbol, IB), collect(Symbol, IC), dims, + conjA, conjB, TA, TB, TC + ) +end + +""" + NetworkSpec(indexlists, conjlist, output, dims, order=nothing, Ts=nothing) + +An `ncon`-style multi-tensor network: `indexlists[k]` gives the signed integer index labels +of the `k`th tensor (positive = contracted, negative = open/output), `dims` maps each *label* +(by absolute value) to its extent, and `order` optionally fixes the contraction order (as a +list of positive labels, in the order they should be contracted) -- `nothing` lets `ncon`'s +default greedy tree builder decide. +""" +struct NetworkSpec <: AbstractCaseSpec + indexlists::Vector{Vector{Int}} + conjlist::Vector{Bool} + output::Vector{Int} + dims::Dict{Int, Int} + order::Union{Nothing, Vector{Int}} + Ts::Union{Nothing, Vector{<:Type}} +end +function NetworkSpec( + indexlists, dims; conjlist = fill(false, length(indexlists)), + output = nothing, order = nothing, Ts = nothing + ) + outputindices = something( + output, sort(unique(l for il in indexlists for l in il if l < 0); rev = true) + ) + return NetworkSpec( + [collect(Int, il) for il in indexlists], collect(Bool, conjlist), + collect(Int, outputindices), dims, order, Ts + ) +end diff --git a/benchmark/src/suite.jl b/benchmark/src/suite.jl new file mode 100644 index 00000000..ded8bc8f --- /dev/null +++ b/benchmark/src/suite.jl @@ -0,0 +1,41 @@ +# Suite assembly: nests a `BenchmarkGroup` as [category][provider label][case id]. Threading +# is deliberately NOT an axis here -- see threading.jl -- since applying it per-case would +# count the thread-count switch as part of the timed operation. `benchmarks.jl` (the +# PkgBenchmark entrypoint) is expected to be a thin wrapper: call `set_threads!` once, construct +# the `AbstractProvider`s to compare, call `build_suite`, assign the result to `const SUITE`. + +""" + build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing) + +Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or the subset in +`categories`) for every provider in `providers` (skipping providers that opt out via +[`supports`](@ref)). + +`sizes` may be `nothing` (use each category's [`default_sizes`](@ref)), a size sweep applied to +every category, or a `Dict{Symbol}` mapping category name to its own size sweep. +""" +function build_suite( + providers::AbstractVector{<:AbstractProvider}; + categories = collect(keys(REGISTRY)), + sizes = nothing + ) + suite = BenchmarkGroup() + for category in categories + generator = REGISTRY[category] + catsizes = _sizes_for(sizes, category) + cases = generator(catsizes) + catgroup = suite[String(category)] = BenchmarkGroup() + for provider in providers + supports(provider, category) || continue + provgroup = catgroup[label(provider)] = BenchmarkGroup() + for case in cases + provgroup[case.id] = make_benchmarkable(case, provider) + end + end + end + return suite +end + +_sizes_for(::Nothing, category::Symbol) = default_sizes(category) +_sizes_for(sizes::AbstractDict, category::Symbol) = get(sizes, category, default_sizes(category)) +_sizes_for(sizes, ::Symbol) = sizes diff --git a/benchmark/src/threading.jl b/benchmark/src/threading.jl new file mode 100644 index 00000000..292a828e --- /dev/null +++ b/benchmark/src/threading.jl @@ -0,0 +1,66 @@ +# Threading is orthogonal to *which* tensor type/backend a provider uses (it applies +# process-wide via BLAS and Strided's own runtime thread counters), so it is not part of the +# `AbstractProvider` interface, and -- importantly -- it is NOT baked into individual +# `@benchmarkable` cases either: doing that would mean every timed sample pays for +# `BLAS.set_num_threads`/`Strided.set_num_threads` plus a closure allocation, polluting the very +# measurement it's supposed to control. Instead, a thread configuration is applied *once*, +# around an entire suite (or PkgBenchmark) run -- see `set_threads!`/`with_threads` below, and +# `benchmarks.jl`, which calls `set_threads!` once at the top level before building `SUITE`. +# +# Note: `Strided.set_num_threads` is capped by `Threads.nthreads()`, which is fixed at Julia +# process startup (`-t`/`JULIA_NUM_THREADS`). Sweeping *above* the process's thread count is not +# possible at runtime; `scripts/run_benchmarks.jl` covers that case by relaunching Julia with a +# different `-t` per outer thread count (via `PkgBenchmark`'s `juliacmd`). + +""" + ThreadConfig(; blas=nothing, strided=nothing) + +A named thread-count configuration. `nothing` for either field means "leave that thread count +untouched". Use [`with_threads`](@ref) to apply one around a block of code. +""" +struct ThreadConfig + blas::Union{Nothing, Int} + strided::Union{Nothing, Int} + name::String +end +function ThreadConfig(; blas::Union{Nothing, Int} = nothing, strided::Union{Nothing, Int} = nothing, name = nothing) + autoname = "blas=$(something(blas, "-")),strided=$(something(strided, "-"))" + return ThreadConfig(blas, strided, something(name, autoname)) +end + +label(cfg::ThreadConfig) = cfg.name + +""" + set_threads!(cfg::ThreadConfig) + +Set `BLAS`/`Strided` thread counts according to `cfg`, permanently (i.e. not restored +afterwards). Intended for one-shot, process-level configuration -- e.g. `benchmarks.jl` calls +this once, before `SUITE` is built, so that the thread-count choice is entirely outside of +anything ever timed. +""" +function set_threads!(cfg::ThreadConfig) + cfg.blas === nothing || BLAS.set_num_threads(cfg.blas) + cfg.strided === nothing || Strided.set_num_threads(cfg.strided) + return nothing +end + +""" + with_threads(f, cfg::ThreadConfig) + +Run `f()` with `BLAS`/`Strided` thread counts set according to `cfg`, restoring the prior +counts afterwards (even if `f` throws). Intended for interactively comparing a couple of +thread counts around a whole `run(suite)`/`benchmarkpkg(...)` call -- never around a single +`@benchmarkable` case, since that would count the thread-count switch itself as part of the +timed operation. +""" +function with_threads(f, cfg::ThreadConfig) + oldblas = BLAS.get_num_threads() + oldstrided = Strided.get_num_threads() + try + set_threads!(cfg) + return f() + finally + BLAS.set_num_threads(oldblas) + Strided.set_num_threads(oldstrided) + end +end From 34ce1d093ea55bcafaacb5d38cd5849f34b08d0c Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:14:51 -0400 Subject: [PATCH 02/18] Add mixed-precision and MPS/MPO benchmark categories mixed_precision: differing input/output element types (e.g. Float32 x Float32 -> Float64, mixed real/complex), exercising promote_add/ promote_contract. mps: the MPS/MPO DMRG effective-Hamiltonian motif (1-site and 2-site "theta" variants), swept over bond dimension D -- the highest-value tensor-network contraction pattern from the literature review. Co-Authored-By: Claude Sonnet 5 --- benchmark/src/TensorOperationsBenchmarks.jl | 2 + benchmark/src/categories/mixed_precision.jl | 31 ++++++++++++ benchmark/src/categories/mps.jl | 54 +++++++++++++++++++++ 3 files changed, 87 insertions(+) create mode 100644 benchmark/src/categories/mixed_precision.jl create mode 100644 benchmark/src/categories/mps.jl diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index 1f6904e1..87e235e5 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -19,6 +19,8 @@ include("report.jl") include("categories/pairwise.jl") include("categories/permute.jl") include("categories/trace.jl") +include("categories/mixed_precision.jl") +include("categories/mps.jl") export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, NetworkSpec export flops, bytes diff --git a/benchmark/src/categories/mixed_precision.jl b/benchmark/src/categories/mixed_precision.jl new file mode 100644 index 00000000..ff07209f --- /dev/null +++ b/benchmark/src/categories/mixed_precision.jl @@ -0,0 +1,31 @@ +# Mixed input/output element-type contractions: TensorOperations.jl's `promote_add`/ +# `promote_contract` (Base.promote_op-based, see src/implementation/allocator.jl) already +# support tensors of differing element types -- e.g. a Float64 tensor traced against a +# ComplexF64 one, or a (Float64, ComplexF64) -> ComplexF32 contraction (both exercised in +# test/methods.jl). This is a real, tested path and gets its own category rather than being +# folded into same-eltype pairwise contractions. + +const MIXED_PRECISION_COMBOS = ( + (Float32, Float32, Float64), # low-precision inputs, high-precision accumulation + (Float64, ComplexF64, ComplexF64), # mixed real/complex + (Float64, ComplexF64, ComplexF32), # mixed real/complex, downcast output +) + +function _mixed_precision_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + for (TA, TB, TC) in MIXED_PRECISION_COMBOS + IA = [:a1, :c1] + IB = [:c1, :b1] + IC = [:a1, :b1] + dims = Dict{Symbol, Int}(:a1 => dim, :b1 => dim, :c1 => dim) + spec = ContractSpec(IA, IB, IC, dims; TA, TB, TC) + within_memory_budget(spec) || continue + id = "dim$(dim)_$(TA)_$(TB)_$(TC)" + push!(cases, BenchmarkCase(:mixed_precision, id, (; dim, TA, TB, TC), spec)) + end + end + return cases +end + +register_category!(:mixed_precision, _mixed_precision_cases; sizes = (32, 128, 512)) diff --git a/benchmark/src/categories/mps.jl b/benchmark/src/categories/mps.jl new file mode 100644 index 00000000..46d9c670 --- /dev/null +++ b/benchmark/src/categories/mps.jl @@ -0,0 +1,54 @@ +# The MPS/MPO DMRG effective-Hamiltonian motif: applying `H_eff = L - W - R` to an MPS +# tensor, i.e. `environment(D,D,w) x MPS(D,d,D) x MPO(w,d,d,w) x environment(D,D,w)`, the +# dominant cost in every DMRG-style sweep (cost ~ O(D^3*d*w + D^2*d^2*w^2)). `sizes` is a list +# of bond dimensions `D` to sweep; physical dimension `d` and MPO bond `w` are held fixed +# (representative of a local spin/Hubbard-like model) since the literature identifies `D` as +# the dominant scaling knob. +# +# Also includes the 2-site "theta" tensor variant (`L - W - W - R` applied to a 2-site ket), +# which is what feeds the SVD/truncation step in 2-site DMRG. + +const MPS_PHYS_DIM = 2 # d: physical dimension (spin-1/2) +const MPS_MPO_BOND = 6 # w: MPO bond dimension (local Hamiltonian) + +function _mps_1site_case(D) + d, w = MPS_PHYS_DIM, MPS_MPO_BOND + # labels: 1=mpoL bond, 2=ket-L bond, 3=phys-in, 4=ket-R bond, 5=mpoR bond + # output (negative): -10=out-L bond, -11=out-phys, -12=out-R bond + indexlists = [ + [-10, 1, 2], # L: (D, w, D) + [2, 3, 4], # ket: (D, d, D) + [1, -11, 3, 5], # MPO: (w, d, d, w) + [4, 5, -12], # R: (D, w, D) + ] + dims = Dict(1 => w, 2 => D, 3 => d, 4 => D, 5 => w, 10 => D, 11 => d, 12 => D) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12]) + return BenchmarkCase(:mps, "1site_D$(D)", (; D, d, w, variant = :onesite), spec) +end + +function _mps_2site_case(D) + d, w = MPS_PHYS_DIM, MPS_MPO_BOND + # labels: 1=mpoL bond, 2=ket-L bond, 3=phys-in(1), 4=ket-mid bond, 5=phys-in(2), + # 6=mpo-mid bond, 7=mpoR bond, 8=ket-R bond + # output (negative): -10=out-L bond, -11=out-phys(1), -13=out-phys(2), -12=out-R bond + indexlists = [ + [-10, 1, 2], # L: (D, w, D) + [2, 3, 4], # ket1: (D, d, D) + [4, 5, 8], # ket2: (D, d, D) + [1, -11, 3, 6], # MPO1: (w, d, d, w) + [6, -13, 5, 7], # MPO2: (w, d, d, w) + [8, 7, -12], # R: (D, w, D) + ] + dims = Dict(1 => w, 2 => D, 3 => d, 4 => D, 5 => d, 6 => w, 7 => w, 8 => D, 10 => D, 11 => d, 12 => D, 13 => d) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -13, -12]) + return BenchmarkCase(:mps, "2site_D$(D)", (; D, d, w, variant = :twosite), spec) +end + +function _mps_cases(sizes) + return vcat( + BenchmarkCase[_mps_1site_case(D) for D in sizes], + BenchmarkCase[_mps_2site_case(D) for D in sizes] + ) +end + +register_category!(:mps, _mps_cases; sizes = (32, 64, 128, 256, 512)) From 92eec4cfcd364a58bc3ee0c277f0db2ddb6c1da3 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:14:59 -0400 Subject: [PATCH 03/18] Add PkgBenchmark entrypoint and CLI scripts benchmarks.jl builds SUITE for PkgBenchmark, applying thread config once via set_threads! before building it. run_benchmarks.jl/show_benchmarks.jl (ArgParse-based CLIs) drive PkgBenchmark runs and plot GFLOP/s-vs-size curves from the results. Co-Authored-By: Claude Sonnet 5 --- benchmark/benchmarks.jl | 20 ++++++++++ benchmark/scripts/run_benchmarks.jl | 60 ++++++++++++++++++++++++++++ benchmark/scripts/show_benchmarks.jl | 56 ++++++++++++++++++++++++++ 3 files changed, 136 insertions(+) create mode 100644 benchmark/benchmarks.jl create mode 100644 benchmark/scripts/run_benchmarks.jl create mode 100644 benchmark/scripts/show_benchmarks.jl diff --git a/benchmark/benchmarks.jl b/benchmark/benchmarks.jl new file mode 100644 index 00000000..fe187db9 --- /dev/null +++ b/benchmark/benchmarks.jl @@ -0,0 +1,20 @@ +# PkgBenchmark entrypoint: `PkgBenchmark.benchmarkpkg` looks for this file and expects a +# top-level `const SUITE`. Thread counts are read from environment variables and applied via +# `set_threads!` *before* `SUITE` is built -- once, at the process level -- rather than swept as +# a suite axis, so that no timed sample ever pays for a `BLAS`/`Strided` thread-count switch +# (see threading.jl). `scripts/run_benchmarks.jl` sets these env vars per outer sweep point. +using TensorOperationsBenchmarks +using TensorOperations: StridedNative, StridedBLAS + +set_threads!( + ThreadConfig(; + blas = tryparse(Int, get(ENV, "TOB_BLAS_THREADS", "")), + strided = tryparse(Int, get(ENV, "TOB_STRIDED_THREADS", "")), + ) +) + +const ELTYPES = (Float64, ComplexF64) +const BACKENDS = (StridedNative(), StridedBLAS()) +const PROVIDERS = [ArrayProvider{T}(; backend) for T in ELTYPES for backend in BACKENDS] + +const SUITE = build_suite(PROVIDERS) diff --git a/benchmark/scripts/run_benchmarks.jl b/benchmark/scripts/run_benchmarks.jl new file mode 100644 index 00000000..7671b47d --- /dev/null +++ b/benchmark/scripts/run_benchmarks.jl @@ -0,0 +1,60 @@ +#!/usr/bin/env julia +# CLI wrapper around PkgBenchmark.benchmarkpkg, mirroring the old `ld/benchmark` runner's +# flags. Run from the `benchmark/` directory, e.g.: +# +# julia --project=. scripts/run_benchmarks.jl --threads 1 2 4 --blas-threads 1 4 --out results +# +# Threads: `--threads` sweeps the *outer* Julia process thread count (`-t`), which is fixed at +# startup and therefore requires relaunching Julia once per value (via PkgBenchmark's +# `juliacmd`). `--blas-threads`/`--strided-threads` sweep the *inner* BLAS/Strided thread +# counts, applied once per run via `TOB_BLAS_THREADS`/`TOB_STRIDED_THREADS` env vars that +# `benchmarks.jl` reads and applies with `set_threads!` before `SUITE` is built -- never as a +# per-case axis (see threading.jl for why). +using Pkg +Pkg.activate(@__DIR__ * "/..") + +using ArgParse +using PkgBenchmark + +function parse_commandline() + s = ArgParseSettings(; description = "Run the TensorOperationsBenchmarks suite via PkgBenchmark.") + @add_arg_table! s begin + "--threads" + help = "outer Julia process thread count(s) to sweep (relaunches Julia per value)" + arg_type = Int + nargs = '*' + default = [Threads.nthreads()] + "--blas-threads" + help = "inner BLAS thread count(s) to sweep" + arg_type = Int + nargs = '*' + default = Int[] + "--strided-threads" + help = "inner Strided.jl thread count(s) to sweep" + arg_type = Int + nargs = '*' + default = Int[] + "--out" + help = "output file prefix (a suffix identifying the thread combo and `.json` are appended)" + default = "results" + end + return parse_args(s) +end + +opts = parse_commandline() +blascounts = isempty(opts["blas-threads"]) ? [nothing] : opts["blas-threads"] +stridedcounts = isempty(opts["strided-threads"]) ? [nothing] : opts["strided-threads"] + +for nthreads in opts["threads"], blas in blascounts, strided in stridedcounts + @info "Running benchmarks" nthreads blas strided + withenv( + "TOB_BLAS_THREADS" => blas === nothing ? "" : string(blas), + "TOB_STRIDED_THREADS" => strided === nothing ? "" : string(strided), + ) do + cfg = BenchmarkConfig(; juliacmd = `julia -t $nthreads -O3`) + results = benchmarkpkg(dirname(@__DIR__), cfg) + outfile = "$(opts["out"])_t$(nthreads)_blas$(blas)_strided$(strided).json" + writeresults(outfile, results) + @info "Wrote $outfile" + end +end diff --git a/benchmark/scripts/show_benchmarks.jl b/benchmark/scripts/show_benchmarks.jl new file mode 100644 index 00000000..e616f2db --- /dev/null +++ b/benchmark/scripts/show_benchmarks.jl @@ -0,0 +1,56 @@ +#!/usr/bin/env julia +# Plots time/GFLOPs-vs-size scaling curves from a PkgBenchmark result JSON (as written by +# `run_benchmarks.jl`). Not part of the main package's dependencies (CairoMakie is heavy) -- +# install it into the active environment yourself first: +# +# julia --project=. -e 'using Pkg; Pkg.add("CairoMakie")' +# julia --project=. scripts/show_benchmarks.jl results_t4_blas1_strided1.json +using Pkg +Pkg.activate(@__DIR__ * "/..") + +using ArgParse +using PkgBenchmark +using CairoMakie +using TensorOperationsBenchmarks + +function parse_commandline() + s = ArgParseSettings(; description = "Plot GFLOP/s-vs-size scaling curves from a PkgBenchmark result.") + @add_arg_table! s begin + "resultfile" + help = "path to a PkgBenchmark result JSON, as written by run_benchmarks.jl" + required = true + "--out" + help = "output image path (default: replace the input's extension with .png)" + default = nothing + end + return parse_args(s) +end + +opts = parse_commandline() +results = PkgBenchmark.readresults(opts["resultfile"]) +group = PkgBenchmark.benchmarkgroup(results) + +rows = resultstable(group) + +fig = Figure(; size = (1000, 800)) +categories = unique(r.category for r in rows) +for (i, category) in enumerate(categories) + ax = Axis( + fig[fldmod1(i, 2)...]; xscale = log2, yscale = log10, + title = category, xlabel = "size", ylabel = "GFLOP/s" + ) + catrows = filter(r -> r.category == category, rows) + for provider in unique(r.provider for r in catrows) + provrows = filter(r -> r.provider == provider, catrows) + sort!(provrows; by = r -> get(r.params, :dim, get(r.params, :D, 0))) + xs = [get(r.params, :dim, get(r.params, :D, 0)) for r in provrows] + ys = [r.gflops for r in provrows] + lines!(ax, xs, ys; label = provider) + scatter!(ax, xs, ys) + end + axislegend(ax) +end + +outfile = something(opts["out"], splitext(opts["resultfile"])[1] * ".png") +save(outfile, fig) +@info "Wrote $outfile" From c0bf8eef563c2be1dd94b23fca146ff33f256ef5 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:15:04 -0400 Subject: [PATCH 04/18] Add test suite for TensorOperationsBenchmarks Covers suite construction/execution for every category, resultstable's flop/byte joins, the network cost model, thread-config scoping, and ArrayProvider's reproducible-but-varying randtensor. Co-Authored-By: Claude Sonnet 5 --- benchmark/test/runtests.jl | 79 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 79 insertions(+) create mode 100644 benchmark/test/runtests.jl diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl new file mode 100644 index 00000000..e94d07eb --- /dev/null +++ b/benchmark/test/runtests.jl @@ -0,0 +1,79 @@ +using Test +using TensorOperationsBenchmarks +using TensorOperations: StridedNative +using BenchmarkTools +using LinearAlgebra: BLAS +using Strided: Strided + +@testset "TensorOperationsBenchmarks" begin + provider = ArrayProvider{Float64}(; backend = StridedNative()) + + @testset "every category builds and runs" begin + suite = build_suite([provider]; categories = collect(keys(REGISTRY)), sizes = (4, 8)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + for category in keys(REGISTRY) + @test haskey(results, String(category)) + end + end + + @testset "resultstable joins timings with flop/byte counts" begin + suite = build_suite([provider]; categories = [:pairwise], sizes = (4, 8)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + rows = resultstable(results; categories = [:pairwise], sizes = (4, 8)) + @test !isempty(rows) + @test all(r -> r.mintime > 0, rows) + @test all(r -> r.gflops isa Float64, rows) + end + + @testset "mixed-precision cases execute" begin + suite = build_suite([provider]; categories = [:mixed_precision], sizes = (8,)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + @test !isempty(results["mixed_precision"][label(provider)]) + end + + @testset "network cost matches ncon's own contraction tree" begin + # for a linear chain A-B-C-D (as in the MPS motif), ncon's greedy tree contracts + # increasing labels first (1, then 2, then 3) -- confirm flops(spec) reflects that + # order rather than an arbitrary pairing. + cases = REGISTRY[:mps]((16,)) + case = only(filter(c -> occursin("1site", c.id), cases)) + @test flops(case.spec) > 0 + end + + @testset "with_threads restores prior thread counts, set_threads! does not" begin + oldblas = BLAS.get_num_threads() + oldstrided = Strided.get_num_threads() + with_threads(() -> nothing, ThreadConfig(; blas = 1, strided = 1)) + @test BLAS.get_num_threads() == oldblas + @test Strided.get_num_threads() == oldstrided + + set_threads!(ThreadConfig(; blas = 1)) + @test BLAS.get_num_threads() == 1 + BLAS.set_num_threads(oldblas) # restore for subsequent tests + end + + @testset "ArrayProvider's randtensor is reproducible across providers, varies within one" begin + p1 = ArrayProvider{Float64}() + p2 = ArrayProvider{Float64}() + A1 = randtensor(p1, [:a, :b], (4, 4), Float64) + A2 = randtensor(p2, [:a, :b], (4, 4), Float64) + @test A1 == A2 # same fixed seed -> same first draw + B1 = randtensor(p1, [:a, :b], (4, 4), Float64) + @test A1 != B1 # stateful rng -> second draw differs from the first + end + + @testset "execute doesn't allocate a fresh output every call (preallocated in setup)" begin + cases = REGISTRY[:pairwise]((8,)) + case = first(cases) + ts = TensorOperationsBenchmarks.maketensors(case.spec, provider) + C_before = ts[end] + C_after = TensorOperationsBenchmarks.execute(case.spec, ts, provider) + @test C_after === C_before + end + + @testset "mps category covers 1-site and 2-site variants" begin + cases = REGISTRY[:mps]((16,)) + @test any(c -> occursin("1site", c.id), cases) + @test any(c -> occursin("2site", c.id), cases) + end +end From 791240fd25759294d1a4a231fb1b1f888fbe0efa Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:15:08 -0400 Subject: [PATCH 05/18] Add README for TensorOperationsBenchmarks Co-Authored-By: Claude Sonnet 5 --- benchmark/README.md | 71 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 71 insertions(+) create mode 100644 benchmark/README.md diff --git a/benchmark/README.md b/benchmark/README.md new file mode 100644 index 00000000..fe80c351 --- /dev/null +++ b/benchmark/README.md @@ -0,0 +1,71 @@ +# TensorOperationsBenchmarks + +Extensible benchmark suite for TensorOperations.jl. Downstream packages (e.g. a +symmetric/block-sparse tensor package) can plug in their own tensor type and run the same +standardized shapes. + +## Running + +```julia +julia --project=. -e ' + using TensorOperationsBenchmarks + using TensorOperations: StridedNative, StridedBLAS + providers = [ArrayProvider{Float64}(; backend=StridedNative()), + ArrayProvider{Float64}(; backend=StridedBLAS())] + suite = build_suite(providers) + results = run(suite) + rows = resultstable(results) +' +``` + +Or for commit-to-commit regression comparison via PkgBenchmark: + +``` +julia --project=. scripts/run_benchmarks.jl --threads 1 4 --blas-threads 1 4 +julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # requires CairoMakie +``` + +## Categories (v1) + +- `:pairwise` -- generic pairwise contractions across open/contracted/batch index splits. +- `:permute` -- permutation-only (`tensorcopy!`) cost. +- `:trace` -- partial and full traces. +- `:mixed_precision` -- differing input/output element types (e.g. `Float32 x Float32 -> + Float64`, mixed real/complex). +- `:mps` -- MPS/MPO DMRG effective-Hamiltonian motif (1-site and 2-site "theta"), swept over + bond dimension `D`. + +Not yet implemented, but addable without a redesign: CTMRG/PEPS, TRG/MERA, +contraction-order/path-finding timing. + +## Adding a category + +New file under `src/categories/`, define `mysizes -> Vector{BenchmarkCase}` building +`ContractSpec`/`TraceSpec`/`AddSpec`/`NetworkSpec` values, `include` it, call +`register_category!(:mycategory, mygenerator)`. Nothing else changes. + +## Plugging in a downstream tensor type + +```julia +struct MyProvider <: AbstractProvider end +TensorOperations.scalartype(::MyProvider) = Float64 +TensorOperationsBenchmarks.randtensor(::MyProvider, labels, dims, T) = # build a random tensor +# optional: backend(p), allocator(p), label(p), supports(p, category), rng(p) +``` + +Then `build_suite([MyProvider(), ArrayProvider{Float64}()])` compares directly against +TensorOperations.jl's own backends. `ArrayProvider` shows the recommended `randtensor` pattern: +allocate via `TensorOperations.tensoralloc` (goes through `allocator`) and fill from a stored, +stateful `rng` field seeded once at construction, so runs are reproducible but tensors within a +run still differ. + +## Threading and precision + +Thread config is not a `build_suite` axis -- setting it per-case would count the switch itself +as part of the timing. Instead `set_threads!` is called once, at the process level, before a +suite is built (`benchmarks.jl` reads `TOB_BLAS_THREADS`/`TOB_STRIDED_THREADS`, set by +`run_benchmarks.jl --blas-threads/--strided-threads`). `with_threads(f, cfg)` is available for +ad hoc comparisons, wrapping a whole `run(suite)` call and restoring afterwards. + +Mixed precision: set `TA`/`TB`/`TC` explicitly on a spec (see `mixed_precision.jl`); other +categories leave them `nothing` (provider's default `scalartype`). From 8b7c208b4253be988088d01324c82ece79ae4a2d Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:40:32 -0400 Subject: [PATCH 06/18] Diversify default size sweeps with non-power-of-two dimensions An all-power-of-two sweep (8,32,64,128,256) hides alignment/padding/ vectorization-boundary effects. Mix in off-by-one and arbitrary sizes (15, 63, 96, 200) for pairwise/permute/trace, and realistic (non-power- of-two) DMRG bond dimensions (100, 300) for mps. Co-Authored-By: Claude Sonnet 5 --- benchmark/src/categories/mixed_precision.jl | 2 +- benchmark/src/categories/mps.jl | 5 +++-- benchmark/src/categories/pairwise.jl | 6 +++++- benchmark/src/categories/permute.jl | 5 +++-- benchmark/src/categories/trace.jl | 2 +- 5 files changed, 13 insertions(+), 7 deletions(-) diff --git a/benchmark/src/categories/mixed_precision.jl b/benchmark/src/categories/mixed_precision.jl index ff07209f..cb8af8af 100644 --- a/benchmark/src/categories/mixed_precision.jl +++ b/benchmark/src/categories/mixed_precision.jl @@ -28,4 +28,4 @@ function _mixed_precision_cases(sizes) return cases end -register_category!(:mixed_precision, _mixed_precision_cases; sizes = (32, 128, 512)) +register_category!(:mixed_precision, _mixed_precision_cases; sizes = (33, 129, 500)) diff --git a/benchmark/src/categories/mps.jl b/benchmark/src/categories/mps.jl index 46d9c670..09987c86 100644 --- a/benchmark/src/categories/mps.jl +++ b/benchmark/src/categories/mps.jl @@ -3,7 +3,8 @@ # dominant cost in every DMRG-style sweep (cost ~ O(D^3*d*w + D^2*d^2*w^2)). `sizes` is a list # of bond dimensions `D` to sweep; physical dimension `d` and MPO bond `w` are held fixed # (representative of a local spin/Hubbard-like model) since the literature identifies `D` as -# the dominant scaling knob. +# the dominant scaling knob. The default sweep mixes power-of-two anchors with bond dimensions +# actually seen in production DMRG (100, 300), rather than an all-power-of-two ladder. # # Also includes the 2-site "theta" tensor variant (`L - W - W - R` applied to a 2-site ket), # which is what feeds the SVD/truncation step in 2-site DMRG. @@ -51,4 +52,4 @@ function _mps_cases(sizes) ) end -register_category!(:mps, _mps_cases; sizes = (32, 64, 128, 256, 512)) +register_category!(:mps, _mps_cases; sizes = (32, 48, 64, 100, 128, 256, 300, 512)) diff --git a/benchmark/src/categories/pairwise.jl b/benchmark/src/categories/pairwise.jl index 4e59138a..12ce3467 100644 --- a/benchmark/src/categories/pairwise.jl +++ b/benchmark/src/categories/pairwise.jl @@ -2,6 +2,10 @@ # not a regex-parsed `.dat` file) from the shapes explored in the stale `ld/benchmark` # prototype. `sizes` is a list of leg dimensions to sweep; every dimension is used at every # `(nopenA, ncontract, nopenB)` shape below, giving a scaling curve per shape. +# +# The default sweep deliberately mixes powers of two with off-by-one and arbitrary sizes +# (15, 63, 96, 200) -- an all-power-of-two sweep hides alignment/padding/vectorization-boundary +# effects that only show up at sizes a SIMD width or cache line doesn't divide evenly. const PAIRWISE_SHAPES = ( (1, 1, 1), # matrix-vector-like @@ -31,4 +35,4 @@ function _pairwise_cases(sizes) return cases end -register_category!(:pairwise, _pairwise_cases; sizes = (8, 32, 64, 128, 256)) +register_category!(:pairwise, _pairwise_cases; sizes = (8, 15, 32, 63, 96, 128, 200, 256)) diff --git a/benchmark/src/categories/permute.jl b/benchmark/src/categories/permute.jl index d1ede9f3..1fb4d840 100644 --- a/benchmark/src/categories/permute.jl +++ b/benchmark/src/categories/permute.jl @@ -4,7 +4,8 @@ # # `sizes` is a list of leg dimensions; for each dimension we benchmark permutations of a # rank-4 tensor across a spread of "how scrambled" the permutation is (identity-adjacent vs. -# fully reversed). +# fully reversed). The default sweep mixes powers of two with non-power-of-two sizes (15, 63, +# 96, 200) to catch alignment/padding effects that a clean power-of-two ladder would hide. const PERMUTE_PATTERNS = ( [1, 2, 3, 4], # identity (still exercises the copy machinery, no real permutation) @@ -29,4 +30,4 @@ function _permute_cases(sizes) return cases end -register_category!(:permute, _permute_cases; sizes = (8, 32, 64, 128, 256)) +register_category!(:permute, _permute_cases; sizes = (8, 15, 32, 63, 96, 128, 200, 256)) diff --git a/benchmark/src/categories/trace.jl b/benchmark/src/categories/trace.jl index 3dc9868d..8e732e15 100644 --- a/benchmark/src/categories/trace.jl +++ b/benchmark/src/categories/trace.jl @@ -32,4 +32,4 @@ function _trace_cases(sizes) return cases end -register_category!(:trace, _trace_cases; sizes = (8, 32, 64, 128, 256)) +register_category!(:trace, _trace_cases; sizes = (8, 15, 32, 63, 96, 128, 200, 256)) From c5ddc3db0cd7f4131b507b0577c3e6054bb06256 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:40:42 -0400 Subject: [PATCH 07/18] Add TCCG quantum-chemistry contraction category 24 real contraction patterns (CCSD, CCSD(T), AO2MO, INTENSLI) sourced from the TCCG benchmark (github.com/HPAC/tccg), a source of real high-rank/irregular-permutation shapes rather than hand-picked ones. Sizes each index uniformly, mirroring TCCG's own approach. Co-Authored-By: Claude Sonnet 5 --- benchmark/README.md | 3 + benchmark/src/TensorOperationsBenchmarks.jl | 1 + benchmark/src/categories/tccg.jl | 62 +++++++++++++++++++++ benchmark/test/runtests.jl | 11 ++++ 4 files changed, 77 insertions(+) create mode 100644 benchmark/src/categories/tccg.jl diff --git a/benchmark/README.md b/benchmark/README.md index fe80c351..0be47a79 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -34,6 +34,9 @@ julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # r Float64`, mixed real/complex). - `:mps` -- MPS/MPO DMRG effective-Hamiltonian motif (1-site and 2-site "theta"), swept over bond dimension `D`. +- `:tccg` -- 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from the + [TCCG benchmark](https://github.com/HPAC/tccg); a source of real high-rank, irregular + index-split shapes rather than hand-picked ones. Not yet implemented, but addable without a redesign: CTMRG/PEPS, TRG/MERA, contraction-order/path-finding timing. diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index 87e235e5..5b7391d5 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -21,6 +21,7 @@ include("categories/permute.jl") include("categories/trace.jl") include("categories/mixed_precision.jl") include("categories/mps.jl") +include("categories/tccg.jl") export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, NetworkSpec export flops, bytes diff --git a/benchmark/src/categories/tccg.jl b/benchmark/src/categories/tccg.jl new file mode 100644 index 00000000..1967a3c3 --- /dev/null +++ b/benchmark/src/categories/tccg.jl @@ -0,0 +1,62 @@ +# Real contraction patterns from the TCCG benchmark (Springer et al., HPAC group, +# github.com/HPAC/tccg -- the "Springer benchmark"), harvested from 4 quantum-chemistry +# papers: CCSD, CCSD(T), AO2MO integral transformation, and INTENSLI. Pulled from +# `benchmark/benchmark.py`'s own "C-A-B" index-string notation (output-A-B), e.g. "ij-ik-kj" +# means `C[i,j] = A[i,k] * B[k,j]`. +# +# These are quantum-chemistry contractions, not tensor-network ones -- included as a source of +# real (rather than synthetic) high-rank, irregular open/contracted/batch index splits, to +# diversify the otherwise hand-picked shapes in `:pairwise`. TCCG itself sizes every index +# uniformly to hit a target total tensor memory, rather than giving occupied/virtual orbitals +# different sizes; `sizes` mirrors that -- a single leg dimension applied to every distinct +# index letter in a given equation. + +const TCCG_CONTRACTIONS = ( + # CCSD + (id = "ccsd_1", C = "ij", A = "ik", B = "kj"), + (id = "ccsd_2", C = "ij", A = "ikl", B = "ljk"), + (id = "ccsd_3", C = "ij", A = "kil", B = "lkj"), + (id = "ccsd_4", C = "ijk", A = "ikl", B = "lj"), + (id = "ccsd_5", C = "ijk", A = "ilk", B = "jl"), + (id = "ccsd_6", C = "ijk", A = "ilmk", B = "mjl"), + (id = "ccsd_7", C = "ijkl", A = "imjn", B = "lnkm"), + (id = "ccsd_8", C = "ijkl", A = "imjn", B = "nlmk"), + (id = "ccsd_9", C = "ijkl", A = "minl", B = "njmk"), + # CCSD(T) + (id = "ccsd_t_1", C = "abcijk", A = "ijma", B = "mkbc"), + (id = "ccsd_t_2", C = "abcijk", A = "ijmb", B = "mkac"), + (id = "ccsd_t_3", C = "abcijk", A = "ijmc", B = "mkab"), + (id = "ccsd_t_4", C = "abcijk", A = "ikmb", B = "mjac"), + # AO2MO integral transformation + (id = "ao2mo_1", C = "aqrs", A = "pa", B = "pqrs"), + (id = "ao2mo_2", C = "abrs", A = "qb", B = "aqrs"), + (id = "ao2mo_3", C = "abcs", A = "rc", B = "abrs"), + # INTENSLI + (id = "intensli_1", C = "abj", A = "bka", B = "kj"), + (id = "intensli_2", C = "ajb", A = "kba", B = "jk"), + (id = "intensli_3", C = "abjc", A = "cbka", B = "kj"), + (id = "intensli_4", C = "ajbc", A = "ckba", B = "jk"), + (id = "intensli_5", C = "abjc", A = "kbac", B = "jk"), + (id = "intensli_6", C = "abjcd", A = "dkbac", B = "kj"), + (id = "intensli_7", C = "adbjc", A = "cbdka", B = "kj"), + (id = "intensli_8", C = "ajbdc", A = "ckbad", B = "jk"), +) + +_tccg_labels(s::AbstractString) = [Symbol(c) for c in s] + +function _tccg_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + for eq in TCCG_CONTRACTIONS + IA, IB, IC = _tccg_labels(eq.A), _tccg_labels(eq.B), _tccg_labels(eq.C) + dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) + spec = ContractSpec(IA, IB, IC, dims) + within_memory_budget(spec) || continue + id = "$(eq.id)_dim$(dim)" + push!(cases, BenchmarkCase(:tccg, id, (; dim, equation = eq.id), spec)) + end + end + return cases +end + +register_category!(:tccg, _tccg_cases; sizes = (4, 8, 12, 16, 24, 32)) diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index e94d07eb..7dafebcf 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -76,4 +76,15 @@ using Strided: Strided @test any(c -> occursin("1site", c.id), cases) @test any(c -> occursin("2site", c.id), cases) end + + @testset "tccg category covers all 4 source groups and executes" begin + cases = REGISTRY[:tccg]((4,)) + @test length(TensorOperationsBenchmarks.TCCG_CONTRACTIONS) == 24 + for prefix in ("ccsd_", "ccsd_t_", "ao2mo_", "intensli_") + @test any(c -> startswith(c.params.equation, prefix), cases) + end + suite = build_suite([provider]; categories = [:tccg], sizes = (4,)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + @test !isempty(results["tccg"][label(provider)]) + end end From 3f08b0d83d0291535b11a565e20b1bc6f2a3f938 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 16:47:04 -0400 Subject: [PATCH 08/18] Trim large comment blocks throughout Cut multi-paragraph file-header prose down to one or two lines across all source, script, and test files. No behavior change. Co-Authored-By: Claude Sonnet 5 --- benchmark/benchmarks.jl | 7 +--- benchmark/scripts/run_benchmarks.jl | 12 +----- benchmark/scripts/show_benchmarks.jl | 8 +--- benchmark/src/categories/mixed_precision.jl | 7 +--- benchmark/src/categories/mps.jl | 12 +----- benchmark/src/categories/pairwise.jl | 10 +---- benchmark/src/categories/permute.jl | 10 +---- benchmark/src/categories/tccg.jl | 15 ++------ benchmark/src/categories/trace.jl | 9 +---- benchmark/src/cost.jl | 25 +++---------- benchmark/src/lowering.jl | 18 ++------- benchmark/src/provider.jl | 41 +++++---------------- benchmark/src/registry.jl | 4 +- benchmark/src/report.jl | 6 +-- benchmark/src/specs.jl | 11 ++---- benchmark/src/suite.jl | 6 +-- benchmark/src/threading.jl | 30 ++++----------- benchmark/test/runtests.jl | 3 -- 18 files changed, 52 insertions(+), 182 deletions(-) diff --git a/benchmark/benchmarks.jl b/benchmark/benchmarks.jl index fe187db9..cb60e373 100644 --- a/benchmark/benchmarks.jl +++ b/benchmark/benchmarks.jl @@ -1,8 +1,5 @@ -# PkgBenchmark entrypoint: `PkgBenchmark.benchmarkpkg` looks for this file and expects a -# top-level `const SUITE`. Thread counts are read from environment variables and applied via -# `set_threads!` *before* `SUITE` is built -- once, at the process level -- rather than swept as -# a suite axis, so that no timed sample ever pays for a `BLAS`/`Strided` thread-count switch -# (see threading.jl). `scripts/run_benchmarks.jl` sets these env vars per outer sweep point. +# PkgBenchmark entrypoint: expects a top-level `const SUITE`. Thread counts come from env vars +# (set by scripts/run_benchmarks.jl) and are applied once, before SUITE is built. using TensorOperationsBenchmarks using TensorOperations: StridedNative, StridedBLAS diff --git a/benchmark/scripts/run_benchmarks.jl b/benchmark/scripts/run_benchmarks.jl index 7671b47d..dcab63c5 100644 --- a/benchmark/scripts/run_benchmarks.jl +++ b/benchmark/scripts/run_benchmarks.jl @@ -1,15 +1,7 @@ #!/usr/bin/env julia -# CLI wrapper around PkgBenchmark.benchmarkpkg, mirroring the old `ld/benchmark` runner's -# flags. Run from the `benchmark/` directory, e.g.: -# +# CLI wrapper around PkgBenchmark.benchmarkpkg. `--threads` relaunches Julia per value (fixed +# at startup); `--blas-threads`/`--strided-threads` set env vars benchmarks.jl reads. # julia --project=. scripts/run_benchmarks.jl --threads 1 2 4 --blas-threads 1 4 --out results -# -# Threads: `--threads` sweeps the *outer* Julia process thread count (`-t`), which is fixed at -# startup and therefore requires relaunching Julia once per value (via PkgBenchmark's -# `juliacmd`). `--blas-threads`/`--strided-threads` sweep the *inner* BLAS/Strided thread -# counts, applied once per run via `TOB_BLAS_THREADS`/`TOB_STRIDED_THREADS` env vars that -# `benchmarks.jl` reads and applies with `set_threads!` before `SUITE` is built -- never as a -# per-case axis (see threading.jl for why). using Pkg Pkg.activate(@__DIR__ * "/..") diff --git a/benchmark/scripts/show_benchmarks.jl b/benchmark/scripts/show_benchmarks.jl index e616f2db..068dd347 100644 --- a/benchmark/scripts/show_benchmarks.jl +++ b/benchmark/scripts/show_benchmarks.jl @@ -1,10 +1,6 @@ #!/usr/bin/env julia -# Plots time/GFLOPs-vs-size scaling curves from a PkgBenchmark result JSON (as written by -# `run_benchmarks.jl`). Not part of the main package's dependencies (CairoMakie is heavy) -- -# install it into the active environment yourself first: -# -# julia --project=. -e 'using Pkg; Pkg.add("CairoMakie")' -# julia --project=. scripts/show_benchmarks.jl results_t4_blas1_strided1.json +# Plots time/GFLOPs-vs-size curves from a run_benchmarks.jl result JSON. Requires CairoMakie +# (`Pkg.add("CairoMakie")` into this environment first -- not a package dependency, it's heavy). using Pkg Pkg.activate(@__DIR__ * "/..") diff --git a/benchmark/src/categories/mixed_precision.jl b/benchmark/src/categories/mixed_precision.jl index cb8af8af..de10e032 100644 --- a/benchmark/src/categories/mixed_precision.jl +++ b/benchmark/src/categories/mixed_precision.jl @@ -1,9 +1,4 @@ -# Mixed input/output element-type contractions: TensorOperations.jl's `promote_add`/ -# `promote_contract` (Base.promote_op-based, see src/implementation/allocator.jl) already -# support tensors of differing element types -- e.g. a Float64 tensor traced against a -# ComplexF64 one, or a (Float64, ComplexF64) -> ComplexF32 contraction (both exercised in -# test/methods.jl). This is a real, tested path and gets its own category rather than being -# folded into same-eltype pairwise contractions. +# Contractions with differing input/output element types (promote_add/promote_contract). const MIXED_PRECISION_COMBOS = ( (Float32, Float32, Float64), # low-precision inputs, high-precision accumulation diff --git a/benchmark/src/categories/mps.jl b/benchmark/src/categories/mps.jl index 09987c86..6d594550 100644 --- a/benchmark/src/categories/mps.jl +++ b/benchmark/src/categories/mps.jl @@ -1,13 +1,5 @@ -# The MPS/MPO DMRG effective-Hamiltonian motif: applying `H_eff = L - W - R` to an MPS -# tensor, i.e. `environment(D,D,w) x MPS(D,d,D) x MPO(w,d,d,w) x environment(D,D,w)`, the -# dominant cost in every DMRG-style sweep (cost ~ O(D^3*d*w + D^2*d^2*w^2)). `sizes` is a list -# of bond dimensions `D` to sweep; physical dimension `d` and MPO bond `w` are held fixed -# (representative of a local spin/Hubbard-like model) since the literature identifies `D` as -# the dominant scaling knob. The default sweep mixes power-of-two anchors with bond dimensions -# actually seen in production DMRG (100, 300), rather than an all-power-of-two ladder. -# -# Also includes the 2-site "theta" tensor variant (`L - W - W - R` applied to a 2-site ket), -# which is what feeds the SVD/truncation step in 2-site DMRG. +# MPS/MPO DMRG effective-Hamiltonian motif (L-MPS-MPO-R, cost ~ O(D^3*d*w)) plus the 2-site +# "theta" variant (L-MPS-MPO-MPO-MPS-R), swept over bond dimension `D`. const MPS_PHYS_DIM = 2 # d: physical dimension (spin-1/2) const MPS_MPO_BOND = 6 # w: MPO bond dimension (local Hamiltonian) diff --git a/benchmark/src/categories/pairwise.jl b/benchmark/src/categories/pairwise.jl index 12ce3467..ec07e8b9 100644 --- a/benchmark/src/categories/pairwise.jl +++ b/benchmark/src/categories/pairwise.jl @@ -1,11 +1,5 @@ -# Generic pairwise contractions of varying rank/dimension, ported (as literal Julia data, -# not a regex-parsed `.dat` file) from the shapes explored in the stale `ld/benchmark` -# prototype. `sizes` is a list of leg dimensions to sweep; every dimension is used at every -# `(nopenA, ncontract, nopenB)` shape below, giving a scaling curve per shape. -# -# The default sweep deliberately mixes powers of two with off-by-one and arbitrary sizes -# (15, 63, 96, 200) -- an all-power-of-two sweep hides alignment/padding/vectorization-boundary -# effects that only show up at sizes a SIMD width or cache line doesn't divide evenly. +# Generic pairwise contractions, swept over leg dimension `sizes` x shape. Sizes mix +# power-of-two with off-by-one/arbitrary values to catch alignment effects. const PAIRWISE_SHAPES = ( (1, 1, 1), # matrix-vector-like diff --git a/benchmark/src/categories/permute.jl b/benchmark/src/categories/permute.jl index 1fb4d840..7f72d01c 100644 --- a/benchmark/src/categories/permute.jl +++ b/benchmark/src/categories/permute.jl @@ -1,11 +1,5 @@ -# Permutation-only benchmarks (transpose cost), a first-class category in its own right: -# TBLIS/TCL-style backends exist precisely because transpose-free vs. transpose-then-GEMM -# strategies differ, so isolating pure `tensorcopy!` cost from contraction cost matters. -# -# `sizes` is a list of leg dimensions; for each dimension we benchmark permutations of a -# rank-4 tensor across a spread of "how scrambled" the permutation is (identity-adjacent vs. -# fully reversed). The default sweep mixes powers of two with non-power-of-two sizes (15, 63, -# 96, 200) to catch alignment/padding effects that a clean power-of-two ladder would hide. +# Permutation-only cost (`tensorcopy!`), isolated from contraction cost. Rank-4 tensor, sweeping +# leg dimension x how scrambled the permutation is. const PERMUTE_PATTERNS = ( [1, 2, 3, 4], # identity (still exercises the copy machinery, no real permutation) diff --git a/benchmark/src/categories/tccg.jl b/benchmark/src/categories/tccg.jl index 1967a3c3..0e3d02a7 100644 --- a/benchmark/src/categories/tccg.jl +++ b/benchmark/src/categories/tccg.jl @@ -1,15 +1,6 @@ -# Real contraction patterns from the TCCG benchmark (Springer et al., HPAC group, -# github.com/HPAC/tccg -- the "Springer benchmark"), harvested from 4 quantum-chemistry -# papers: CCSD, CCSD(T), AO2MO integral transformation, and INTENSLI. Pulled from -# `benchmark/benchmark.py`'s own "C-A-B" index-string notation (output-A-B), e.g. "ij-ik-kj" -# means `C[i,j] = A[i,k] * B[k,j]`. -# -# These are quantum-chemistry contractions, not tensor-network ones -- included as a source of -# real (rather than synthetic) high-rank, irregular open/contracted/batch index splits, to -# diversify the otherwise hand-picked shapes in `:pairwise`. TCCG itself sizes every index -# uniformly to hit a target total tensor memory, rather than giving occupied/virtual orbitals -# different sizes; `sizes` mirrors that -- a single leg dimension applied to every distinct -# index letter in a given equation. +# Real quantum-chemistry contractions from the TCCG benchmark (github.com/HPAC/tccg): CCSD, +# CCSD(T), AO2MO, INTENSLI, as "C-A-B" index strings (e.g. "ij-ik-kj" = C[i,j]=A[i,k]*B[k,j]). +# `sizes` applies one leg dimension uniformly to every index letter, as TCCG itself does. const TCCG_CONTRACTIONS = ( # CCSD diff --git a/benchmark/src/categories/trace.jl b/benchmark/src/categories/trace.jl index 8e732e15..38cf8e32 100644 --- a/benchmark/src/categories/trace.jl +++ b/benchmark/src/categories/trace.jl @@ -1,14 +1,8 @@ -# Trace-heavy benchmarks: partial traces (some legs traced, some kept open) and full traces -# (reduce all the way to a scalar), as distinct from the pairwise-contraction category. -# -# `sizes` is a list of leg dimensions; for each dimension we benchmark a rank-6 tensor traced -# down to rank-2 (partial trace) and a rank-4 tensor traced all the way to a scalar (full -# trace). +# Partial trace (rank 6 -> rank 2) and full trace (rank 4 -> scalar), swept over leg dimension. function _trace_cases(sizes) cases = BenchmarkCase[] for dim in sizes - # partial trace: rank 6 -> rank 2, trace 2 pairs, keep 2 open IA6 = [:o1, :o2, :t1, :t1, :t2, :t2] IC6 = [:o1, :o2] dims6 = Dict{Symbol, Int}(l => dim for l in unique(IA6)) @@ -20,7 +14,6 @@ function _trace_cases(sizes) ) end - # full trace: rank 4 -> scalar IA4 = [:t1, :t1, :t2, :t2] IC4 = Symbol[] dims4 = Dict{Symbol, Int}(l => dim for l in unique(IA4)) diff --git a/benchmark/src/cost.jl b/benchmark/src/cost.jl index 28e33062..babdaf3a 100644 --- a/benchmark/src/cost.jl +++ b/benchmark/src/cost.jl @@ -1,13 +1,7 @@ -# Analytical flop/byte counts computed directly from a spec (no tensor objects needed), so -# that timings can be reported as GFLOP/s or GB/s scaling curves instead of raw wall-clock time. +# Analytical flop/byte counts computed from a spec alone, so timings can be reported as +# GFLOP/s / GB/s scaling curves. -# A spec's `TA`/`TB`/`TC` (or `Ts`) are `nothing` unless the mixed-precision category set them -# explicitly -- meaning "whatever the provider's default `scalartype` turns out to be", which -# isn't known until a provider is chosen. For sizing purposes (both for reporting GB/s *before* -# a provider is picked, and for the `within_memory_budget` safety check in registry.jl, which -# runs at case-generation time) we assume the common case of `Float64`/`ComplexF64`-sized -# (8-byte) elements; this can only ever *underestimate* real memory use for a provider using a -# larger element type, never overestimate it into skipping a case that would actually fit. +# `nothing` eltype means "provider not chosen yet"; assume 8 bytes (never overestimates memory). _elsize(::Nothing) = sizeof(Float64) _elsize(T::Type) = sizeof(T) @@ -44,8 +38,7 @@ function bytes(spec::TraceSpec) return nA * _elsize(spec.TA) + nC * _elsize(spec.TC) end -# Pairwise contraction: 2 * (open-A) * (open-B) * (contracted), the standard GEMM-equivalent -# flop count, using multiply-add pairs. +# standard GEMM-equivalent flop count: 2 * open-A * open-B * contracted function flops(spec::ContractSpec) contracted = intersect(spec.IA, spec.IB) openA = setdiff(spec.IA, contracted) @@ -62,10 +55,7 @@ function bytes(spec::ContractSpec) return nA * _elsize(spec.TA) + nB * _elsize(spec.TB) + nC * _elsize(spec.TC) end -# Network: walk the *actual* pairwise contraction tree that `ncon` would build for this -# network (`TensorOperations.ncontree`/`indexordertree`, the same functions `ncon` itself -# calls), rather than an arbitrary/greedy pairing -- so the reported cost matches what -# `execute(spec::NetworkSpec, ...)` actually runs, not a guess at it. +# walks the same tree ncon itself builds (ncontree/indexordertree), not an arbitrary pairing. function flops(spec::NetworkSpec) tree = spec.order === nothing ? TensorOperations.ncontree(spec.indexlists) : TensorOperations.indexordertree(spec.indexlists, spec.order) @@ -85,10 +75,7 @@ function bytes(spec::NetworkSpec) ) end -# Recursively walks a `ncontree`/`indexordertree` result (leaves are `Int` indices into -# `spec.indexlists`, nodes are `Any[left, right]`), returning `(survivinglabels, totalcost)`, -# calling `f(labelsA, labelsB, contractedlabels)` at every pairwise step -- mirroring exactly -# how `ncon`'s own `contracttree` combines subtrees (`IC = symdiff(IA, IB)`). +# tree: leaves are Int indices into spec.indexlists, nodes are Any[left, right] (ncontree's format). function _tree_cost(f, spec::NetworkSpec, tree) tree isa Int && return spec.indexlists[tree], 0 labelsA, costA = _tree_cost(f, spec, tree[1]) diff --git a/benchmark/src/lowering.jl b/benchmark/src/lowering.jl index b2d20816..b84724e0 100644 --- a/benchmark/src/lowering.jl +++ b/benchmark/src/lowering.jl @@ -1,17 +1,7 @@ -# Turns a (spec, provider) pair into an executable BenchmarkTools benchmark. Everything that -# isn't the operation itself -- input tensors, the *output* tensor, and the low-level -# `Index2Tuple` index-permutation computation -- is built fresh in the `setup=` block (once per -# sample, not per eval), so it doesn't count against the timed operation. In particular, `C` is -# preallocated (via `tensoralloc_add`/`tensoralloc_contract`, so it goes through the provider's -# configured allocator) and execution uses the mutating `tensorcopy!`/`tensortrace!`/ -# `tensorcontract!`, so the timed region is exactly the compute kernel, not an output -# allocation. -# -# `NetworkSpec` is the one exception: `ncon` has no public in-place variant (it always -# allocates its final and intermediate results internally), so its timed region does include -# allocation. Reimplementing `ncon`'s tree contraction manually with preallocated buffers would -# let us avoid that, but is out of scope for v1 -- `ncon`-based network cases should be read as -# "cost of ncon", allocation included, not "cost of the raw contraction kernel". +# Input/output tensors are built in setup=, not timed. Output `C` is preallocated via +# tensoralloc_add/tensoralloc_contract and execution uses the mutating tensor*!, so the timed +# region is the compute kernel, not an allocation -- except NetworkSpec/ncon, which has no +# in-place variant, so its timing includes allocation. _scalartype_or(::Nothing, provider) = TensorOperations.scalartype(provider) _scalartype_or(T::Type, provider) = T diff --git a/benchmark/src/provider.jl b/benchmark/src/provider.jl index 0cde390c..8ebf9f8a 100644 --- a/benchmark/src/provider.jl +++ b/benchmark/src/provider.jl @@ -1,29 +1,13 @@ -# The downstream extension point: a `AbstractProvider` supplies the tensor type, default -# element type, backend and allocator to use when executing a (backend-agnostic) spec. This is -# deliberately a minimal interface (four required-ish methods) so that a downstream package -# (e.g. a symmetric/block-sparse tensor package) can add its own provider without touching -# anything else in this package. - """ AbstractProvider -Supertype for the downstream extension point of the benchmark suite. A provider ties a spec -(pure index/dimension data) to an actual tensor type, backend, and allocator to run it with. - -Required methods: -- `scalartype(provider)`: the provider's default element type. -- `randtensor(provider, labels, dims, T=scalartype(provider))`: build a random tensor with the - given (ordered) `labels`/`dims` and element type `T`. +The downstream extension point: ties a spec (pure index/dimension data) to an actual tensor +type, backend, and allocator. -Optional methods (with sensible defaults): -- `backend(provider) = DefaultBackend()` -- `allocator(provider) = DefaultAllocator()` -- `label(provider) = string(nameof(typeof(provider)))` -- `supports(provider, category::Symbol) = true` -- `rng(provider) = Random.default_rng()`: the RNG `randtensor` should draw from. Override - with a *stored, stateful* RNG seeded once at construction (as [`ArrayProvider`](@ref) does) - to make repeated suite runs reproducible -- a fresh RNG re-seeded on every call would just - make every tensor within a run identical, which isn't the same thing. +Required: `scalartype(provider)`, `randtensor(provider, labels, dims, T=scalartype(provider))`. +Optional: `backend`, `allocator`, `label`, `supports(provider, category)`, `rng` (defaults +below) -- override `rng` with a *stored, stateful* RNG (as [`ArrayProvider`](@ref) does) so +repeated suite runs are reproducible without every tensor in one run being identical. """ abstract type AbstractProvider end @@ -48,15 +32,10 @@ rng(p::AbstractProvider) = Random.default_rng() """ ArrayProvider{T}(; backend=DefaultBackend(), allocator=DefaultAllocator(), rng=Random.Xoshiro(0x5eed5eed5eed5eed)) -The reference provider: plain dense `Array`s of element type `T`, exercising -TensorOperations.jl's own backends (`StridedNative`, `StridedBLAS`, `cuTENSORBackend`, ...). -This is how TensorOperations.jl dogfoods its own suite. - -Tensors are allocated via `TensorOperations.tensoralloc` (so they go through `allocator`, not -a bare `Array` constructor) and filled via `randn!(rng, ...)`. `rng` is stored (not -reconstructed per call), so it advances across successive `randtensor` calls within one suite -build -- but is seeded identically across separate `ArrayProvider()` constructions, so two runs -of the same suite see the same input data. +The reference provider: plain dense `Array{T}`, exercising TensorOperations.jl's own backends. +Allocates via `TensorOperations.tensoralloc` (goes through `allocator`) and fills via +`randn!(rng, ...)`; `rng` is stored, not reseeded per call, so separate `ArrayProvider()`s +reproduce the same data while tensors within one run still differ. """ struct ArrayProvider{T, B <: AbstractBackend, A, R <: Random.AbstractRNG} <: AbstractProvider backend::B diff --git a/benchmark/src/registry.jl b/benchmark/src/registry.jl index 0377384e..c8a18598 100644 --- a/benchmark/src/registry.jl +++ b/benchmark/src/registry.jl @@ -1,6 +1,4 @@ -# A category is registered as a single generator function `sizes -> Vector{BenchmarkCase}`. -# Adding a new category to the suite is exactly: write one file defining a generator, `include` -# it, and call `register_category!` -- nothing else in the framework changes. +# A category is a generator function `sizes -> Vector{BenchmarkCase}`, registered by name. """ BenchmarkCase(category, id, params, spec) diff --git a/benchmark/src/report.jl b/benchmark/src/report.jl index 666bb607..b5fd8da0 100644 --- a/benchmark/src/report.jl +++ b/benchmark/src/report.jl @@ -1,7 +1,5 @@ -# Turns a BenchmarkTools/PkgBenchmark result `BenchmarkGroup` (as produced by running a -# `build_suite` suite) into a flat table of rows, joining the raw timings back against the -# originating specs (regenerated from `REGISTRY`) so GFLOP/s and bandwidth can be reported -# alongside wall-clock time. +# Flattens a benchmark result into rows, joining timings back against specs (regenerated from +# REGISTRY) for GFLOP/s and bandwidth. """ ResultRow diff --git a/benchmark/src/specs.jl b/benchmark/src/specs.jl index 287b3ce2..14affc81 100644 --- a/benchmark/src/specs.jl +++ b/benchmark/src/specs.jl @@ -1,11 +1,6 @@ -# Pure-data contraction/network specifications. -# -# Specs carry index labels and leg extents only -- never actual tensor objects -- so that a -# single spec can be shared across providers (different tensor types/backends) and so that -# `cost.jl` can compute flop/byte counts analytically from the spec alone. -# -# `TA`/`TB`/`TC` (and `Ts` for `NetworkSpec`) default to `nothing`, meaning "ask the provider -# for its default `scalartype`". Only the mixed-precision category sets them explicitly. +# Pure-data contraction/network specs: labels and leg extents only, no tensor objects, so a +# spec is shareable across providers. `TA`/`TB`/`TC`/`Ts` default to `nothing` ("provider's +# default scalartype"); only mixed_precision.jl sets them explicitly. abstract type AbstractCaseSpec end diff --git a/benchmark/src/suite.jl b/benchmark/src/suite.jl index ded8bc8f..2fb36a49 100644 --- a/benchmark/src/suite.jl +++ b/benchmark/src/suite.jl @@ -1,8 +1,4 @@ -# Suite assembly: nests a `BenchmarkGroup` as [category][provider label][case id]. Threading -# is deliberately NOT an axis here -- see threading.jl -- since applying it per-case would -# count the thread-count switch as part of the timed operation. `benchmarks.jl` (the -# PkgBenchmark entrypoint) is expected to be a thin wrapper: call `set_threads!` once, construct -# the `AbstractProvider`s to compare, call `build_suite`, assign the result to `const SUITE`. +# Nests a BenchmarkGroup as [category][provider label][case id]. No threading axis -- see threading.jl. """ build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing) diff --git a/benchmark/src/threading.jl b/benchmark/src/threading.jl index 292a828e..b8ea56e1 100644 --- a/benchmark/src/threading.jl +++ b/benchmark/src/threading.jl @@ -1,16 +1,7 @@ -# Threading is orthogonal to *which* tensor type/backend a provider uses (it applies -# process-wide via BLAS and Strided's own runtime thread counters), so it is not part of the -# `AbstractProvider` interface, and -- importantly -- it is NOT baked into individual -# `@benchmarkable` cases either: doing that would mean every timed sample pays for -# `BLAS.set_num_threads`/`Strided.set_num_threads` plus a closure allocation, polluting the very -# measurement it's supposed to control. Instead, a thread configuration is applied *once*, -# around an entire suite (or PkgBenchmark) run -- see `set_threads!`/`with_threads` below, and -# `benchmarks.jl`, which calls `set_threads!` once at the top level before building `SUITE`. -# -# Note: `Strided.set_num_threads` is capped by `Threads.nthreads()`, which is fixed at Julia -# process startup (`-t`/`JULIA_NUM_THREADS`). Sweeping *above* the process's thread count is not -# possible at runtime; `scripts/run_benchmarks.jl` covers that case by relaunching Julia with a -# different `-t` per outer thread count (via `PkgBenchmark`'s `juliacmd`). +# Threading is process-wide, not a provider/spec concern, and deliberately not baked into +# per-case timing (that would count the thread-count switch as part of the measurement). +# `Strided.set_num_threads` is capped by `Threads.nthreads()` (fixed at Julia startup); sweeping +# above it needs relaunching Julia -- see scripts/run_benchmarks.jl. """ ThreadConfig(; blas=nothing, strided=nothing) @@ -33,10 +24,8 @@ label(cfg::ThreadConfig) = cfg.name """ set_threads!(cfg::ThreadConfig) -Set `BLAS`/`Strided` thread counts according to `cfg`, permanently (i.e. not restored -afterwards). Intended for one-shot, process-level configuration -- e.g. `benchmarks.jl` calls -this once, before `SUITE` is built, so that the thread-count choice is entirely outside of -anything ever timed. +Set `BLAS`/`Strided` thread counts according to `cfg`, permanently. For one-shot, +process-level configuration (`benchmarks.jl` calls this before building `SUITE`). """ function set_threads!(cfg::ThreadConfig) cfg.blas === nothing || BLAS.set_num_threads(cfg.blas) @@ -47,11 +36,8 @@ end """ with_threads(f, cfg::ThreadConfig) -Run `f()` with `BLAS`/`Strided` thread counts set according to `cfg`, restoring the prior -counts afterwards (even if `f` throws). Intended for interactively comparing a couple of -thread counts around a whole `run(suite)`/`benchmarkpkg(...)` call -- never around a single -`@benchmarkable` case, since that would count the thread-count switch itself as part of the -timed operation. +Run `f()` under `cfg`, restoring prior thread counts afterwards. For comparing a couple of +configs around a whole `run(suite)` call -- never around a single `@benchmarkable` case. """ function with_threads(f, cfg::ThreadConfig) oldblas = BLAS.get_num_threads() diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 7dafebcf..d02aaf77 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -32,9 +32,6 @@ using Strided: Strided end @testset "network cost matches ncon's own contraction tree" begin - # for a linear chain A-B-C-D (as in the MPS motif), ncon's greedy tree contracts - # increasing labels first (1, then 2, then 3) -- confirm flops(spec) reflects that - # order rather than an arbitrary pairing. cases = REGISTRY[:mps]((16,)) case = only(filter(c -> occursin("1site", c.id), cases)) @test flops(case.spec) > 0 From 528dd85d8bf886c734a207732a7cb591a3691dd3 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 17:00:38 -0400 Subject: [PATCH 09/18] Add CTMRG/PEPS and TRG benchmark categories ctmrg: corner-growth step (C-T-T-a) of the boundary-MPS/CTMRG method for 2D PEPS, swept over environment bond chi, cost ~O(chi^3*D^6). trg: plaquette contraction (4-ring of rank-3 tensors from splitting neighboring TRG tensors), swept over bond chi, cost ~O(chi^6) -- confirmed numerically (6x chi -> ~45500x flops vs theoretical 46656x). Co-Authored-By: Claude Sonnet 5 --- benchmark/README.md | 5 +++-- benchmark/src/TensorOperationsBenchmarks.jl | 2 ++ benchmark/src/categories/ctmrg.jl | 22 +++++++++++++++++++++ benchmark/src/categories/trg.jl | 19 ++++++++++++++++++ benchmark/test/runtests.jl | 8 ++++++++ 5 files changed, 54 insertions(+), 2 deletions(-) create mode 100644 benchmark/src/categories/ctmrg.jl create mode 100644 benchmark/src/categories/trg.jl diff --git a/benchmark/README.md b/benchmark/README.md index 0be47a79..14637c0e 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -37,9 +37,10 @@ julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # r - `:tccg` -- 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from the [TCCG benchmark](https://github.com/HPAC/tccg); a source of real high-rank, irregular index-split shapes rather than hand-picked ones. +- `:ctmrg` -- CTMRG corner-growth step (2D PEPS boundary-MPS), swept over environment bond `chi`. +- `:trg` -- TRG plaquette contraction (4-ring of rank-3 tensors), swept over bond `chi`. -Not yet implemented, but addable without a redesign: CTMRG/PEPS, TRG/MERA, -contraction-order/path-finding timing. +Not yet implemented, but addable without a redesign: MERA, contraction-order/path-finding timing. ## Adding a category diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index 5b7391d5..b3e86d6b 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -22,6 +22,8 @@ include("categories/trace.jl") include("categories/mixed_precision.jl") include("categories/mps.jl") include("categories/tccg.jl") +include("categories/ctmrg.jl") +include("categories/trg.jl") export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, NetworkSpec export flops, bytes diff --git a/benchmark/src/categories/ctmrg.jl b/benchmark/src/categories/ctmrg.jl new file mode 100644 index 00000000..73546afa --- /dev/null +++ b/benchmark/src/categories/ctmrg.jl @@ -0,0 +1,22 @@ +# CTMRG corner-growth step (boundary-MPS method for 2D PEPS): C-T-T-a, producing an unfused +# rank-4 corner (chi,chi,D2,D2). Cost ~ O(chi^3*D2^3) (literature O(chi^3*D^6), D2=D^2). +# `sizes` sweeps environment bond `chi`; PEPS bond `D` is fixed. + +const CTMRG_PEPS_BOND = 3 # D + +function _ctmrg_case(chi) + D2 = CTMRG_PEPS_BOND^2 + indexlists = [ + [1, 2], # C: (chi, chi) + [1, 3, -10], # T_left: (chi, D2, chi) + [2, 4, -11], # T_top: (chi, D2, chi) + [3, -12, 4, -13], # a (double-layer PEPS tensor): (D2, D2, D2, D2) + ] + dims = Dict(1 => chi, 2 => chi, 3 => D2, 4 => D2, 10 => chi, 11 => chi, 12 => D2, 13 => D2) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12, -13]) + return BenchmarkCase(:ctmrg, "corner_chi$(chi)", (; chi, D = CTMRG_PEPS_BOND), spec) +end + +_ctmrg_cases(sizes) = [_ctmrg_case(chi) for chi in sizes] + +register_category!(:ctmrg, _ctmrg_cases; sizes = (16, 24, 32, 48, 64, 100)) diff --git a/benchmark/src/categories/trg.jl b/benchmark/src/categories/trg.jl new file mode 100644 index 00000000..94224aa3 --- /dev/null +++ b/benchmark/src/categories/trg.jl @@ -0,0 +1,19 @@ +# TRG plaquette contraction: a 4-ring of rank-3 tensors (from SVD-splitting neighboring rank-4 +# TRG tensors), producing the coarse-grained rank-4 tensor. Cost ~ O(chi^6). `sizes` sweeps the +# bond dimension `chi`. + +function _trg_case(chi) + indexlists = [ + [-10, 1, 2], + [2, -11, 3], + [3, -12, 4], + [4, -13, 1], + ] + dims = Dict(1 => chi, 2 => chi, 3 => chi, 4 => chi, 10 => chi, 11 => chi, 12 => chi, 13 => chi) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12, -13]) + return BenchmarkCase(:trg, "plaquette_chi$(chi)", (; chi), spec) +end + +_trg_cases(sizes) = [_trg_case(chi) for chi in sizes] + +register_category!(:trg, _trg_cases; sizes = (16, 24, 32, 48, 64, 96)) diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index d02aaf77..c1a4fb58 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -84,4 +84,12 @@ using Strided: Strided results = run(suite; samples = 1, evals = 1, seconds = 5) @test !isempty(results["tccg"][label(provider)]) end + + @testset "ctmrg and trg categories execute" begin + for category in (:ctmrg, :trg) + suite = build_suite([provider]; categories = [category], sizes = (8,)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + @test !isempty(results[String(category)][label(provider)]) + end + end end From 5c94d6af20b090c81c78892c541736ff0b77bef7 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Tue, 15 Sep 2026 17:27:05 -0400 Subject: [PATCH 10/18] Fix Runic format-check failure: @__DIR__ * string ambiguity @__DIR__ followed directly by an infix operator is ambiguous macro-call syntax that Runic's parser rejects (base Julia's parser happens to accept it). Use joinpath(@__DIR__, "..") instead, which is also more portable (Windows CI runs this repo too). Co-Authored-By: Claude Sonnet 5 --- benchmark/scripts/run_benchmarks.jl | 2 +- benchmark/scripts/show_benchmarks.jl | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/benchmark/scripts/run_benchmarks.jl b/benchmark/scripts/run_benchmarks.jl index dcab63c5..0327ab77 100644 --- a/benchmark/scripts/run_benchmarks.jl +++ b/benchmark/scripts/run_benchmarks.jl @@ -3,7 +3,7 @@ # at startup); `--blas-threads`/`--strided-threads` set env vars benchmarks.jl reads. # julia --project=. scripts/run_benchmarks.jl --threads 1 2 4 --blas-threads 1 4 --out results using Pkg -Pkg.activate(@__DIR__ * "/..") +Pkg.activate(joinpath(@__DIR__, "..")) using ArgParse using PkgBenchmark diff --git a/benchmark/scripts/show_benchmarks.jl b/benchmark/scripts/show_benchmarks.jl index 068dd347..2d2e1c58 100644 --- a/benchmark/scripts/show_benchmarks.jl +++ b/benchmark/scripts/show_benchmarks.jl @@ -2,7 +2,7 @@ # Plots time/GFLOPs-vs-size curves from a run_benchmarks.jl result JSON. Requires CairoMakie # (`Pkg.add("CairoMakie")` into this environment first -- not a package dependency, it's heavy). using Pkg -Pkg.activate(@__DIR__ * "/..") +Pkg.activate(joinpath(@__DIR__, "..")) using ArgParse using PkgBenchmark From 7492e2edb0387062065fc5d7403f2117f98b9483 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Wed, 16 Sep 2026 11:02:38 -0400 Subject: [PATCH 11/18] Merge :tccg into :pairwise, filterable via casefilter TCCG cases were structurally just ContractSpec, same as :pairwise -- no separate execution path or spec type justified a separate category. Tag each case's params.source (:synthetic/:tccg) instead, and add a casefilter predicate to build_suite to select either subset without a separate top-level category. Co-Authored-By: Claude Sonnet 5 --- benchmark/README.md | 8 ++++---- benchmark/src/TensorOperationsBenchmarks.jl | 2 +- benchmark/src/categories/pairwise.jl | 18 +++++++++++++----- benchmark/src/categories/tccg.jl | 6 +++--- benchmark/src/suite.jl | 10 ++++++++-- benchmark/test/runtests.jl | 16 +++++++++++----- 6 files changed, 40 insertions(+), 20 deletions(-) diff --git a/benchmark/README.md b/benchmark/README.md index 14637c0e..514748a4 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -27,16 +27,16 @@ julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # r ## Categories (v1) -- `:pairwise` -- generic pairwise contractions across open/contracted/batch index splits. +- `:pairwise` -- generic pairwise contractions: a synthetic parametric shape family + (`params.source == :synthetic`) plus 24 real quantum-chemistry contractions (CCSD, CCSD(T), + AO2MO, INTENSLI) from the [TCCG benchmark](https://github.com/HPAC/tccg) + (`params.source == :tccg`). Use `build_suite`'s `casefilter` to select either subset. - `:permute` -- permutation-only (`tensorcopy!`) cost. - `:trace` -- partial and full traces. - `:mixed_precision` -- differing input/output element types (e.g. `Float32 x Float32 -> Float64`, mixed real/complex). - `:mps` -- MPS/MPO DMRG effective-Hamiltonian motif (1-site and 2-site "theta"), swept over bond dimension `D`. -- `:tccg` -- 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from the - [TCCG benchmark](https://github.com/HPAC/tccg); a source of real high-rank, irregular - index-split shapes rather than hand-picked ones. - `:ctmrg` -- CTMRG corner-growth step (2D PEPS boundary-MPS), swept over environment bond `chi`. - `:trg` -- TRG plaquette contraction (4-ring of rank-3 tensors), swept over bond `chi`. diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index b3e86d6b..ce9ced56 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -16,12 +16,12 @@ include("lowering.jl") include("suite.jl") include("report.jl") +include("categories/tccg.jl") # defines _tccg_cases, merged into :pairwise below include("categories/pairwise.jl") include("categories/permute.jl") include("categories/trace.jl") include("categories/mixed_precision.jl") include("categories/mps.jl") -include("categories/tccg.jl") include("categories/ctmrg.jl") include("categories/trg.jl") diff --git a/benchmark/src/categories/pairwise.jl b/benchmark/src/categories/pairwise.jl index ec07e8b9..6b5a2de6 100644 --- a/benchmark/src/categories/pairwise.jl +++ b/benchmark/src/categories/pairwise.jl @@ -1,5 +1,7 @@ -# Generic pairwise contractions, swept over leg dimension `sizes` x shape. Sizes mix -# power-of-two with off-by-one/arbitrary values to catch alignment effects. +# Pairwise contractions: a synthetic parametric shape family, plus the real TCCG equations +# (tccg.jl) -- both just `ContractSpec`, merged into one category and distinguished via +# `params.source` (`:synthetic`/`:tccg`) so `build_suite`'s `casefilter` can select either. +# Sizes mix power-of-two with off-by-one/arbitrary values to catch alignment effects. const PAIRWISE_SHAPES = ( (1, 1, 1), # matrix-vector-like @@ -9,7 +11,7 @@ const PAIRWISE_SHAPES = ( (1, 0, 1), # pure outer product, no contraction ) -function _pairwise_cases(sizes) +function _synthetic_pairwise_cases(sizes) cases = BenchmarkCase[] for dim in sizes for (nopenA, ncontract, nopenB) in PAIRWISE_SHAPES @@ -23,10 +25,16 @@ function _pairwise_cases(sizes) spec = ContractSpec(IA, IB, IC, dims) within_memory_budget(spec) || continue id = "dim$(dim)_$(nopenA)_$(ncontract)_$(nopenB)" - push!(cases, BenchmarkCase(:pairwise, id, (; dim, nopenA, ncontract, nopenB), spec)) + params = (; dim, nopenA, ncontract, nopenB, source = :synthetic) + push!(cases, BenchmarkCase(:pairwise, id, params, spec)) end end return cases end -register_category!(:pairwise, _pairwise_cases; sizes = (8, 15, 32, 63, 96, 128, 200, 256)) +_pairwise_cases(sizes) = vcat(_synthetic_pairwise_cases(sizes), _tccg_cases(sizes)) + +register_category!( + :pairwise, _pairwise_cases; + sizes = (8, 12, 15, 16, 24, 32, 63, 96, 128, 200, 256) +) diff --git a/benchmark/src/categories/tccg.jl b/benchmark/src/categories/tccg.jl index 0e3d02a7..a23674ea 100644 --- a/benchmark/src/categories/tccg.jl +++ b/benchmark/src/categories/tccg.jl @@ -1,6 +1,7 @@ # Real quantum-chemistry contractions from the TCCG benchmark (github.com/HPAC/tccg): CCSD, # CCSD(T), AO2MO, INTENSLI, as "C-A-B" index strings (e.g. "ij-ik-kj" = C[i,j]=A[i,k]*B[k,j]). # `sizes` applies one leg dimension uniformly to every index letter, as TCCG itself does. +# Merged into `:pairwise` (see pairwise.jl) tagged `params.source = :tccg`, not its own category. const TCCG_CONTRACTIONS = ( # CCSD @@ -44,10 +45,9 @@ function _tccg_cases(sizes) spec = ContractSpec(IA, IB, IC, dims) within_memory_budget(spec) || continue id = "$(eq.id)_dim$(dim)" - push!(cases, BenchmarkCase(:tccg, id, (; dim, equation = eq.id), spec)) + params = (; dim, equation = eq.id, source = :tccg) + push!(cases, BenchmarkCase(:pairwise, id, params, spec)) end end return cases end - -register_category!(:tccg, _tccg_cases; sizes = (4, 8, 12, 16, 24, 32)) diff --git a/benchmark/src/suite.jl b/benchmark/src/suite.jl index 2fb36a49..f4c918df 100644 --- a/benchmark/src/suite.jl +++ b/benchmark/src/suite.jl @@ -1,7 +1,7 @@ # Nests a BenchmarkGroup as [category][provider label][case id]. No threading axis -- see threading.jl. """ - build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing) + build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing, casefilter=nothing) Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or the subset in `categories`) for every provider in `providers` (skipping providers that opt out via @@ -9,17 +9,23 @@ Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or t `sizes` may be `nothing` (use each category's [`default_sizes`](@ref)), a size sweep applied to every category, or a `Dict{Symbol}` mapping category name to its own size sweep. + +`casefilter` (a `BenchmarkCase -> Bool` predicate, or `nothing`) selects a subset of cases +within each category -- e.g. `casefilter = c -> c.params.source == :tccg` runs only the +TCCG-sourced cases within `:pairwise`, without needing a separate category. """ function build_suite( providers::AbstractVector{<:AbstractProvider}; categories = collect(keys(REGISTRY)), - sizes = nothing + sizes = nothing, + casefilter = nothing ) suite = BenchmarkGroup() for category in categories generator = REGISTRY[category] catsizes = _sizes_for(sizes, category) cases = generator(catsizes) + casefilter === nothing || (cases = filter(casefilter, cases)) catgroup = suite[String(category)] = BenchmarkGroup() for provider in providers supports(provider, category) || continue diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index c1a4fb58..38105d89 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -74,15 +74,21 @@ using Strided: Strided @test any(c -> occursin("2site", c.id), cases) end - @testset "tccg category covers all 4 source groups and executes" begin - cases = REGISTRY[:tccg]((4,)) + @testset "tccg cases are merged into :pairwise, filterable by params.source" begin + cases = REGISTRY[:pairwise]((4,)) + tccg_cases = filter(c -> c.params.source == :tccg, cases) @test length(TensorOperationsBenchmarks.TCCG_CONTRACTIONS) == 24 + @test any(c -> c.params.source == :synthetic, cases) for prefix in ("ccsd_", "ccsd_t_", "ao2mo_", "intensli_") - @test any(c -> startswith(c.params.equation, prefix), cases) + @test any(c -> startswith(c.params.equation, prefix), tccg_cases) end - suite = build_suite([provider]; categories = [:tccg], sizes = (4,)) + suite = build_suite( + [provider]; categories = [:pairwise], sizes = (4,), + casefilter = c -> c.params.source == :tccg + ) results = run(suite; samples = 1, evals = 1, seconds = 5) - @test !isempty(results["tccg"][label(provider)]) + @test !isempty(results["pairwise"][label(provider)]) + @test length(results["pairwise"][label(provider)]) == length(tccg_cases) end @testset "ctmrg and trg categories execute" begin From 9f5d81c08883872ad2a6f154da597da35c8d3415 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Wed, 16 Sep 2026 13:31:07 -0400 Subject: [PATCH 12/18] Add permuted-stride pairwise layouts; filter via native BenchmarkTools tags Each synthetic pairwise shape now generates up to 4 label-order layouts (gemm_ready/a_permuted/b_permuted/both_permuted): gemm_ready reshapes directly to a BLAS call, the others interleave open/contracted labels so no reshape or transpose flag suffices -- forcing a real permutation, which is what actually separates StridedNative from StridedBLAS. Structurally-redundant layouts (e.g. shapes with <=1 contracted index) are deduplicated automatically. Replace the bespoke `casefilter` predicate with BenchmarkTools' own `@tagged` mechanism: each case is now wrapped as a tagged BenchmarkGroup (category + every Symbol-valued params entry, via the new `casetags`), so `suite[@tagged "tccg"]` or `suite[@tagged "pairwise" && "both_permuted"]` filter natively, including boolean combinations, on the built suite -- no bespoke filtering API to learn or maintain. Co-Authored-By: Claude Sonnet 5 --- benchmark/README.md | 35 ++++++++++++-- benchmark/src/TensorOperationsBenchmarks.jl | 3 +- benchmark/src/categories/pairwise.jl | 51 ++++++++++++++++----- benchmark/src/registry.jl | 11 +++++ benchmark/src/report.jl | 4 +- benchmark/src/suite.jl | 20 ++++---- benchmark/test/runtests.jl | 28 ++++++++--- 7 files changed, 118 insertions(+), 34 deletions(-) diff --git a/benchmark/README.md b/benchmark/README.md index 514748a4..8cf9c2b3 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -18,7 +18,8 @@ julia --project=. -e ' ' ``` -Or for commit-to-commit regression comparison via PkgBenchmark: +Or for commit-to-commit regression comparison via PkgBenchmark. Both scripts are `@main` apps +(Julia 1.11+), so `--help` works and `ARGS` are parsed the normal way: ``` julia --project=. scripts/run_benchmarks.jl --threads 1 4 --blas-threads 1 4 @@ -27,10 +28,13 @@ julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # r ## Categories (v1) -- `:pairwise` -- generic pairwise contractions: a synthetic parametric shape family - (`params.source == :synthetic`) plus 24 real quantum-chemistry contractions (CCSD, CCSD(T), - AO2MO, INTENSLI) from the [TCCG benchmark](https://github.com/HPAC/tccg) - (`params.source == :tccg`). Use `build_suite`'s `casefilter` to select either subset. +- `:pairwise` -- generic pairwise contractions: a synthetic parametric shape family (tagged + `synthetic`) plus 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from + the [TCCG benchmark](https://github.com/HPAC/tccg) (tagged `tccg`). Each synthetic shape also + comes in up to 4 label-order layouts (tagged `gemm_ready`/`a_permuted`/`b_permuted`/ + `both_permuted`): `gemm_ready` is directly reshapeable to a BLAS call, the others interleave + open/contracted labels so no reshape or transpose flag suffices -- a real permutation is + required, which is what actually separates `StridedNative` from `StridedBLAS`. - `:permute` -- permutation-only (`tensorcopy!`) cost. - `:trace` -- partial and full traces. - `:mixed_precision` -- differing input/output element types (e.g. `Float32 x Float32 -> @@ -48,6 +52,27 @@ New file under `src/categories/`, define `mysizes -> Vector{BenchmarkCase}` buil `ContractSpec`/`TraceSpec`/`AddSpec`/`NetworkSpec` values, `include` it, call `register_category!(:mycategory, mygenerator)`. Nothing else changes. +## `BenchmarkCase` vs. plain `BenchmarkTools` + +A `BenchmarkTools.Benchmark`/`Trial` only knows how to run a closure and record timings -- it +carries no metadata about *why* that closure exists. `BenchmarkCase` is our own struct that +keeps the `AbstractCaseSpec` (needed for the `flops`/`bytes` cost model) and `params` +(the sweep values that produced it) alongside each case, so `resultstable` can join timings +back against cost figures after the fact. `build_suite` consumes a `Vector{BenchmarkCase}` and +produces an ordinary `BenchmarkGroup`; nothing downstream of that ever sees `BenchmarkCase` +again. + +Filtering uses `BenchmarkGroup`'s native tags, not a bespoke mechanism: each case is wrapped as +`BenchmarkGroup(casetags(case), "benchmark" => ...)`, where `casetags` is the category name plus +every `Symbol`-valued `params` entry (`source`, `kind`, `variant`, `layout`, ...). Filter the +*built* suite with `@tagged` (re-exported from BenchmarkTools) before running it: + +```julia +suite = build_suite(providers) +run(suite[@tagged "tccg"]) # only the TCCG-sourced pairwise cases +run(suite[@tagged "pairwise" && "both_permuted"]) # boolean tag expressions work +``` + ## Plugging in a downstream tensor type ```julia diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index ce9ced56..9d852d04 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -30,8 +30,9 @@ export flops, bytes export AbstractProvider, ArrayProvider, scalartype, randtensor, backend, allocator, label, supports, rng export ThreadConfig, with_threads, set_threads! -export BenchmarkCase, register_category!, REGISTRY, default_sizes +export BenchmarkCase, register_category!, REGISTRY, default_sizes, casetags export build_suite export resultstable +export @tagged end # module diff --git a/benchmark/src/categories/pairwise.jl b/benchmark/src/categories/pairwise.jl index 6b5a2de6..062cf249 100644 --- a/benchmark/src/categories/pairwise.jl +++ b/benchmark/src/categories/pairwise.jl @@ -1,7 +1,12 @@ # Pairwise contractions: a synthetic parametric shape family, plus the real TCCG equations -# (tccg.jl) -- both just `ContractSpec`, merged into one category and distinguished via -# `params.source` (`:synthetic`/`:tccg`) so `build_suite`'s `casefilter` can select either. -# Sizes mix power-of-two with off-by-one/arbitrary values to catch alignment effects. +# (tccg.jl) -- both just `ContractSpec`, merged into one category, tagged (`:synthetic`/`:tccg`) +# for filtering via `@tagged` (see registry.jl). Sizes mix power-of-two with off-by-one values. +# +# Each shape gets up to 4 label-order "layouts": `(openA...,contract...)`/`(contract...,openB...)` +# is directly reshapeable to GEMM (no data movement); interleaving open and contracted labels +# (e.g. `[a1,c1,a2,c2]`) cannot be expressed as a single reshape or BLAS transpose flag, forcing +# a real permutation -- exactly the case that separates StridedNative from StridedBLAS. Layouts +# that coincide with the GEMM one (e.g. when a shape has ≤1 contracted index) are skipped. const PAIRWISE_SHAPES = ( (1, 1, 1), # matrix-vector-like @@ -11,6 +16,30 @@ const PAIRWISE_SHAPES = ( (1, 0, 1), # pure outer product, no contraction ) +function _interleave(a::Vector{Symbol}, b::Vector{Symbol}) + n = min(length(a), length(b)) + return vcat((Symbol[a[i], b[i]] for i in 1:n)..., a[(n + 1):end], b[(n + 1):end]) +end + +function _pairwise_layouts(openA, contract, openB) + gemmA, gemmB = vcat(openA, contract), vcat(contract, openB) + permA, permB = _interleave(openA, contract), _interleave(contract, openB) + candidates = ( + (:gemm_ready, gemmA, gemmB), + (:a_permuted, permA, gemmB), + (:b_permuted, gemmA, permB), + (:both_permuted, permA, permB), + ) + seen = Set{Tuple{Vector{Symbol}, Vector{Symbol}}}() + layouts = Tuple{Symbol, Vector{Symbol}, Vector{Symbol}}[] + for (layout, IA, IB) in candidates + (IA, IB) in seen && continue + push!(seen, (IA, IB)) + push!(layouts, (layout, IA, IB)) + end + return layouts +end + function _synthetic_pairwise_cases(sizes) cases = BenchmarkCase[] for dim in sizes @@ -18,15 +47,15 @@ function _synthetic_pairwise_cases(sizes) openA = [Symbol("a", i) for i in 1:nopenA] contract = [Symbol("c", i) for i in 1:ncontract] openB = [Symbol("b", i) for i in 1:nopenB] - IA = vcat(openA, contract) - IB = vcat(contract, openB) IC = vcat(openA, openB) - dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) - spec = ContractSpec(IA, IB, IC, dims) - within_memory_budget(spec) || continue - id = "dim$(dim)_$(nopenA)_$(ncontract)_$(nopenB)" - params = (; dim, nopenA, ncontract, nopenB, source = :synthetic) - push!(cases, BenchmarkCase(:pairwise, id, params, spec)) + for (layout, IA, IB) in _pairwise_layouts(openA, contract, openB) + dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) + spec = ContractSpec(IA, IB, IC, dims) + within_memory_budget(spec) || continue + id = "dim$(dim)_$(nopenA)_$(ncontract)_$(nopenB)_$(layout)" + params = (; dim, nopenA, ncontract, nopenB, layout, source = :synthetic) + push!(cases, BenchmarkCase(:pairwise, id, params, spec)) + end end end return cases diff --git a/benchmark/src/registry.jl b/benchmark/src/registry.jl index c8a18598..c25e05e0 100644 --- a/benchmark/src/registry.jl +++ b/benchmark/src/registry.jl @@ -14,6 +14,17 @@ struct BenchmarkCase spec::AbstractCaseSpec end +""" + casetags(case::BenchmarkCase) -> Vector + +Tags for `BenchmarkTools.@tagged`-based filtering (see suite.jl): the category name, plus every +`Symbol`-valued entry of `params` (e.g. `source`, `kind`, `variant`, `layout`). A category +generator opts a `params` field into tag-filtering just by giving it a `Symbol` value. +""" +function casetags(case::BenchmarkCase) + return Any[String(case.category); [string(v) for v in values(case.params) if v isa Symbol]] +end + const REGISTRY = Dict{Symbol, Function}() const DEFAULT_SIZES = Dict{Symbol, Any}() diff --git a/benchmark/src/report.jl b/benchmark/src/report.jl index b5fd8da0..8b95b675 100644 --- a/benchmark/src/report.jl +++ b/benchmark/src/report.jl @@ -23,7 +23,7 @@ end """ resultstable(results::BenchmarkGroup; categories=collect(keys(REGISTRY)), sizes=nothing) -Flatten a benchmark-run result (with the same `[category][provider][id]` nesting +Flatten a benchmark-run result (with the same `[category][provider][id]["benchmark"]` nesting `build_suite` produces) into a `Vector{ResultRow}`. `categories`/`sizes` must match what was passed to the `build_suite` call that produced `results`, since specs (and therefore flop/byte counts) are regenerated from `REGISTRY` rather than stored in the result itself. @@ -40,7 +40,7 @@ function resultstable( for providerlabel in keys(catgroup) provgroup = catgroup[providerlabel] for id in keys(provgroup) - trial = provgroup[id] + trial = provgroup[id]["benchmark"] case = casesbyid[id] mintime = minimum(trial.times) memory = trial.memory diff --git a/benchmark/src/suite.jl b/benchmark/src/suite.jl index f4c918df..fe00e17c 100644 --- a/benchmark/src/suite.jl +++ b/benchmark/src/suite.jl @@ -1,7 +1,10 @@ -# Nests a BenchmarkGroup as [category][provider label][case id]. No threading axis -- see threading.jl. +# Nests a BenchmarkGroup as [category][provider label][case id]["benchmark"]. Each case-id leaf +# is itself a tagged BenchmarkGroup (tags from `casetags`: category + Symbol-valued params), so +# `suite[BenchmarkTools.@tagged "tccg"]` (or any boolean tag expression) selects a subset of +# cases natively -- no bespoke filter predicate needed. No threading axis -- see threading.jl. """ - build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing, casefilter=nothing) + build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing) Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or the subset in `categories`) for every provider in `providers` (skipping providers that opt out via @@ -10,28 +13,27 @@ Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or t `sizes` may be `nothing` (use each category's [`default_sizes`](@ref)), a size sweep applied to every category, or a `Dict{Symbol}` mapping category name to its own size sweep. -`casefilter` (a `BenchmarkCase -> Bool` predicate, or `nothing`) selects a subset of cases -within each category -- e.g. `casefilter = c -> c.params.source == :tccg` runs only the -TCCG-sourced cases within `:pairwise`, without needing a separate category. +To run only a tagged subset, filter *after* building: `suite[@tagged "tccg"]` (re-exported from +BenchmarkTools), or combine tags: `suite[@tagged "pairwise" && "tccg"]`. """ function build_suite( providers::AbstractVector{<:AbstractProvider}; categories = collect(keys(REGISTRY)), - sizes = nothing, - casefilter = nothing + sizes = nothing ) suite = BenchmarkGroup() for category in categories generator = REGISTRY[category] catsizes = _sizes_for(sizes, category) cases = generator(catsizes) - casefilter === nothing || (cases = filter(casefilter, cases)) catgroup = suite[String(category)] = BenchmarkGroup() for provider in providers supports(provider, category) || continue provgroup = catgroup[label(provider)] = BenchmarkGroup() for case in cases - provgroup[case.id] = make_benchmarkable(case, provider) + provgroup[case.id] = BenchmarkGroup( + casetags(case), "benchmark" => make_benchmarkable(case, provider) + ) end end end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 38105d89..ba138276 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -74,7 +74,7 @@ using Strided: Strided @test any(c -> occursin("2site", c.id), cases) end - @testset "tccg cases are merged into :pairwise, filterable by params.source" begin + @testset "tccg cases are merged into :pairwise, filterable via @tagged" begin cases = REGISTRY[:pairwise]((4,)) tccg_cases = filter(c -> c.params.source == :tccg, cases) @test length(TensorOperationsBenchmarks.TCCG_CONTRACTIONS) == 24 @@ -82,15 +82,31 @@ using Strided: Strided for prefix in ("ccsd_", "ccsd_t_", "ao2mo_", "intensli_") @test any(c -> startswith(c.params.equation, prefix), tccg_cases) end - suite = build_suite( - [provider]; categories = [:pairwise], sizes = (4,), - casefilter = c -> c.params.source == :tccg - ) - results = run(suite; samples = 1, evals = 1, seconds = 5) + suite = build_suite([provider]; categories = [:pairwise], sizes = (4,)) + results = run(suite[@tagged "tccg"]; samples = 1, evals = 1, seconds = 5) @test !isempty(results["pairwise"][label(provider)]) @test length(results["pairwise"][label(provider)]) == length(tccg_cases) end + @testset "casetags derives from category + Symbol-valued params" begin + case = first(REGISTRY[:trace]((8,))) + tags = casetags(case) + @test "trace" in tags + @test string(case.params.kind) in tags + end + + @testset "pairwise permuted-stride layouts are structurally distinct and filterable" begin + cases = REGISTRY[:pairwise]((8,)) + layouts = unique(c.params.layout for c in cases if c.params.source == :synthetic) + @test :gemm_ready in layouts + @test :both_permuted in layouts + + suite = build_suite([provider]; categories = [:pairwise], sizes = (8,)) + results = run(suite[@tagged "both_permuted"]; samples = 1, evals = 1, seconds = 5) + n_expected = count(c -> c.params.source == :synthetic && c.params.layout == :both_permuted, cases) + @test length(results["pairwise"][label(provider)]) == n_expected + end + @testset "ctmrg and trg categories execute" begin for category in (:ctmrg, :trg) suite = build_suite([provider]; categories = [category], sizes = (8,)) From 38821fc6817e555bbebce84b1a546d0f9ed7e2e0 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Wed, 16 Sep 2026 13:31:17 -0400 Subject: [PATCH 13/18] Use @main entrypoints for CLI scripts Wrap run_benchmarks.jl/show_benchmarks.jl's logic in a main(args) function marked @main (Julia 1.11+), so ARGS flow through the normal argument-parsing path instead of being read as a top-level side effect. Bumps the julia compat bound to 1.11 accordingly. Co-Authored-By: Claude Sonnet 5 --- benchmark/Project.toml | 2 +- benchmark/scripts/run_benchmarks.jl | 37 +++++++++++-------- benchmark/scripts/show_benchmarks.jl | 55 +++++++++++++++------------- 3 files changed, 52 insertions(+), 42 deletions(-) diff --git a/benchmark/Project.toml b/benchmark/Project.toml index da297f82..3e053269 100644 --- a/benchmark/Project.toml +++ b/benchmark/Project.toml @@ -21,7 +21,7 @@ Random = "1.10" Strided = "2.6" TensorOperations = "5.8" Test = "1" -julia = "1.10" +julia = "1.11" [extras] Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" diff --git a/benchmark/scripts/run_benchmarks.jl b/benchmark/scripts/run_benchmarks.jl index 0327ab77..5a7b6385 100644 --- a/benchmark/scripts/run_benchmarks.jl +++ b/benchmark/scripts/run_benchmarks.jl @@ -8,7 +8,7 @@ Pkg.activate(joinpath(@__DIR__, "..")) using ArgParse using PkgBenchmark -function parse_commandline() +function parse_commandline(args) s = ArgParseSettings(; description = "Run the TensorOperationsBenchmarks suite via PkgBenchmark.") @add_arg_table! s begin "--threads" @@ -30,23 +30,28 @@ function parse_commandline() help = "output file prefix (a suffix identifying the thread combo and `.json` are appended)" default = "results" end - return parse_args(s) + return parse_args(args, s) end -opts = parse_commandline() -blascounts = isempty(opts["blas-threads"]) ? [nothing] : opts["blas-threads"] -stridedcounts = isempty(opts["strided-threads"]) ? [nothing] : opts["strided-threads"] +function main(args) + opts = parse_commandline(args) + blascounts = isempty(opts["blas-threads"]) ? [nothing] : opts["blas-threads"] + stridedcounts = isempty(opts["strided-threads"]) ? [nothing] : opts["strided-threads"] -for nthreads in opts["threads"], blas in blascounts, strided in stridedcounts - @info "Running benchmarks" nthreads blas strided - withenv( - "TOB_BLAS_THREADS" => blas === nothing ? "" : string(blas), - "TOB_STRIDED_THREADS" => strided === nothing ? "" : string(strided), - ) do - cfg = BenchmarkConfig(; juliacmd = `julia -t $nthreads -O3`) - results = benchmarkpkg(dirname(@__DIR__), cfg) - outfile = "$(opts["out"])_t$(nthreads)_blas$(blas)_strided$(strided).json" - writeresults(outfile, results) - @info "Wrote $outfile" + for nthreads in opts["threads"], blas in blascounts, strided in stridedcounts + @info "Running benchmarks" nthreads blas strided + withenv( + "TOB_BLAS_THREADS" => blas === nothing ? "" : string(blas), + "TOB_STRIDED_THREADS" => strided === nothing ? "" : string(strided), + ) do + cfg = BenchmarkConfig(; juliacmd = `julia -t $nthreads -O3`) + results = benchmarkpkg(dirname(@__DIR__), cfg) + outfile = "$(opts["out"])_t$(nthreads)_blas$(blas)_strided$(strided).json" + writeresults(outfile, results) + @info "Wrote $outfile" + end end + return 0 end + +@main diff --git a/benchmark/scripts/show_benchmarks.jl b/benchmark/scripts/show_benchmarks.jl index 2d2e1c58..7b6c94af 100644 --- a/benchmark/scripts/show_benchmarks.jl +++ b/benchmark/scripts/show_benchmarks.jl @@ -9,7 +9,7 @@ using PkgBenchmark using CairoMakie using TensorOperationsBenchmarks -function parse_commandline() +function parse_commandline(args) s = ArgParseSettings(; description = "Plot GFLOP/s-vs-size scaling curves from a PkgBenchmark result.") @add_arg_table! s begin "resultfile" @@ -19,34 +19,39 @@ function parse_commandline() help = "output image path (default: replace the input's extension with .png)" default = nothing end - return parse_args(s) + return parse_args(args, s) end -opts = parse_commandline() -results = PkgBenchmark.readresults(opts["resultfile"]) -group = PkgBenchmark.benchmarkgroup(results) +function main(args) + opts = parse_commandline(args) + results = PkgBenchmark.readresults(opts["resultfile"]) + group = PkgBenchmark.benchmarkgroup(results) -rows = resultstable(group) + rows = resultstable(group) -fig = Figure(; size = (1000, 800)) -categories = unique(r.category for r in rows) -for (i, category) in enumerate(categories) - ax = Axis( - fig[fldmod1(i, 2)...]; xscale = log2, yscale = log10, - title = category, xlabel = "size", ylabel = "GFLOP/s" - ) - catrows = filter(r -> r.category == category, rows) - for provider in unique(r.provider for r in catrows) - provrows = filter(r -> r.provider == provider, catrows) - sort!(provrows; by = r -> get(r.params, :dim, get(r.params, :D, 0))) - xs = [get(r.params, :dim, get(r.params, :D, 0)) for r in provrows] - ys = [r.gflops for r in provrows] - lines!(ax, xs, ys; label = provider) - scatter!(ax, xs, ys) + fig = Figure(; size = (1000, 800)) + categories = unique(r.category for r in rows) + for (i, category) in enumerate(categories) + ax = Axis( + fig[fldmod1(i, 2)...]; xscale = log2, yscale = log10, + title = category, xlabel = "size", ylabel = "GFLOP/s" + ) + catrows = filter(r -> r.category == category, rows) + for provider in unique(r.provider for r in catrows) + provrows = filter(r -> r.provider == provider, catrows) + sort!(provrows; by = r -> get(r.params, :dim, get(r.params, :D, 0))) + xs = [get(r.params, :dim, get(r.params, :D, 0)) for r in provrows] + ys = [r.gflops for r in provrows] + lines!(ax, xs, ys; label = provider) + scatter!(ax, xs, ys) + end + axislegend(ax) end - axislegend(ax) + + outfile = something(opts["out"], splitext(opts["resultfile"])[1] * ".png") + save(outfile, fig) + @info "Wrote $outfile" + return 0 end -outfile = something(opts["out"], splitext(opts["resultfile"])[1] * ".png") -save(outfile, fig) -@info "Wrote $outfile" +@main From 90dfa5bf8776d3963e0dbc86046a49a05f04134a Mon Sep 17 00:00:00 2001 From: lkdvos Date: Wed, 16 Sep 2026 16:03:33 -0400 Subject: [PATCH 14/18] Rename :pairwise to :contract; merge mps/ctmrg/trg into tagged :network :contract better matches TensorOperations.jl's own tensorcontract API naming than the more incidental "pairwise" description. mps/ctmrg/trg were all NetworkSpec-based motifs with no structural difference beyond which topology they encode -- merged into one :network category, distinguished via params.topic (:mps/:ctmrg/:trg), filterable through the same @tagged mechanism already used for :contract's synthetic/tccg split. Co-Authored-By: Claude Sonnet 5 --- benchmark/src/TensorOperationsBenchmarks.jl | 17 ++++++++++++++--- .../categories/{pairwise.jl => contract.jl} | 18 +++++++++--------- benchmark/src/categories/ctmrg.jl | 8 ++++---- benchmark/src/categories/mps.jl | 11 ++++++----- benchmark/src/categories/network.jl | 11 +++++++++++ benchmark/src/categories/tccg.jl | 6 +++--- benchmark/src/categories/trg.jl | 7 +++---- 7 files changed, 50 insertions(+), 28 deletions(-) rename benchmark/src/categories/{pairwise.jl => contract.jl} (84%) create mode 100644 benchmark/src/categories/network.jl diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index 9d852d04..eafe8382 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -3,34 +3,45 @@ module TensorOperationsBenchmarks using LinearAlgebra: BLAS using Strided: Strided using Random: Random, randn! +using DataFrames: DataFrame using BenchmarkTools using TensorOperations using TensorOperations: DefaultBackend, DefaultAllocator, AbstractBackend +# Specs (pure data) and the cost model computed from them. include("specs.jl") include("cost.jl") + +# The downstream extension points: what tensor type to run against, and how many threads. include("provider.jl") include("threading.jl") + +# The case registry, and turning a (spec, provider) pair into an executable benchmark. include("registry.jl") include("lowering.jl") + +# Suite assembly and reporting. include("suite.jl") include("report.jl") -include("categories/tccg.jl") # defines _tccg_cases, merged into :pairwise below -include("categories/pairwise.jl") +# Categories: tccg.jl/mps.jl/ctmrg.jl/trg.jl define generators merged by contract.jl/network.jl +# (include order doesn't matter -- generators are only called after the whole module loads). +include("categories/tccg.jl") +include("categories/contract.jl") include("categories/permute.jl") include("categories/trace.jl") include("categories/mixed_precision.jl") include("categories/mps.jl") include("categories/ctmrg.jl") include("categories/trg.jl") +include("categories/network.jl") export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, NetworkSpec export flops, bytes export AbstractProvider, ArrayProvider, scalartype, randtensor, backend, allocator, label, supports, rng export ThreadConfig, with_threads, set_threads! -export BenchmarkCase, register_category!, REGISTRY, default_sizes, casetags +export BenchmarkCase, register_category!, REGISTRY, default_sizes, within_memory_budget export build_suite export resultstable export @tagged diff --git a/benchmark/src/categories/pairwise.jl b/benchmark/src/categories/contract.jl similarity index 84% rename from benchmark/src/categories/pairwise.jl rename to benchmark/src/categories/contract.jl index 062cf249..c5dcc532 100644 --- a/benchmark/src/categories/pairwise.jl +++ b/benchmark/src/categories/contract.jl @@ -8,7 +8,7 @@ # a real permutation -- exactly the case that separates StridedNative from StridedBLAS. Layouts # that coincide with the GEMM one (e.g. when a shape has ≤1 contracted index) are skipped. -const PAIRWISE_SHAPES = ( +const CONTRACT_SHAPES = ( (1, 1, 1), # matrix-vector-like (2, 1, 2), # single shared bond, several open legs each side (2, 2, 2), # GEMM-like, rank 4 total @@ -21,7 +21,7 @@ function _interleave(a::Vector{Symbol}, b::Vector{Symbol}) return vcat((Symbol[a[i], b[i]] for i in 1:n)..., a[(n + 1):end], b[(n + 1):end]) end -function _pairwise_layouts(openA, contract, openB) +function _contract_layouts(openA, contract, openB) gemmA, gemmB = vcat(openA, contract), vcat(contract, openB) permA, permB = _interleave(openA, contract), _interleave(contract, openB) candidates = ( @@ -40,30 +40,30 @@ function _pairwise_layouts(openA, contract, openB) return layouts end -function _synthetic_pairwise_cases(sizes) +function _synthetic_contract_cases(sizes) cases = BenchmarkCase[] for dim in sizes - for (nopenA, ncontract, nopenB) in PAIRWISE_SHAPES + for (nopenA, ncontract, nopenB) in CONTRACT_SHAPES openA = [Symbol("a", i) for i in 1:nopenA] contract = [Symbol("c", i) for i in 1:ncontract] openB = [Symbol("b", i) for i in 1:nopenB] IC = vcat(openA, openB) - for (layout, IA, IB) in _pairwise_layouts(openA, contract, openB) + for (layout, IA, IB) in _contract_layouts(openA, contract, openB) dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) spec = ContractSpec(IA, IB, IC, dims) - within_memory_budget(spec) || continue id = "dim$(dim)_$(nopenA)_$(ncontract)_$(nopenB)_$(layout)" + within_memory_budget(spec, id) || continue params = (; dim, nopenA, ncontract, nopenB, layout, source = :synthetic) - push!(cases, BenchmarkCase(:pairwise, id, params, spec)) + push!(cases, BenchmarkCase(:contract, id, params, spec)) end end end return cases end -_pairwise_cases(sizes) = vcat(_synthetic_pairwise_cases(sizes), _tccg_cases(sizes)) +_contract_cases(sizes) = vcat(_synthetic_contract_cases(sizes), _tccg_cases(sizes)) register_category!( - :pairwise, _pairwise_cases; + :contract, _contract_cases; sizes = (8, 12, 15, 16, 24, 32, 63, 96, 128, 200, 256) ) diff --git a/benchmark/src/categories/ctmrg.jl b/benchmark/src/categories/ctmrg.jl index 73546afa..be228bf0 100644 --- a/benchmark/src/categories/ctmrg.jl +++ b/benchmark/src/categories/ctmrg.jl @@ -1,6 +1,7 @@ # CTMRG corner-growth step (boundary-MPS method for 2D PEPS): C-T-T-a, producing an unfused # rank-4 corner (chi,chi,D2,D2). Cost ~ O(chi^3*D2^3) (literature O(chi^3*D^6), D2=D^2). -# `sizes` sweeps environment bond `chi`; PEPS bond `D` is fixed. +# `sizes` sweeps environment bond `chi`; PEPS bond `D` is fixed. Merged into `:network` (see +# network.jl) tagged `params.topic = :ctmrg`. const CTMRG_PEPS_BOND = 3 # D @@ -14,9 +15,8 @@ function _ctmrg_case(chi) ] dims = Dict(1 => chi, 2 => chi, 3 => D2, 4 => D2, 10 => chi, 11 => chi, 12 => D2, 13 => D2) spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12, -13]) - return BenchmarkCase(:ctmrg, "corner_chi$(chi)", (; chi, D = CTMRG_PEPS_BOND), spec) + params = (; chi, D = CTMRG_PEPS_BOND, topic = :ctmrg) + return BenchmarkCase(:network, "ctmrg_corner_chi$(chi)", params, spec) end _ctmrg_cases(sizes) = [_ctmrg_case(chi) for chi in sizes] - -register_category!(:ctmrg, _ctmrg_cases; sizes = (16, 24, 32, 48, 64, 100)) diff --git a/benchmark/src/categories/mps.jl b/benchmark/src/categories/mps.jl index 6d594550..3927902f 100644 --- a/benchmark/src/categories/mps.jl +++ b/benchmark/src/categories/mps.jl @@ -1,5 +1,6 @@ # MPS/MPO DMRG effective-Hamiltonian motif (L-MPS-MPO-R, cost ~ O(D^3*d*w)) plus the 2-site -# "theta" variant (L-MPS-MPO-MPO-MPS-R), swept over bond dimension `D`. +# "theta" variant (L-MPS-MPO-MPO-MPS-R), swept over bond dimension `D`. Merged into `:network` +# (see network.jl) tagged `params.topic = :mps`. const MPS_PHYS_DIM = 2 # d: physical dimension (spin-1/2) const MPS_MPO_BOND = 6 # w: MPO bond dimension (local Hamiltonian) @@ -16,7 +17,8 @@ function _mps_1site_case(D) ] dims = Dict(1 => w, 2 => D, 3 => d, 4 => D, 5 => w, 10 => D, 11 => d, 12 => D) spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12]) - return BenchmarkCase(:mps, "1site_D$(D)", (; D, d, w, variant = :onesite), spec) + params = (; D, d, w, topic = :mps, variant = :onesite) + return BenchmarkCase(:network, "mps_1site_D$(D)", params, spec) end function _mps_2site_case(D) @@ -34,7 +36,8 @@ function _mps_2site_case(D) ] dims = Dict(1 => w, 2 => D, 3 => d, 4 => D, 5 => d, 6 => w, 7 => w, 8 => D, 10 => D, 11 => d, 12 => D, 13 => d) spec = NetworkSpec(indexlists, dims; output = [-10, -11, -13, -12]) - return BenchmarkCase(:mps, "2site_D$(D)", (; D, d, w, variant = :twosite), spec) + params = (; D, d, w, topic = :mps, variant = :twosite) + return BenchmarkCase(:network, "mps_2site_D$(D)", params, spec) end function _mps_cases(sizes) @@ -43,5 +46,3 @@ function _mps_cases(sizes) BenchmarkCase[_mps_2site_case(D) for D in sizes] ) end - -register_category!(:mps, _mps_cases; sizes = (32, 48, 64, 100, 128, 256, 300, 512)) diff --git a/benchmark/src/categories/network.jl b/benchmark/src/categories/network.jl new file mode 100644 index 00000000..16dfdc61 --- /dev/null +++ b/benchmark/src/categories/network.jl @@ -0,0 +1,11 @@ +# Merges the NetworkSpec-based motifs (mps.jl/ctmrg.jl/trg.jl) into one `:network` category, +# tagged by `params.topic` (`:mps`/`:ctmrg`/`:trg`) for `@tagged`-based filtering -- same pattern +# as `:contract` merging synthetic shapes with TCCG. `sizes` is the shared bond-dimension knob +# (`D` for mps, `chi` for ctmrg/trg); each topic interprets it independently. + +_network_cases(sizes) = vcat(_mps_cases(sizes), _ctmrg_cases(sizes), _trg_cases(sizes)) + +register_category!( + :network, _network_cases; + sizes = (16, 24, 32, 48, 64, 96, 100, 128, 256, 300, 512) +) diff --git a/benchmark/src/categories/tccg.jl b/benchmark/src/categories/tccg.jl index a23674ea..3f1641aa 100644 --- a/benchmark/src/categories/tccg.jl +++ b/benchmark/src/categories/tccg.jl @@ -1,7 +1,7 @@ # Real quantum-chemistry contractions from the TCCG benchmark (github.com/HPAC/tccg): CCSD, # CCSD(T), AO2MO, INTENSLI, as "C-A-B" index strings (e.g. "ij-ik-kj" = C[i,j]=A[i,k]*B[k,j]). # `sizes` applies one leg dimension uniformly to every index letter, as TCCG itself does. -# Merged into `:pairwise` (see pairwise.jl) tagged `params.source = :tccg`, not its own category. +# Merged into `:contract` (see contract.jl) tagged `params.source = :tccg`, not its own category. const TCCG_CONTRACTIONS = ( # CCSD @@ -43,10 +43,10 @@ function _tccg_cases(sizes) IA, IB, IC = _tccg_labels(eq.A), _tccg_labels(eq.B), _tccg_labels(eq.C) dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) spec = ContractSpec(IA, IB, IC, dims) - within_memory_budget(spec) || continue id = "$(eq.id)_dim$(dim)" + within_memory_budget(spec, id) || continue params = (; dim, equation = eq.id, source = :tccg) - push!(cases, BenchmarkCase(:pairwise, id, params, spec)) + push!(cases, BenchmarkCase(:contract, id, params, spec)) end end return cases diff --git a/benchmark/src/categories/trg.jl b/benchmark/src/categories/trg.jl index 94224aa3..433f75d6 100644 --- a/benchmark/src/categories/trg.jl +++ b/benchmark/src/categories/trg.jl @@ -1,6 +1,6 @@ # TRG plaquette contraction: a 4-ring of rank-3 tensors (from SVD-splitting neighboring rank-4 # TRG tensors), producing the coarse-grained rank-4 tensor. Cost ~ O(chi^6). `sizes` sweeps the -# bond dimension `chi`. +# bond dimension `chi`. Merged into `:network` (see network.jl) tagged `params.topic = :trg`. function _trg_case(chi) indexlists = [ @@ -11,9 +11,8 @@ function _trg_case(chi) ] dims = Dict(1 => chi, 2 => chi, 3 => chi, 4 => chi, 10 => chi, 11 => chi, 12 => chi, 13 => chi) spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12, -13]) - return BenchmarkCase(:trg, "plaquette_chi$(chi)", (; chi), spec) + params = (; chi, topic = :trg) + return BenchmarkCase(:network, "trg_plaquette_chi$(chi)", params, spec) end _trg_cases(sizes) = [_trg_case(chi) for chi in sizes] - -register_category!(:trg, _trg_cases; sizes = (16, 24, 32, 48, 64, 96)) From 619d79be028b4b93cf43f6bde792a50e16622fb2 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Wed, 16 Sep 2026 16:03:46 -0400 Subject: [PATCH 15/18] Address remaining review comments MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - BenchmarkCase now stores tags explicitly (computed once at construction) rather than re-deriving them via a separate casetags function; suite.jl uses case.tags directly. - within_memory_budget(spec, id) warns (naming id) when skipping an oversized case, so a sparse sweep is diagnosable, not silent. MAX_CASE_BYTES now scales with host memory (Sys.total_memory() ÷ 64) rather than a fixed 256 MiB. - lowering.jl's maketensors/execute tuples reordered to match tensorcopy!/tensortrace!/tensorcontract!'s own (C, A, ...) argument order. - AddSpec/TraceSpec/ContractSpec/NetworkSpec get compact einsum-style show methods, e.g. `ContractSpec: C[i,j] = A[i,k] * B[k,j] (dim=64)`. - resultstable now returns a DataFrame (added DataFrames dependency) instead of a bespoke ResultRow/Vector; show_benchmarks.jl updated to iterate via eachrow. - @main moved onto the function definition itself (function (@main)(args) ... end), the documented placement, rather than a trailing bare @main. - README: dropped the "(v1)" version qualifier, and removed the BenchmarkCase-vs-BenchmarkTools writeup in favor of the equivalent (already up to date) docstrings in registry.jl/suite.jl. Co-Authored-By: Claude Sonnet 5 --- benchmark/Project.toml | 2 + benchmark/README.md | 37 ++++---------- benchmark/scripts/run_benchmarks.jl | 4 +- benchmark/scripts/show_benchmarks.jl | 12 ++--- benchmark/src/lowering.jl | 12 ++--- benchmark/src/registry.jl | 40 +++++++++------- benchmark/src/report.jl | 45 ++++++----------- benchmark/src/specs.jl | 36 ++++++++++++++ benchmark/src/suite.jl | 10 ++-- benchmark/test/runtests.jl | 72 +++++++++++++++------------- 10 files changed, 142 insertions(+), 128 deletions(-) diff --git a/benchmark/Project.toml b/benchmark/Project.toml index 3e053269..ed4ce259 100644 --- a/benchmark/Project.toml +++ b/benchmark/Project.toml @@ -6,6 +6,7 @@ version = "0.1.0" [deps] ArgParse = "c7e460c6-2fb9-53a9-8c5b-16f535851c63" BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" +DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0" LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" PkgBenchmark = "32113eaa-f34f-5b0d-bd6c-c81e245fc73d" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" @@ -15,6 +16,7 @@ TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" [compat] ArgParse = "1" BenchmarkTools = "1" +DataFrames = "1" LinearAlgebra = "1.10" PkgBenchmark = "0.2" Random = "1.10" diff --git a/benchmark/README.md b/benchmark/README.md index 8cf9c2b3..09b7751c 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -26,9 +26,9 @@ julia --project=. scripts/run_benchmarks.jl --threads 1 4 --blas-threads 1 4 julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # requires CairoMakie ``` -## Categories (v1) +## Categories -- `:pairwise` -- generic pairwise contractions: a synthetic parametric shape family (tagged +- `:contract` -- generic pairwise contractions: a synthetic parametric shape family (tagged `synthetic`) plus 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from the [TCCG benchmark](https://github.com/HPAC/tccg) (tagged `tccg`). Each synthetic shape also comes in up to 4 label-order layouts (tagged `gemm_ready`/`a_permuted`/`b_permuted`/ @@ -39,10 +39,10 @@ julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # r - `:trace` -- partial and full traces. - `:mixed_precision` -- differing input/output element types (e.g. `Float32 x Float32 -> Float64`, mixed real/complex). -- `:mps` -- MPS/MPO DMRG effective-Hamiltonian motif (1-site and 2-site "theta"), swept over - bond dimension `D`. -- `:ctmrg` -- CTMRG corner-growth step (2D PEPS boundary-MPS), swept over environment bond `chi`. -- `:trg` -- TRG plaquette contraction (4-ring of rank-3 tensors), swept over bond `chi`. +- `:network` -- multi-tensor-network motifs, tagged by `topic`: `mps` (MPS/MPO DMRG + effective-Hamiltonian, 1-site and 2-site "theta", swept over bond `D`), `ctmrg` (CTMRG + corner-growth step for 2D PEPS, swept over environment bond `chi`), `trg` (TRG plaquette + contraction, swept over bond `chi`). Not yet implemented, but addable without a redesign: MERA, contraction-order/path-finding timing. @@ -50,28 +50,9 @@ Not yet implemented, but addable without a redesign: MERA, contraction-order/pat New file under `src/categories/`, define `mysizes -> Vector{BenchmarkCase}` building `ContractSpec`/`TraceSpec`/`AddSpec`/`NetworkSpec` values, `include` it, call -`register_category!(:mycategory, mygenerator)`. Nothing else changes. - -## `BenchmarkCase` vs. plain `BenchmarkTools` - -A `BenchmarkTools.Benchmark`/`Trial` only knows how to run a closure and record timings -- it -carries no metadata about *why* that closure exists. `BenchmarkCase` is our own struct that -keeps the `AbstractCaseSpec` (needed for the `flops`/`bytes` cost model) and `params` -(the sweep values that produced it) alongside each case, so `resultstable` can join timings -back against cost figures after the fact. `build_suite` consumes a `Vector{BenchmarkCase}` and -produces an ordinary `BenchmarkGroup`; nothing downstream of that ever sees `BenchmarkCase` -again. - -Filtering uses `BenchmarkGroup`'s native tags, not a bespoke mechanism: each case is wrapped as -`BenchmarkGroup(casetags(case), "benchmark" => ...)`, where `casetags` is the category name plus -every `Symbol`-valued `params` entry (`source`, `kind`, `variant`, `layout`, ...). Filter the -*built* suite with `@tagged` (re-exported from BenchmarkTools) before running it: - -```julia -suite = build_suite(providers) -run(suite[@tagged "tccg"]) # only the TCCG-sourced pairwise cases -run(suite[@tagged "pairwise" && "both_permuted"]) # boolean tag expressions work -``` +`register_category!(:mycategory, mygenerator)`. Nothing else changes. Filtering uses +`BenchmarkGroup`'s native tags (`suite[@tagged "..."]`, see `BenchmarkCase`'s docstring for how +tags are derived) -- no bespoke filter API to learn. ## Plugging in a downstream tensor type diff --git a/benchmark/scripts/run_benchmarks.jl b/benchmark/scripts/run_benchmarks.jl index 5a7b6385..525d707c 100644 --- a/benchmark/scripts/run_benchmarks.jl +++ b/benchmark/scripts/run_benchmarks.jl @@ -33,7 +33,7 @@ function parse_commandline(args) return parse_args(args, s) end -function main(args) +function (@main)(args) opts = parse_commandline(args) blascounts = isempty(opts["blas-threads"]) ? [nothing] : opts["blas-threads"] stridedcounts = isempty(opts["strided-threads"]) ? [nothing] : opts["strided-threads"] @@ -53,5 +53,3 @@ function main(args) end return 0 end - -@main diff --git a/benchmark/scripts/show_benchmarks.jl b/benchmark/scripts/show_benchmarks.jl index 7b6c94af..27cac4b3 100644 --- a/benchmark/scripts/show_benchmarks.jl +++ b/benchmark/scripts/show_benchmarks.jl @@ -22,7 +22,7 @@ function parse_commandline(args) return parse_args(args, s) end -function main(args) +function (@main)(args) opts = parse_commandline(args) results = PkgBenchmark.readresults(opts["resultfile"]) group = PkgBenchmark.benchmarkgroup(results) @@ -30,18 +30,18 @@ function main(args) rows = resultstable(group) fig = Figure(; size = (1000, 800)) - categories = unique(r.category for r in rows) + categories = unique(r.category for r in eachrow(rows)) for (i, category) in enumerate(categories) ax = Axis( fig[fldmod1(i, 2)...]; xscale = log2, yscale = log10, title = category, xlabel = "size", ylabel = "GFLOP/s" ) catrows = filter(r -> r.category == category, rows) - for provider in unique(r.provider for r in catrows) + for provider in unique(r.provider for r in eachrow(catrows)) provrows = filter(r -> r.provider == provider, catrows) sort!(provrows; by = r -> get(r.params, :dim, get(r.params, :D, 0))) - xs = [get(r.params, :dim, get(r.params, :D, 0)) for r in provrows] - ys = [r.gflops for r in provrows] + xs = [get(r.params, :dim, get(r.params, :D, 0)) for r in eachrow(provrows)] + ys = [r.gflops for r in eachrow(provrows)] lines!(ax, xs, ys; label = provider) scatter!(ax, xs, ys) end @@ -53,5 +53,3 @@ function main(args) @info "Wrote $outfile" return 0 end - -@main diff --git a/benchmark/src/lowering.jl b/benchmark/src/lowering.jl index b84724e0..0a7d3a3a 100644 --- a/benchmark/src/lowering.jl +++ b/benchmark/src/lowering.jl @@ -13,7 +13,7 @@ function maketensors(spec::AddSpec, provider) A = randtensor(provider, spec.IA, dims, TA) pA = TensorOperations.add_indices(spec.IA, spec.IC) C = TensorOperations.tensoralloc_add(TC, A, pA, spec.conjA, Val(false), allocator(provider)) - return (A, pA, C) + return (C, A, pA) end function maketensors(spec::TraceSpec, provider) @@ -23,7 +23,7 @@ function maketensors(spec::TraceSpec, provider) A = randtensor(provider, spec.IA, dims, TA) p, q = TensorOperations.trace_indices(spec.IA, spec.IC) C = TensorOperations.tensoralloc_add(TC, A, p, spec.conjA, Val(false), allocator(provider)) - return (A, p, q, C) + return (C, A, p, q) end function maketensors(spec::ContractSpec, provider) @@ -38,7 +38,7 @@ function maketensors(spec::ContractSpec, provider) C = TensorOperations.tensoralloc_contract( TC, A, pA, spec.conjA, B, pB, spec.conjB, pAB, Val(false), allocator(provider) ) - return (A, B, pA, pB, pAB, C) + return (C, A, pA, B, pB, pAB) end function maketensors(spec::NetworkSpec, provider) @@ -49,20 +49,20 @@ function maketensors(spec::NetworkSpec, provider) end end -function execute(spec::AddSpec, (A, pA, C), provider) +function execute(spec::AddSpec, (C, A, pA), provider) return tensorcopy!( C, A, pA, spec.conjA, one(eltype(C)), backend(provider), allocator(provider) ) end -function execute(spec::TraceSpec, (A, p, q, C), provider) +function execute(spec::TraceSpec, (C, A, p, q), provider) return tensortrace!( C, A, p, q, spec.conjA, one(eltype(C)), zero(eltype(C)), backend(provider), allocator(provider) ) end -function execute(spec::ContractSpec, (A, B, pA, pB, pAB, C), provider) +function execute(spec::ContractSpec, (C, A, pA, B, pB, pAB), provider) return tensorcontract!( C, A, pA, spec.conjA, B, pB, spec.conjB, pAB, one(eltype(C)), zero(eltype(C)), backend(provider), allocator(provider) diff --git a/benchmark/src/registry.jl b/benchmark/src/registry.jl index c25e05e0..5ac4b9d0 100644 --- a/benchmark/src/registry.jl +++ b/benchmark/src/registry.jl @@ -3,26 +3,24 @@ """ BenchmarkCase(category, id, params, spec) -One concrete benchmark case: `category` groups it (e.g. `:pairwise`), `id` is a short unique +One concrete benchmark case: `category` groups it (e.g. `:contract`), `id` is a short unique label within the category, `params` records the sweep parameters that produced it (e.g. -`(; D=64)`), and `spec` is the [`AbstractCaseSpec`](@ref) to execute. +`(; D=64)`), and `spec` is the [`AbstractCaseSpec`](@ref) to execute. `tags` (for +`BenchmarkTools.@tagged`-based filtering, see suite.jl) is computed once here and stored, +rather than re-derived on demand: the category name, plus every `Symbol`-valued entry of +`params` (e.g. `source`, `kind`, `variant`, `layout`, `topic`) -- a generator opts a `params` +field into tag-filtering just by giving it a `Symbol` value. """ struct BenchmarkCase category::Symbol id::String params::NamedTuple + tags::Vector{Any} spec::AbstractCaseSpec end - -""" - casetags(case::BenchmarkCase) -> Vector - -Tags for `BenchmarkTools.@tagged`-based filtering (see suite.jl): the category name, plus every -`Symbol`-valued entry of `params` (e.g. `source`, `kind`, `variant`, `layout`). A category -generator opts a `params` field into tag-filtering just by giving it a `Symbol` value. -""" -function casetags(case::BenchmarkCase) - return Any[String(case.category); [string(v) for v in values(case.params) if v isa Symbol]] +function BenchmarkCase(category::Symbol, id::AbstractString, params::NamedTuple, spec::AbstractCaseSpec) + tags = Any[String(category); [string(v) for v in values(params) if v isa Symbol]] + return BenchmarkCase(category, String(id), params, tags, spec) end const REGISTRY = Dict{Symbol, Function}() @@ -50,12 +48,22 @@ Each category interprets `sizes` in its own way (a list of ranks, of bond dimens default_sizes(category::Symbol) = get(DEFAULT_SIZES, category, (4, 8, 16, 32, 64, 128)) """ - within_memory_budget(spec::AbstractCaseSpec; maxbytes=MAX_CASE_BYTES) + within_memory_budget(spec::AbstractCaseSpec, [id]; maxbytes=MAX_CASE_BYTES) Whether executing `spec` would stay within `maxbytes` of total tensor memory, assuming (in the absence of an explicit `TA`/`TB`/`TC`/`Ts` override) worst-case `Float64`-sized elements. -Category generators use this to silently skip dimension/shape combinations that would -otherwise OOM the benchmark process, rather than hand-tuning per-shape size ceilings. +Category generators use this to skip dimension/shape combinations that would otherwise OOM the +benchmark process, rather than hand-tuning per-shape size ceilings. Passing `id` additionally +`@warn`s (naming `id`) when a case is skipped, so an unexpectedly sparse sweep is diagnosable +rather than silent. + +`MAX_CASE_BYTES` scales with the host's total memory (`Sys.total_memory() ÷ 64`) rather than a +fixed constant, since this ranges from CI runners to large shared workstations. """ -const MAX_CASE_BYTES = 2^28 # 256 MiB +const MAX_CASE_BYTES = Sys.total_memory() ÷ 64 within_memory_budget(spec::AbstractCaseSpec; maxbytes = MAX_CASE_BYTES) = bytes(spec) <= maxbytes +function within_memory_budget(spec::AbstractCaseSpec, id; maxbytes = MAX_CASE_BYTES) + ok = within_memory_budget(spec; maxbytes) + ok || @warn "Skipping benchmark case: exceeds memory budget" id maxbytes bytes = bytes(spec) + return ok +end diff --git a/benchmark/src/report.jl b/benchmark/src/report.jl index 8b95b675..1c6c0149 100644 --- a/benchmark/src/report.jl +++ b/benchmark/src/report.jl @@ -1,37 +1,21 @@ -# Flattens a benchmark result into rows, joining timings back against specs (regenerated from -# REGISTRY) for GFLOP/s and bandwidth. - -""" - ResultRow - -One row of [`resultstable`](@ref): `category`, `provider`, `id`, `params`, `mintime` (ns), -`allocs`, `memory` (bytes), `gflops` (`flops(spec) / mintime`, or `missing` if `mintime` is -zero), and `gbps` (`bytes(spec) / mintime`). -""" -struct ResultRow - category::String - provider::String - id::String - params::NamedTuple - mintime::Float64 - allocs::Int - memory::Int - gflops::Union{Float64, Missing} - gbps::Union{Float64, Missing} -end +# Flattens a benchmark result into a DataFrame, joining timings back against specs +# (regenerated from REGISTRY) for GFLOP/s and bandwidth. """ resultstable(results::BenchmarkGroup; categories=collect(keys(REGISTRY)), sizes=nothing) Flatten a benchmark-run result (with the same `[category][provider][id]["benchmark"]` nesting -`build_suite` produces) into a `Vector{ResultRow}`. `categories`/`sizes` must match what was -passed to the `build_suite` call that produced `results`, since specs (and therefore -flop/byte counts) are regenerated from `REGISTRY` rather than stored in the result itself. +`build_suite` produces) into a `DataFrame` with columns `category`, `provider`, `id`, `params` +(the originating `NamedTuple`), `mintime` (ns), `allocs`, `memory` (bytes), `gflops` +(`flops(spec) / mintime`, or `missing` if `mintime` is zero), and `gbps` (`bytes(spec) / +mintime`). `categories`/`sizes` must match what was passed to the `build_suite` call that +produced `results`, since specs (and therefore flop/byte counts) are regenerated from +`REGISTRY` rather than stored in the result itself. """ function resultstable( results::BenchmarkGroup; categories = collect(keys(REGISTRY)), sizes = nothing ) - rows = ResultRow[] + rows = NamedTuple[] for category in categories haskey(results, String(category)) || continue cases = REGISTRY[category](_sizes_for(sizes, category)) @@ -43,21 +27,20 @@ function resultstable( trial = provgroup[id]["benchmark"] case = casesbyid[id] mintime = minimum(trial.times) - memory = trial.memory - allocs = trial.allocs fl = flops(case.spec) by = bytes(case.spec) gflops = mintime > 0 ? fl / mintime : missing gbps = mintime > 0 ? by / mintime : missing push!( rows, - ResultRow( - String(category), providerlabel, id, case.params, - mintime, allocs, memory, gflops, gbps + (; + category = String(category), provider = providerlabel, id, + params = case.params, mintime, allocs = trial.allocs, + memory = trial.memory, gflops, gbps, ) ) end end end - return rows + return DataFrame(rows) end diff --git a/benchmark/src/specs.jl b/benchmark/src/specs.jl index 14affc81..1efeedda 100644 --- a/benchmark/src/specs.jl +++ b/benchmark/src/specs.jl @@ -97,3 +97,39 @@ function NetworkSpec( collect(Int, outputindices), dims, order, Ts ) end + +# Compact einsum-style show methods, e.g. `ContractSpec: C[i,j] = A[i,k] * B[k,j] (dim=64)`. +_dimsnote(dims::Dict) = (vals = unique(values(dims)); length(vals) == 1 ? " (dim=$(only(vals)))" : " (dims=$dims)") +_opstr(label, conj) = conj ? "conj($label)" : label + +function Base.show(io::IO, spec::AddSpec) + return print( + io, "AddSpec: C[", join(spec.IC, ","), "] = ", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::TraceSpec) + return print( + io, "TraceSpec: C[", join(spec.IC, ","), "] = tr(", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), ")", _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::ContractSpec) + return print( + io, "ContractSpec: C[", join(spec.IC, ","), "] = ", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), " * ", + _opstr("B[$(join(spec.IB, ","))]", spec.conjB), _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::NetworkSpec) + tensors = join( + ( + _opstr("T$k[$(join(il, ","))]", conj) + for (k, (il, conj)) in enumerate(zip(spec.indexlists, spec.conjlist)) + ), " * " + ) + return print(io, "NetworkSpec: C[", join(spec.output, ","), "] = ", tensors, _dimsnote(spec.dims)) +end diff --git a/benchmark/src/suite.jl b/benchmark/src/suite.jl index fe00e17c..9f944b45 100644 --- a/benchmark/src/suite.jl +++ b/benchmark/src/suite.jl @@ -1,7 +1,7 @@ # Nests a BenchmarkGroup as [category][provider label][case id]["benchmark"]. Each case-id leaf -# is itself a tagged BenchmarkGroup (tags from `casetags`: category + Symbol-valued params), so -# `suite[BenchmarkTools.@tagged "tccg"]` (or any boolean tag expression) selects a subset of -# cases natively -- no bespoke filter predicate needed. No threading axis -- see threading.jl. +# is itself a tagged BenchmarkGroup (using `case.tags`), so `suite[BenchmarkTools.@tagged +# "tccg"]` (or any boolean tag expression) selects a subset of cases natively -- no bespoke +# filter predicate needed. No threading axis -- see threading.jl. """ build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing) @@ -14,7 +14,7 @@ Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or t every category, or a `Dict{Symbol}` mapping category name to its own size sweep. To run only a tagged subset, filter *after* building: `suite[@tagged "tccg"]` (re-exported from -BenchmarkTools), or combine tags: `suite[@tagged "pairwise" && "tccg"]`. +BenchmarkTools), or combine tags: `suite[@tagged "contract" && "tccg"]`. """ function build_suite( providers::AbstractVector{<:AbstractProvider}; @@ -32,7 +32,7 @@ function build_suite( provgroup = catgroup[label(provider)] = BenchmarkGroup() for case in cases provgroup[case.id] = BenchmarkGroup( - casetags(case), "benchmark" => make_benchmarkable(case, provider) + case.tags, "benchmark" => make_benchmarkable(case, provider) ) end end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index ba138276..0ee93df1 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -2,6 +2,7 @@ using Test using TensorOperationsBenchmarks using TensorOperations: StridedNative using BenchmarkTools +using DataFrames: nrow using LinearAlgebra: BLAS using Strided: Strided @@ -17,12 +18,12 @@ using Strided: Strided end @testset "resultstable joins timings with flop/byte counts" begin - suite = build_suite([provider]; categories = [:pairwise], sizes = (4, 8)) + suite = build_suite([provider]; categories = [:contract], sizes = (4, 8)) results = run(suite; samples = 1, evals = 1, seconds = 5) - rows = resultstable(results; categories = [:pairwise], sizes = (4, 8)) - @test !isempty(rows) - @test all(r -> r.mintime > 0, rows) - @test all(r -> r.gflops isa Float64, rows) + rows = resultstable(results; categories = [:contract], sizes = (4, 8)) + @test nrow(rows) > 0 + @test all(>(0), rows.mintime) + @test all(x -> x isa Float64, rows.gflops) end @testset "mixed-precision cases execute" begin @@ -32,8 +33,8 @@ using Strided: Strided end @testset "network cost matches ncon's own contraction tree" begin - cases = REGISTRY[:mps]((16,)) - case = only(filter(c -> occursin("1site", c.id), cases)) + cases = REGISTRY[:network]((16,)) + case = only(filter(c -> occursin("mps_1site", c.id), cases)) @test flops(case.spec) > 0 end @@ -60,58 +61,65 @@ using Strided: Strided end @testset "execute doesn't allocate a fresh output every call (preallocated in setup)" begin - cases = REGISTRY[:pairwise]((8,)) + cases = REGISTRY[:contract]((8,)) case = first(cases) ts = TensorOperationsBenchmarks.maketensors(case.spec, provider) - C_before = ts[end] + C_before = ts[1] C_after = TensorOperationsBenchmarks.execute(case.spec, ts, provider) @test C_after === C_before end - @testset "mps category covers 1-site and 2-site variants" begin - cases = REGISTRY[:mps]((16,)) - @test any(c -> occursin("1site", c.id), cases) - @test any(c -> occursin("2site", c.id), cases) + @testset "network category covers mps/ctmrg/trg topics" begin + cases = REGISTRY[:network]((16,)) + @test any(c -> c.params.topic == :mps && c.params.variant == :onesite, cases) + @test any(c -> c.params.topic == :mps && c.params.variant == :twosite, cases) + @test any(c -> c.params.topic == :ctmrg, cases) + @test any(c -> c.params.topic == :trg, cases) end - @testset "tccg cases are merged into :pairwise, filterable via @tagged" begin - cases = REGISTRY[:pairwise]((4,)) + @testset "tccg cases are merged into :contract, filterable via @tagged" begin + cases = REGISTRY[:contract]((4,)) tccg_cases = filter(c -> c.params.source == :tccg, cases) @test length(TensorOperationsBenchmarks.TCCG_CONTRACTIONS) == 24 @test any(c -> c.params.source == :synthetic, cases) for prefix in ("ccsd_", "ccsd_t_", "ao2mo_", "intensli_") @test any(c -> startswith(c.params.equation, prefix), tccg_cases) end - suite = build_suite([provider]; categories = [:pairwise], sizes = (4,)) + suite = build_suite([provider]; categories = [:contract], sizes = (4,)) results = run(suite[@tagged "tccg"]; samples = 1, evals = 1, seconds = 5) - @test !isempty(results["pairwise"][label(provider)]) - @test length(results["pairwise"][label(provider)]) == length(tccg_cases) + @test !isempty(results["contract"][label(provider)]) + @test length(results["contract"][label(provider)]) == length(tccg_cases) end - @testset "casetags derives from category + Symbol-valued params" begin + @testset "BenchmarkCase stores tags derived from category + Symbol-valued params" begin case = first(REGISTRY[:trace]((8,))) - tags = casetags(case) - @test "trace" in tags - @test string(case.params.kind) in tags + @test "trace" in case.tags + @test string(case.params.kind) in case.tags end - @testset "pairwise permuted-stride layouts are structurally distinct and filterable" begin - cases = REGISTRY[:pairwise]((8,)) + @testset "contract permuted-stride layouts are structurally distinct and filterable" begin + cases = REGISTRY[:contract]((8,)) layouts = unique(c.params.layout for c in cases if c.params.source == :synthetic) @test :gemm_ready in layouts @test :both_permuted in layouts - suite = build_suite([provider]; categories = [:pairwise], sizes = (8,)) + suite = build_suite([provider]; categories = [:contract], sizes = (8,)) results = run(suite[@tagged "both_permuted"]; samples = 1, evals = 1, seconds = 5) n_expected = count(c -> c.params.source == :synthetic && c.params.layout == :both_permuted, cases) - @test length(results["pairwise"][label(provider)]) == n_expected + @test length(results["contract"][label(provider)]) == n_expected end - @testset "ctmrg and trg categories execute" begin - for category in (:ctmrg, :trg) - suite = build_suite([provider]; categories = [category], sizes = (8,)) - results = run(suite; samples = 1, evals = 1, seconds = 5) - @test !isempty(results[String(category)][label(provider)]) - end + @testset "within_memory_budget(spec, id) warns when skipping an oversized case" begin + spec = ContractSpec([:a1, :a2], [:a2, :b1], [:a1, :b1], Dict(:a1 => 8, :a2 => 8, :b1 => 8)) + # override maxbytes (rather than relying on the host's memory-scaled default) so this + # test is deterministic regardless of how much RAM the machine running it has. + @test_logs (:warn, r"exceeds memory budget") within_memory_budget(spec, "tiny_budget_test"; maxbytes = 10) + end + + @testset "specs have informative show methods" begin + spec = ContractSpec([:i, :k], [:k, :j], [:i, :j], Dict(:i => 4, :j => 4, :k => 4)) + @test occursin("C[i,j]", sprint(show, spec)) + @test occursin("A[i,k]", sprint(show, spec)) + @test occursin("B[k,j]", sprint(show, spec)) end end From 129ea45521901d2a4e974e722f3d4ee5c3110a71 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Sat, 26 Sep 2026 09:19:26 -0400 Subject: [PATCH 16/18] Add GEMM-resistant contract shapes, batched cases, blas tag, arithmetic intensity Prior :contract shapes were all reshapeable to a single GEMM call. Add: - a contract_scrambled layout (contracted indices contiguous but differently ordered between operands, per TAPP's index taxonomy) that still isn't reshapeable despite looking like it should be - two higher-rank/asymmetric shapes (environment-into-tensor, 8-leg bundle) - a BatchedContractSpec (tenferro-rs's bij,bjk->bik motif): independent per-slice tensorcontract! calls, since no fused batched-GEMM primitive exists, isolating per-call dispatch overhead as its own regime Also add a structural isblasequivalent(spec) check, auto-tagging every case (any category) "blas" when it collapses to a single BLAS call after only a reshape, and expose intensity(spec) = flops/bytes as a resultstable column. Co-Authored-By: Claude Sonnet 5 --- benchmark/README.md | 20 +++++--- benchmark/src/TensorOperationsBenchmarks.jl | 4 +- benchmark/src/categories/contract.jl | 54 +++++++++++++++++---- benchmark/src/cost.jl | 51 +++++++++++++++++++ benchmark/src/lowering.jl | 30 +++++++++++- benchmark/src/registry.jl | 4 +- benchmark/src/report.jl | 12 +++-- benchmark/src/specs.jl | 39 +++++++++++++++ benchmark/test/runtests.jl | 41 ++++++++++++++++ 9 files changed, 230 insertions(+), 25 deletions(-) diff --git a/benchmark/README.md b/benchmark/README.md index 09b7751c..607facfa 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -14,7 +14,7 @@ julia --project=. -e ' ArrayProvider{Float64}(; backend=StridedBLAS())] suite = build_suite(providers) results = run(suite) - rows = resultstable(results) + rows = resultstable(results) # includes gflops/gbps (measured) and intensity (flops/bytes, static) ' ``` @@ -29,12 +29,18 @@ julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # r ## Categories - `:contract` -- generic pairwise contractions: a synthetic parametric shape family (tagged - `synthetic`) plus 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from - the [TCCG benchmark](https://github.com/HPAC/tccg) (tagged `tccg`). Each synthetic shape also - comes in up to 4 label-order layouts (tagged `gemm_ready`/`a_permuted`/`b_permuted`/ - `both_permuted`): `gemm_ready` is directly reshapeable to a BLAS call, the others interleave - open/contracted labels so no reshape or transpose flag suffices -- a real permutation is - required, which is what actually separates `StridedNative` from `StridedBLAS`. + `synthetic`), 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from the + [TCCG benchmark](https://github.com/HPAC/tccg) (tagged `tccg`), and a batch of independent + small contractions (tagged `batched`, tenferro-rs's `bij,bjk->bik` motif -- no fused + batched-GEMM primitive exists here, so it's real per-call dispatch overhead, not one bigger + call). Each synthetic shape also comes in up to 5 label-order layouts: `gemm_ready` is directly + reshapeable to a BLAS call; `a_permuted`/`b_permuted`/`both_permuted` interleave open and + contracted labels so no reshape or transpose flag suffices; `contract_scrambled` keeps the + contracted indices contiguous in both operands but in a *different relative order* between + them (TAPP's `D[a,d,e] = A[a,b,c]*B[c,d,e,b]`), which still isn't reshapeable despite looking + like it should be. Every case (any category) is auto-tagged `blas` when `isblasequivalent` + holds -- exactly `gemm_ready` among the above, but computed structurally, so e.g. TCCG + equations that happen to be GEMM-ready get it too. - `:permute` -- permutation-only (`tensorcopy!`) cost. - `:trace` -- partial and full traces. - `:mixed_precision` -- differing input/output element types (e.g. `Float32 x Float32 -> diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl index eafe8382..f00e6f68 100644 --- a/benchmark/src/TensorOperationsBenchmarks.jl +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -36,8 +36,8 @@ include("categories/ctmrg.jl") include("categories/trg.jl") include("categories/network.jl") -export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, NetworkSpec -export flops, bytes +export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, BatchedContractSpec, NetworkSpec +export flops, bytes, intensity, isblasequivalent export AbstractProvider, ArrayProvider, scalartype, randtensor, backend, allocator, label, supports, rng export ThreadConfig, with_threads, set_threads! diff --git a/benchmark/src/categories/contract.jl b/benchmark/src/categories/contract.jl index c5dcc532..11456a05 100644 --- a/benchmark/src/categories/contract.jl +++ b/benchmark/src/categories/contract.jl @@ -1,12 +1,21 @@ -# Pairwise contractions: a synthetic parametric shape family, plus the real TCCG equations -# (tccg.jl) -- both just `ContractSpec`, merged into one category, tagged (`:synthetic`/`:tccg`) -# for filtering via `@tagged` (see registry.jl). Sizes mix power-of-two with off-by-one values. +# Pairwise contractions: a synthetic parametric shape family, the real TCCG equations +# (tccg.jl), and a batch of independent small contractions -- all just `ContractSpec`/ +# `BatchedContractSpec`, merged into one category, tagged (`:synthetic`/`:tccg`/`:batched`) for +# filtering via `@tagged` (see registry.jl). Sizes mix power-of-two with off-by-one values. # -# Each shape gets up to 4 label-order "layouts": `(openA...,contract...)`/`(contract...,openB...)` -# is directly reshapeable to GEMM (no data movement); interleaving open and contracted labels -# (e.g. `[a1,c1,a2,c2]`) cannot be expressed as a single reshape or BLAS transpose flag, forcing -# a real permutation -- exactly the case that separates StridedNative from StridedBLAS. Layouts -# that coincide with the GEMM one (e.g. when a shape has ≤1 contracted index) are skipped. +# Each shape gets up to 5 label-order "layouts" (see tenferro-rs's binary_diagnostic suite and +# TAPP's index taxonomy for the shapes this is drawn from): +# - `:gemm_ready`: `(openA...,contract...)`/`(contract...,openB...)`, reshapeable to GEMM with +# no data movement. +# - `:a_permuted`/`:b_permuted`/`:both_permuted`: open and contracted labels interleaved (e.g. +# `[a1,c1,a2,c2]`), which cannot be expressed as a single reshape or BLAS transpose flag, +# forcing a real permutation -- exactly what separates StridedNative from StridedBLAS. +# - `:contract_scrambled`: contracted indices form a contiguous block in *both* operands (so +# naively "looks" reshapeable) but in a *different relative order* between A and B -- e.g. +# TAPP's `D[a,d,e] = A[a,b,c]*B[c,d,e,b]`. No single reshape+transpose pair fixes this since +# the shared multi-index isn't flattened consistently between operands. +# Layouts that coincide with an earlier one (e.g. when a shape has ≤1 contracted index) are +# skipped by the dedup below. const CONTRACT_SHAPES = ( (1, 1, 1), # matrix-vector-like @@ -14,6 +23,8 @@ const CONTRACT_SHAPES = ( (2, 2, 2), # GEMM-like, rank 4 total (1, 3, 1), # trace-heavy: many contracted, few open (1, 0, 1), # pure outer product, no contraction + (0, 2, 3), # environment-into-tensor: A fully absorbed, no open legs of its own + (3, 2, 3), # high total rank (8), multiple bonds -- TBLIS/TAPP-style bundle-of-dims shape ) function _interleave(a::Vector{Symbol}, b::Vector{Symbol}) @@ -24,11 +35,13 @@ end function _contract_layouts(openA, contract, openB) gemmA, gemmB = vcat(openA, contract), vcat(contract, openB) permA, permB = _interleave(openA, contract), _interleave(contract, openB) + scrambledB = vcat(reverse(contract), openB) candidates = ( (:gemm_ready, gemmA, gemmB), (:a_permuted, permA, gemmB), (:b_permuted, gemmA, permB), (:both_permuted, permA, permB), + (:contract_scrambled, gemmA, scrambledB), ) seen = Set{Tuple{Vector{Symbol}, Vector{Symbol}}}() layouts = Tuple{Symbol, Vector{Symbol}, Vector{Symbol}}[] @@ -61,7 +74,30 @@ function _synthetic_contract_cases(sizes) return cases end -_contract_cases(sizes) = vcat(_synthetic_contract_cases(sizes), _tccg_cases(sizes)) +# Batch of independent small contractions (tenferro-rs's `bij,bjk->bik`-style motif): no fused +# batched-GEMM primitive exists here, so this is `batch` real dispatches, not one bigger call -- +# a distinct, call-overhead-dominated regime from the single large `ContractSpec` cases above. +const BATCH_SIZES = (4, 16, 64) + +function _batched_contract_cases(sizes) + cases = BenchmarkCase[] + IA, IB, IC = [:a1, :c1], [:c1, :b1], [:a1, :b1] + for dim in sizes + dims = Dict(:a1 => dim, :b1 => dim, :c1 => dim) + for batch in BATCH_SIZES + spec = BatchedContractSpec(batch, IA, IB, IC, dims) + id = "batched_dim$(dim)_batch$(batch)" + within_memory_budget(spec, id) || continue + params = (; dim, batch, source = :batched) + push!(cases, BenchmarkCase(:contract, id, params, spec)) + end + end + return cases +end + +_contract_cases(sizes) = vcat( + _synthetic_contract_cases(sizes), _tccg_cases(sizes), _batched_contract_cases(sizes) +) register_category!( :contract, _contract_cases; diff --git a/benchmark/src/cost.jl b/benchmark/src/cost.jl index babdaf3a..6c98d01f 100644 --- a/benchmark/src/cost.jl +++ b/benchmark/src/cost.jl @@ -21,6 +21,40 @@ Approximate number of bytes moved (all tensors read once, output written once) t """ function bytes end +""" + intensity(spec::AbstractCaseSpec) -> Float64 + +Arithmetic intensity `flops(spec) / bytes(spec)`, in flops/byte. Low intensity (permutations, +outer-product-free traces) is memory-bound; high intensity (GEMM-like contractions) is +compute-bound -- useful for separating "this got slower because of bandwidth" from "this got +slower because of compute" across the case set. +""" +intensity(spec::AbstractCaseSpec) = flops(spec) / bytes(spec) + +""" + isblasequivalent(spec::AbstractCaseSpec) -> Bool + +Whether `spec` can be executed as a single BLAS call after only a reshape, with no data +permutation: an identity `AddSpec`, or a `ContractSpec` whose contracted indices form a +contiguous block at the tail of `IA` and the head of `IB`, in the same relative order in both +(so a single flatten of the shared dimension is consistent between operands -- a `ContractSpec` +where the contracted indices appear in a *different* relative order in `IA` vs `IB`, e.g. TAPP's +`D[a,d,e] = A[a,b,c]*B[c,d,e,b]`, is contiguous in both operands yet still not BLAS-equivalent). +Used to auto-tag cases `"blas"` (see registry.jl); a `BatchedContractSpec` is never +BLAS-equivalent regardless of its per-slice layout, since it has no fused batched-GEMM +primitive to call as one BLAS operation. +""" +isblasequivalent(::AbstractCaseSpec) = false +isblasequivalent(spec::AddSpec) = spec.IA == spec.IC +function isblasequivalent(spec::ContractSpec) + contractedA = intersect(spec.IA, spec.IB) + contractedB = intersect(spec.IB, spec.IA) + contractedA == contractedB || return false + openA = setdiff(spec.IA, contractedA) + openB = setdiff(spec.IB, contractedB) + return spec.IA == vcat(openA, contractedA) && spec.IB == vcat(contractedB, openB) +end + # Permutation: no arithmetic, just data movement (one read + one write of every element). flops(::AddSpec) = 0 function bytes(spec::AddSpec) @@ -55,6 +89,23 @@ function bytes(spec::ContractSpec) return nA * _elsize(spec.TA) + nB * _elsize(spec.TB) + nC * _elsize(spec.TC) end +# batch independent tensorcontract! calls: batch x per-slice cost, not one bigger contraction. +function flops(spec::BatchedContractSpec) + contracted = intersect(spec.IA, spec.IB) + openA = setdiff(spec.IA, contracted) + openB = setdiff(spec.IB, contracted) + nopenA = prod((spec.dims[l] for l in openA); init = 1) + nopenB = prod((spec.dims[l] for l in openB); init = 1) + ncontracted = prod((spec.dims[l] for l in contracted); init = 1) + return spec.batch * 2 * nopenA * nopenB * ncontracted +end +function bytes(spec::BatchedContractSpec) + nA = prod((spec.dims[l] for l in spec.IA); init = 1) + nB = prod((spec.dims[l] for l in spec.IB); init = 1) + nC = prod((spec.dims[l] for l in spec.IC); init = 1) + return spec.batch * (nA * _elsize(spec.TA) + nB * _elsize(spec.TB) + nC * _elsize(spec.TC)) +end + # walks the same tree ncon itself builds (ncontree/indexordertree), not an arbitrary pairing. function flops(spec::NetworkSpec) tree = spec.order === nothing ? TensorOperations.ncontree(spec.indexlists) : diff --git a/benchmark/src/lowering.jl b/benchmark/src/lowering.jl index 0a7d3a3a..8dd20671 100644 --- a/benchmark/src/lowering.jl +++ b/benchmark/src/lowering.jl @@ -1,7 +1,8 @@ # Input/output tensors are built in setup=, not timed. Output `C` is preallocated via # tensoralloc_add/tensoralloc_contract and execution uses the mutating tensor*!, so the timed # region is the compute kernel, not an allocation -- except NetworkSpec/ncon, which has no -# in-place variant, so its timing includes allocation. +# in-place variant, so its timing includes allocation. BatchedContractSpec's timed region is +# `batch` separate tensorcontract! calls, intentionally (that per-call overhead is the point). _scalartype_or(::Nothing, provider) = TensorOperations.scalartype(provider) _scalartype_or(T::Type, provider) = T @@ -41,6 +42,23 @@ function maketensors(spec::ContractSpec, provider) return (C, A, pA, B, pB, pAB) end +function maketensors(spec::BatchedContractSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TB = _scalartype_or(spec.TB, provider) + TC = _scalartype_or(spec.TC, provider) + dimsA = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + dimsB = ntuple(i -> spec.dims[spec.IB[i]], length(spec.IB)) + pA, pB, pAB = TensorOperations.contract_indices(spec.IA, spec.IB, spec.IC) + As = [randtensor(provider, spec.IA, dimsA, TA) for _ in 1:spec.batch] + Bs = [randtensor(provider, spec.IB, dimsB, TB) for _ in 1:spec.batch] + Cs = [ + TensorOperations.tensoralloc_contract( + TC, As[b], pA, spec.conjA, Bs[b], pB, spec.conjB, pAB, Val(false), allocator(provider) + ) for b in 1:spec.batch + ] + return (Cs, As, pA, Bs, pB, pAB) +end + function maketensors(spec::NetworkSpec, provider) Ts = something(spec.Ts, fill(TensorOperations.scalartype(provider), length(spec.indexlists))) return map(spec.indexlists, Ts) do il, T @@ -69,6 +87,16 @@ function execute(spec::ContractSpec, (C, A, pA, B, pB, pAB), provider) ) end +function execute(spec::BatchedContractSpec, (Cs, As, pA, Bs, pB, pAB), provider) + for b in eachindex(Cs) + tensorcontract!( + Cs[b], As[b], pA, spec.conjA, Bs[b], pB, spec.conjB, pAB, + one(eltype(Cs[b])), zero(eltype(Cs[b])), backend(provider), allocator(provider) + ) + end + return Cs +end + function execute(spec::NetworkSpec, tensors, provider) return ncon( tensors, spec.indexlists, spec.conjlist; diff --git a/benchmark/src/registry.jl b/benchmark/src/registry.jl index 5ac4b9d0..573d349a 100644 --- a/benchmark/src/registry.jl +++ b/benchmark/src/registry.jl @@ -9,7 +9,8 @@ label within the category, `params` records the sweep parameters that produced i `BenchmarkTools.@tagged`-based filtering, see suite.jl) is computed once here and stored, rather than re-derived on demand: the category name, plus every `Symbol`-valued entry of `params` (e.g. `source`, `kind`, `variant`, `layout`, `topic`) -- a generator opts a `params` -field into tag-filtering just by giving it a `Symbol` value. +field into tag-filtering just by giving it a `Symbol` value. Also gets a `"blas"` tag when +[`isblasequivalent`](@ref) holds for `spec`, regardless of category. """ struct BenchmarkCase category::Symbol @@ -20,6 +21,7 @@ struct BenchmarkCase end function BenchmarkCase(category::Symbol, id::AbstractString, params::NamedTuple, spec::AbstractCaseSpec) tags = Any[String(category); [string(v) for v in values(params) if v isa Symbol]] + isblasequivalent(spec) && push!(tags, "blas") return BenchmarkCase(category, String(id), params, tags, spec) end diff --git a/benchmark/src/report.jl b/benchmark/src/report.jl index 1c6c0149..7bf03030 100644 --- a/benchmark/src/report.jl +++ b/benchmark/src/report.jl @@ -7,10 +7,12 @@ Flatten a benchmark-run result (with the same `[category][provider][id]["benchmark"]` nesting `build_suite` produces) into a `DataFrame` with columns `category`, `provider`, `id`, `params` (the originating `NamedTuple`), `mintime` (ns), `allocs`, `memory` (bytes), `gflops` -(`flops(spec) / mintime`, or `missing` if `mintime` is zero), and `gbps` (`bytes(spec) / -mintime`). `categories`/`sizes` must match what was passed to the `build_suite` call that -produced `results`, since specs (and therefore flop/byte counts) are regenerated from -`REGISTRY` rather than stored in the result itself. +(`flops(spec) / mintime`, or `missing` if `mintime` is zero), `gbps` (`bytes(spec) / mintime`), +and `intensity` (`flops(spec) / bytes(spec)`, flops/byte -- a static property of the case, not +the measurement, useful for a roofline-style plot against `gflops`). `categories`/`sizes` must +match what was passed to the `build_suite` call that produced `results`, since specs (and +therefore flop/byte counts) are regenerated from `REGISTRY` rather than stored in the result +itself. """ function resultstable( results::BenchmarkGroup; categories = collect(keys(REGISTRY)), sizes = nothing @@ -36,7 +38,7 @@ function resultstable( (; category = String(category), provider = providerlabel, id, params = case.params, mintime, allocs = trial.allocs, - memory = trial.memory, gflops, gbps, + memory = trial.memory, gflops, gbps, intensity = intensity(case.spec), ) ) end diff --git a/benchmark/src/specs.jl b/benchmark/src/specs.jl index 1efeedda..e8fcc798 100644 --- a/benchmark/src/specs.jl +++ b/benchmark/src/specs.jl @@ -68,6 +68,37 @@ function ContractSpec( ) end +""" + BatchedContractSpec(batch, IA, IB, IC, dims, conjA=false, conjB=false, TA=nothing, TB=nothing, TC=nothing) + +`batch` independent pairwise contractions, each with the same label structure, executed as +`batch` separate `tensorcontract!` calls -- TensorOperations has no fused batched-GEMM +primitive, so this genuinely cannot collapse to one BLAS call, unlike a `ContractSpec` with a +`:gemm_ready` layout. Models a "many small contractions" regime (e.g. attention-style batched +matmuls) where per-call dispatch overhead dominates rather than raw FLOPs. +""" +struct BatchedContractSpec <: AbstractCaseSpec + batch::Int + IA::Vector{Symbol} + IB::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + conjB::Bool + TA::Union{Nothing, Type} + TB::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function BatchedContractSpec( + batch, IA, IB, IC, dims; conjA::Bool = false, conjB::Bool = false, + TA = nothing, TB = nothing, TC = nothing + ) + return BatchedContractSpec( + Int(batch), collect(Symbol, IA), collect(Symbol, IB), collect(Symbol, IC), dims, + conjA, conjB, TA, TB, TC + ) +end + """ NetworkSpec(indexlists, conjlist, output, dims, order=nothing, Ts=nothing) @@ -124,6 +155,14 @@ function Base.show(io::IO, spec::ContractSpec) ) end +function Base.show(io::IO, spec::BatchedContractSpec) + return print( + io, "BatchedContractSpec: ", spec.batch, " x C[", join(spec.IC, ","), "] = ", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), " * ", + _opstr("B[$(join(spec.IB, ","))]", spec.conjB), _dimsnote(spec.dims) + ) +end + function Base.show(io::IO, spec::NetworkSpec) tensors = join( ( diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl index 0ee93df1..ed3f4bd1 100644 --- a/benchmark/test/runtests.jl +++ b/benchmark/test/runtests.jl @@ -116,6 +116,47 @@ using Strided: Strided @test_logs (:warn, r"exceeds memory budget") within_memory_budget(spec, "tiny_budget_test"; maxbytes = 10) end + @testset "isblasequivalent tags gemm-ready contractions, not permuted/scrambled ones" begin + cases = REGISTRY[:contract]((8,)) + gemmcase = only( + filter( + c -> c.params.source == :synthetic && c.params.layout == :gemm_ready && + c.params.nopenA == 2 && c.params.ncontract == 1 && c.params.nopenB == 2, + cases + ) + ) + @test "blas" in gemmcase.tags + scrambledcases = filter( + c -> c.params.source == :synthetic && c.params.layout == :contract_scrambled, cases + ) + @test !isempty(scrambledcases) + @test all(c -> !("blas" in c.tags), scrambledcases) + permutedcases = filter( + c -> c.params.source == :synthetic && c.params.layout == :both_permuted, cases + ) + @test !isempty(permutedcases) + @test all(c -> !("blas" in c.tags), permutedcases) + end + + @testset "batched contraction cases run as independent per-slice tensorcontract! calls" begin + cases = REGISTRY[:contract]((8,)) + batchedcases = filter(c -> c.params.source == :batched, cases) + @test !isempty(batchedcases) + @test all(c -> !("blas" in c.tags), batchedcases) + case = first(batchedcases) + ts = TensorOperationsBenchmarks.maketensors(case.spec, provider) + Cs = TensorOperationsBenchmarks.execute(case.spec, ts, provider) + @test length(Cs) == case.spec.batch + end + + @testset "resultstable reports a static arithmetic-intensity column" begin + suite = build_suite([provider]; categories = [:contract], sizes = (8,)) + results = run(suite[@tagged "gemm_ready"]; samples = 1, evals = 1, seconds = 5) + rows = resultstable(results; categories = [:contract], sizes = (8,)) + @test nrow(rows) > 0 + @test all(>(0), rows.intensity) + end + @testset "specs have informative show methods" begin spec = ContractSpec([:i, :k], [:k, :j], [:i, :j], Dict(:i => 4, :j => 4, :k => 4)) @test occursin("C[i,j]", sprint(show, spec)) From 065b78f1809335b7930b208d26b79a3125c90180 Mon Sep 17 00:00:00 2001 From: lkdvos Date: Sat, 26 Sep 2026 09:27:26 -0400 Subject: [PATCH 17/18] Add dim=4,6 to contract/permute/trace sweeps for L1-resident coverage Only 8.3% of cases fit in a typical 32 KiB L1 cache (median footprint ~2.5 MB, already past L2). Adding two smaller leg dimensions brings L1-resident coverage to 20.8% and pulls the median down to ~0.8 MB, without touching the large end of the sweep (still tops out near the memory-budget ceiling). Co-Authored-By: Claude Sonnet 5 --- benchmark/src/categories/contract.jl | 2 +- benchmark/src/categories/permute.jl | 2 +- benchmark/src/categories/trace.jl | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/benchmark/src/categories/contract.jl b/benchmark/src/categories/contract.jl index 11456a05..101cdbcb 100644 --- a/benchmark/src/categories/contract.jl +++ b/benchmark/src/categories/contract.jl @@ -101,5 +101,5 @@ _contract_cases(sizes) = vcat( register_category!( :contract, _contract_cases; - sizes = (8, 12, 15, 16, 24, 32, 63, 96, 128, 200, 256) + sizes = (4, 6, 8, 12, 15, 16, 24, 32, 63, 96, 128, 200, 256) ) diff --git a/benchmark/src/categories/permute.jl b/benchmark/src/categories/permute.jl index 7f72d01c..fb776af6 100644 --- a/benchmark/src/categories/permute.jl +++ b/benchmark/src/categories/permute.jl @@ -24,4 +24,4 @@ function _permute_cases(sizes) return cases end -register_category!(:permute, _permute_cases; sizes = (8, 15, 32, 63, 96, 128, 200, 256)) +register_category!(:permute, _permute_cases; sizes = (4, 6, 8, 15, 32, 63, 96, 128, 200, 256)) diff --git a/benchmark/src/categories/trace.jl b/benchmark/src/categories/trace.jl index 38cf8e32..adebaa2b 100644 --- a/benchmark/src/categories/trace.jl +++ b/benchmark/src/categories/trace.jl @@ -25,4 +25,4 @@ function _trace_cases(sizes) return cases end -register_category!(:trace, _trace_cases; sizes = (8, 15, 32, 63, 96, 128, 200, 256)) +register_category!(:trace, _trace_cases; sizes = (4, 6, 8, 15, 32, 63, 96, 128, 200, 256)) From d1ace864bb2d5a8e5db6d1d6d9d51c14ba19369e Mon Sep 17 00:00:00 2001 From: lkdvos Date: Sat, 26 Sep 2026 09:33:11 -0400 Subject: [PATCH 18/18] Drop (1,1,1) synthetic shape: exact duplicate of TCCG's ccsd_1 With 1 open leg per side and 1 contracted leg, every layout candidate collapses to gemm_ready (nowhere to permute a single element), which is then structurally identical to ccsd_1 (C[i,j]=A[i,k]*B[k,j]) at every swept dim -- pure duplication contributing no distinct coverage. Also note that TCCG's own ccsd_t_1..4 share one abstract shape by design (kept: each is independently citable, unlike our synthetic case). Co-Authored-By: Claude Sonnet 5 --- benchmark/src/categories/contract.jl | 4 +++- benchmark/src/categories/tccg.jl | 5 ++++- 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/benchmark/src/categories/contract.jl b/benchmark/src/categories/contract.jl index 101cdbcb..ba07161d 100644 --- a/benchmark/src/categories/contract.jl +++ b/benchmark/src/categories/contract.jl @@ -18,7 +18,9 @@ # skipped by the dedup below. const CONTRACT_SHAPES = ( - (1, 1, 1), # matrix-vector-like + # No (1,1,1): with 1 open leg each side and 1 contracted leg, every layout candidate below + # collapses to `gemm_ready` (nowhere else to put a single element), which is then exactly + # TCCG's ccsd_1 (C[i,j]=A[i,k]*B[k,j]) -- pure duplication, so this shape is dropped. (2, 1, 2), # single shared bond, several open legs each side (2, 2, 2), # GEMM-like, rank 4 total (1, 3, 1), # trace-heavy: many contracted, few open diff --git a/benchmark/src/categories/tccg.jl b/benchmark/src/categories/tccg.jl index 3f1641aa..4cababc1 100644 --- a/benchmark/src/categories/tccg.jl +++ b/benchmark/src/categories/tccg.jl @@ -14,7 +14,10 @@ const TCCG_CONTRACTIONS = ( (id = "ccsd_7", C = "ijkl", A = "imjn", B = "lnkm"), (id = "ccsd_8", C = "ijkl", A = "imjn", B = "nlmk"), (id = "ccsd_9", C = "ijkl", A = "minl", B = "njmk"), - # CCSD(T) + # CCSD(T) -- ccsd_t_1..4 are 4 separately-named equations from the theory that happen to + # share one abstract contraction shape (structurally identical up to relabeling); kept as 4 + # since they're each independently citable, not because they exercise different performance + # regimes. (id = "ccsd_t_1", C = "abcijk", A = "ijma", B = "mkbc"), (id = "ccsd_t_2", C = "abcijk", A = "ijmb", B = "mkac"), (id = "ccsd_t_3", C = "abcijk", A = "ijmc", B = "mkab"),