diff --git a/benchmark/Project.toml b/benchmark/Project.toml new file mode 100644 index 00000000..ed4ce259 --- /dev/null +++ b/benchmark/Project.toml @@ -0,0 +1,32 @@ +name = "TensorOperationsBenchmarks" +uuid = "d983fc97-4e87-46ba-abd9-b4864e69d4dd" +authors = ["Lukas Devos "] +version = "0.1.0" + +[deps] +ArgParse = "c7e460c6-2fb9-53a9-8c5b-16f535851c63" +BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" +DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0" +LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" +PkgBenchmark = "32113eaa-f34f-5b0d-bd6c-c81e245fc73d" +Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" +Strided = "5e0ebb24-38b0-5f93-81fe-25c709ecae67" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" + +[compat] +ArgParse = "1" +BenchmarkTools = "1" +DataFrames = "1" +LinearAlgebra = "1.10" +PkgBenchmark = "0.2" +Random = "1.10" +Strided = "2.6" +TensorOperations = "5.8" +Test = "1" +julia = "1.11" + +[extras] +Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" + +[targets] +test = ["Test"] diff --git a/benchmark/README.md b/benchmark/README.md new file mode 100644 index 00000000..607facfa --- /dev/null +++ b/benchmark/README.md @@ -0,0 +1,87 @@ +# TensorOperationsBenchmarks + +Extensible benchmark suite for TensorOperations.jl. Downstream packages (e.g. a +symmetric/block-sparse tensor package) can plug in their own tensor type and run the same +standardized shapes. + +## Running + +```julia +julia --project=. -e ' + using TensorOperationsBenchmarks + using TensorOperations: StridedNative, StridedBLAS + providers = [ArrayProvider{Float64}(; backend=StridedNative()), + ArrayProvider{Float64}(; backend=StridedBLAS())] + suite = build_suite(providers) + results = run(suite) + rows = resultstable(results) # includes gflops/gbps (measured) and intensity (flops/bytes, static) +' +``` + +Or for commit-to-commit regression comparison via PkgBenchmark. Both scripts are `@main` apps +(Julia 1.11+), so `--help` works and `ARGS` are parsed the normal way: + +``` +julia --project=. scripts/run_benchmarks.jl --threads 1 4 --blas-threads 1 4 +julia --project=. scripts/show_benchmarks.jl results_t4_blas4_strided.json # requires CairoMakie +``` + +## Categories + +- `:contract` -- generic pairwise contractions: a synthetic parametric shape family (tagged + `synthetic`), 24 real quantum-chemistry contractions (CCSD, CCSD(T), AO2MO, INTENSLI) from the + [TCCG benchmark](https://github.com/HPAC/tccg) (tagged `tccg`), and a batch of independent + small contractions (tagged `batched`, tenferro-rs's `bij,bjk->bik` motif -- no fused + batched-GEMM primitive exists here, so it's real per-call dispatch overhead, not one bigger + call). Each synthetic shape also comes in up to 5 label-order layouts: `gemm_ready` is directly + reshapeable to a BLAS call; `a_permuted`/`b_permuted`/`both_permuted` interleave open and + contracted labels so no reshape or transpose flag suffices; `contract_scrambled` keeps the + contracted indices contiguous in both operands but in a *different relative order* between + them (TAPP's `D[a,d,e] = A[a,b,c]*B[c,d,e,b]`), which still isn't reshapeable despite looking + like it should be. Every case (any category) is auto-tagged `blas` when `isblasequivalent` + holds -- exactly `gemm_ready` among the above, but computed structurally, so e.g. TCCG + equations that happen to be GEMM-ready get it too. +- `:permute` -- permutation-only (`tensorcopy!`) cost. +- `:trace` -- partial and full traces. +- `:mixed_precision` -- differing input/output element types (e.g. `Float32 x Float32 -> + Float64`, mixed real/complex). +- `:network` -- multi-tensor-network motifs, tagged by `topic`: `mps` (MPS/MPO DMRG + effective-Hamiltonian, 1-site and 2-site "theta", swept over bond `D`), `ctmrg` (CTMRG + corner-growth step for 2D PEPS, swept over environment bond `chi`), `trg` (TRG plaquette + contraction, swept over bond `chi`). + +Not yet implemented, but addable without a redesign: MERA, contraction-order/path-finding timing. + +## Adding a category + +New file under `src/categories/`, define `mysizes -> Vector{BenchmarkCase}` building +`ContractSpec`/`TraceSpec`/`AddSpec`/`NetworkSpec` values, `include` it, call +`register_category!(:mycategory, mygenerator)`. Nothing else changes. Filtering uses +`BenchmarkGroup`'s native tags (`suite[@tagged "..."]`, see `BenchmarkCase`'s docstring for how +tags are derived) -- no bespoke filter API to learn. + +## Plugging in a downstream tensor type + +```julia +struct MyProvider <: AbstractProvider end +TensorOperations.scalartype(::MyProvider) = Float64 +TensorOperationsBenchmarks.randtensor(::MyProvider, labels, dims, T) = # build a random tensor +# optional: backend(p), allocator(p), label(p), supports(p, category), rng(p) +``` + +Then `build_suite([MyProvider(), ArrayProvider{Float64}()])` compares directly against +TensorOperations.jl's own backends. `ArrayProvider` shows the recommended `randtensor` pattern: +allocate via `TensorOperations.tensoralloc` (goes through `allocator`) and fill from a stored, +stateful `rng` field seeded once at construction, so runs are reproducible but tensors within a +run still differ. + +## Threading and precision + +Thread config is not a `build_suite` axis -- setting it per-case would count the switch itself +as part of the timing. Instead `set_threads!` is called once, at the process level, before a +suite is built (`benchmarks.jl` reads `TOB_BLAS_THREADS`/`TOB_STRIDED_THREADS`, set by +`run_benchmarks.jl --blas-threads/--strided-threads`). `with_threads(f, cfg)` is available for +ad hoc comparisons, wrapping a whole `run(suite)` call and restoring afterwards. + +Mixed precision: set `TA`/`TB`/`TC` explicitly on a spec (see `mixed_precision.jl`); other +categories leave them `nothing` (provider's default `scalartype`). diff --git a/benchmark/benchmarks.jl b/benchmark/benchmarks.jl new file mode 100644 index 00000000..cb60e373 --- /dev/null +++ b/benchmark/benchmarks.jl @@ -0,0 +1,17 @@ +# PkgBenchmark entrypoint: expects a top-level `const SUITE`. Thread counts come from env vars +# (set by scripts/run_benchmarks.jl) and are applied once, before SUITE is built. +using TensorOperationsBenchmarks +using TensorOperations: StridedNative, StridedBLAS + +set_threads!( + ThreadConfig(; + blas = tryparse(Int, get(ENV, "TOB_BLAS_THREADS", "")), + strided = tryparse(Int, get(ENV, "TOB_STRIDED_THREADS", "")), + ) +) + +const ELTYPES = (Float64, ComplexF64) +const BACKENDS = (StridedNative(), StridedBLAS()) +const PROVIDERS = [ArrayProvider{T}(; backend) for T in ELTYPES for backend in BACKENDS] + +const SUITE = build_suite(PROVIDERS) diff --git a/benchmark/scripts/run_benchmarks.jl b/benchmark/scripts/run_benchmarks.jl new file mode 100644 index 00000000..525d707c --- /dev/null +++ b/benchmark/scripts/run_benchmarks.jl @@ -0,0 +1,55 @@ +#!/usr/bin/env julia +# CLI wrapper around PkgBenchmark.benchmarkpkg. `--threads` relaunches Julia per value (fixed +# at startup); `--blas-threads`/`--strided-threads` set env vars benchmarks.jl reads. +# julia --project=. scripts/run_benchmarks.jl --threads 1 2 4 --blas-threads 1 4 --out results +using Pkg +Pkg.activate(joinpath(@__DIR__, "..")) + +using ArgParse +using PkgBenchmark + +function parse_commandline(args) + s = ArgParseSettings(; description = "Run the TensorOperationsBenchmarks suite via PkgBenchmark.") + @add_arg_table! s begin + "--threads" + help = "outer Julia process thread count(s) to sweep (relaunches Julia per value)" + arg_type = Int + nargs = '*' + default = [Threads.nthreads()] + "--blas-threads" + help = "inner BLAS thread count(s) to sweep" + arg_type = Int + nargs = '*' + default = Int[] + "--strided-threads" + help = "inner Strided.jl thread count(s) to sweep" + arg_type = Int + nargs = '*' + default = Int[] + "--out" + help = "output file prefix (a suffix identifying the thread combo and `.json` are appended)" + default = "results" + end + return parse_args(args, s) +end + +function (@main)(args) + opts = parse_commandline(args) + blascounts = isempty(opts["blas-threads"]) ? [nothing] : opts["blas-threads"] + stridedcounts = isempty(opts["strided-threads"]) ? [nothing] : opts["strided-threads"] + + for nthreads in opts["threads"], blas in blascounts, strided in stridedcounts + @info "Running benchmarks" nthreads blas strided + withenv( + "TOB_BLAS_THREADS" => blas === nothing ? "" : string(blas), + "TOB_STRIDED_THREADS" => strided === nothing ? "" : string(strided), + ) do + cfg = BenchmarkConfig(; juliacmd = `julia -t $nthreads -O3`) + results = benchmarkpkg(dirname(@__DIR__), cfg) + outfile = "$(opts["out"])_t$(nthreads)_blas$(blas)_strided$(strided).json" + writeresults(outfile, results) + @info "Wrote $outfile" + end + end + return 0 +end diff --git a/benchmark/scripts/show_benchmarks.jl b/benchmark/scripts/show_benchmarks.jl new file mode 100644 index 00000000..27cac4b3 --- /dev/null +++ b/benchmark/scripts/show_benchmarks.jl @@ -0,0 +1,55 @@ +#!/usr/bin/env julia +# Plots time/GFLOPs-vs-size curves from a run_benchmarks.jl result JSON. Requires CairoMakie +# (`Pkg.add("CairoMakie")` into this environment first -- not a package dependency, it's heavy). +using Pkg +Pkg.activate(joinpath(@__DIR__, "..")) + +using ArgParse +using PkgBenchmark +using CairoMakie +using TensorOperationsBenchmarks + +function parse_commandline(args) + s = ArgParseSettings(; description = "Plot GFLOP/s-vs-size scaling curves from a PkgBenchmark result.") + @add_arg_table! s begin + "resultfile" + help = "path to a PkgBenchmark result JSON, as written by run_benchmarks.jl" + required = true + "--out" + help = "output image path (default: replace the input's extension with .png)" + default = nothing + end + return parse_args(args, s) +end + +function (@main)(args) + opts = parse_commandline(args) + results = PkgBenchmark.readresults(opts["resultfile"]) + group = PkgBenchmark.benchmarkgroup(results) + + rows = resultstable(group) + + fig = Figure(; size = (1000, 800)) + categories = unique(r.category for r in eachrow(rows)) + for (i, category) in enumerate(categories) + ax = Axis( + fig[fldmod1(i, 2)...]; xscale = log2, yscale = log10, + title = category, xlabel = "size", ylabel = "GFLOP/s" + ) + catrows = filter(r -> r.category == category, rows) + for provider in unique(r.provider for r in eachrow(catrows)) + provrows = filter(r -> r.provider == provider, catrows) + sort!(provrows; by = r -> get(r.params, :dim, get(r.params, :D, 0))) + xs = [get(r.params, :dim, get(r.params, :D, 0)) for r in eachrow(provrows)] + ys = [r.gflops for r in eachrow(provrows)] + lines!(ax, xs, ys; label = provider) + scatter!(ax, xs, ys) + end + axislegend(ax) + end + + outfile = something(opts["out"], splitext(opts["resultfile"])[1] * ".png") + save(outfile, fig) + @info "Wrote $outfile" + return 0 +end diff --git a/benchmark/src/TensorOperationsBenchmarks.jl b/benchmark/src/TensorOperationsBenchmarks.jl new file mode 100644 index 00000000..f00e6f68 --- /dev/null +++ b/benchmark/src/TensorOperationsBenchmarks.jl @@ -0,0 +1,49 @@ +module TensorOperationsBenchmarks + +using LinearAlgebra: BLAS +using Strided: Strided +using Random: Random, randn! +using DataFrames: DataFrame +using BenchmarkTools +using TensorOperations +using TensorOperations: DefaultBackend, DefaultAllocator, AbstractBackend + +# Specs (pure data) and the cost model computed from them. +include("specs.jl") +include("cost.jl") + +# The downstream extension points: what tensor type to run against, and how many threads. +include("provider.jl") +include("threading.jl") + +# The case registry, and turning a (spec, provider) pair into an executable benchmark. +include("registry.jl") +include("lowering.jl") + +# Suite assembly and reporting. +include("suite.jl") +include("report.jl") + +# Categories: tccg.jl/mps.jl/ctmrg.jl/trg.jl define generators merged by contract.jl/network.jl +# (include order doesn't matter -- generators are only called after the whole module loads). +include("categories/tccg.jl") +include("categories/contract.jl") +include("categories/permute.jl") +include("categories/trace.jl") +include("categories/mixed_precision.jl") +include("categories/mps.jl") +include("categories/ctmrg.jl") +include("categories/trg.jl") +include("categories/network.jl") + +export AbstractCaseSpec, AddSpec, TraceSpec, ContractSpec, BatchedContractSpec, NetworkSpec +export flops, bytes, intensity, isblasequivalent +export AbstractProvider, ArrayProvider, scalartype, randtensor, backend, allocator, label, + supports, rng +export ThreadConfig, with_threads, set_threads! +export BenchmarkCase, register_category!, REGISTRY, default_sizes, within_memory_budget +export build_suite +export resultstable +export @tagged + +end # module diff --git a/benchmark/src/categories/contract.jl b/benchmark/src/categories/contract.jl new file mode 100644 index 00000000..ba07161d --- /dev/null +++ b/benchmark/src/categories/contract.jl @@ -0,0 +1,107 @@ +# Pairwise contractions: a synthetic parametric shape family, the real TCCG equations +# (tccg.jl), and a batch of independent small contractions -- all just `ContractSpec`/ +# `BatchedContractSpec`, merged into one category, tagged (`:synthetic`/`:tccg`/`:batched`) for +# filtering via `@tagged` (see registry.jl). Sizes mix power-of-two with off-by-one values. +# +# Each shape gets up to 5 label-order "layouts" (see tenferro-rs's binary_diagnostic suite and +# TAPP's index taxonomy for the shapes this is drawn from): +# - `:gemm_ready`: `(openA...,contract...)`/`(contract...,openB...)`, reshapeable to GEMM with +# no data movement. +# - `:a_permuted`/`:b_permuted`/`:both_permuted`: open and contracted labels interleaved (e.g. +# `[a1,c1,a2,c2]`), which cannot be expressed as a single reshape or BLAS transpose flag, +# forcing a real permutation -- exactly what separates StridedNative from StridedBLAS. +# - `:contract_scrambled`: contracted indices form a contiguous block in *both* operands (so +# naively "looks" reshapeable) but in a *different relative order* between A and B -- e.g. +# TAPP's `D[a,d,e] = A[a,b,c]*B[c,d,e,b]`. No single reshape+transpose pair fixes this since +# the shared multi-index isn't flattened consistently between operands. +# Layouts that coincide with an earlier one (e.g. when a shape has ≤1 contracted index) are +# skipped by the dedup below. + +const CONTRACT_SHAPES = ( + # No (1,1,1): with 1 open leg each side and 1 contracted leg, every layout candidate below + # collapses to `gemm_ready` (nowhere else to put a single element), which is then exactly + # TCCG's ccsd_1 (C[i,j]=A[i,k]*B[k,j]) -- pure duplication, so this shape is dropped. + (2, 1, 2), # single shared bond, several open legs each side + (2, 2, 2), # GEMM-like, rank 4 total + (1, 3, 1), # trace-heavy: many contracted, few open + (1, 0, 1), # pure outer product, no contraction + (0, 2, 3), # environment-into-tensor: A fully absorbed, no open legs of its own + (3, 2, 3), # high total rank (8), multiple bonds -- TBLIS/TAPP-style bundle-of-dims shape +) + +function _interleave(a::Vector{Symbol}, b::Vector{Symbol}) + n = min(length(a), length(b)) + return vcat((Symbol[a[i], b[i]] for i in 1:n)..., a[(n + 1):end], b[(n + 1):end]) +end + +function _contract_layouts(openA, contract, openB) + gemmA, gemmB = vcat(openA, contract), vcat(contract, openB) + permA, permB = _interleave(openA, contract), _interleave(contract, openB) + scrambledB = vcat(reverse(contract), openB) + candidates = ( + (:gemm_ready, gemmA, gemmB), + (:a_permuted, permA, gemmB), + (:b_permuted, gemmA, permB), + (:both_permuted, permA, permB), + (:contract_scrambled, gemmA, scrambledB), + ) + seen = Set{Tuple{Vector{Symbol}, Vector{Symbol}}}() + layouts = Tuple{Symbol, Vector{Symbol}, Vector{Symbol}}[] + for (layout, IA, IB) in candidates + (IA, IB) in seen && continue + push!(seen, (IA, IB)) + push!(layouts, (layout, IA, IB)) + end + return layouts +end + +function _synthetic_contract_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + for (nopenA, ncontract, nopenB) in CONTRACT_SHAPES + openA = [Symbol("a", i) for i in 1:nopenA] + contract = [Symbol("c", i) for i in 1:ncontract] + openB = [Symbol("b", i) for i in 1:nopenB] + IC = vcat(openA, openB) + for (layout, IA, IB) in _contract_layouts(openA, contract, openB) + dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) + spec = ContractSpec(IA, IB, IC, dims) + id = "dim$(dim)_$(nopenA)_$(ncontract)_$(nopenB)_$(layout)" + within_memory_budget(spec, id) || continue + params = (; dim, nopenA, ncontract, nopenB, layout, source = :synthetic) + push!(cases, BenchmarkCase(:contract, id, params, spec)) + end + end + end + return cases +end + +# Batch of independent small contractions (tenferro-rs's `bij,bjk->bik`-style motif): no fused +# batched-GEMM primitive exists here, so this is `batch` real dispatches, not one bigger call -- +# a distinct, call-overhead-dominated regime from the single large `ContractSpec` cases above. +const BATCH_SIZES = (4, 16, 64) + +function _batched_contract_cases(sizes) + cases = BenchmarkCase[] + IA, IB, IC = [:a1, :c1], [:c1, :b1], [:a1, :b1] + for dim in sizes + dims = Dict(:a1 => dim, :b1 => dim, :c1 => dim) + for batch in BATCH_SIZES + spec = BatchedContractSpec(batch, IA, IB, IC, dims) + id = "batched_dim$(dim)_batch$(batch)" + within_memory_budget(spec, id) || continue + params = (; dim, batch, source = :batched) + push!(cases, BenchmarkCase(:contract, id, params, spec)) + end + end + return cases +end + +_contract_cases(sizes) = vcat( + _synthetic_contract_cases(sizes), _tccg_cases(sizes), _batched_contract_cases(sizes) +) + +register_category!( + :contract, _contract_cases; + sizes = (4, 6, 8, 12, 15, 16, 24, 32, 63, 96, 128, 200, 256) +) diff --git a/benchmark/src/categories/ctmrg.jl b/benchmark/src/categories/ctmrg.jl new file mode 100644 index 00000000..be228bf0 --- /dev/null +++ b/benchmark/src/categories/ctmrg.jl @@ -0,0 +1,22 @@ +# CTMRG corner-growth step (boundary-MPS method for 2D PEPS): C-T-T-a, producing an unfused +# rank-4 corner (chi,chi,D2,D2). Cost ~ O(chi^3*D2^3) (literature O(chi^3*D^6), D2=D^2). +# `sizes` sweeps environment bond `chi`; PEPS bond `D` is fixed. Merged into `:network` (see +# network.jl) tagged `params.topic = :ctmrg`. + +const CTMRG_PEPS_BOND = 3 # D + +function _ctmrg_case(chi) + D2 = CTMRG_PEPS_BOND^2 + indexlists = [ + [1, 2], # C: (chi, chi) + [1, 3, -10], # T_left: (chi, D2, chi) + [2, 4, -11], # T_top: (chi, D2, chi) + [3, -12, 4, -13], # a (double-layer PEPS tensor): (D2, D2, D2, D2) + ] + dims = Dict(1 => chi, 2 => chi, 3 => D2, 4 => D2, 10 => chi, 11 => chi, 12 => D2, 13 => D2) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12, -13]) + params = (; chi, D = CTMRG_PEPS_BOND, topic = :ctmrg) + return BenchmarkCase(:network, "ctmrg_corner_chi$(chi)", params, spec) +end + +_ctmrg_cases(sizes) = [_ctmrg_case(chi) for chi in sizes] diff --git a/benchmark/src/categories/mixed_precision.jl b/benchmark/src/categories/mixed_precision.jl new file mode 100644 index 00000000..de10e032 --- /dev/null +++ b/benchmark/src/categories/mixed_precision.jl @@ -0,0 +1,26 @@ +# Contractions with differing input/output element types (promote_add/promote_contract). + +const MIXED_PRECISION_COMBOS = ( + (Float32, Float32, Float64), # low-precision inputs, high-precision accumulation + (Float64, ComplexF64, ComplexF64), # mixed real/complex + (Float64, ComplexF64, ComplexF32), # mixed real/complex, downcast output +) + +function _mixed_precision_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + for (TA, TB, TC) in MIXED_PRECISION_COMBOS + IA = [:a1, :c1] + IB = [:c1, :b1] + IC = [:a1, :b1] + dims = Dict{Symbol, Int}(:a1 => dim, :b1 => dim, :c1 => dim) + spec = ContractSpec(IA, IB, IC, dims; TA, TB, TC) + within_memory_budget(spec) || continue + id = "dim$(dim)_$(TA)_$(TB)_$(TC)" + push!(cases, BenchmarkCase(:mixed_precision, id, (; dim, TA, TB, TC), spec)) + end + end + return cases +end + +register_category!(:mixed_precision, _mixed_precision_cases; sizes = (33, 129, 500)) diff --git a/benchmark/src/categories/mps.jl b/benchmark/src/categories/mps.jl new file mode 100644 index 00000000..3927902f --- /dev/null +++ b/benchmark/src/categories/mps.jl @@ -0,0 +1,48 @@ +# MPS/MPO DMRG effective-Hamiltonian motif (L-MPS-MPO-R, cost ~ O(D^3*d*w)) plus the 2-site +# "theta" variant (L-MPS-MPO-MPO-MPS-R), swept over bond dimension `D`. Merged into `:network` +# (see network.jl) tagged `params.topic = :mps`. + +const MPS_PHYS_DIM = 2 # d: physical dimension (spin-1/2) +const MPS_MPO_BOND = 6 # w: MPO bond dimension (local Hamiltonian) + +function _mps_1site_case(D) + d, w = MPS_PHYS_DIM, MPS_MPO_BOND + # labels: 1=mpoL bond, 2=ket-L bond, 3=phys-in, 4=ket-R bond, 5=mpoR bond + # output (negative): -10=out-L bond, -11=out-phys, -12=out-R bond + indexlists = [ + [-10, 1, 2], # L: (D, w, D) + [2, 3, 4], # ket: (D, d, D) + [1, -11, 3, 5], # MPO: (w, d, d, w) + [4, 5, -12], # R: (D, w, D) + ] + dims = Dict(1 => w, 2 => D, 3 => d, 4 => D, 5 => w, 10 => D, 11 => d, 12 => D) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12]) + params = (; D, d, w, topic = :mps, variant = :onesite) + return BenchmarkCase(:network, "mps_1site_D$(D)", params, spec) +end + +function _mps_2site_case(D) + d, w = MPS_PHYS_DIM, MPS_MPO_BOND + # labels: 1=mpoL bond, 2=ket-L bond, 3=phys-in(1), 4=ket-mid bond, 5=phys-in(2), + # 6=mpo-mid bond, 7=mpoR bond, 8=ket-R bond + # output (negative): -10=out-L bond, -11=out-phys(1), -13=out-phys(2), -12=out-R bond + indexlists = [ + [-10, 1, 2], # L: (D, w, D) + [2, 3, 4], # ket1: (D, d, D) + [4, 5, 8], # ket2: (D, d, D) + [1, -11, 3, 6], # MPO1: (w, d, d, w) + [6, -13, 5, 7], # MPO2: (w, d, d, w) + [8, 7, -12], # R: (D, w, D) + ] + dims = Dict(1 => w, 2 => D, 3 => d, 4 => D, 5 => d, 6 => w, 7 => w, 8 => D, 10 => D, 11 => d, 12 => D, 13 => d) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -13, -12]) + params = (; D, d, w, topic = :mps, variant = :twosite) + return BenchmarkCase(:network, "mps_2site_D$(D)", params, spec) +end + +function _mps_cases(sizes) + return vcat( + BenchmarkCase[_mps_1site_case(D) for D in sizes], + BenchmarkCase[_mps_2site_case(D) for D in sizes] + ) +end diff --git a/benchmark/src/categories/network.jl b/benchmark/src/categories/network.jl new file mode 100644 index 00000000..16dfdc61 --- /dev/null +++ b/benchmark/src/categories/network.jl @@ -0,0 +1,11 @@ +# Merges the NetworkSpec-based motifs (mps.jl/ctmrg.jl/trg.jl) into one `:network` category, +# tagged by `params.topic` (`:mps`/`:ctmrg`/`:trg`) for `@tagged`-based filtering -- same pattern +# as `:contract` merging synthetic shapes with TCCG. `sizes` is the shared bond-dimension knob +# (`D` for mps, `chi` for ctmrg/trg); each topic interprets it independently. + +_network_cases(sizes) = vcat(_mps_cases(sizes), _ctmrg_cases(sizes), _trg_cases(sizes)) + +register_category!( + :network, _network_cases; + sizes = (16, 24, 32, 48, 64, 96, 100, 128, 256, 300, 512) +) diff --git a/benchmark/src/categories/permute.jl b/benchmark/src/categories/permute.jl new file mode 100644 index 00000000..fb776af6 --- /dev/null +++ b/benchmark/src/categories/permute.jl @@ -0,0 +1,27 @@ +# Permutation-only cost (`tensorcopy!`), isolated from contraction cost. Rank-4 tensor, sweeping +# leg dimension x how scrambled the permutation is. + +const PERMUTE_PATTERNS = ( + [1, 2, 3, 4], # identity (still exercises the copy machinery, no real permutation) + [2, 1, 3, 4], # single adjacent swap + [4, 3, 2, 1], # full reversal + [3, 1, 4, 2], # scrambled +) + +function _permute_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + IA = [Symbol("a", i) for i in 1:4] + dims = Dict{Symbol, Int}(l => dim for l in IA) + for pattern in PERMUTE_PATTERNS + IC = IA[pattern] + spec = AddSpec(IA, IC, dims) + within_memory_budget(spec) || continue + id = "dim$(dim)_perm$(join(pattern))" + push!(cases, BenchmarkCase(:permute, id, (; dim, pattern), spec)) + end + end + return cases +end + +register_category!(:permute, _permute_cases; sizes = (4, 6, 8, 15, 32, 63, 96, 128, 200, 256)) diff --git a/benchmark/src/categories/tccg.jl b/benchmark/src/categories/tccg.jl new file mode 100644 index 00000000..4cababc1 --- /dev/null +++ b/benchmark/src/categories/tccg.jl @@ -0,0 +1,56 @@ +# Real quantum-chemistry contractions from the TCCG benchmark (github.com/HPAC/tccg): CCSD, +# CCSD(T), AO2MO, INTENSLI, as "C-A-B" index strings (e.g. "ij-ik-kj" = C[i,j]=A[i,k]*B[k,j]). +# `sizes` applies one leg dimension uniformly to every index letter, as TCCG itself does. +# Merged into `:contract` (see contract.jl) tagged `params.source = :tccg`, not its own category. + +const TCCG_CONTRACTIONS = ( + # CCSD + (id = "ccsd_1", C = "ij", A = "ik", B = "kj"), + (id = "ccsd_2", C = "ij", A = "ikl", B = "ljk"), + (id = "ccsd_3", C = "ij", A = "kil", B = "lkj"), + (id = "ccsd_4", C = "ijk", A = "ikl", B = "lj"), + (id = "ccsd_5", C = "ijk", A = "ilk", B = "jl"), + (id = "ccsd_6", C = "ijk", A = "ilmk", B = "mjl"), + (id = "ccsd_7", C = "ijkl", A = "imjn", B = "lnkm"), + (id = "ccsd_8", C = "ijkl", A = "imjn", B = "nlmk"), + (id = "ccsd_9", C = "ijkl", A = "minl", B = "njmk"), + # CCSD(T) -- ccsd_t_1..4 are 4 separately-named equations from the theory that happen to + # share one abstract contraction shape (structurally identical up to relabeling); kept as 4 + # since they're each independently citable, not because they exercise different performance + # regimes. + (id = "ccsd_t_1", C = "abcijk", A = "ijma", B = "mkbc"), + (id = "ccsd_t_2", C = "abcijk", A = "ijmb", B = "mkac"), + (id = "ccsd_t_3", C = "abcijk", A = "ijmc", B = "mkab"), + (id = "ccsd_t_4", C = "abcijk", A = "ikmb", B = "mjac"), + # AO2MO integral transformation + (id = "ao2mo_1", C = "aqrs", A = "pa", B = "pqrs"), + (id = "ao2mo_2", C = "abrs", A = "qb", B = "aqrs"), + (id = "ao2mo_3", C = "abcs", A = "rc", B = "abrs"), + # INTENSLI + (id = "intensli_1", C = "abj", A = "bka", B = "kj"), + (id = "intensli_2", C = "ajb", A = "kba", B = "jk"), + (id = "intensli_3", C = "abjc", A = "cbka", B = "kj"), + (id = "intensli_4", C = "ajbc", A = "ckba", B = "jk"), + (id = "intensli_5", C = "abjc", A = "kbac", B = "jk"), + (id = "intensli_6", C = "abjcd", A = "dkbac", B = "kj"), + (id = "intensli_7", C = "adbjc", A = "cbdka", B = "kj"), + (id = "intensli_8", C = "ajbdc", A = "ckbad", B = "jk"), +) + +_tccg_labels(s::AbstractString) = [Symbol(c) for c in s] + +function _tccg_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + for eq in TCCG_CONTRACTIONS + IA, IB, IC = _tccg_labels(eq.A), _tccg_labels(eq.B), _tccg_labels(eq.C) + dims = Dict{Symbol, Int}(l => dim for l in vcat(IA, IB)) + spec = ContractSpec(IA, IB, IC, dims) + id = "$(eq.id)_dim$(dim)" + within_memory_budget(spec, id) || continue + params = (; dim, equation = eq.id, source = :tccg) + push!(cases, BenchmarkCase(:contract, id, params, spec)) + end + end + return cases +end diff --git a/benchmark/src/categories/trace.jl b/benchmark/src/categories/trace.jl new file mode 100644 index 00000000..adebaa2b --- /dev/null +++ b/benchmark/src/categories/trace.jl @@ -0,0 +1,28 @@ +# Partial trace (rank 6 -> rank 2) and full trace (rank 4 -> scalar), swept over leg dimension. + +function _trace_cases(sizes) + cases = BenchmarkCase[] + for dim in sizes + IA6 = [:o1, :o2, :t1, :t1, :t2, :t2] + IC6 = [:o1, :o2] + dims6 = Dict{Symbol, Int}(l => dim for l in unique(IA6)) + partialspec = TraceSpec(IA6, IC6, dims6) + if within_memory_budget(partialspec) + push!( + cases, + BenchmarkCase(:trace, "partial_dim$(dim)", (; dim, kind = :partial), partialspec) + ) + end + + IA4 = [:t1, :t1, :t2, :t2] + IC4 = Symbol[] + dims4 = Dict{Symbol, Int}(l => dim for l in unique(IA4)) + fullspec = TraceSpec(IA4, IC4, dims4) + if within_memory_budget(fullspec) + push!(cases, BenchmarkCase(:trace, "full_dim$(dim)", (; dim, kind = :full), fullspec)) + end + end + return cases +end + +register_category!(:trace, _trace_cases; sizes = (4, 6, 8, 15, 32, 63, 96, 128, 200, 256)) diff --git a/benchmark/src/categories/trg.jl b/benchmark/src/categories/trg.jl new file mode 100644 index 00000000..433f75d6 --- /dev/null +++ b/benchmark/src/categories/trg.jl @@ -0,0 +1,18 @@ +# TRG plaquette contraction: a 4-ring of rank-3 tensors (from SVD-splitting neighboring rank-4 +# TRG tensors), producing the coarse-grained rank-4 tensor. Cost ~ O(chi^6). `sizes` sweeps the +# bond dimension `chi`. Merged into `:network` (see network.jl) tagged `params.topic = :trg`. + +function _trg_case(chi) + indexlists = [ + [-10, 1, 2], + [2, -11, 3], + [3, -12, 4], + [4, -13, 1], + ] + dims = Dict(1 => chi, 2 => chi, 3 => chi, 4 => chi, 10 => chi, 11 => chi, 12 => chi, 13 => chi) + spec = NetworkSpec(indexlists, dims; output = [-10, -11, -12, -13]) + params = (; chi, topic = :trg) + return BenchmarkCase(:network, "trg_plaquette_chi$(chi)", params, spec) +end + +_trg_cases(sizes) = [_trg_case(chi) for chi in sizes] diff --git a/benchmark/src/cost.jl b/benchmark/src/cost.jl new file mode 100644 index 00000000..6c98d01f --- /dev/null +++ b/benchmark/src/cost.jl @@ -0,0 +1,136 @@ +# Analytical flop/byte counts computed from a spec alone, so timings can be reported as +# GFLOP/s / GB/s scaling curves. + +# `nothing` eltype means "provider not chosen yet"; assume 8 bytes (never overestimates memory). +_elsize(::Nothing) = sizeof(Float64) +_elsize(T::Type) = sizeof(T) + +""" + flops(spec::AbstractCaseSpec) -> Int + +Approximate number of floating point operations (multiply + add counted together) needed to +execute `spec`. +""" +function flops end + +""" + bytes(spec::AbstractCaseSpec) -> Int + +Approximate number of bytes moved (all tensors read once, output written once) to execute +`spec`. +""" +function bytes end + +""" + intensity(spec::AbstractCaseSpec) -> Float64 + +Arithmetic intensity `flops(spec) / bytes(spec)`, in flops/byte. Low intensity (permutations, +outer-product-free traces) is memory-bound; high intensity (GEMM-like contractions) is +compute-bound -- useful for separating "this got slower because of bandwidth" from "this got +slower because of compute" across the case set. +""" +intensity(spec::AbstractCaseSpec) = flops(spec) / bytes(spec) + +""" + isblasequivalent(spec::AbstractCaseSpec) -> Bool + +Whether `spec` can be executed as a single BLAS call after only a reshape, with no data +permutation: an identity `AddSpec`, or a `ContractSpec` whose contracted indices form a +contiguous block at the tail of `IA` and the head of `IB`, in the same relative order in both +(so a single flatten of the shared dimension is consistent between operands -- a `ContractSpec` +where the contracted indices appear in a *different* relative order in `IA` vs `IB`, e.g. TAPP's +`D[a,d,e] = A[a,b,c]*B[c,d,e,b]`, is contiguous in both operands yet still not BLAS-equivalent). +Used to auto-tag cases `"blas"` (see registry.jl); a `BatchedContractSpec` is never +BLAS-equivalent regardless of its per-slice layout, since it has no fused batched-GEMM +primitive to call as one BLAS operation. +""" +isblasequivalent(::AbstractCaseSpec) = false +isblasequivalent(spec::AddSpec) = spec.IA == spec.IC +function isblasequivalent(spec::ContractSpec) + contractedA = intersect(spec.IA, spec.IB) + contractedB = intersect(spec.IB, spec.IA) + contractedA == contractedB || return false + openA = setdiff(spec.IA, contractedA) + openB = setdiff(spec.IB, contractedB) + return spec.IA == vcat(openA, contractedA) && spec.IB == vcat(contractedB, openB) +end + +# Permutation: no arithmetic, just data movement (one read + one write of every element). +flops(::AddSpec) = 0 +function bytes(spec::AddSpec) + n = prod((spec.dims[l] for l in spec.IA); init = 1) + return n * (_elsize(spec.TA) + _elsize(spec.TC)) +end + +# Trace: every element of A is read and accumulated once into the (smaller) output. +function flops(spec::TraceSpec) + return prod((spec.dims[l] for l in spec.IA); init = 1) +end +function bytes(spec::TraceSpec) + nA = prod((spec.dims[l] for l in spec.IA); init = 1) + nC = prod((spec.dims[l] for l in spec.IC); init = 1) + return nA * _elsize(spec.TA) + nC * _elsize(spec.TC) +end + +# standard GEMM-equivalent flop count: 2 * open-A * open-B * contracted +function flops(spec::ContractSpec) + contracted = intersect(spec.IA, spec.IB) + openA = setdiff(spec.IA, contracted) + openB = setdiff(spec.IB, contracted) + nopenA = prod((spec.dims[l] for l in openA); init = 1) + nopenB = prod((spec.dims[l] for l in openB); init = 1) + ncontracted = prod((spec.dims[l] for l in contracted); init = 1) + return 2 * nopenA * nopenB * ncontracted +end +function bytes(spec::ContractSpec) + nA = prod((spec.dims[l] for l in spec.IA); init = 1) + nB = prod((spec.dims[l] for l in spec.IB); init = 1) + nC = prod((spec.dims[l] for l in spec.IC); init = 1) + return nA * _elsize(spec.TA) + nB * _elsize(spec.TB) + nC * _elsize(spec.TC) +end + +# batch independent tensorcontract! calls: batch x per-slice cost, not one bigger contraction. +function flops(spec::BatchedContractSpec) + contracted = intersect(spec.IA, spec.IB) + openA = setdiff(spec.IA, contracted) + openB = setdiff(spec.IB, contracted) + nopenA = prod((spec.dims[l] for l in openA); init = 1) + nopenB = prod((spec.dims[l] for l in openB); init = 1) + ncontracted = prod((spec.dims[l] for l in contracted); init = 1) + return spec.batch * 2 * nopenA * nopenB * ncontracted +end +function bytes(spec::BatchedContractSpec) + nA = prod((spec.dims[l] for l in spec.IA); init = 1) + nB = prod((spec.dims[l] for l in spec.IB); init = 1) + nC = prod((spec.dims[l] for l in spec.IC); init = 1) + return spec.batch * (nA * _elsize(spec.TA) + nB * _elsize(spec.TB) + nC * _elsize(spec.TC)) +end + +# walks the same tree ncon itself builds (ncontree/indexordertree), not an arbitrary pairing. +function flops(spec::NetworkSpec) + tree = spec.order === nothing ? TensorOperations.ncontree(spec.indexlists) : + TensorOperations.indexordertree(spec.indexlists, spec.order) + _, total = _tree_cost(spec, tree) do labelsA, labelsB, contracted + nopenA = prod((spec.dims[abs(l)] for l in labelsA if !(l in contracted)); init = 1) + nopenB = prod((spec.dims[abs(l)] for l in labelsB if !(l in contracted)); init = 1) + ncontracted = prod((spec.dims[abs(l)] for l in contracted); init = 1) + return 2 * nopenA * nopenB * ncontracted + end + return total +end +function bytes(spec::NetworkSpec) + Ts = something(spec.Ts, fill(nothing, length(spec.indexlists))) + return sum( + prod((spec.dims[abs(l)] for l in il); init = 1) * _elsize(T) + for (il, T) in zip(spec.indexlists, Ts) + ) +end + +# tree: leaves are Int indices into spec.indexlists, nodes are Any[left, right] (ncontree's format). +function _tree_cost(f, spec::NetworkSpec, tree) + tree isa Int && return spec.indexlists[tree], 0 + labelsA, costA = _tree_cost(f, spec, tree[1]) + labelsB, costB = _tree_cost(f, spec, tree[2]) + contracted = intersect(labelsA, labelsB) + return symdiff(labelsA, labelsB), costA + costB + f(labelsA, labelsB, contracted) +end diff --git a/benchmark/src/lowering.jl b/benchmark/src/lowering.jl new file mode 100644 index 00000000..8dd20671 --- /dev/null +++ b/benchmark/src/lowering.jl @@ -0,0 +1,120 @@ +# Input/output tensors are built in setup=, not timed. Output `C` is preallocated via +# tensoralloc_add/tensoralloc_contract and execution uses the mutating tensor*!, so the timed +# region is the compute kernel, not an allocation -- except NetworkSpec/ncon, which has no +# in-place variant, so its timing includes allocation. BatchedContractSpec's timed region is +# `batch` separate tensorcontract! calls, intentionally (that per-call overhead is the point). + +_scalartype_or(::Nothing, provider) = TensorOperations.scalartype(provider) +_scalartype_or(T::Type, provider) = T + +function maketensors(spec::AddSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TC = _scalartype_or(spec.TC, provider) + dims = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + A = randtensor(provider, spec.IA, dims, TA) + pA = TensorOperations.add_indices(spec.IA, spec.IC) + C = TensorOperations.tensoralloc_add(TC, A, pA, spec.conjA, Val(false), allocator(provider)) + return (C, A, pA) +end + +function maketensors(spec::TraceSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TC = _scalartype_or(spec.TC, provider) + dims = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + A = randtensor(provider, spec.IA, dims, TA) + p, q = TensorOperations.trace_indices(spec.IA, spec.IC) + C = TensorOperations.tensoralloc_add(TC, A, p, spec.conjA, Val(false), allocator(provider)) + return (C, A, p, q) +end + +function maketensors(spec::ContractSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TB = _scalartype_or(spec.TB, provider) + TC = _scalartype_or(spec.TC, provider) + dimsA = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + dimsB = ntuple(i -> spec.dims[spec.IB[i]], length(spec.IB)) + A = randtensor(provider, spec.IA, dimsA, TA) + B = randtensor(provider, spec.IB, dimsB, TB) + pA, pB, pAB = TensorOperations.contract_indices(spec.IA, spec.IB, spec.IC) + C = TensorOperations.tensoralloc_contract( + TC, A, pA, spec.conjA, B, pB, spec.conjB, pAB, Val(false), allocator(provider) + ) + return (C, A, pA, B, pB, pAB) +end + +function maketensors(spec::BatchedContractSpec, provider) + TA = _scalartype_or(spec.TA, provider) + TB = _scalartype_or(spec.TB, provider) + TC = _scalartype_or(spec.TC, provider) + dimsA = ntuple(i -> spec.dims[spec.IA[i]], length(spec.IA)) + dimsB = ntuple(i -> spec.dims[spec.IB[i]], length(spec.IB)) + pA, pB, pAB = TensorOperations.contract_indices(spec.IA, spec.IB, spec.IC) + As = [randtensor(provider, spec.IA, dimsA, TA) for _ in 1:spec.batch] + Bs = [randtensor(provider, spec.IB, dimsB, TB) for _ in 1:spec.batch] + Cs = [ + TensorOperations.tensoralloc_contract( + TC, As[b], pA, spec.conjA, Bs[b], pB, spec.conjB, pAB, Val(false), allocator(provider) + ) for b in 1:spec.batch + ] + return (Cs, As, pA, Bs, pB, pAB) +end + +function maketensors(spec::NetworkSpec, provider) + Ts = something(spec.Ts, fill(TensorOperations.scalartype(provider), length(spec.indexlists))) + return map(spec.indexlists, Ts) do il, T + dims = ntuple(i -> spec.dims[abs(il[i])], length(il)) + randtensor(provider, il, dims, T) + end +end + +function execute(spec::AddSpec, (C, A, pA), provider) + return tensorcopy!( + C, A, pA, spec.conjA, one(eltype(C)), backend(provider), allocator(provider) + ) +end + +function execute(spec::TraceSpec, (C, A, p, q), provider) + return tensortrace!( + C, A, p, q, spec.conjA, one(eltype(C)), zero(eltype(C)), + backend(provider), allocator(provider) + ) +end + +function execute(spec::ContractSpec, (C, A, pA, B, pB, pAB), provider) + return tensorcontract!( + C, A, pA, spec.conjA, B, pB, spec.conjB, pAB, one(eltype(C)), zero(eltype(C)), + backend(provider), allocator(provider) + ) +end + +function execute(spec::BatchedContractSpec, (Cs, As, pA, Bs, pB, pAB), provider) + for b in eachindex(Cs) + tensorcontract!( + Cs[b], As[b], pA, spec.conjA, Bs[b], pB, spec.conjB, pAB, + one(eltype(Cs[b])), zero(eltype(Cs[b])), backend(provider), allocator(provider) + ) + end + return Cs +end + +function execute(spec::NetworkSpec, tensors, provider) + return ncon( + tensors, spec.indexlists, spec.conjlist; + order = spec.order, output = spec.output, + backend = backend(provider), allocator = allocator(provider) + ) +end + +""" + make_benchmarkable(case::BenchmarkCase, provider::AbstractProvider) + +Build a `BenchmarkTools.Benchmark` for `case` run against `provider`, constructing fresh +input tensors and a fresh (preallocated, uninitialized) output tensor before every sample. +""" +function make_benchmarkable(case::BenchmarkCase, provider::AbstractProvider) + spec = case.spec + return @benchmarkable( + execute($spec, ts, $provider), + setup = (ts = maketensors($spec, $provider)) + ) +end diff --git a/benchmark/src/provider.jl b/benchmark/src/provider.jl new file mode 100644 index 00000000..8ebf9f8a --- /dev/null +++ b/benchmark/src/provider.jl @@ -0,0 +1,60 @@ +""" + AbstractProvider + +The downstream extension point: ties a spec (pure index/dimension data) to an actual tensor +type, backend, and allocator. + +Required: `scalartype(provider)`, `randtensor(provider, labels, dims, T=scalartype(provider))`. +Optional: `backend`, `allocator`, `label`, `supports(provider, category)`, `rng` (defaults +below) -- override `rng` with a *stored, stateful* RNG (as [`ArrayProvider`](@ref) does) so +repeated suite runs are reproducible without every tensor in one run being identical. +""" +abstract type AbstractProvider end + +function TensorOperations.scalartype(p::AbstractProvider) + error("`scalartype` not implemented for provider $(typeof(p))") +end + +""" + randtensor(provider, labels, dims, T=scalartype(provider)) + +Instantiate a random tensor with index order `labels` (a `Vector{Symbol}` or `Vector{Int}`, +only used for bookkeeping), size `dims` (matching `labels` elementwise), and element type `T`. +""" +function randtensor end + +backend(p::AbstractProvider) = DefaultBackend() +allocator(p::AbstractProvider) = DefaultAllocator() +label(p::AbstractProvider) = string(nameof(typeof(p))) +supports(p::AbstractProvider, category::Symbol) = true +rng(p::AbstractProvider) = Random.default_rng() + +""" + ArrayProvider{T}(; backend=DefaultBackend(), allocator=DefaultAllocator(), rng=Random.Xoshiro(0x5eed5eed5eed5eed)) + +The reference provider: plain dense `Array{T}`, exercising TensorOperations.jl's own backends. +Allocates via `TensorOperations.tensoralloc` (goes through `allocator`) and fills via +`randn!(rng, ...)`; `rng` is stored, not reseeded per call, so separate `ArrayProvider()`s +reproduce the same data while tensors within one run still differ. +""" +struct ArrayProvider{T, B <: AbstractBackend, A, R <: Random.AbstractRNG} <: AbstractProvider + backend::B + allocator::A + rng::R +end +function ArrayProvider{T}(; + backend::AbstractBackend = DefaultBackend(), allocator = DefaultAllocator(), + rng::Random.AbstractRNG = Random.Xoshiro(0x5eed5eed5eed5eed) + ) where {T} + return ArrayProvider{T, typeof(backend), typeof(allocator), typeof(rng)}(backend, allocator, rng) +end + +TensorOperations.scalartype(::ArrayProvider{T}) where {T} = T +function randtensor(p::ArrayProvider, labels, dims, T = TensorOperations.scalartype(p)) + C = TensorOperations.tensoralloc(Array{T, length(dims)}, dims, Val(false), allocator(p)) + return randn!(rng(p), C) +end +backend(p::ArrayProvider) = p.backend +allocator(p::ArrayProvider) = p.allocator +label(p::ArrayProvider{T}) where {T} = "$(nameof(typeof(p.backend)))/$T" +rng(p::ArrayProvider) = p.rng diff --git a/benchmark/src/registry.jl b/benchmark/src/registry.jl new file mode 100644 index 00000000..573d349a --- /dev/null +++ b/benchmark/src/registry.jl @@ -0,0 +1,71 @@ +# A category is a generator function `sizes -> Vector{BenchmarkCase}`, registered by name. + +""" + BenchmarkCase(category, id, params, spec) + +One concrete benchmark case: `category` groups it (e.g. `:contract`), `id` is a short unique +label within the category, `params` records the sweep parameters that produced it (e.g. +`(; D=64)`), and `spec` is the [`AbstractCaseSpec`](@ref) to execute. `tags` (for +`BenchmarkTools.@tagged`-based filtering, see suite.jl) is computed once here and stored, +rather than re-derived on demand: the category name, plus every `Symbol`-valued entry of +`params` (e.g. `source`, `kind`, `variant`, `layout`, `topic`) -- a generator opts a `params` +field into tag-filtering just by giving it a `Symbol` value. Also gets a `"blas"` tag when +[`isblasequivalent`](@ref) holds for `spec`, regardless of category. +""" +struct BenchmarkCase + category::Symbol + id::String + params::NamedTuple + tags::Vector{Any} + spec::AbstractCaseSpec +end +function BenchmarkCase(category::Symbol, id::AbstractString, params::NamedTuple, spec::AbstractCaseSpec) + tags = Any[String(category); [string(v) for v in values(params) if v isa Symbol]] + isblasequivalent(spec) && push!(tags, "blas") + return BenchmarkCase(category, String(id), params, tags, spec) +end + +const REGISTRY = Dict{Symbol, Function}() +const DEFAULT_SIZES = Dict{Symbol, Any}() + +""" + register_category!(name::Symbol, generator::Function; sizes=(4, 8, 16, 32, 64, 128)) + +Register `generator(sizes) -> Vector{BenchmarkCase}` under category `name`, along with the +default size sweep to use when the caller doesn't supply one. Calling this again for the same +`name` overwrites the previous generator/default. +""" +function register_category!(name::Symbol, generator::Function; sizes = (4, 8, 16, 32, 64, 128)) + REGISTRY[name] = generator + DEFAULT_SIZES[name] = sizes + return nothing +end + +""" + default_sizes(category::Symbol) + +The default size sweep passed to `category`'s generator when the caller doesn't supply one. +Each category interprets `sizes` in its own way (a list of ranks, of bond dimensions, ...). +""" +default_sizes(category::Symbol) = get(DEFAULT_SIZES, category, (4, 8, 16, 32, 64, 128)) + +""" + within_memory_budget(spec::AbstractCaseSpec, [id]; maxbytes=MAX_CASE_BYTES) + +Whether executing `spec` would stay within `maxbytes` of total tensor memory, assuming (in the +absence of an explicit `TA`/`TB`/`TC`/`Ts` override) worst-case `Float64`-sized elements. +Category generators use this to skip dimension/shape combinations that would otherwise OOM the +benchmark process, rather than hand-tuning per-shape size ceilings. Passing `id` additionally +`@warn`s (naming `id`) when a case is skipped, so an unexpectedly sparse sweep is diagnosable +rather than silent. + +`MAX_CASE_BYTES` scales with the host's total memory (`Sys.total_memory() ÷ 64`) rather than a +fixed constant, since this ranges from CI runners to large shared workstations. +""" +const MAX_CASE_BYTES = Sys.total_memory() ÷ 64 +within_memory_budget(spec::AbstractCaseSpec; maxbytes = MAX_CASE_BYTES) = bytes(spec) <= maxbytes +function within_memory_budget(spec::AbstractCaseSpec, id; maxbytes = MAX_CASE_BYTES) + ok = within_memory_budget(spec; maxbytes) + ok || @warn "Skipping benchmark case: exceeds memory budget" id maxbytes bytes = bytes(spec) + return ok +end diff --git a/benchmark/src/report.jl b/benchmark/src/report.jl new file mode 100644 index 00000000..7bf03030 --- /dev/null +++ b/benchmark/src/report.jl @@ -0,0 +1,48 @@ +# Flattens a benchmark result into a DataFrame, joining timings back against specs +# (regenerated from REGISTRY) for GFLOP/s and bandwidth. + +""" + resultstable(results::BenchmarkGroup; categories=collect(keys(REGISTRY)), sizes=nothing) + +Flatten a benchmark-run result (with the same `[category][provider][id]["benchmark"]` nesting +`build_suite` produces) into a `DataFrame` with columns `category`, `provider`, `id`, `params` +(the originating `NamedTuple`), `mintime` (ns), `allocs`, `memory` (bytes), `gflops` +(`flops(spec) / mintime`, or `missing` if `mintime` is zero), `gbps` (`bytes(spec) / mintime`), +and `intensity` (`flops(spec) / bytes(spec)`, flops/byte -- a static property of the case, not +the measurement, useful for a roofline-style plot against `gflops`). `categories`/`sizes` must +match what was passed to the `build_suite` call that produced `results`, since specs (and +therefore flop/byte counts) are regenerated from `REGISTRY` rather than stored in the result +itself. +""" +function resultstable( + results::BenchmarkGroup; categories = collect(keys(REGISTRY)), sizes = nothing + ) + rows = NamedTuple[] + for category in categories + haskey(results, String(category)) || continue + cases = REGISTRY[category](_sizes_for(sizes, category)) + casesbyid = Dict(c.id => c for c in cases) + catgroup = results[String(category)] + for providerlabel in keys(catgroup) + provgroup = catgroup[providerlabel] + for id in keys(provgroup) + trial = provgroup[id]["benchmark"] + case = casesbyid[id] + mintime = minimum(trial.times) + fl = flops(case.spec) + by = bytes(case.spec) + gflops = mintime > 0 ? fl / mintime : missing + gbps = mintime > 0 ? by / mintime : missing + push!( + rows, + (; + category = String(category), provider = providerlabel, id, + params = case.params, mintime, allocs = trial.allocs, + memory = trial.memory, gflops, gbps, intensity = intensity(case.spec), + ) + ) + end + end + end + return DataFrame(rows) +end diff --git a/benchmark/src/specs.jl b/benchmark/src/specs.jl new file mode 100644 index 00000000..e8fcc798 --- /dev/null +++ b/benchmark/src/specs.jl @@ -0,0 +1,174 @@ +# Pure-data contraction/network specs: labels and leg extents only, no tensor objects, so a +# spec is shareable across providers. `TA`/`TB`/`TC`/`Ts` default to `nothing` ("provider's +# default scalartype"); only mixed_precision.jl sets them explicitly. + +abstract type AbstractCaseSpec end + +""" + AddSpec(IA, IC, dims, conjA, TA=nothing, TC=nothing) + +Specification of a permutation `C = permutedims(opA(A), pA)` (executed via `tensorcopy`), +where `pA` is derived from matching labels in `IA` to `IC`. +""" +struct AddSpec <: AbstractCaseSpec + IA::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + TA::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function AddSpec(IA, IC, dims; conjA::Bool = false, TA = nothing, TC = nothing) + return AddSpec(collect(Symbol, IA), collect(Symbol, IC), dims, conjA, TA, TC) +end + +""" + TraceSpec(IA, IC, dims, conjA, TA=nothing, TC=nothing) + +Specification of a (partial) trace `C = permutedims(trace(opA(A)), p)` (executed via +`tensortrace`). Labels appearing twice in `IA` are traced; labels appearing once and also +in `IC` are kept. +""" +struct TraceSpec <: AbstractCaseSpec + IA::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + TA::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function TraceSpec(IA, IC, dims; conjA::Bool = false, TA = nothing, TC = nothing) + return TraceSpec(collect(Symbol, IA), collect(Symbol, IC), dims, conjA, TA, TC) +end + +""" + ContractSpec(IA, IB, IC, dims, conjA, conjB, TA=nothing, TB=nothing, TC=nothing) + +Specification of a pairwise contraction `C = contract(opA(A), opB(B))` (executed via +`tensorcontract`). +""" +struct ContractSpec <: AbstractCaseSpec + IA::Vector{Symbol} + IB::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + conjB::Bool + TA::Union{Nothing, Type} + TB::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function ContractSpec( + IA, IB, IC, dims; conjA::Bool = false, conjB::Bool = false, + TA = nothing, TB = nothing, TC = nothing + ) + return ContractSpec( + collect(Symbol, IA), collect(Symbol, IB), collect(Symbol, IC), dims, + conjA, conjB, TA, TB, TC + ) +end + +""" + BatchedContractSpec(batch, IA, IB, IC, dims, conjA=false, conjB=false, TA=nothing, TB=nothing, TC=nothing) + +`batch` independent pairwise contractions, each with the same label structure, executed as +`batch` separate `tensorcontract!` calls -- TensorOperations has no fused batched-GEMM +primitive, so this genuinely cannot collapse to one BLAS call, unlike a `ContractSpec` with a +`:gemm_ready` layout. Models a "many small contractions" regime (e.g. attention-style batched +matmuls) where per-call dispatch overhead dominates rather than raw FLOPs. +""" +struct BatchedContractSpec <: AbstractCaseSpec + batch::Int + IA::Vector{Symbol} + IB::Vector{Symbol} + IC::Vector{Symbol} + dims::Dict{Symbol, Int} + conjA::Bool + conjB::Bool + TA::Union{Nothing, Type} + TB::Union{Nothing, Type} + TC::Union{Nothing, Type} +end +function BatchedContractSpec( + batch, IA, IB, IC, dims; conjA::Bool = false, conjB::Bool = false, + TA = nothing, TB = nothing, TC = nothing + ) + return BatchedContractSpec( + Int(batch), collect(Symbol, IA), collect(Symbol, IB), collect(Symbol, IC), dims, + conjA, conjB, TA, TB, TC + ) +end + +""" + NetworkSpec(indexlists, conjlist, output, dims, order=nothing, Ts=nothing) + +An `ncon`-style multi-tensor network: `indexlists[k]` gives the signed integer index labels +of the `k`th tensor (positive = contracted, negative = open/output), `dims` maps each *label* +(by absolute value) to its extent, and `order` optionally fixes the contraction order (as a +list of positive labels, in the order they should be contracted) -- `nothing` lets `ncon`'s +default greedy tree builder decide. +""" +struct NetworkSpec <: AbstractCaseSpec + indexlists::Vector{Vector{Int}} + conjlist::Vector{Bool} + output::Vector{Int} + dims::Dict{Int, Int} + order::Union{Nothing, Vector{Int}} + Ts::Union{Nothing, Vector{<:Type}} +end +function NetworkSpec( + indexlists, dims; conjlist = fill(false, length(indexlists)), + output = nothing, order = nothing, Ts = nothing + ) + outputindices = something( + output, sort(unique(l for il in indexlists for l in il if l < 0); rev = true) + ) + return NetworkSpec( + [collect(Int, il) for il in indexlists], collect(Bool, conjlist), + collect(Int, outputindices), dims, order, Ts + ) +end + +# Compact einsum-style show methods, e.g. `ContractSpec: C[i,j] = A[i,k] * B[k,j] (dim=64)`. +_dimsnote(dims::Dict) = (vals = unique(values(dims)); length(vals) == 1 ? " (dim=$(only(vals)))" : " (dims=$dims)") +_opstr(label, conj) = conj ? "conj($label)" : label + +function Base.show(io::IO, spec::AddSpec) + return print( + io, "AddSpec: C[", join(spec.IC, ","), "] = ", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::TraceSpec) + return print( + io, "TraceSpec: C[", join(spec.IC, ","), "] = tr(", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), ")", _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::ContractSpec) + return print( + io, "ContractSpec: C[", join(spec.IC, ","), "] = ", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), " * ", + _opstr("B[$(join(spec.IB, ","))]", spec.conjB), _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::BatchedContractSpec) + return print( + io, "BatchedContractSpec: ", spec.batch, " x C[", join(spec.IC, ","), "] = ", + _opstr("A[$(join(spec.IA, ","))]", spec.conjA), " * ", + _opstr("B[$(join(spec.IB, ","))]", spec.conjB), _dimsnote(spec.dims) + ) +end + +function Base.show(io::IO, spec::NetworkSpec) + tensors = join( + ( + _opstr("T$k[$(join(il, ","))]", conj) + for (k, (il, conj)) in enumerate(zip(spec.indexlists, spec.conjlist)) + ), " * " + ) + return print(io, "NetworkSpec: C[", join(spec.output, ","), "] = ", tensors, _dimsnote(spec.dims)) +end diff --git a/benchmark/src/suite.jl b/benchmark/src/suite.jl new file mode 100644 index 00000000..9f944b45 --- /dev/null +++ b/benchmark/src/suite.jl @@ -0,0 +1,45 @@ +# Nests a BenchmarkGroup as [category][provider label][case id]["benchmark"]. Each case-id leaf +# is itself a tagged BenchmarkGroup (using `case.tags`), so `suite[BenchmarkTools.@tagged +# "tccg"]` (or any boolean tag expression) selects a subset of cases natively -- no bespoke +# filter predicate needed. No threading axis -- see threading.jl. + +""" + build_suite(providers; categories=collect(keys(REGISTRY)), sizes=nothing) + +Build a `BenchmarkTools.BenchmarkGroup` covering every registered category (or the subset in +`categories`) for every provider in `providers` (skipping providers that opt out via +[`supports`](@ref)). + +`sizes` may be `nothing` (use each category's [`default_sizes`](@ref)), a size sweep applied to +every category, or a `Dict{Symbol}` mapping category name to its own size sweep. + +To run only a tagged subset, filter *after* building: `suite[@tagged "tccg"]` (re-exported from +BenchmarkTools), or combine tags: `suite[@tagged "contract" && "tccg"]`. +""" +function build_suite( + providers::AbstractVector{<:AbstractProvider}; + categories = collect(keys(REGISTRY)), + sizes = nothing + ) + suite = BenchmarkGroup() + for category in categories + generator = REGISTRY[category] + catsizes = _sizes_for(sizes, category) + cases = generator(catsizes) + catgroup = suite[String(category)] = BenchmarkGroup() + for provider in providers + supports(provider, category) || continue + provgroup = catgroup[label(provider)] = BenchmarkGroup() + for case in cases + provgroup[case.id] = BenchmarkGroup( + case.tags, "benchmark" => make_benchmarkable(case, provider) + ) + end + end + end + return suite +end + +_sizes_for(::Nothing, category::Symbol) = default_sizes(category) +_sizes_for(sizes::AbstractDict, category::Symbol) = get(sizes, category, default_sizes(category)) +_sizes_for(sizes, ::Symbol) = sizes diff --git a/benchmark/src/threading.jl b/benchmark/src/threading.jl new file mode 100644 index 00000000..b8ea56e1 --- /dev/null +++ b/benchmark/src/threading.jl @@ -0,0 +1,52 @@ +# Threading is process-wide, not a provider/spec concern, and deliberately not baked into +# per-case timing (that would count the thread-count switch as part of the measurement). +# `Strided.set_num_threads` is capped by `Threads.nthreads()` (fixed at Julia startup); sweeping +# above it needs relaunching Julia -- see scripts/run_benchmarks.jl. + +""" + ThreadConfig(; blas=nothing, strided=nothing) + +A named thread-count configuration. `nothing` for either field means "leave that thread count +untouched". Use [`with_threads`](@ref) to apply one around a block of code. +""" +struct ThreadConfig + blas::Union{Nothing, Int} + strided::Union{Nothing, Int} + name::String +end +function ThreadConfig(; blas::Union{Nothing, Int} = nothing, strided::Union{Nothing, Int} = nothing, name = nothing) + autoname = "blas=$(something(blas, "-")),strided=$(something(strided, "-"))" + return ThreadConfig(blas, strided, something(name, autoname)) +end + +label(cfg::ThreadConfig) = cfg.name + +""" + set_threads!(cfg::ThreadConfig) + +Set `BLAS`/`Strided` thread counts according to `cfg`, permanently. For one-shot, +process-level configuration (`benchmarks.jl` calls this before building `SUITE`). +""" +function set_threads!(cfg::ThreadConfig) + cfg.blas === nothing || BLAS.set_num_threads(cfg.blas) + cfg.strided === nothing || Strided.set_num_threads(cfg.strided) + return nothing +end + +""" + with_threads(f, cfg::ThreadConfig) + +Run `f()` under `cfg`, restoring prior thread counts afterwards. For comparing a couple of +configs around a whole `run(suite)` call -- never around a single `@benchmarkable` case. +""" +function with_threads(f, cfg::ThreadConfig) + oldblas = BLAS.get_num_threads() + oldstrided = Strided.get_num_threads() + try + set_threads!(cfg) + return f() + finally + BLAS.set_num_threads(oldblas) + Strided.set_num_threads(oldstrided) + end +end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl new file mode 100644 index 00000000..ed3f4bd1 --- /dev/null +++ b/benchmark/test/runtests.jl @@ -0,0 +1,166 @@ +using Test +using TensorOperationsBenchmarks +using TensorOperations: StridedNative +using BenchmarkTools +using DataFrames: nrow +using LinearAlgebra: BLAS +using Strided: Strided + +@testset "TensorOperationsBenchmarks" begin + provider = ArrayProvider{Float64}(; backend = StridedNative()) + + @testset "every category builds and runs" begin + suite = build_suite([provider]; categories = collect(keys(REGISTRY)), sizes = (4, 8)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + for category in keys(REGISTRY) + @test haskey(results, String(category)) + end + end + + @testset "resultstable joins timings with flop/byte counts" begin + suite = build_suite([provider]; categories = [:contract], sizes = (4, 8)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + rows = resultstable(results; categories = [:contract], sizes = (4, 8)) + @test nrow(rows) > 0 + @test all(>(0), rows.mintime) + @test all(x -> x isa Float64, rows.gflops) + end + + @testset "mixed-precision cases execute" begin + suite = build_suite([provider]; categories = [:mixed_precision], sizes = (8,)) + results = run(suite; samples = 1, evals = 1, seconds = 5) + @test !isempty(results["mixed_precision"][label(provider)]) + end + + @testset "network cost matches ncon's own contraction tree" begin + cases = REGISTRY[:network]((16,)) + case = only(filter(c -> occursin("mps_1site", c.id), cases)) + @test flops(case.spec) > 0 + end + + @testset "with_threads restores prior thread counts, set_threads! does not" begin + oldblas = BLAS.get_num_threads() + oldstrided = Strided.get_num_threads() + with_threads(() -> nothing, ThreadConfig(; blas = 1, strided = 1)) + @test BLAS.get_num_threads() == oldblas + @test Strided.get_num_threads() == oldstrided + + set_threads!(ThreadConfig(; blas = 1)) + @test BLAS.get_num_threads() == 1 + BLAS.set_num_threads(oldblas) # restore for subsequent tests + end + + @testset "ArrayProvider's randtensor is reproducible across providers, varies within one" begin + p1 = ArrayProvider{Float64}() + p2 = ArrayProvider{Float64}() + A1 = randtensor(p1, [:a, :b], (4, 4), Float64) + A2 = randtensor(p2, [:a, :b], (4, 4), Float64) + @test A1 == A2 # same fixed seed -> same first draw + B1 = randtensor(p1, [:a, :b], (4, 4), Float64) + @test A1 != B1 # stateful rng -> second draw differs from the first + end + + @testset "execute doesn't allocate a fresh output every call (preallocated in setup)" begin + cases = REGISTRY[:contract]((8,)) + case = first(cases) + ts = TensorOperationsBenchmarks.maketensors(case.spec, provider) + C_before = ts[1] + C_after = TensorOperationsBenchmarks.execute(case.spec, ts, provider) + @test C_after === C_before + end + + @testset "network category covers mps/ctmrg/trg topics" begin + cases = REGISTRY[:network]((16,)) + @test any(c -> c.params.topic == :mps && c.params.variant == :onesite, cases) + @test any(c -> c.params.topic == :mps && c.params.variant == :twosite, cases) + @test any(c -> c.params.topic == :ctmrg, cases) + @test any(c -> c.params.topic == :trg, cases) + end + + @testset "tccg cases are merged into :contract, filterable via @tagged" begin + cases = REGISTRY[:contract]((4,)) + tccg_cases = filter(c -> c.params.source == :tccg, cases) + @test length(TensorOperationsBenchmarks.TCCG_CONTRACTIONS) == 24 + @test any(c -> c.params.source == :synthetic, cases) + for prefix in ("ccsd_", "ccsd_t_", "ao2mo_", "intensli_") + @test any(c -> startswith(c.params.equation, prefix), tccg_cases) + end + suite = build_suite([provider]; categories = [:contract], sizes = (4,)) + results = run(suite[@tagged "tccg"]; samples = 1, evals = 1, seconds = 5) + @test !isempty(results["contract"][label(provider)]) + @test length(results["contract"][label(provider)]) == length(tccg_cases) + end + + @testset "BenchmarkCase stores tags derived from category + Symbol-valued params" begin + case = first(REGISTRY[:trace]((8,))) + @test "trace" in case.tags + @test string(case.params.kind) in case.tags + end + + @testset "contract permuted-stride layouts are structurally distinct and filterable" begin + cases = REGISTRY[:contract]((8,)) + layouts = unique(c.params.layout for c in cases if c.params.source == :synthetic) + @test :gemm_ready in layouts + @test :both_permuted in layouts + + suite = build_suite([provider]; categories = [:contract], sizes = (8,)) + results = run(suite[@tagged "both_permuted"]; samples = 1, evals = 1, seconds = 5) + n_expected = count(c -> c.params.source == :synthetic && c.params.layout == :both_permuted, cases) + @test length(results["contract"][label(provider)]) == n_expected + end + + @testset "within_memory_budget(spec, id) warns when skipping an oversized case" begin + spec = ContractSpec([:a1, :a2], [:a2, :b1], [:a1, :b1], Dict(:a1 => 8, :a2 => 8, :b1 => 8)) + # override maxbytes (rather than relying on the host's memory-scaled default) so this + # test is deterministic regardless of how much RAM the machine running it has. + @test_logs (:warn, r"exceeds memory budget") within_memory_budget(spec, "tiny_budget_test"; maxbytes = 10) + end + + @testset "isblasequivalent tags gemm-ready contractions, not permuted/scrambled ones" begin + cases = REGISTRY[:contract]((8,)) + gemmcase = only( + filter( + c -> c.params.source == :synthetic && c.params.layout == :gemm_ready && + c.params.nopenA == 2 && c.params.ncontract == 1 && c.params.nopenB == 2, + cases + ) + ) + @test "blas" in gemmcase.tags + scrambledcases = filter( + c -> c.params.source == :synthetic && c.params.layout == :contract_scrambled, cases + ) + @test !isempty(scrambledcases) + @test all(c -> !("blas" in c.tags), scrambledcases) + permutedcases = filter( + c -> c.params.source == :synthetic && c.params.layout == :both_permuted, cases + ) + @test !isempty(permutedcases) + @test all(c -> !("blas" in c.tags), permutedcases) + end + + @testset "batched contraction cases run as independent per-slice tensorcontract! calls" begin + cases = REGISTRY[:contract]((8,)) + batchedcases = filter(c -> c.params.source == :batched, cases) + @test !isempty(batchedcases) + @test all(c -> !("blas" in c.tags), batchedcases) + case = first(batchedcases) + ts = TensorOperationsBenchmarks.maketensors(case.spec, provider) + Cs = TensorOperationsBenchmarks.execute(case.spec, ts, provider) + @test length(Cs) == case.spec.batch + end + + @testset "resultstable reports a static arithmetic-intensity column" begin + suite = build_suite([provider]; categories = [:contract], sizes = (8,)) + results = run(suite[@tagged "gemm_ready"]; samples = 1, evals = 1, seconds = 5) + rows = resultstable(results; categories = [:contract], sizes = (8,)) + @test nrow(rows) > 0 + @test all(>(0), rows.intensity) + end + + @testset "specs have informative show methods" begin + spec = ContractSpec([:i, :k], [:k, :j], [:i, :j], Dict(:i => 4, :j => 4, :k => 4)) + @test occursin("C[i,j]", sprint(show, spec)) + @test occursin("A[i,k]", sprint(show, spec)) + @test occursin("B[k,j]", sprint(show, spec)) + end +end