diff --git a/.github/workflows/performance-regression.yml b/.github/workflows/performance-regression.yml index 309abfc4..b877b4fa 100644 --- a/.github/workflows/performance-regression.yml +++ b/.github/workflows/performance-regression.yml @@ -1,7 +1,16 @@ name: Performance -# Checks out `inputs.ref` when a caller supplies one, otherwise the commit the run was triggered for (github.sha). -# A manual run measures the branch it is dispatched from; no input selects code from another ref. +# Measures the tested commit (`inputs.ref`, otherwise the commit the run was triggered for) against a base revision +# built on the same runner. Base and head launches are interleaved on one pinned CPU, so the result does not depend on +# which hosted-runner hardware the job lands on, and there is no checked-in measurement that a data-source update can +# make stale. The base is selected by `inputs.base`: +# anchor (default) the release tag in src/performance-harness/drift-anchor.txt. The tested commit is compared +# with the last accepted release, so the cost of one change and any slow drift accumulated on main since +# that release are both visible. Advance the tag to accept the accumulated change. +# parent the tested commit's first parent - the target branch for a pull request merge commit, the previous +# commit for a push. Isolates what one change costs. +# any other git revision, for manual investigation. +# Callers decide whether the result gates them: pull requests and pushes gate on it, the release workflow does not. on: workflow_call: inputs: @@ -9,15 +18,25 @@ on: description: Commit to test; defaults to the commit the calling run was triggered for required: false type: string + base: + description: Revision to compare against - `anchor`, `parent`, or any git revision + required: false + type: string + default: anchor performance-gate: - description: Fail the workflow when the expected-performance check fails + description: Fail the workflow when the performance check fails required: false type: boolean default: true workflow_dispatch: inputs: + base: + description: Revision to compare against - `anchor`, `parent`, or any git revision + required: false + type: string + default: anchor performance-gate: - description: Fail the workflow when the expected-performance check fails + description: Fail the workflow when the performance check fails required: false type: boolean default: true @@ -33,12 +52,17 @@ jobs: check: name: Check expected performance runs-on: ubuntu-latest - timeout-minutes: 30 + timeout-minutes: 60 + env: + BASE: ${{ inputs.base || 'anchor' }} + DRIFT_ANCHOR_FILE: src/performance-harness/drift-anchor.txt steps: - name: Checkout repository uses: actions/checkout@v6 with: ref: ${{ inputs.ref || github.sha }} + # Lets `base: parent` resolve without another fetch. + fetch-depth: 2 - name: Load build configs id: configs @@ -49,24 +73,92 @@ jobs: with: toolchain: ${{ steps.configs.outputs.rust-toolchain }} - - name: Test performance assertions + - name: Test performance harness working-directory: ${{ env.WORKING_DIR }} run: cargo test --locked --release -p performance-harness - - name: Check expected performance + - name: Build head harness + working-directory: ${{ env.WORKING_DIR }} + run: | + set -euo pipefail + cargo build --locked --release -p performance-harness + install -D target/release/performance-harness "$RUNNER_TEMP/performance-harness/head" + + - name: Resolve base revision + id: base + run: | + set -euo pipefail + case "$BASE" in + anchor) + anchor=$(<"$DRIFT_ANCHOR_FILE") + if [[ ! $anchor =~ ^[0-9]+\.[0-9]+\.[0-9]+(-beta)?$ ]]; then + echo "::error file=$DRIFT_ANCHOR_FILE::expected the file to contain one release tag, found: $anchor" + exit 1 + fi + git fetch --no-tags --depth=1 origin "refs/tags/$anchor:refs/tags/$anchor" + base_sha=$(git rev-parse --verify "refs/tags/$anchor^{commit}") + title="Performance since release $anchor" + ;; + parent) + base_sha=$(git rev-parse --verify 'HEAD^1^{commit}') + title="Performance comparison against parent" + ;; + *) + git fetch --no-tags --depth=1 origin "$BASE" + base_sha=$(git rev-parse --verify 'FETCH_HEAD^{commit}') + title="Performance comparison against $BASE" + ;; + esac + if [[ $base_sha == "$(git rev-parse HEAD)" ]]; then + if [[ $BASE == anchor ]]; then + # Re-validating the anchored release commit itself has nothing to compare. + echo "::notice::the tested commit is release $anchor itself; skipping the comparison" + echo "skip=true" >> "$GITHUB_OUTPUT" + echo "Performance check skipped: the tested commit is release \`$anchor\` itself." >> "$GITHUB_STEP_SUMMARY" + exit 0 + fi + echo "::error::base $BASE resolves to the tested commit itself" + exit 1 + fi + echo "Comparing $(git rev-parse HEAD) against $BASE base $base_sha" + echo "sha=$base_sha" >> "$GITHUB_OUTPUT" + echo "title=$title" >> "$GITHUB_OUTPUT" + + # The base harness supplies only the `measure` worker; the head harness defines the workloads and evaluation. + # Reusing the head target directory keeps the unchanged third-party dependencies from being rebuilt. + - name: Build base harness + if: ${{ steps.base.outputs.skip != 'true' }} + env: + BASE_SHA: ${{ steps.base.outputs.sha }} + run: | + set -euo pipefail + base_dir="$RUNNER_TEMP/performance-base" + mkdir -p "$base_dir" + git archive "$BASE_SHA" | tar -x -C "$base_dir" + target_dir="$GITHUB_WORKSPACE/$WORKING_DIR/target" + (cd "$base_dir/$WORKING_DIR" && CARGO_TARGET_DIR="$target_dir" cargo build --locked --release -p performance-harness) + install -D "$target_dir/release/performance-harness" "$RUNNER_TEMP/performance-harness/base" + + - name: Compare performance against base + if: ${{ steps.base.outputs.skip != 'true' }} continue-on-error: ${{ !inputs.performance-gate }} working-directory: ${{ env.WORKING_DIR }} + env: + BASE_SHA: ${{ steps.base.outputs.sha }} + TITLE: ${{ steps.base.outputs.title }} run: | set -euo pipefail - cargo run --locked --release -p performance-harness -- \ - check \ + "$RUNNER_TEMP/performance-harness/head" compare \ + --base-executable "$RUNNER_TEMP/performance-harness/base" \ + --base-revision "$BASE_SHA" \ + --title "$TITLE" \ --output-dir "$RUNNER_TEMP/performance-check" - name: Add performance result to job summary if: always() run: | - if [[ -f "$RUNNER_TEMP/performance-check/performance-results.md" ]]; then - cat "$RUNNER_TEMP/performance-check/performance-results.md" >> "$GITHUB_STEP_SUMMARY" + if [[ -f "$RUNNER_TEMP/performance-check/performance-comparison.md" ]]; then + cat "$RUNNER_TEMP/performance-check/performance-comparison.md" >> "$GITHUB_STEP_SUMMARY" fi - name: Upload performance evidence @@ -75,8 +167,7 @@ jobs: with: name: performance-${{ github.event.pull_request.number || github.run_id }} path: | - ${{ runner.temp }}/performance-check/performance-results.json - ${{ runner.temp }}/performance-check/performance-results.md - ${{ runner.temp }}/performance-check/performance-candidate-baseline.json + ${{ runner.temp }}/performance-check/performance-comparison.json + ${{ runner.temp }}/performance-check/performance-comparison.md if-no-files-found: warn retention-days: 14 diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml index 01ca673c..e7093f8a 100644 --- a/.github/workflows/validate.yml +++ b/.github/workflows/validate.yml @@ -22,7 +22,7 @@ on: type: boolean default: true performance-gate: - description: 'Fail validation when the expected-performance check fails' + description: 'Fail validation when the performance check against the anchored release fails' required: false type: boolean default: true @@ -40,7 +40,7 @@ on: type: boolean default: true performance-gate: - description: 'Fail validation when the expected-performance check fails' + description: 'Fail validation when the performance check against the anchored release fails' required: false type: boolean default: true diff --git a/.kiro/steering/structure.md b/.kiro/steering/structure.md index 3570d31c..e3ca389a 100644 --- a/.kiro/steering/structure.md +++ b/.kiro/steering/structure.md @@ -39,8 +39,9 @@ src/ ├── guard-translator/ # Guard DSL evaluation via the Guard evaluator (cloudformation-guard-lang) against the │ # authored template; produces engine-agnostic findings that validation-engine maps to │ # diagnostics through one GuardRuleSet every engine calls -├── performance-harness/ # Performance regression harness (`check`/`update`) with per-environment -│ └── expected/ # baseline profiles for GitHub x64 runners and the reference Apple Silicon Mac +├── performance-harness/ # Performance regression harness: CI `compare` measures head against the release tag +│ # in `drift-anchor.txt` (or, on manual runs, a parent/any revision) built on the same +│ # runner; nothing measured is checked in ├── bindings-wasm/ # WASM bindings (wasm-bindgen) for Node.js embedding │ ├── ts/ # TypeScript wrapper + type definitions │ ├── tests/ # Node test suite (vitest, run.sh) diff --git a/src/Cargo.lock b/src/Cargo.lock index 604b4d8d..ceb9ae08 100644 --- a/src/Cargo.lock +++ b/src/Cargo.lock @@ -238,7 +238,7 @@ checksum = "4372b9543397a4b86050cc5e7ee36953edf4bac9518e8a774c2da694977fb6e4" [[package]] name = "bindings-go" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-cel-engine", "cloudformation-validate-composite-engine", @@ -257,7 +257,7 @@ dependencies = [ [[package]] name = "bindings-jvm" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-cel-engine", "cloudformation-validate-composite-engine", @@ -274,7 +274,7 @@ dependencies = [ [[package]] name = "bindings-python" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-cel-engine", "cloudformation-validate-composite-engine", @@ -291,7 +291,7 @@ dependencies = [ [[package]] name = "bindings-wasm" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-cel-engine", "cloudformation-validate-composite-engine", @@ -487,7 +487,7 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "cfn-validate" -version = "1.12.1" +version = "1.14.0" dependencies = [ "chrono", "cloudformation-validate-cel-engine", @@ -585,7 +585,7 @@ dependencies = [ [[package]] name = "cloudformation-validate" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-cel-engine", "cloudformation-validate-composite-engine", @@ -600,7 +600,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-cel-engine" -version = "1.12.1" +version = "1.14.0" dependencies = [ "anyhow", "cel-interpreter", @@ -620,7 +620,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-composite-engine" -version = "1.12.1" +version = "1.14.0" dependencies = [ "anyhow", "cloudformation-validate-cel-engine", @@ -635,7 +635,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-data-source" -version = "1.12.1" +version = "1.14.0" dependencies = [ "anyhow", "chrono", @@ -656,7 +656,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-diagnostics" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-rules", "cloudformation-validate-template-model", @@ -671,7 +671,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-guard-translator" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-guard-lang", "serde_json", @@ -679,7 +679,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-rego-engine" -version = "1.12.1" +version = "1.14.0" dependencies = [ "anyhow", "cloudformation-validate-data-source", @@ -699,7 +699,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-rules" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-template-model", "log", @@ -713,7 +713,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-schema-validator" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-data-source", "cloudformation-validate-diagnostics", @@ -731,7 +731,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-template-model" -version = "1.12.1" +version = "1.14.0" dependencies = [ "base64", "env_logger", @@ -750,7 +750,7 @@ dependencies = [ [[package]] name = "cloudformation-validate-validation-engine" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-data-source", "cloudformation-validate-diagnostics", @@ -1501,7 +1501,7 @@ checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" [[package]] name = "performance-harness" -version = "1.12.1" +version = "1.14.0" dependencies = [ "cloudformation-validate-cel-engine", "cloudformation-validate-composite-engine", @@ -1688,7 +1688,7 @@ dependencies = [ [[package]] name = "resources" -version = "1.12.1" +version = "1.14.0" dependencies = [ "serde_json", ] diff --git a/src/Cargo.toml b/src/Cargo.toml index dd1b5feb..0d6987c6 100644 --- a/src/Cargo.toml +++ b/src/Cargo.toml @@ -25,7 +25,7 @@ resolver = "2" name = "cloudformation-validate" [workspace.package] -version = "1.12.1" +version = "1.14.0" edition = "2024" rust-version = "1.96" license = "Apache-2.0" @@ -46,16 +46,16 @@ categories = [ ] [workspace.dependencies] -cel-engine = { package = "cloudformation-validate-cel-engine", path = "cel-engine", version = "=1.12.1" } -composite-engine = { package = "cloudformation-validate-composite-engine", path = "composite-engine", version = "=1.12.1" } -data-source = { package = "cloudformation-validate-data-source", path = "data-source", version = "=1.12.1", default-features = false } -diagnostics = { package = "cloudformation-validate-diagnostics", path = "diagnostics", version = "=1.12.1" } -guard-translator = { package = "cloudformation-validate-guard-translator", path = "guard-translator", version = "=1.12.1" } -rego-engine = { package = "cloudformation-validate-rego-engine", path = "rego-engine", version = "=1.12.1" } -rules = { package = "cloudformation-validate-rules", path = "rules", version = "=1.12.1" } -schema-validator = { package = "cloudformation-validate-schema-validator", path = "schema-validator", version = "=1.12.1" } -template-model = { package = "cloudformation-validate-template-model", path = "template-model", version = "=1.12.1" } -validation-engine = { package = "cloudformation-validate-validation-engine", path = "validation-engine", version = "=1.12.1" } +cel-engine = { package = "cloudformation-validate-cel-engine", path = "cel-engine", version = "=1.14.0" } +composite-engine = { package = "cloudformation-validate-composite-engine", path = "composite-engine", version = "=1.14.0" } +data-source = { package = "cloudformation-validate-data-source", path = "data-source", version = "=1.14.0", default-features = false } +diagnostics = { package = "cloudformation-validate-diagnostics", path = "diagnostics", version = "=1.14.0" } +guard-translator = { package = "cloudformation-validate-guard-translator", path = "guard-translator", version = "=1.14.0" } +rego-engine = { package = "cloudformation-validate-rego-engine", path = "rego-engine", version = "=1.14.0" } +rules = { package = "cloudformation-validate-rules", path = "rules", version = "=1.14.0" } +schema-validator = { package = "cloudformation-validate-schema-validator", path = "schema-validator", version = "=1.14.0" } +template-model = { package = "cloudformation-validate-template-model", path = "template-model", version = "=1.14.0" } +validation-engine = { package = "cloudformation-validate-validation-engine", path = "validation-engine", version = "=1.14.0" } uniffi = "=0.32.0" [profile.dev] diff --git a/src/performance-harness/README.md b/src/performance-harness/README.md index 75bd74d1..b37123cb 100644 --- a/src/performance-harness/README.md +++ b/src/performance-harness/README.md @@ -1,46 +1,81 @@ # Performance harness -Performance is checked against a versioned environment profile, never against another Git revision. +CI checks performance by comparing the tested revision with the last release, both built and measured on the same +runner. No measurement is checked in, so a new runner CPU model or a schema/data-source update cannot make an +expectation stale. -* `expected/github-ubuntu-x64-amd-epyc-7763.json`, `expected/github-ubuntu-x64-amd-epyc-9v74.json`, - `expected/github-ubuntu-x64-intel-xeon-6973pc.json`, and - `expected/github-ubuntu-x64-intel-xeon-platinum-8573c.json` are separate tight contracts for CPU models used by - GitHub-hosted `ubuntu-latest` x64 runners. The harness derives and enforces the matching model automatically. -* `expected/local-macos-arm64.json` is the contract for the recorded reference Apple Silicon Mac. The harness rejects a different Mac model instead of comparing unlike hardware. +## How a comparison runs -The `check` command spawns the current release executable for all three engines (rego, cel, composite) across synthetic, real-template, and security workloads. Each case discards its first process launch, then uses the median of five independent launches. An apparent failure receives four additional samples and is evaluated again over the combined set. +The `performance-regression` workflow builds the harness twice - once from the tested commit (head) and once from the +base revision - and runs: -Only robust end-to-end metrics are enforced: initialization plus first validation, warm validation time per call above the profile's stability floor, and peak resident memory. Per-case ratios are normalized by the run-wide geometric-mean ratio for that metric, removing common GitHub host-speed shifts; the raw aggregate separately fails broad changes that exceed its explicit band. The long cross-reference-fanout timing case is aggregate-only because identical-tree GitHub runs showed workload-specific variance beyond the normal residual range. +```bash +cd src +cargo run --locked --release -p performance-harness -- \ + compare --base-executable [--base-revision ] [--title ] [--output-dir ] +``` -The checked-in two-sided limits are intentionally tight: +The head harness defines the workloads, templates, and evaluation; the base executable supplies only its `measure` +worker, so the `measure` command-line interface and its JSON output must stay backward compatible. -* GitHub: normalized per-case timing ±15%, raw aggregate timing ±10%, normalized RSS ±3%, aggregate RSS ±1%. -* Reference Mac: normalized per-case timing ±8%, raw aggregate init ±7%, raw aggregate warm ±6%, normalized RSS ±2%, aggregate RSS ±1%. +For every engine (rego, cel, composite) and every synthetic, real-template, and security workload, the harness +discards one launch of each side, then runs five base/head launch pairs back to back on one pinned CPU, alternating +which side starts each pair. A metric's ratio is the median of the per-pair head/base ratios, which cancels host speed +changes shared by both launches of a pair. An apparent regression receives four more pairs and is evaluated again. -Crossing an upper bound is a regression. Crossing a lower bound is an unexpectedly large improvement and also fails so the baseline cannot silently become stale. Apparent failures are evaluated again after confirmation samples. +The enforced metrics are initialization plus first validation, warm validation time per call, and peak resident +memory. A metric is gated only when the base measurement is above its stability floor (5 ms, 0.30 ms, and 16 MiB); the +long cross-reference-fanout workload gates only memory per case, and the deep-nesting workloads gate only warm time. +The limits are: -## Run locally +| Metric | Per case | Aggregate (geometric mean) | +|-------------------------|---------:|---------------------------:| +| Init + first, warm/call | 1.20x | 1.08x | +| Peak RSS | 1.10x | 1.05x | -On the recorded reference Mac, the profile is selected automatically: +Exceeding a limit fails the check. Improvements and changed diagnostics are reported but never fail, because there is +no checked-in expectation to keep current. Results are written to `performance-comparison.md` (also added to the job +summary) and `performance-comparison.json`. -```bash -cd src -cargo run --locked --release -p performance-harness -- check -``` +## Where the check runs and what it compares against -Results and a ready-to-review candidate baseline are written under `tmp/performance-check/` at the repository root. +`Check expected performance` runs on every pull request and push to `main`, on manual runs of the `Validate` and +`Performance` workflows, and during a release. Pull requests and pushes gate on it; the release workflow runs it with +`performance-gate: false`, so the report is attached to the release run but never blocks publishing. -## Update an expected file +The base is the release tag recorded in [`drift-anchor.txt`](drift-anchor.txt). Comparing every tested commit with the +last accepted release shows the cost of the change under review together with everything that has accumulated on `main` +since that release, so small regressions that would each pass a commit-to-commit comparison cannot add up unnoticed. +When the check fails on a pull request that did not itself change performance, `main` has drifted past the budget since +the release: fix the regression, or advance the anchor as described below. The check is skipped when the tested commit +is the anchored release itself. -Update the local reference profile only after confirming an intentional performance change: +A manual run of the `Performance` workflow can set `base` to `parent` (the tested commit's first parent, which isolates +one change) or to any git revision for investigation. + +### Advancing the drift anchor + +`drift-anchor.txt` holds one release tag (`MAJOR.MINOR.PATCH`, optionally `-beta`) and nothing else. Bundled schema +data only grows, so peak memory trends upward with every data-source update and the check will eventually fail even +without a code regression; one measured update cost about 1% aggregate peak memory and under 0.5% aggregate time, so +the 1.05x memory limit absorbs several of them. Replace the tag in a reviewed change once the accumulated change since +it is understood and accepted - normally with each new release, so that every release is checked against the previous +one. The anchor is a tag rather than recorded numbers, so it is independent of runner hardware and never needs +measurements from a GitHub runner to update. + +## Run locally + +Build the base harness from a clean export of the base revision, then pass it to `compare`. Reusing the workspace +target directory avoids rebuilding the unchanged third-party dependencies: ```bash +mkdir -p tmp/perf-base && git archive main | tar -x -C tmp/perf-base +(cd tmp/perf-base/src && CARGO_TARGET_DIR="$PWD/../../../src/target" cargo build --locked --release -p performance-harness) +cp src/target/release/performance-harness tmp/perf-base-harness cd src -cargo run --locked --release -p performance-harness -- update +cargo run --locked --release -p performance-harness -- compare --base-executable ../tmp/perf-base-harness ``` -GitHub expectations must be measured on GitHub-hosted runners. Every workflow run uploads `performance-candidate-baseline.json`; after a confirmed improvement or an intentional regression, use the candidate to update the matching CPU-specific profile in review. Never copy local Linux measurements into a GitHub profile. - -A previously unseen GitHub CPU still fails the required check. Instead of failing before measurement, the harness collects nine independent launches per case and uploads a CPU-enforced candidate plus raw results. Validate repeated hosted-runner evidence before adding the candidate under its suggested deterministic profile name; the unknown hardware cannot pass until that profile is checked in. - -Changing workloads causes an exact case-set mismatch and requires an intentional baseline regeneration. Environment mismatches fail before measurement rather than producing misleading performance results. +Results are written under `tmp/performance-check/` at the repository root. A full comparison takes about 15-20 +minutes; run it on an otherwise idle machine, because a background load that hits only one launch of a pair shows up +as noise. diff --git a/src/performance-harness/drift-anchor.txt b/src/performance-harness/drift-anchor.txt new file mode 100644 index 00000000..f8f4f03b --- /dev/null +++ b/src/performance-harness/drift-anchor.txt @@ -0,0 +1 @@ +1.12.1 diff --git a/src/performance-harness/expected/github-ubuntu-x64-amd-epyc-7763.json b/src/performance-harness/expected/github-ubuntu-x64-amd-epyc-7763.json deleted file mode 100644 index 82f8a078..00000000 --- a/src/performance-harness/expected/github-ubuntu-x64-amd-epyc-7763.json +++ /dev/null @@ -1,860 +0,0 @@ -{ - "schemaVersion": 1, - "profile": "github-ubuntu-x64-amd-epyc-7763", - "environment": { - "context": "github-actions", - "system": "Linux", - "architecture": "x86_64", - "machineModel": null, - "cpuModel": "AMD EPYC 7763 64-Core Processor", - "logicalCpuCount": 4, - "pageSizeBytes": 4096, - "enforceMachineModel": false, - "enforceCpuModel": true - }, - "measurement": { - "sampleCount": 5, - "confirmationSampleCount": 4, - "discardedLaunchCount": 1, - "warmupIterations": 2 - }, - "thresholds": { - "initAndFirstMs": { - "minimumExpected": 5.0, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "warmPerCallMs": { - "minimumExpected": 0.3, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "peakRssBytes": { - "minimumExpected": 16777216.0, - "regressionFactor": 1.03, - "improvementFactor": 0.97, - "aggregateRegressionFactor": 1.01, - "aggregateImprovementFactor": 0.99 - } - }, - "provenance": { - "gitSha": "ac74dc99db80cfd003a180754ab207b8faacaba4", - "rustVersion": "rustc 1.96.0 (ac68faa20 2026-05-25)", - "workingTreeDirty": false - }, - "cases": { - "cel/conditional-100": { - "initAndFirstMs": 247.498, - "warmPerCallMs": 11.723, - "peakRssBytes": 238383104.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/duplicate-500": { - "initAndFirstMs": 294.474, - "warmPerCallMs": 56.119, - "peakRssBytes": 255053824.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/mixed-real": { - "initAndFirstMs": 256.14, - "warmPerCallMs": 15.612, - "peakRssBytes": 242614272.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-combined-conditions": { - "initAndFirstMs": 1259.06, - "warmPerCallMs": 1039.005, - "peakRssBytes": 238546944.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-boundary": { - "initAndFirstMs": 350.86, - "warmPerCallMs": 116.353, - "peakRssBytes": 237277184.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-wide": { - "initAndFirstMs": 265.541, - "warmPerCallMs": 30.552, - "peakRssBytes": 237645824.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-fusion": { - "initAndFirstMs": 1010.286, - "warmPerCallMs": 784.326, - "peakRssBytes": 244101120.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-reference-fanout": { - "initAndFirstMs": 2056.32, - "warmPerCallMs": 1813.762, - "peakRssBytes": 310206464.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-resource-scale": { - "initAndFirstMs": 291.45, - "warmPerCallMs": 53.418, - "peakRssBytes": 256094208.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-intrinsic-resolution": { - "initAndFirstMs": 237.457, - "warmPerCallMs": 2.851, - "peakRssBytes": 236679168.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-nesting": { - "initAndFirstMs": 234.478, - "warmPerCallMs": 0.026, - "peakRssBytes": 235048960.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "cel/security-deep-yaml-nesting": { - "initAndFirstMs": 239.037, - "warmPerCallMs": 3.764, - "peakRssBytes": 236548096.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "cel/security-foreach-branch-explosion": { - "initAndFirstMs": 414.227, - "warmPerCallMs": 172.748, - "peakRssBytes": 269504512.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-conditions": { - "initAndFirstMs": 342.967, - "warmPerCallMs": 109.234, - "peakRssBytes": 237645824.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-resources": { - "initAndFirstMs": 258.348, - "warmPerCallMs": 22.375, - "peakRssBytes": 241045504.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-pathological-conditions": { - "initAndFirstMs": 313.575, - "warmPerCallMs": 78.332, - "peakRssBytes": 237195264.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-scenario-assignment-budget": { - "initAndFirstMs": 991.879, - "warmPerCallMs": 756.916, - "peakRssBytes": 261210112.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/tiny": { - "initAndFirstMs": 234.897, - "warmPerCallMs": 0.09, - "peakRssBytes": 236347392.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "cel/unique-500": { - "initAndFirstMs": 269.248, - "warmPerCallMs": 31.402, - "peakRssBytes": 244391936.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/conditional-100": { - "initAndFirstMs": 247.077, - "warmPerCallMs": 11.747, - "peakRssBytes": 238379008.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/duplicate-500": { - "initAndFirstMs": 294.874, - "warmPerCallMs": 56.255, - "peakRssBytes": 255111168.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/mixed-real": { - "initAndFirstMs": 256.633, - "warmPerCallMs": 15.659, - "peakRssBytes": 242610176.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-combined-conditions": { - "initAndFirstMs": 1258.516, - "warmPerCallMs": 1036.912, - "peakRssBytes": 238575616.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-boundary": { - "initAndFirstMs": 350.281, - "warmPerCallMs": 116.974, - "peakRssBytes": 237277184.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-wide": { - "initAndFirstMs": 264.959, - "warmPerCallMs": 30.661, - "peakRssBytes": 237785088.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-fusion": { - "initAndFirstMs": 1010.0, - "warmPerCallMs": 783.108, - "peakRssBytes": 244117504.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-reference-fanout": { - "initAndFirstMs": 2064.478, - "warmPerCallMs": 1822.238, - "peakRssBytes": 310296576.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-resource-scale": { - "initAndFirstMs": 291.254, - "warmPerCallMs": 53.631, - "peakRssBytes": 256192512.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-intrinsic-resolution": { - "initAndFirstMs": 237.142, - "warmPerCallMs": 2.848, - "peakRssBytes": 236666880.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-nesting": { - "initAndFirstMs": 234.559, - "warmPerCallMs": 0.026, - "peakRssBytes": 234979328.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "composite/security-deep-yaml-nesting": { - "initAndFirstMs": 239.423, - "warmPerCallMs": 3.759, - "peakRssBytes": 236613632.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "composite/security-foreach-branch-explosion": { - "initAndFirstMs": 414.569, - "warmPerCallMs": 172.651, - "peakRssBytes": 270069760.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-conditions": { - "initAndFirstMs": 343.794, - "warmPerCallMs": 109.225, - "peakRssBytes": 237568000.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-resources": { - "initAndFirstMs": 260.74, - "warmPerCallMs": 22.588, - "peakRssBytes": 241410048.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-pathological-conditions": { - "initAndFirstMs": 313.2, - "warmPerCallMs": 78.322, - "peakRssBytes": 237146112.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-scenario-assignment-budget": { - "initAndFirstMs": 989.867, - "warmPerCallMs": 756.832, - "peakRssBytes": 261332992.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/tiny": { - "initAndFirstMs": 235.42, - "warmPerCallMs": 0.09, - "peakRssBytes": 236290048.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "composite/unique-500": { - "initAndFirstMs": 268.349, - "warmPerCallMs": 31.254, - "peakRssBytes": 244465664.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/conditional-100": { - "initAndFirstMs": 300.086, - "warmPerCallMs": 38.972, - "peakRssBytes": 244318208.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/duplicate-500": { - "initAndFirstMs": 439.543, - "warmPerCallMs": 173.053, - "peakRssBytes": 267812864.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/mixed-real": { - "initAndFirstMs": 327.274, - "warmPerCallMs": 67.889, - "peakRssBytes": 252416000.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-combined-conditions": { - "initAndFirstMs": 1401.891, - "warmPerCallMs": 1150.477, - "peakRssBytes": 244436992.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-boundary": { - "initAndFirstMs": 400.179, - "warmPerCallMs": 139.777, - "peakRssBytes": 242823168.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-wide": { - "initAndFirstMs": 317.718, - "warmPerCallMs": 58.005, - "peakRssBytes": 243245056.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-fusion": { - "initAndFirstMs": 1161.111, - "warmPerCallMs": 905.627, - "peakRssBytes": 251179008.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-reference-fanout": { - "initAndFirstMs": 4529.401, - "warmPerCallMs": 4231.798, - "peakRssBytes": 345055232.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-resource-scale": { - "initAndFirstMs": 412.341, - "warmPerCallMs": 145.103, - "peakRssBytes": 267882496.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-intrinsic-resolution": { - "initAndFirstMs": 267.602, - "warmPerCallMs": 6.952, - "peakRssBytes": 242110464.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-nesting": { - "initAndFirstMs": 259.997, - "warmPerCallMs": 0.026, - "peakRssBytes": 239595520.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "rego/security-deep-yaml-nesting": { - "initAndFirstMs": 264.563, - "warmPerCallMs": 3.715, - "peakRssBytes": 241672192.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "rego/security-foreach-branch-explosion": { - "initAndFirstMs": 442.046, - "warmPerCallMs": 176.75, - "peakRssBytes": 276475904.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-conditions": { - "initAndFirstMs": 460.499, - "warmPerCallMs": 200.252, - "peakRssBytes": 243499008.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-resources": { - "initAndFirstMs": 363.993, - "warmPerCallMs": 101.233, - "peakRssBytes": 249483264.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-pathological-conditions": { - "initAndFirstMs": 372.947, - "warmPerCallMs": 111.928, - "peakRssBytes": 242683904.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-scenario-assignment-budget": { - "initAndFirstMs": 1037.007, - "warmPerCallMs": 777.365, - "peakRssBytes": 266113024.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/tiny": { - "initAndFirstMs": 262.088, - "warmPerCallMs": 2.42, - "peakRssBytes": 241434624.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/unique-500": { - "initAndFirstMs": 403.41, - "warmPerCallMs": 140.726, - "peakRssBytes": 252940288.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - } - } -} diff --git a/src/performance-harness/expected/github-ubuntu-x64-amd-epyc-9v74.json b/src/performance-harness/expected/github-ubuntu-x64-amd-epyc-9v74.json deleted file mode 100644 index b6d10a42..00000000 --- a/src/performance-harness/expected/github-ubuntu-x64-amd-epyc-9v74.json +++ /dev/null @@ -1,860 +0,0 @@ -{ - "schemaVersion": 1, - "profile": "github-ubuntu-x64-amd-epyc-9v74", - "environment": { - "context": "github-actions", - "system": "Linux", - "architecture": "x86_64", - "machineModel": null, - "cpuModel": "AMD EPYC 9V74 80-Core Processor", - "logicalCpuCount": 4, - "pageSizeBytes": 4096, - "enforceMachineModel": false, - "enforceCpuModel": true - }, - "measurement": { - "sampleCount": 5, - "confirmationSampleCount": 4, - "discardedLaunchCount": 1, - "warmupIterations": 2 - }, - "thresholds": { - "initAndFirstMs": { - "minimumExpected": 5.0, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "warmPerCallMs": { - "minimumExpected": 0.3, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "peakRssBytes": { - "minimumExpected": 16777216.0, - "regressionFactor": 1.03, - "improvementFactor": 0.97, - "aggregateRegressionFactor": 1.01, - "aggregateImprovementFactor": 0.99 - } - }, - "provenance": { - "gitSha": "2b3ac47d97cfeb4e594dcc7df0345798b361fd73", - "rustVersion": "rustc 1.96.0 (ac68faa20 2026-05-25)", - "workingTreeDirty": false - }, - "cases": { - "cel/conditional-100": { - "initAndFirstMs": 226.011, - "warmPerCallMs": 9.609, - "peakRssBytes": 238497792.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/duplicate-500": { - "initAndFirstMs": 267.188, - "warmPerCallMs": 48.495, - "peakRssBytes": 255139840.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/mixed-real": { - "initAndFirstMs": 236.088, - "warmPerCallMs": 14.521, - "peakRssBytes": 242634752.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-combined-conditions": { - "initAndFirstMs": 1016.324, - "warmPerCallMs": 809.327, - "peakRssBytes": 238682112.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-boundary": { - "initAndFirstMs": 305.095, - "warmPerCallMs": 92.91, - "peakRssBytes": 237342720.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-wide": { - "initAndFirstMs": 239.645, - "warmPerCallMs": 24.416, - "peakRssBytes": 237916160.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-fusion": { - "initAndFirstMs": 830.355, - "warmPerCallMs": 622.731, - "peakRssBytes": 244207616.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-reference-fanout": { - "initAndFirstMs": 1568.29, - "warmPerCallMs": 1329.702, - "peakRssBytes": 310259712.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-resource-scale": { - "initAndFirstMs": 264.748, - "warmPerCallMs": 45.516, - "peakRssBytes": 256118784.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-intrinsic-resolution": { - "initAndFirstMs": 217.98, - "warmPerCallMs": 2.354, - "peakRssBytes": 236662784.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-nesting": { - "initAndFirstMs": 215.693, - "warmPerCallMs": 0.021, - "peakRssBytes": 235192320.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "cel/security-deep-yaml-nesting": { - "initAndFirstMs": 222.425, - "warmPerCallMs": 3.391, - "peakRssBytes": 236539904.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "cel/security-foreach-branch-explosion": { - "initAndFirstMs": 355.649, - "warmPerCallMs": 137.244, - "peakRssBytes": 270102528.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-conditions": { - "initAndFirstMs": 295.129, - "warmPerCallMs": 81.327, - "peakRssBytes": 237658112.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-resources": { - "initAndFirstMs": 237.815, - "warmPerCallMs": 18.68, - "peakRssBytes": 241156096.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-pathological-conditions": { - "initAndFirstMs": 274.063, - "warmPerCallMs": 59.664, - "peakRssBytes": 237182976.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-scenario-assignment-budget": { - "initAndFirstMs": 783.335, - "warmPerCallMs": 565.063, - "peakRssBytes": 261361664.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/tiny": { - "initAndFirstMs": 214.754, - "warmPerCallMs": 0.054, - "peakRssBytes": 236392448.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "cel/unique-500": { - "initAndFirstMs": 245.66, - "warmPerCallMs": 28.236, - "peakRssBytes": 244596736.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/conditional-100": { - "initAndFirstMs": 226.973, - "warmPerCallMs": 9.578, - "peakRssBytes": 238530560.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/duplicate-500": { - "initAndFirstMs": 268.75, - "warmPerCallMs": 48.701, - "peakRssBytes": 255008768.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/mixed-real": { - "initAndFirstMs": 236.032, - "warmPerCallMs": 14.687, - "peakRssBytes": 242778112.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-combined-conditions": { - "initAndFirstMs": 1015.744, - "warmPerCallMs": 811.102, - "peakRssBytes": 238813184.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-boundary": { - "initAndFirstMs": 307.772, - "warmPerCallMs": 92.487, - "peakRssBytes": 237314048.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-wide": { - "initAndFirstMs": 240.417, - "warmPerCallMs": 24.464, - "peakRssBytes": 237772800.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-fusion": { - "initAndFirstMs": 832.237, - "warmPerCallMs": 621.916, - "peakRssBytes": 244277248.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-reference-fanout": { - "initAndFirstMs": 1573.911, - "warmPerCallMs": 1349.062, - "peakRssBytes": 310317056.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-resource-scale": { - "initAndFirstMs": 264.166, - "warmPerCallMs": 45.196, - "peakRssBytes": 256253952.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-intrinsic-resolution": { - "initAndFirstMs": 218.752, - "warmPerCallMs": 2.352, - "peakRssBytes": 236617728.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-nesting": { - "initAndFirstMs": 214.582, - "warmPerCallMs": 0.021, - "peakRssBytes": 235032576.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "composite/security-deep-yaml-nesting": { - "initAndFirstMs": 220.746, - "warmPerCallMs": 3.364, - "peakRssBytes": 236605440.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "composite/security-foreach-branch-explosion": { - "initAndFirstMs": 354.463, - "warmPerCallMs": 136.319, - "peakRssBytes": 270516224.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-conditions": { - "initAndFirstMs": 300.157, - "warmPerCallMs": 81.761, - "peakRssBytes": 237752320.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-resources": { - "initAndFirstMs": 235.423, - "warmPerCallMs": 18.549, - "peakRssBytes": 241369088.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-pathological-conditions": { - "initAndFirstMs": 274.976, - "warmPerCallMs": 59.767, - "peakRssBytes": 237285376.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-scenario-assignment-budget": { - "initAndFirstMs": 780.359, - "warmPerCallMs": 564.794, - "peakRssBytes": 261414912.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/tiny": { - "initAndFirstMs": 216.947, - "warmPerCallMs": 0.054, - "peakRssBytes": 236437504.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "composite/unique-500": { - "initAndFirstMs": 248.604, - "warmPerCallMs": 28.624, - "peakRssBytes": 244531200.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/conditional-100": { - "initAndFirstMs": 273.203, - "warmPerCallMs": 33.02, - "peakRssBytes": 244367360.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/duplicate-500": { - "initAndFirstMs": 396.352, - "warmPerCallMs": 151.007, - "peakRssBytes": 267960320.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/mixed-real": { - "initAndFirstMs": 296.52, - "warmPerCallMs": 58.055, - "peakRssBytes": 252489728.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-combined-conditions": { - "initAndFirstMs": 1129.001, - "warmPerCallMs": 905.411, - "peakRssBytes": 244551680.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-boundary": { - "initAndFirstMs": 349.905, - "warmPerCallMs": 113.73, - "peakRssBytes": 242921472.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-wide": { - "initAndFirstMs": 284.942, - "warmPerCallMs": 47.099, - "peakRssBytes": 243347456.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-fusion": { - "initAndFirstMs": 955.813, - "warmPerCallMs": 724.987, - "peakRssBytes": 251158528.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-reference-fanout": { - "initAndFirstMs": 3636.335, - "warmPerCallMs": 3348.761, - "peakRssBytes": 344977408.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-resource-scale": { - "initAndFirstMs": 372.576, - "warmPerCallMs": 127.934, - "peakRssBytes": 269209600.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-intrinsic-resolution": { - "initAndFirstMs": 245.738, - "warmPerCallMs": 6.809, - "peakRssBytes": 242118656.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-nesting": { - "initAndFirstMs": 238.665, - "warmPerCallMs": 0.02, - "peakRssBytes": 239579136.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "rego/security-deep-yaml-nesting": { - "initAndFirstMs": 242.755, - "warmPerCallMs": 3.281, - "peakRssBytes": 241573888.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "rego/security-foreach-branch-explosion": { - "initAndFirstMs": 385.668, - "warmPerCallMs": 139.471, - "peakRssBytes": 277061632.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-conditions": { - "initAndFirstMs": 393.515, - "warmPerCallMs": 153.049, - "peakRssBytes": 243781632.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-resources": { - "initAndFirstMs": 329.714, - "warmPerCallMs": 87.863, - "peakRssBytes": 249561088.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-pathological-conditions": { - "initAndFirstMs": 325.499, - "warmPerCallMs": 86.262, - "peakRssBytes": 242790400.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-scenario-assignment-budget": { - "initAndFirstMs": 830.943, - "warmPerCallMs": 585.018, - "peakRssBytes": 266121216.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/tiny": { - "initAndFirstMs": 241.709, - "warmPerCallMs": 2.585, - "peakRssBytes": 241430528.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/unique-500": { - "initAndFirstMs": 365.796, - "warmPerCallMs": 123.991, - "peakRssBytes": 252895232.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - } - } -} diff --git a/src/performance-harness/expected/github-ubuntu-x64-intel-xeon-6973pc.json b/src/performance-harness/expected/github-ubuntu-x64-intel-xeon-6973pc.json deleted file mode 100644 index 5ee67f15..00000000 --- a/src/performance-harness/expected/github-ubuntu-x64-intel-xeon-6973pc.json +++ /dev/null @@ -1,860 +0,0 @@ -{ - "schemaVersion": 1, - "profile": "github-ubuntu-x64-intel-xeon-6973pc", - "environment": { - "context": "github-actions", - "system": "Linux", - "architecture": "x86_64", - "machineModel": null, - "cpuModel": "Intel(R) Xeon(R) 6973P-C", - "logicalCpuCount": 4, - "pageSizeBytes": 4096, - "enforceMachineModel": false, - "enforceCpuModel": true - }, - "measurement": { - "sampleCount": 5, - "confirmationSampleCount": 4, - "discardedLaunchCount": 1, - "warmupIterations": 2 - }, - "thresholds": { - "initAndFirstMs": { - "minimumExpected": 5.0, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "warmPerCallMs": { - "minimumExpected": 0.3, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "peakRssBytes": { - "minimumExpected": 16777216.0, - "regressionFactor": 1.03, - "improvementFactor": 0.97, - "aggregateRegressionFactor": 1.01, - "aggregateImprovementFactor": 0.99 - } - }, - "provenance": { - "gitSha": "509ecdc00f1e0ceb28bd42c3a0542a13260f94cb", - "rustVersion": "rustc 1.96.0 (ac68faa20 2026-05-25)", - "workingTreeDirty": false - }, - "cases": { - "cel/conditional-100": { - "initAndFirstMs": 176.224, - "warmPerCallMs": 7.16, - "peakRssBytes": 238510080.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/duplicate-500": { - "initAndFirstMs": 207.418, - "warmPerCallMs": 33.637, - "peakRssBytes": 255016960.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/mixed-real": { - "initAndFirstMs": 183.065, - "warmPerCallMs": 11.106, - "peakRssBytes": 242679808.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-combined-conditions": { - "initAndFirstMs": 841.38, - "warmPerCallMs": 677.821, - "peakRssBytes": 238772224.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-boundary": { - "initAndFirstMs": 235.989, - "warmPerCallMs": 68.262, - "peakRssBytes": 237309952.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-wide": { - "initAndFirstMs": 187.603, - "warmPerCallMs": 19.304, - "peakRssBytes": 237764608.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-fusion": { - "initAndFirstMs": 670.64, - "warmPerCallMs": 503.545, - "peakRssBytes": 244215808.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-reference-fanout": { - "initAndFirstMs": 1835.794, - "warmPerCallMs": 1667.228, - "peakRssBytes": 310235136.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-resource-scale": { - "initAndFirstMs": 204.144, - "warmPerCallMs": 31.613, - "peakRssBytes": 256167936.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-intrinsic-resolution": { - "initAndFirstMs": 170.286, - "warmPerCallMs": 1.823, - "peakRssBytes": 236666880.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-nesting": { - "initAndFirstMs": 168.048, - "warmPerCallMs": 0.019, - "peakRssBytes": 235134976.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "cel/security-deep-yaml-nesting": { - "initAndFirstMs": 171.113, - "warmPerCallMs": 2.637, - "peakRssBytes": 236605440.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "cel/security-foreach-branch-explosion": { - "initAndFirstMs": 276.377, - "warmPerCallMs": 100.795, - "peakRssBytes": 270303232.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-conditions": { - "initAndFirstMs": 234.854, - "warmPerCallMs": 67.005, - "peakRssBytes": 237707264.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-resources": { - "initAndFirstMs": 182.415, - "warmPerCallMs": 12.863, - "peakRssBytes": 241541120.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-pathological-conditions": { - "initAndFirstMs": 219.099, - "warmPerCallMs": 48.35, - "peakRssBytes": 237338624.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-scenario-assignment-budget": { - "initAndFirstMs": 627.641, - "warmPerCallMs": 456.577, - "peakRssBytes": 261201920.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/tiny": { - "initAndFirstMs": 169.286, - "warmPerCallMs": 0.043, - "peakRssBytes": 236417024.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "cel/unique-500": { - "initAndFirstMs": 191.185, - "warmPerCallMs": 19.302, - "peakRssBytes": 244498432.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/conditional-100": { - "initAndFirstMs": 177.193, - "warmPerCallMs": 7.125, - "peakRssBytes": 238579712.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/duplicate-500": { - "initAndFirstMs": 206.614, - "warmPerCallMs": 33.865, - "peakRssBytes": 255070208.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/mixed-real": { - "initAndFirstMs": 185.722, - "warmPerCallMs": 11.087, - "peakRssBytes": 242696192.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-combined-conditions": { - "initAndFirstMs": 821.829, - "warmPerCallMs": 654.176, - "peakRssBytes": 238784512.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-boundary": { - "initAndFirstMs": 234.901, - "warmPerCallMs": 68.324, - "peakRssBytes": 237322240.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-wide": { - "initAndFirstMs": 188.959, - "warmPerCallMs": 19.328, - "peakRssBytes": 237899776.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-fusion": { - "initAndFirstMs": 669.111, - "warmPerCallMs": 505.765, - "peakRssBytes": 244232192.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-reference-fanout": { - "initAndFirstMs": 1832.287, - "warmPerCallMs": 1652.255, - "peakRssBytes": 310276096.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-resource-scale": { - "initAndFirstMs": 229.618, - "warmPerCallMs": 36.273, - "peakRssBytes": 256053248.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-intrinsic-resolution": { - "initAndFirstMs": 173.593, - "warmPerCallMs": 1.8, - "peakRssBytes": 236724224.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-nesting": { - "initAndFirstMs": 168.81, - "warmPerCallMs": 0.019, - "peakRssBytes": 235147264.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "composite/security-deep-yaml-nesting": { - "initAndFirstMs": 172.482, - "warmPerCallMs": 2.594, - "peakRssBytes": 236498944.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "composite/security-foreach-branch-explosion": { - "initAndFirstMs": 280.816, - "warmPerCallMs": 109.815, - "peakRssBytes": 270454784.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-conditions": { - "initAndFirstMs": 236.656, - "warmPerCallMs": 67.28, - "peakRssBytes": 237752320.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-resources": { - "initAndFirstMs": 184.596, - "warmPerCallMs": 13.001, - "peakRssBytes": 241336320.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-pathological-conditions": { - "initAndFirstMs": 218.959, - "warmPerCallMs": 48.352, - "peakRssBytes": 237223936.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-scenario-assignment-budget": { - "initAndFirstMs": 626.408, - "warmPerCallMs": 454.649, - "peakRssBytes": 261337088.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/tiny": { - "initAndFirstMs": 169.307, - "warmPerCallMs": 0.043, - "peakRssBytes": 236421120.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "composite/unique-500": { - "initAndFirstMs": 190.323, - "warmPerCallMs": 19.177, - "peakRssBytes": 244486144.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/conditional-100": { - "initAndFirstMs": 238.728, - "warmPerCallMs": 29.171, - "peakRssBytes": 244256768.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/duplicate-500": { - "initAndFirstMs": 308.203, - "warmPerCallMs": 114.004, - "peakRssBytes": 267702272.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/mixed-real": { - "initAndFirstMs": 234.619, - "warmPerCallMs": 46.802, - "peakRssBytes": 252399616.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-combined-conditions": { - "initAndFirstMs": 914.247, - "warmPerCallMs": 752.62, - "peakRssBytes": 244510720.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-boundary": { - "initAndFirstMs": 269.202, - "warmPerCallMs": 85.478, - "peakRssBytes": 242880512.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-wide": { - "initAndFirstMs": 224.834, - "warmPerCallMs": 38.205, - "peakRssBytes": 243212288.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-fusion": { - "initAndFirstMs": 772.95, - "warmPerCallMs": 589.441, - "peakRssBytes": 251281408.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-reference-fanout": { - "initAndFirstMs": 3829.367, - "warmPerCallMs": 3620.617, - "peakRssBytes": 345100288.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-resource-scale": { - "initAndFirstMs": 306.295, - "warmPerCallMs": 95.762, - "peakRssBytes": 268308480.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-intrinsic-resolution": { - "initAndFirstMs": 189.314, - "warmPerCallMs": 4.633, - "peakRssBytes": 242147328.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-nesting": { - "initAndFirstMs": 185.028, - "warmPerCallMs": 0.019, - "peakRssBytes": 239648768.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "rego/security-deep-yaml-nesting": { - "initAndFirstMs": 188.079, - "warmPerCallMs": 2.514, - "peakRssBytes": 241553408.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "rego/security-foreach-branch-explosion": { - "initAndFirstMs": 299.496, - "warmPerCallMs": 103.255, - "peakRssBytes": 276627456.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-conditions": { - "initAndFirstMs": 313.722, - "warmPerCallMs": 127.785, - "peakRssBytes": 243572736.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-resources": { - "initAndFirstMs": 256.701, - "warmPerCallMs": 67.66, - "peakRssBytes": 249532416.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-pathological-conditions": { - "initAndFirstMs": 256.506, - "warmPerCallMs": 70.652, - "peakRssBytes": 242724864.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-scenario-assignment-budget": { - "initAndFirstMs": 657.464, - "warmPerCallMs": 467.843, - "peakRssBytes": 266010624.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/tiny": { - "initAndFirstMs": 189.404, - "warmPerCallMs": 1.756, - "peakRssBytes": 241385472.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/unique-500": { - "initAndFirstMs": 284.764, - "warmPerCallMs": 93.615, - "peakRssBytes": 252899328.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - } - } -} diff --git a/src/performance-harness/expected/github-ubuntu-x64-intel-xeon-platinum-8573c.json b/src/performance-harness/expected/github-ubuntu-x64-intel-xeon-platinum-8573c.json deleted file mode 100644 index 1f545679..00000000 --- a/src/performance-harness/expected/github-ubuntu-x64-intel-xeon-platinum-8573c.json +++ /dev/null @@ -1,860 +0,0 @@ -{ - "schemaVersion": 1, - "profile": "github-ubuntu-x64-intel-xeon-platinum-8573c", - "environment": { - "context": "github-actions", - "system": "Linux", - "architecture": "x86_64", - "machineModel": null, - "cpuModel": "INTEL(R) XEON(R) PLATINUM 8573C", - "logicalCpuCount": 4, - "pageSizeBytes": 4096, - "enforceMachineModel": false, - "enforceCpuModel": true - }, - "measurement": { - "sampleCount": 5, - "confirmationSampleCount": 4, - "discardedLaunchCount": 1, - "warmupIterations": 2 - }, - "thresholds": { - "initAndFirstMs": { - "minimumExpected": 5.0, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "warmPerCallMs": { - "minimumExpected": 0.3, - "regressionFactor": 1.15, - "improvementFactor": 0.85, - "aggregateRegressionFactor": 1.1, - "aggregateImprovementFactor": 0.9 - }, - "peakRssBytes": { - "minimumExpected": 16777216.0, - "regressionFactor": 1.03, - "improvementFactor": 0.97, - "aggregateRegressionFactor": 1.01, - "aggregateImprovementFactor": 0.99 - } - }, - "provenance": { - "gitSha": "c73519b5accc3ee6148dd6f66d9440718f9f9dfd", - "rustVersion": "rustc 1.96.0 (ac68faa20 2026-05-25)", - "workingTreeDirty": false - }, - "cases": { - "cel/conditional-100": { - "initAndFirstMs": 189.727, - "warmPerCallMs": 8.574, - "peakRssBytes": 238460928.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/duplicate-500": { - "initAndFirstMs": 224.464, - "warmPerCallMs": 39.961, - "peakRssBytes": 255086592.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/mixed-real": { - "initAndFirstMs": 198.245, - "warmPerCallMs": 12.867, - "peakRssBytes": 242663424.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-combined-conditions": { - "initAndFirstMs": 939.914, - "warmPerCallMs": 760.135, - "peakRssBytes": 238837760.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-boundary": { - "initAndFirstMs": 259.97, - "warmPerCallMs": 79.441, - "peakRssBytes": 237301760.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-wide": { - "initAndFirstMs": 200.896, - "warmPerCallMs": 22.86, - "peakRssBytes": 237875200.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-fusion": { - "initAndFirstMs": 773.331, - "warmPerCallMs": 592.475, - "peakRssBytes": 244191232.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-reference-fanout": { - "initAndFirstMs": 2057.3, - "warmPerCallMs": 1852.863, - "peakRssBytes": 310329344.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-resource-scale": { - "initAndFirstMs": 222.382, - "warmPerCallMs": 37.693, - "peakRssBytes": 256151552.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-intrinsic-resolution": { - "initAndFirstMs": 179.719, - "warmPerCallMs": 2.177, - "peakRssBytes": 236589056.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-nesting": { - "initAndFirstMs": 176.362, - "warmPerCallMs": 0.024, - "peakRssBytes": 235180032.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "cel/security-deep-yaml-nesting": { - "initAndFirstMs": 180.832, - "warmPerCallMs": 3.35, - "peakRssBytes": 236564480.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "cel/security-foreach-branch-explosion": { - "initAndFirstMs": 312.672, - "warmPerCallMs": 124.351, - "peakRssBytes": 270045184.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-conditions": { - "initAndFirstMs": 261.412, - "warmPerCallMs": 79.471, - "peakRssBytes": 237789184.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-resources": { - "initAndFirstMs": 196.242, - "warmPerCallMs": 14.89, - "peakRssBytes": 241364992.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-pathological-conditions": { - "initAndFirstMs": 236.825, - "warmPerCallMs": 56.785, - "peakRssBytes": 237215744.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-scenario-assignment-budget": { - "initAndFirstMs": 714.297, - "warmPerCallMs": 525.406, - "peakRssBytes": 261312512.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/tiny": { - "initAndFirstMs": 179.022, - "warmPerCallMs": 0.051, - "peakRssBytes": 236408832.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "cel/unique-500": { - "initAndFirstMs": 203.813, - "warmPerCallMs": 22.149, - "peakRssBytes": 244449280.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/conditional-100": { - "initAndFirstMs": 187.366, - "warmPerCallMs": 8.576, - "peakRssBytes": 238485504.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/duplicate-500": { - "initAndFirstMs": 223.996, - "warmPerCallMs": 40.107, - "peakRssBytes": 254992384.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/mixed-real": { - "initAndFirstMs": 197.159, - "warmPerCallMs": 12.901, - "peakRssBytes": 242728960.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-combined-conditions": { - "initAndFirstMs": 939.615, - "warmPerCallMs": 757.623, - "peakRssBytes": 238882816.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-boundary": { - "initAndFirstMs": 260.559, - "warmPerCallMs": 79.879, - "peakRssBytes": 237379584.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-wide": { - "initAndFirstMs": 203.583, - "warmPerCallMs": 22.882, - "peakRssBytes": 237903872.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-fusion": { - "initAndFirstMs": 771.532, - "warmPerCallMs": 593.45, - "peakRssBytes": 244228096.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-reference-fanout": { - "initAndFirstMs": 2059.63, - "warmPerCallMs": 1857.906, - "peakRssBytes": 310255616.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-resource-scale": { - "initAndFirstMs": 220.284, - "warmPerCallMs": 37.576, - "peakRssBytes": 256155648.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-intrinsic-resolution": { - "initAndFirstMs": 179.15, - "warmPerCallMs": 2.179, - "peakRssBytes": 236650496.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-nesting": { - "initAndFirstMs": 176.892, - "warmPerCallMs": 0.024, - "peakRssBytes": 235220992.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "composite/security-deep-yaml-nesting": { - "initAndFirstMs": 180.707, - "warmPerCallMs": 3.351, - "peakRssBytes": 236535808.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "composite/security-foreach-branch-explosion": { - "initAndFirstMs": 313.664, - "warmPerCallMs": 124.203, - "peakRssBytes": 270139392.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-conditions": { - "initAndFirstMs": 261.352, - "warmPerCallMs": 79.77, - "peakRssBytes": 237715456.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-resources": { - "initAndFirstMs": 194.644, - "warmPerCallMs": 15.071, - "peakRssBytes": 241631232.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-pathological-conditions": { - "initAndFirstMs": 236.72, - "warmPerCallMs": 56.888, - "peakRssBytes": 237281280.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-scenario-assignment-budget": { - "initAndFirstMs": 717.871, - "warmPerCallMs": 525.33, - "peakRssBytes": 261349376.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/tiny": { - "initAndFirstMs": 178.505, - "warmPerCallMs": 0.052, - "peakRssBytes": 236302336.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "composite/unique-500": { - "initAndFirstMs": 203.336, - "warmPerCallMs": 22.104, - "peakRssBytes": 244592640.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/conditional-100": { - "initAndFirstMs": 231.694, - "warmPerCallMs": 30.361, - "peakRssBytes": 244248576.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/duplicate-500": { - "initAndFirstMs": 343.091, - "warmPerCallMs": 133.725, - "peakRssBytes": 267804672.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/mixed-real": { - "initAndFirstMs": 256.135, - "warmPerCallMs": 53.355, - "peakRssBytes": 252514304.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-combined-conditions": { - "initAndFirstMs": 1053.277, - "warmPerCallMs": 856.547, - "peakRssBytes": 244514816.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-boundary": { - "initAndFirstMs": 298.524, - "warmPerCallMs": 98.305, - "peakRssBytes": 242982912.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-wide": { - "initAndFirstMs": 244.706, - "warmPerCallMs": 45.238, - "peakRssBytes": 243310592.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-fusion": { - "initAndFirstMs": 891.886, - "warmPerCallMs": 690.23, - "peakRssBytes": 251256832.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-reference-fanout": { - "initAndFirstMs": 4481.843, - "warmPerCallMs": 4241.732, - "peakRssBytes": 345153536.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-resource-scale": { - "initAndFirstMs": 321.498, - "warmPerCallMs": 111.986, - "peakRssBytes": 268279808.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-intrinsic-resolution": { - "initAndFirstMs": 204.67, - "warmPerCallMs": 5.534, - "peakRssBytes": 242114560.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-nesting": { - "initAndFirstMs": 196.926, - "warmPerCallMs": 0.023, - "peakRssBytes": 239718400.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "rego/security-deep-yaml-nesting": { - "initAndFirstMs": 199.567, - "warmPerCallMs": 3.267, - "peakRssBytes": 241590272.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "rego/security-foreach-branch-explosion": { - "initAndFirstMs": 334.825, - "warmPerCallMs": 126.038, - "peakRssBytes": 276123648.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-conditions": { - "initAndFirstMs": 355.82, - "warmPerCallMs": 152.676, - "peakRssBytes": 243658752.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-resources": { - "initAndFirstMs": 284.419, - "warmPerCallMs": 79.563, - "peakRssBytes": 249479168.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-pathological-conditions": { - "initAndFirstMs": 286.513, - "warmPerCallMs": 83.563, - "peakRssBytes": 242843648.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-scenario-assignment-budget": { - "initAndFirstMs": 753.493, - "warmPerCallMs": 545.516, - "peakRssBytes": 266129408.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/tiny": { - "initAndFirstMs": 203.703, - "warmPerCallMs": 2.039, - "peakRssBytes": 241467392.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/unique-500": { - "initAndFirstMs": 317.853, - "warmPerCallMs": 110.955, - "peakRssBytes": 252874752.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - } - } -} diff --git a/src/performance-harness/expected/local-macos-arm64.json b/src/performance-harness/expected/local-macos-arm64.json deleted file mode 100644 index 63b3f168..00000000 --- a/src/performance-harness/expected/local-macos-arm64.json +++ /dev/null @@ -1,860 +0,0 @@ -{ - "schemaVersion": 1, - "profile": "local-macos-arm64", - "environment": { - "context": "local", - "system": "Darwin", - "architecture": "arm64", - "machineModel": "Mac14,9", - "cpuModel": "Apple M2 Pro", - "logicalCpuCount": 12, - "pageSizeBytes": 16384, - "enforceMachineModel": true, - "enforceCpuModel": false - }, - "measurement": { - "sampleCount": 5, - "confirmationSampleCount": 4, - "discardedLaunchCount": 1, - "warmupIterations": 2 - }, - "thresholds": { - "initAndFirstMs": { - "minimumExpected": 5.0, - "regressionFactor": 1.08, - "improvementFactor": 0.92, - "aggregateRegressionFactor": 1.07, - "aggregateImprovementFactor": 0.93 - }, - "warmPerCallMs": { - "minimumExpected": 0.3, - "regressionFactor": 1.08, - "improvementFactor": 0.92, - "aggregateRegressionFactor": 1.06, - "aggregateImprovementFactor": 0.94 - }, - "peakRssBytes": { - "minimumExpected": 16777216.0, - "regressionFactor": 1.02, - "improvementFactor": 0.98, - "aggregateRegressionFactor": 1.01, - "aggregateImprovementFactor": 0.99 - } - }, - "provenance": { - "gitSha": "2a2deba01f604805bdd93619155158252c542a5b", - "rustVersion": "rustc 1.96.0 (ac68faa20 2026-05-25)", - "workingTreeDirty": true - }, - "cases": { - "cel/conditional-100": { - "initAndFirstMs": 109.919, - "warmPerCallMs": 8.347, - "peakRssBytes": 263913472.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/duplicate-500": { - "initAndFirstMs": 138.838, - "warmPerCallMs": 37.678, - "peakRssBytes": 285802496.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/mixed-real": { - "initAndFirstMs": 119.402, - "warmPerCallMs": 10.126, - "peakRssBytes": 268124160.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-combined-conditions": { - "initAndFirstMs": 895.256, - "warmPerCallMs": 799.248, - "peakRssBytes": 264994816.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-boundary": { - "initAndFirstMs": 189.739, - "warmPerCallMs": 89.679, - "peakRssBytes": 263208960.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-chain-wide": { - "initAndFirstMs": 122.716, - "warmPerCallMs": 22.32, - "peakRssBytes": 263405568.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-condition-fusion": { - "initAndFirstMs": 690.599, - "warmPerCallMs": 588.27, - "peakRssBytes": 269631488.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-reference-fanout": { - "initAndFirstMs": 1567.197, - "warmPerCallMs": 1454.827, - "peakRssBytes": 345653248.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-cross-resource-scale": { - "initAndFirstMs": 141.341, - "warmPerCallMs": 39.415, - "peakRssBytes": 292012032.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-intrinsic-resolution": { - "initAndFirstMs": 103.366, - "warmPerCallMs": 1.871, - "peakRssBytes": 262602752.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-deep-nesting": { - "initAndFirstMs": 99.146, - "warmPerCallMs": 0.027, - "peakRssBytes": 259948544.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "cel/security-deep-yaml-nesting": { - "initAndFirstMs": 104.662, - "warmPerCallMs": 3.502, - "peakRssBytes": 262078464.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "cel/security-foreach-branch-explosion": { - "initAndFirstMs": 255.111, - "warmPerCallMs": 153.013, - "peakRssBytes": 336412672.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-conditions": { - "initAndFirstMs": 185.166, - "warmPerCallMs": 84.032, - "peakRssBytes": 263700480.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-many-resources": { - "initAndFirstMs": 112.495, - "warmPerCallMs": 12.399, - "peakRssBytes": 266731520.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-pathological-conditions": { - "initAndFirstMs": 157.439, - "warmPerCallMs": 56.911, - "peakRssBytes": 262897664.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/security-scenario-assignment-budget": { - "initAndFirstMs": 598.031, - "warmPerCallMs": 492.028, - "peakRssBytes": 290406400.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "cel/tiny": { - "initAndFirstMs": 100.496, - "warmPerCallMs": 0.039, - "peakRssBytes": 261586944.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "cel/unique-500": { - "initAndFirstMs": 119.58, - "warmPerCallMs": 18.191, - "peakRssBytes": 272072704.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/conditional-100": { - "initAndFirstMs": 114.265, - "warmPerCallMs": 8.254, - "peakRssBytes": 264011776.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/duplicate-500": { - "initAndFirstMs": 143.263, - "warmPerCallMs": 37.369, - "peakRssBytes": 285720576.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/mixed-real": { - "initAndFirstMs": 116.815, - "warmPerCallMs": 10.24, - "peakRssBytes": 268091392.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-combined-conditions": { - "initAndFirstMs": 897.775, - "warmPerCallMs": 797.887, - "peakRssBytes": 264929280.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-boundary": { - "initAndFirstMs": 189.771, - "warmPerCallMs": 89.302, - "peakRssBytes": 263094272.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-chain-wide": { - "initAndFirstMs": 123.03, - "warmPerCallMs": 22.288, - "peakRssBytes": 263553024.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-condition-fusion": { - "initAndFirstMs": 692.364, - "warmPerCallMs": 591.709, - "peakRssBytes": 269123584.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-reference-fanout": { - "initAndFirstMs": 1559.365, - "warmPerCallMs": 1457.858, - "peakRssBytes": 345341952.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-cross-resource-scale": { - "initAndFirstMs": 140.616, - "warmPerCallMs": 39.037, - "peakRssBytes": 292012032.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-intrinsic-resolution": { - "initAndFirstMs": 102.754, - "warmPerCallMs": 1.845, - "peakRssBytes": 262488064.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-deep-nesting": { - "initAndFirstMs": 104.642, - "warmPerCallMs": 0.027, - "peakRssBytes": 259260416.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "composite/security-deep-yaml-nesting": { - "initAndFirstMs": 107.215, - "warmPerCallMs": 3.515, - "peakRssBytes": 262193152.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "composite/security-foreach-branch-explosion": { - "initAndFirstMs": 253.883, - "warmPerCallMs": 151.864, - "peakRssBytes": 337133568.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-conditions": { - "initAndFirstMs": 185.549, - "warmPerCallMs": 83.766, - "peakRssBytes": 263569408.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-many-resources": { - "initAndFirstMs": 117.541, - "warmPerCallMs": 12.377, - "peakRssBytes": 265977856.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-pathological-conditions": { - "initAndFirstMs": 157.523, - "warmPerCallMs": 56.708, - "peakRssBytes": 262832128.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/security-scenario-assignment-budget": { - "initAndFirstMs": 600.732, - "warmPerCallMs": 493.974, - "peakRssBytes": 291094528.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "composite/tiny": { - "initAndFirstMs": 100.754, - "warmPerCallMs": 0.039, - "peakRssBytes": 261554176.0, - "gatedMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "peakRssBytes" - ] - }, - "composite/unique-500": { - "initAndFirstMs": 123.353, - "warmPerCallMs": 18.21, - "peakRssBytes": 272121856.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/conditional-100": { - "initAndFirstMs": 144.927, - "warmPerCallMs": 25.319, - "peakRssBytes": 277594112.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/duplicate-500": { - "initAndFirstMs": 233.157, - "warmPerCallMs": 113.589, - "peakRssBytes": 313589760.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/mixed-real": { - "initAndFirstMs": 160.48, - "warmPerCallMs": 40.365, - "peakRssBytes": 286081024.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-combined-conditions": { - "initAndFirstMs": 991.19, - "warmPerCallMs": 869.922, - "peakRssBytes": 279314432.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-boundary": { - "initAndFirstMs": 222.003, - "warmPerCallMs": 102.279, - "peakRssBytes": 275677184.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-chain-wide": { - "initAndFirstMs": 159.513, - "warmPerCallMs": 38.743, - "peakRssBytes": 276086784.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-condition-fusion": { - "initAndFirstMs": 776.211, - "warmPerCallMs": 652.349, - "peakRssBytes": 284786688.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-reference-fanout": { - "initAndFirstMs": 2852.036, - "warmPerCallMs": 2717.369, - "peakRssBytes": 397262848.0, - "gatedMetrics": [ - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-cross-resource-scale": { - "initAndFirstMs": 217.904, - "warmPerCallMs": 96.761, - "peakRssBytes": 315752448.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-intrinsic-resolution": { - "initAndFirstMs": 123.107, - "warmPerCallMs": 4.133, - "peakRssBytes": 273924096.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-deep-nesting": { - "initAndFirstMs": 117.719, - "warmPerCallMs": 0.027, - "peakRssBytes": 271253504.0, - "gatedMetrics": [], - "aggregateMetrics": [] - }, - "rego/security-deep-yaml-nesting": { - "initAndFirstMs": 122.516, - "warmPerCallMs": 3.456, - "peakRssBytes": 273694720.0, - "gatedMetrics": [ - "warmPerCallMs" - ], - "aggregateMetrics": [ - "warmPerCallMs" - ] - }, - "rego/security-foreach-branch-explosion": { - "initAndFirstMs": 273.9, - "warmPerCallMs": 152.827, - "peakRssBytes": 338673664.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-conditions": { - "initAndFirstMs": 258.563, - "warmPerCallMs": 137.97, - "peakRssBytes": 276840448.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-many-resources": { - "initAndFirstMs": 183.092, - "warmPerCallMs": 62.681, - "peakRssBytes": 282673152.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-pathological-conditions": { - "initAndFirstMs": 195.597, - "warmPerCallMs": 75.758, - "peakRssBytes": 275333120.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/security-scenario-assignment-budget": { - "initAndFirstMs": 639.371, - "warmPerCallMs": 513.669, - "peakRssBytes": 304168960.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/tiny": { - "initAndFirstMs": 121.522, - "warmPerCallMs": 1.359, - "peakRssBytes": 273268736.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - }, - "rego/unique-500": { - "initAndFirstMs": 209.794, - "warmPerCallMs": 90.456, - "peakRssBytes": 289619968.0, - "gatedMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ], - "aggregateMetrics": [ - "initAndFirstMs", - "warmPerCallMs", - "peakRssBytes" - ] - } - } -} diff --git a/src/performance-harness/src/baseline.rs b/src/performance-harness/src/baseline.rs deleted file mode 100644 index 0b709c3e..00000000 --- a/src/performance-harness/src/baseline.rs +++ /dev/null @@ -1,1704 +0,0 @@ -use crate::worker::Measurement; -use serde::{Deserialize, Serialize}; -use serde_json::{Value, json}; -use std::collections::{BTreeMap, BTreeSet}; -use std::env; -use std::fs; -use std::path::{Path, PathBuf}; -use std::process::{Command, Output}; - -const SCHEMA_VERSION: u32 = 1; -const ENGINES: [&str; 3] = ["rego", "cel", "composite"]; - -#[derive(Debug, Clone)] -struct Workload { - name: String, - templates: Vec, - iterations: usize, - gate_process_lifecycle: bool, -} - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub struct Environment { - context: String, - system: String, - architecture: String, - machine_model: Option, - cpu_model: Option, - logical_cpu_count: Option, - page_size_bytes: Option, - enforce_machine_model: bool, - #[serde(default)] - enforce_cpu_model: bool, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub struct MeasurementConfig { - sample_count: usize, - confirmation_sample_count: usize, - discarded_launch_count: usize, - warmup_iterations: usize, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub struct MetricThreshold { - minimum_expected: f64, - regression_factor: f64, - improvement_factor: f64, - aggregate_regression_factor: f64, - aggregate_improvement_factor: f64, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub struct Thresholds { - init_and_first_ms: MetricThreshold, - warm_per_call_ms: MetricThreshold, - peak_rss_bytes: MetricThreshold, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub struct ExpectedCase { - init_and_first_ms: f64, - warm_per_call_ms: f64, - peak_rss_bytes: f64, - gated_metrics: Vec, - aggregate_metrics: Vec, -} - -#[derive(Debug, Clone, Serialize, Deserialize)] -#[serde(rename_all = "camelCase", deny_unknown_fields)] -pub struct Baseline { - schema_version: u32, - profile: String, - environment: Environment, - measurement: MeasurementConfig, - thresholds: Thresholds, - provenance: Value, - cases: BTreeMap, -} - -#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq)] -#[serde(rename_all = "camelCase")] -pub struct Summary { - init_and_first_ms: f64, - warm_per_call_ms: f64, - peak_rss_bytes: f64, -} - -#[derive(Debug, Clone, Copy, Serialize, PartialEq, Eq)] -#[serde(rename_all = "lowercase")] -enum EvaluationStatus { - Info, - Pass, - Regression, - Improvement, -} - -#[derive(Debug, Clone, Serialize)] -#[serde(rename_all = "camelCase")] -struct MetricEvaluation { - case: String, - metric: String, - expected: f64, - actual: f64, - raw_ratio: f64, - normalized_ratio: f64, - lower_ratio: f64, - upper_ratio: f64, - status: EvaluationStatus, -} - -#[derive(Debug, Clone, Serialize)] -#[serde(rename_all = "camelCase")] -struct AggregateEvaluation { - metric: String, - ratio: f64, - lower_bound: f64, - upper_bound: f64, - case_count: usize, - status: EvaluationStatus, -} - -#[derive(Debug, Serialize)] -#[serde(rename_all = "camelCase")] -struct Results<'a> { - schema_version: u32, - profile: &'a str, - revision: String, - expected_file: String, - environment: &'a Environment, - sample_counts: BTreeMap, - summaries: &'a BTreeMap, - evaluations: &'a [MetricEvaluation], - aggregate_evaluations: &'a [AggregateEvaluation], - failures: &'a [String], - measurements: &'a BTreeMap>, -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] -enum Metric { - InitAndFirst, - WarmPerCall, - PeakRss, -} - -impl Metric { - const ALL: [Self; 3] = [Self::InitAndFirst, Self::WarmPerCall, Self::PeakRss]; - - fn key(self) -> &'static str { - match self { - Self::InitAndFirst => "initAndFirstMs", - Self::WarmPerCall => "warmPerCallMs", - Self::PeakRss => "peakRssBytes", - } - } - - fn label(self) -> &'static str { - match self { - Self::InitAndFirst => "Init + first", - Self::WarmPerCall => "Warm / call", - Self::PeakRss => "Peak RSS", - } - } - - fn measurement_value(self, measurement: &Measurement) -> f64 { - match self { - Self::InitAndFirst => measurement.init_total_ms + measurement.first_validation.wall_ms, - Self::WarmPerCall => measurement.warm.per_call_total_ms, - Self::PeakRss => measurement.peak_rss_bytes as f64, - } - } - - fn summary_value(self, summary: &Summary) -> f64 { - match self { - Self::InitAndFirst => summary.init_and_first_ms, - Self::WarmPerCall => summary.warm_per_call_ms, - Self::PeakRss => summary.peak_rss_bytes, - } - } - - fn expected_value(self, expected: &ExpectedCase) -> f64 { - match self { - Self::InitAndFirst => expected.init_and_first_ms, - Self::WarmPerCall => expected.warm_per_call_ms, - Self::PeakRss => expected.peak_rss_bytes, - } - } - - fn threshold(self, thresholds: &Thresholds) -> &MetricThreshold { - match self { - Self::InitAndFirst => &thresholds.init_and_first_ms, - Self::WarmPerCall => &thresholds.warm_per_call_ms, - Self::PeakRss => &thresholds.peak_rss_bytes, - } - } - - fn format_value(self, value: f64) -> String { - match self { - Self::PeakRss => format!("{:.1} MiB", value / (1024.0 * 1024.0)), - _ if value < 1.0 => format!("{value:.3} ms"), - _ => format!("{value:.2} ms"), - } - } -} - -#[derive(Debug)] -struct EvaluationOutcome { - metrics: Vec, - aggregates: Vec, - failures: Vec, - confirmation_cases: BTreeSet, -} - -fn project_root() -> PathBuf { - PathBuf::from(env!("CARGO_MANIFEST_DIR")) - .parent() - .and_then(Path::parent) - .map(Path::to_path_buf) - .unwrap_or_else(|| PathBuf::from(".")) -} - -fn expected_directory() -> PathBuf { - PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("expected") -} - -fn canonical_architecture(architecture: &str) -> String { - match architecture.to_ascii_lowercase().as_str() { - "amd64" | "x86_64" => "x86_64".into(), - "aarch64" | "arm64" => "arm64".into(), - other => other.into(), - } -} - -fn command_text(program: &str, arguments: &[&str]) -> Option { - let output = Command::new(program).args(arguments).output().ok()?; - if !output.status.success() { - return None; - } - let text = String::from_utf8(output.stdout).ok()?.trim().to_string(); - (!text.is_empty()).then_some(text) -} - -fn linux_cpu_model() -> Option { - let cpuinfo = fs::read_to_string("/proc/cpuinfo").ok()?; - cpuinfo.lines().find_map(|line| { - let (name, value) = line.split_once(':')?; - name.trim().eq_ignore_ascii_case("model name").then(|| value.trim().to_string()) - }) -} - -pub fn detect_environment() -> Environment { - let system = match env::consts::OS { - "macos" => "Darwin", - "linux" => "Linux", - other => other, - } - .to_string(); - let machine_model = (system == "Darwin").then(|| command_text("sysctl", &["-n", "hw.model"])).flatten(); - let cpu_model = if system == "Darwin" { - command_text("sysctl", &["-n", "machdep.cpu.brand_string"]).or_else(|| machine_model.clone()) - } else { - linux_cpu_model() - }; - let page_size_bytes = if system == "Darwin" { - command_text("sysctl", &["-n", "hw.pagesize"]).and_then(|value| value.parse().ok()) - } else { - command_text("getconf", &["PAGESIZE"]).and_then(|value| value.parse().ok()) - }; - Environment { - context: if env::var("GITHUB_ACTIONS").is_ok_and(|value| value.eq_ignore_ascii_case("true")) { - "github-actions".into() - } else { - "local".into() - }, - system, - architecture: canonical_architecture(env::consts::ARCH), - machine_model, - cpu_model, - logical_cpu_count: std::thread::available_parallelism().ok().map(usize::from), - page_size_bytes, - enforce_machine_model: false, - enforce_cpu_model: false, - } -} - -fn github_profile_name(cpu_model: &str) -> Result { - let normalized = cpu_model.to_ascii_lowercase().replace("(r)", ""); - let profile_identifiers: Vec = normalized - .split_whitespace() - .take_while(|part| !matches!(*part, "cpu" | "processor" | "@")) - .filter(|part| !part.ends_with("-core")) - .map(|part| part.chars().filter(|character| character.is_ascii_alphanumeric()).collect()) - .filter(|part: &String| !part.is_empty()) - .collect(); - if profile_identifiers.is_empty() { - return Err(format!("GitHub runner CPU model {cpu_model:?} has no usable profile identifier")); - } - Ok(format!("github-ubuntu-x64-{}", profile_identifiers.join("-"))) -} - -fn github_expected_file(cpu_model: Option<&str>) -> Result { - let cpu_model = cpu_model.ok_or_else(|| "GitHub runner CPU model could not be detected".to_string())?; - Ok(expected_directory().join(format!("{}.json", github_profile_name(cpu_model)?))) -} - -fn default_expected_path(environment: &Environment) -> Result { - if environment.context == "github-actions" { - if environment.system != "Linux" || environment.architecture != "x86_64" { - return Err("the checked-in GitHub baseline supports only Linux x86_64".into()); - } - return github_expected_file(environment.cpu_model.as_deref()); - } - if environment.system == "Darwin" && environment.architecture == "arm64" { - return Ok(expected_directory().join("local-macos-arm64.json")); - } - Err("no default performance profile for this environment; pass --expected explicitly".into()) -} - -pub fn default_expected_file(environment: &Environment) -> Result { - let expected_file = default_expected_path(environment)?; - if environment.context == "github-actions" && !expected_file.is_file() { - let cpu_model = environment - .cpu_model - .as_deref() - .ok_or_else(|| "GitHub runner CPU model could not be detected".to_string())?; - return Err(format!( - "no checked-in GitHub performance profile for CPU model {cpu_model:?}; add the calibrated profile {}", - expected_file.display() - )); - } - Ok(expected_file) -} - -fn validate_environment(expected: &Environment, actual: &Environment) -> Result<(), String> { - for (name, expected_value, actual_value) in [ - ("context", expected.context.as_str(), actual.context.as_str()), - ("system", expected.system.as_str(), actual.system.as_str()), - ("architecture", expected.architecture.as_str(), actual.architecture.as_str()), - ] { - if expected_value != actual_value { - return Err(format!( - "performance environment mismatch for {name}: expected={expected_value:?} actual={actual_value:?}" - )); - } - } - if expected.enforce_machine_model { - let expected_model = expected - .machine_model - .as_deref() - .ok_or_else(|| "baseline enforces machine model without recording one".to_string())?; - if Some(expected_model) != actual.machine_model.as_deref() { - return Err(format!( - "local baseline is for machine model {expected_model:?}, not {:?}", - actual.machine_model - )); - } - } - if expected.enforce_cpu_model { - let expected_cpu = expected - .cpu_model - .as_deref() - .ok_or_else(|| "baseline enforces CPU model without recording one".to_string())?; - if Some(expected_cpu) != actual.cpu_model.as_deref() { - return Err(format!("performance baseline is for CPU model {expected_cpu:?}, not {:?}", actual.cpu_model)); - } - } - Ok(()) -} - -fn validate_threshold(metric: Metric, threshold: &MetricThreshold) -> Result<(), String> { - if !threshold.minimum_expected.is_finite() || threshold.minimum_expected < 0.0 { - return Err(format!("{} minimumExpected must be finite and non-negative", metric.key())); - } - for (name, value) in [ - ("regressionFactor", threshold.regression_factor), - ("aggregateRegressionFactor", threshold.aggregate_regression_factor), - ] { - if !value.is_finite() || value <= 1.0 { - return Err(format!("{} {name} must be finite and greater than 1", metric.key())); - } - } - for (name, value) in [ - ("improvementFactor", threshold.improvement_factor), - ("aggregateImprovementFactor", threshold.aggregate_improvement_factor), - ] { - if !value.is_finite() || !(0.0..1.0).contains(&value) { - return Err(format!("{} {name} must be finite and between 0 and 1", metric.key())); - } - } - Ok(()) -} - -fn validated_metric_set<'a>(case: &str, scope: &str, metrics: &'a [String]) -> Result, String> { - let mut unique = BTreeSet::new(); - for metric_name in metrics { - if !Metric::ALL.iter().any(|metric| metric.key() == metric_name) { - return Err(format!("{case} {scope} includes unknown metric {metric_name}")); - } - if !unique.insert(metric_name.as_str()) { - return Err(format!("{case} {scope} includes {metric_name} more than once")); - } - } - Ok(unique) -} - -fn validate_baseline(baseline: &Baseline) -> Result<(), String> { - if baseline.schema_version != SCHEMA_VERSION { - return Err(format!( - "unsupported baseline schemaVersion {}; expected {SCHEMA_VERSION}", - baseline.schema_version - )); - } - if baseline.profile.is_empty() || baseline.cases.is_empty() { - return Err("baseline profile and cases must not be empty".into()); - } - for (name, value) in [ - ("sampleCount", baseline.measurement.sample_count), - ("confirmationSampleCount", baseline.measurement.confirmation_sample_count), - ("discardedLaunchCount", baseline.measurement.discarded_launch_count), - ("warmupIterations", baseline.measurement.warmup_iterations), - ] { - if value == 0 { - return Err(format!("{name} must be positive")); - } - } - for metric in Metric::ALL { - validate_threshold(metric, metric.threshold(&baseline.thresholds))?; - } - for (case, expected) in &baseline.cases { - let gated_metrics = validated_metric_set(case, "gatedMetrics", &expected.gated_metrics)?; - let aggregate_metrics = validated_metric_set(case, "aggregateMetrics", &expected.aggregate_metrics)?; - if !gated_metrics.is_subset(&aggregate_metrics) { - return Err(format!("{case} gatedMetrics must be a subset of aggregateMetrics")); - } - for metric in Metric::ALL { - let value = metric.expected_value(expected); - if !value.is_finite() || value <= 0.0 { - return Err(format!("{case} {} must be finite and positive", metric.key())); - } - if (gated_metrics.contains(metric.key()) || aggregate_metrics.contains(metric.key())) - && value < metric.threshold(&baseline.thresholds).minimum_expected - { - return Err(format!("{case} includes {} below its stability floor", metric.key())); - } - } - } - Ok(()) -} - -pub fn load_baseline(path: &Path) -> Result { - let bytes = fs::read(path).map_err(|error| format!("could not read {}: {error}", path.display()))?; - let baseline: Baseline = - serde_json::from_slice(&bytes).map_err(|error| format!("invalid baseline {}: {error}", path.display()))?; - validate_baseline(&baseline)?; - Ok(baseline) -} - -fn default_profile(profile: &str) -> Result<(MeasurementConfig, Thresholds), String> { - let measurement = MeasurementConfig { - sample_count: 5, - confirmation_sample_count: 4, - discarded_launch_count: 1, - warmup_iterations: 2, - }; - let thresholds = match profile { - profile if profile.starts_with("github-ubuntu-x64-") => Thresholds { - init_and_first_ms: MetricThreshold { - minimum_expected: 5.0, - regression_factor: 1.15, - improvement_factor: 0.85, - aggregate_regression_factor: 1.10, - aggregate_improvement_factor: 0.90, - }, - warm_per_call_ms: MetricThreshold { - minimum_expected: 0.30, - regression_factor: 1.15, - improvement_factor: 0.85, - aggregate_regression_factor: 1.10, - aggregate_improvement_factor: 0.90, - }, - peak_rss_bytes: MetricThreshold { - minimum_expected: 16.0 * 1024.0 * 1024.0, - regression_factor: 1.03, - improvement_factor: 0.97, - aggregate_regression_factor: 1.01, - aggregate_improvement_factor: 0.99, - }, - }, - "local-macos-arm64" => Thresholds { - init_and_first_ms: MetricThreshold { - minimum_expected: 5.0, - regression_factor: 1.08, - improvement_factor: 0.92, - aggregate_regression_factor: 1.07, - aggregate_improvement_factor: 0.93, - }, - warm_per_call_ms: MetricThreshold { - minimum_expected: 0.30, - regression_factor: 1.08, - improvement_factor: 0.92, - aggregate_regression_factor: 1.06, - aggregate_improvement_factor: 0.94, - }, - peak_rss_bytes: MetricThreshold { - minimum_expected: 16.0 * 1024.0 * 1024.0, - regression_factor: 1.02, - improvement_factor: 0.98, - aggregate_regression_factor: 1.01, - aggregate_improvement_factor: 0.99, - }, - }, - _ => return Err(format!("unknown performance profile {profile:?}")), - }; - Ok((measurement, thresholds)) -} - -fn write_buckets(path: &Path, count: usize, duplicate: bool) -> Result<(), String> { - let mut text = String::from("AWSTemplateFormatVersion: '2010-09-09'\nResources:\n"); - for index in 0..count { - let bucket_name = - if duplicate { "shared-performance-id".to_string() } else { format!("unique-performance-id-{index}") }; - text.push_str(&format!( - " Bucket{index}:\n Type: AWS::S3::Bucket\n Properties:\n BucketName: {bucket_name}\n" - )); - } - fs::write(path, text).map_err(|error| format!("could not write {}: {error}", path.display())) -} - -fn generate_fixtures(directory: &Path) -> Result, String> { - fs::create_dir_all(directory).map_err(|error| format!("could not create {}: {error}", directory.display()))?; - let tiny = directory.join("tiny.yaml"); - fs::write( - &tiny, - "AWSTemplateFormatVersion: '2010-09-09'\nResources:\n Bucket:\n Type: AWS::S3::Bucket\n Properties:\n BucketName: performance-baseline-bucket\n", - ) - .map_err(|error| format!("could not write {}: {error}", tiny.display()))?; - let unique = directory.join("unique-500.yaml"); - let duplicate = directory.join("duplicate-500.yaml"); - write_buckets(&unique, 500, false)?; - write_buckets(&duplicate, 500, true)?; - - let conditional = directory.join("conditional-100.yaml"); - let mut conditional_text = String::from( - "AWSTemplateFormatVersion: '2010-09-09'\nParameters:\n Environment:\n Type: String\n AllowedValues: [a, b]\nConditions:\n IsA: !Equals [!Ref Environment, a]\n IsB: !Equals [!Ref Environment, b]\nResources:\n", - ); - for index in 0..100 { - let condition = if index % 2 == 0 { "IsA" } else { "IsB" }; - conditional_text.push_str(&format!( - " Bucket{index}:\n Type: AWS::S3::Bucket\n Condition: {condition}\n Properties:\n BucketName: shared-conditional-id\n" - )); - } - fs::write(&conditional, conditional_text) - .map_err(|error| format!("could not write {}: {error}", conditional.display()))?; - Ok(BTreeMap::from([("tiny", tiny), ("unique", unique), ("duplicate", duplicate), ("conditional", conditional)])) -} - -fn collect_template_paths(directory: &Path, output: &mut Vec) -> Result<(), String> { - for entry in fs::read_dir(directory).map_err(|error| format!("could not read {}: {error}", directory.display()))? { - let entry = entry.map_err(|error| format!("could not read entry in {}: {error}", directory.display()))?; - let path = entry.path(); - if path.is_dir() { - collect_template_paths(&path, output)?; - } else if path - .extension() - .and_then(|extension| extension.to_str()) - .is_some_and(|extension| matches!(extension.to_ascii_lowercase().as_str(), "json" | "yaml" | "yml")) - { - output.push(path); - } - } - Ok(()) -} - -fn security_workloads(directory: &Path) -> Result, String> { - let iteration_counts = BTreeMap::from([ - ("condition_fusion.yaml", 2), - ("cross_reference_fanout.yaml", 1), - ("cross_resource_scale.yaml", 3), - ("deep_intrinsic_resolution.yaml", 2), - ("deep_nesting.json", 1), - ("deep_yaml_nesting.yaml", 1), - ("many_resources.yaml", 5), - ("pathological_conditions.yaml", 3), - ("scenario_assignment_budget.yaml", 2), - ]); - let mut templates = Vec::new(); - collect_template_paths(directory, &mut templates)?; - templates.sort(); - templates - .into_iter() - .map(|template| { - let relative = template - .strip_prefix(directory) - .map_err(|error| format!("security path {} is invalid: {error}", template.display()))?; - let mut parts: Vec = relative - .components() - .map(|component| component.as_os_str().to_string_lossy().replace('_', "-")) - .collect(); - let last = parts.last_mut().ok_or_else(|| "security template has no file name".to_string())?; - if let Some((stem, _)) = last.rsplit_once('.') { - *last = stem.to_string(); - } - let file_name = template.file_name().and_then(|name| name.to_str()).unwrap_or_default().to_string(); - Ok(Workload { - name: format!("security-{}", parts.join("-")), - templates: vec![template], - iterations: iteration_counts.get(file_name.as_str()).copied().unwrap_or(2), - gate_process_lifecycle: !matches!(file_name.as_str(), "deep_nesting.json" | "deep_yaml_nesting.yaml"), - }) - }) - .collect() -} - -fn workload_matrix(fixtures: &BTreeMap<&str, PathBuf>) -> Result, String> { - let fixture = - |name: &str| fixtures.get(name).cloned().ok_or_else(|| format!("generated fixture {name:?} is missing")); - let root = project_root(); - let templates = root.join("src/resources/templates"); - let mut workloads = vec![ - Workload { - name: "tiny".into(), - templates: vec![fixture("tiny")?], - iterations: 151, - gate_process_lifecycle: true, - }, - Workload { - name: "unique-500".into(), - templates: vec![fixture("unique")?], - iterations: 9, - gate_process_lifecycle: true, - }, - Workload { - name: "duplicate-500".into(), - templates: vec![fixture("duplicate")?], - iterations: 7, - gate_process_lifecycle: true, - }, - Workload { - name: "conditional-100".into(), - templates: vec![fixture("conditional")?], - iterations: 15, - gate_process_lifecycle: true, - }, - Workload { - name: "mixed-real".into(), - templates: vec![ - templates.join("cdk/codepipeline-build-deploy--CodepipelineBuildDeployStack.template.json"), - templates.join("quickstart/vpc.json"), - ], - iterations: 7, - gate_process_lifecycle: true, - }, - ]; - workloads.extend(security_workloads(&root.join("src/resources/security"))?); - Ok(workloads) -} - -fn time_arguments() -> Result, String> { - if !Path::new("/usr/bin/time").exists() { - return Err("/usr/bin/time is required for peak RSS measurement".into()); - } - match env::consts::OS { - "linux" => Ok(vec!["-v"]), - "macos" => Ok(vec!["-l"]), - other => Err(format!("performance measurement is not supported on {other}")), - } -} - -fn first_allowed_cpu() -> Option { - let status = fs::read_to_string("/proc/self/status").ok()?; - let value = status.lines().find_map(|line| line.strip_prefix("Cpus_allowed_list:"))?.trim(); - let first = value.split(',').next()?.split('-').next()?.trim(); - (!first.is_empty()).then(|| first.to_string()) -} - -fn taskset_prefix() -> Option<(PathBuf, String)> { - let taskset = ["/usr/bin/taskset", "/bin/taskset"].into_iter().map(PathBuf::from).find(|path| path.exists())?; - Some((taskset, first_allowed_cpu()?)) -} - -fn parse_peak_rss(stderr: &str) -> Result { - for line in stderr.lines() { - if let Some((_, value)) = line.split_once("Maximum resident set size (kbytes):") { - return value - .trim() - .parse::() - .map(|kilobytes| kilobytes * 1024) - .map_err(|error| format!("invalid Linux peak RSS: {error}")); - } - if let Some(value) = line.trim().strip_suffix("maximum resident set size") { - return value.trim().parse::().map_err(|error| format!("invalid macOS peak RSS: {error}")); - } - } - Err("maximum resident set size was not present in /usr/bin/time output".into()) -} - -fn failed_output(command: &str, output: &Output) -> String { - format!( - "{command} failed with {}\nstdout:\n{}\nstderr:\n{}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - ) -} - -fn run_measurement( - executable: &Path, - engine: &str, - workload: &Workload, - warmup_iterations: usize, - sample: i32, -) -> Result { - let mut command = Command::new("/usr/bin/time"); - command.args(time_arguments()?); - if let Some((taskset, cpu)) = taskset_prefix() { - command.arg(taskset).args(["-c", &cpu]); - } - command - .arg(executable) - .arg("measure") - .arg(engine) - .arg(workload.iterations.to_string()) - .arg(warmup_iterations.to_string()) - .arg(&workload.name) - .args(&workload.templates); - let output = command.output().map_err(|error| format!("could not run {}: {error}", workload.name))?; - if !output.status.success() { - return Err(failed_output(&format!("{engine}/{}", workload.name), &output)); - } - let stdout = - String::from_utf8(output.stdout).map_err(|error| format!("measurement output was not UTF-8: {error}"))?; - let json_line = stdout - .lines() - .rev() - .find(|line| !line.trim().is_empty()) - .ok_or_else(|| format!("measurement emitted no JSON for {engine}/{}", workload.name))?; - let mut measurement: Measurement = serde_json::from_str(json_line) - .map_err(|error| format!("measurement JSON was invalid for {engine}/{}: {error}", workload.name))?; - let stderr = String::from_utf8_lossy(&output.stderr); - measurement.peak_rss_bytes = parse_peak_rss(&stderr)?; - measurement.sample = sample; - measurement.gate_process_lifecycle = workload.gate_process_lifecycle; - Ok(measurement) -} - -fn collect_measurements( - executable: &Path, - workloads: &[Workload], - sample_count: usize, - config: &MeasurementConfig, - measurements: &mut BTreeMap>, - selected_cases: Option<&BTreeSet>, -) -> Result<(), String> { - for engine in ENGINES { - for workload in workloads { - let case = format!("{engine}/{}", workload.name); - if selected_cases.is_some_and(|selected| !selected.contains(&case)) { - continue; - } - let samples = measurements.entry(case.clone()).or_default(); - if samples.is_empty() { - for discarded in 0..config.discarded_launch_count { - eprintln!("Discarding launch {}/{} for {case}", discarded + 1, config.discarded_launch_count); - run_measurement(executable, engine, workload, config.warmup_iterations, -((discarded + 1) as i32))?; - } - } - for _ in 0..sample_count { - let sample = samples.len() as i32; - eprintln!("Measuring {case} sample {}", sample + 1); - samples.push(run_measurement(executable, engine, workload, config.warmup_iterations, sample)?); - } - } - } - Ok(()) -} - -fn median(mut values: Vec) -> Result { - if values.is_empty() || values.iter().any(|value| !value.is_finite()) { - return Err("median requires finite samples".into()); - } - values.sort_by(f64::total_cmp); - let middle = values.len() / 2; - if values.len().is_multiple_of(2) { Ok((values[middle - 1] + values[middle]) / 2.0) } else { Ok(values[middle]) } -} - -fn summarize_measurements( - measurements: &BTreeMap>, -) -> Result, String> { - measurements - .iter() - .map(|(case, samples)| { - let summary = Summary { - init_and_first_ms: median( - samples.iter().map(|sample| Metric::InitAndFirst.measurement_value(sample)).collect(), - )?, - warm_per_call_ms: median( - samples.iter().map(|sample| Metric::WarmPerCall.measurement_value(sample)).collect(), - )?, - peak_rss_bytes: median( - samples.iter().map(|sample| Metric::PeakRss.measurement_value(sample)).collect(), - )?, - }; - Ok((case.clone(), summary)) - }) - .collect() -} - -fn diagnostic_signature(measurement: &Measurement) -> Result { - let fingerprints: Vec<_> = measurement - .fingerprints - .iter() - .map(|item| { - ( - Path::new(&item.path).file_name().and_then(|name| name.to_str()).unwrap_or_default(), - item.fingerprint.as_str(), - item.diagnostics, - &item.status, - ) - }) - .collect(); - serde_json::to_string(&(measurement.first_validation.fingerprint.as_str(), fingerprints)) - .map_err(|error| format!("diagnostic signature could not be serialized: {error}")) -} - -fn diagnostic_failures(measurements: &BTreeMap>) -> Result, String> { - let mut failures = Vec::new(); - for (case, samples) in measurements { - let Some(first) = samples.first() else { - return Err(format!("no samples collected for {case}")); - }; - let expected = diagnostic_signature(first)?; - for sample in samples.iter().skip(1) { - if diagnostic_signature(sample)? != expected { - failures.push(format!("{case}: diagnostic fingerprints changed across performance samples")); - break; - } - } - } - Ok(failures) -} - -fn expected_case_names(workloads: &[Workload]) -> BTreeSet { - ENGINES - .into_iter() - .flat_map(|engine| workloads.iter().map(move |workload| format!("{engine}/{}", workload.name))) - .collect() -} - -fn validate_case_set(baseline: &Baseline, workloads: &[Workload]) -> Result<(), String> { - let expected: BTreeSet<_> = baseline.cases.keys().cloned().collect(); - let actual = expected_case_names(workloads); - if expected == actual { - return Ok(()); - } - let missing: Vec<_> = actual.difference(&expected).cloned().collect(); - let obsolete: Vec<_> = expected.difference(&actual).cloned().collect(); - Err(format!( - "performance workload set differs from the baseline; update it intentionally. missing={missing:?} obsolete={obsolete:?}" - )) -} - -fn evaluation_status(ratio: f64, improvement_factor: f64, regression_factor: f64) -> EvaluationStatus { - if ratio > regression_factor { - EvaluationStatus::Regression - } else if ratio < improvement_factor { - EvaluationStatus::Improvement - } else { - EvaluationStatus::Pass - } -} - -fn geometric_mean(values: &[f64]) -> Result { - if values.is_empty() || values.iter().any(|value| !value.is_finite() || *value <= 0.0) { - return Err("geometric mean requires finite positive ratios".into()); - } - Ok((values.iter().map(|value| value.ln()).sum::() / values.len() as f64).exp()) -} - -fn evaluate_performance( - baseline: &Baseline, - summaries: &BTreeMap, -) -> Result { - let mut aggregate_ratios: BTreeMap> = BTreeMap::new(); - for (case, expected_case) in &baseline.cases { - let summary = summaries.get(case).ok_or_else(|| format!("no measured summary for {case}"))?; - let aggregate_metrics: BTreeSet<_> = expected_case.aggregate_metrics.iter().map(String::as_str).collect(); - for metric in Metric::ALL { - if aggregate_metrics.contains(metric.key()) { - let ratio = metric.summary_value(summary) / metric.expected_value(expected_case); - aggregate_ratios.entry(metric).or_default().push((case.clone(), ratio)); - } - } - } - - let mut aggregates = Vec::new(); - let mut normalizers = BTreeMap::new(); - let mut failures = Vec::new(); - let mut confirmation_cases = BTreeSet::new(); - for metric in Metric::ALL { - let ratios = aggregate_ratios.get(&metric).cloned().unwrap_or_default(); - if ratios.is_empty() { - continue; - } - let ratio = geometric_mean(&ratios.iter().map(|(_, value)| *value).collect::>())?; - normalizers.insert(metric, ratio); - let threshold = metric.threshold(&baseline.thresholds); - let status = - evaluation_status(ratio, threshold.aggregate_improvement_factor, threshold.aggregate_regression_factor); - aggregates.push(AggregateEvaluation { - metric: metric.key().into(), - ratio, - lower_bound: threshold.aggregate_improvement_factor, - upper_bound: threshold.aggregate_regression_factor, - case_count: ratios.len(), - status, - }); - match status { - EvaluationStatus::Regression => { - failures.push(format!( - "aggregate {} regressed to {ratio:.3}x expected across {} cases", - metric.label(), - ratios.len() - )); - confirmation_cases.extend(ratios.iter().map(|(case, _)| case.clone())); - } - EvaluationStatus::Improvement => { - failures.push(format!( - "aggregate {} improved to {ratio:.3}x expected across {} cases; confirm it and update the expected file", - metric.label(), - ratios.len() - )); - confirmation_cases.extend(ratios.iter().map(|(case, _)| case.clone())); - } - EvaluationStatus::Info | EvaluationStatus::Pass => {} - } - } - - let mut metrics = Vec::new(); - for (case, expected_case) in &baseline.cases { - let summary = summaries.get(case).ok_or_else(|| format!("no measured summary for {case}"))?; - let gated_metrics: BTreeSet<_> = expected_case.gated_metrics.iter().map(String::as_str).collect(); - for metric in Metric::ALL { - let expected = metric.expected_value(expected_case); - let actual = metric.summary_value(summary); - let raw_ratio = actual / expected; - let normalized_ratio = raw_ratio / normalizers.get(&metric).copied().unwrap_or(1.0); - let threshold = metric.threshold(&baseline.thresholds); - let status = if gated_metrics.contains(metric.key()) { - evaluation_status(normalized_ratio, threshold.improvement_factor, threshold.regression_factor) - } else { - EvaluationStatus::Info - }; - metrics.push(MetricEvaluation { - case: case.clone(), - metric: metric.key().into(), - expected, - actual, - raw_ratio, - normalized_ratio, - lower_ratio: threshold.improvement_factor, - upper_ratio: threshold.regression_factor, - status, - }); - match status { - EvaluationStatus::Regression => { - failures.push(format!( - "{case}: {} regressed to {normalized_ratio:.3}x normalized ({raw_ratio:.3}x raw; {} vs {} expected)", - metric.label(), - metric.format_value(actual), - metric.format_value(expected) - )); - confirmation_cases.insert(case.clone()); - } - EvaluationStatus::Improvement => { - failures.push(format!( - "{case}: {} improved to {normalized_ratio:.3}x normalized ({raw_ratio:.3}x raw); confirm it and update the expected file", - metric.label() - )); - confirmation_cases.insert(case.clone()); - } - EvaluationStatus::Info | EvaluationStatus::Pass => {} - } - } - } - Ok(EvaluationOutcome { metrics, aggregates, failures, confirmation_cases }) -} - -fn workload_gates(workload: &Workload, summary: &Summary, thresholds: &Thresholds) -> Vec { - workload_aggregate_metrics(workload, summary, thresholds) - .into_iter() - .filter(|metric| workload.name != "security-cross-reference-fanout" || metric == Metric::PeakRss.key()) - .collect() -} - -fn workload_aggregate_metrics(workload: &Workload, summary: &Summary, thresholds: &Thresholds) -> Vec { - Metric::ALL - .into_iter() - .filter(|metric| metric.summary_value(summary) >= metric.threshold(thresholds).minimum_expected) - .filter(|metric| workload.gate_process_lifecycle || !matches!(metric, Metric::InitAndFirst | Metric::PeakRss)) - .map(|metric| metric.key().to_string()) - .collect() -} - -fn command_output(program: &str, arguments: &[&str], directory: &Path) -> Result { - let output = Command::new(program) - .args(arguments) - .current_dir(directory) - .output() - .map_err(|error| format!("could not run {program}: {error}"))?; - if !output.status.success() { - return Err(failed_output(program, &output)); - } - String::from_utf8(output.stdout) - .map(|value| value.trim().to_string()) - .map_err(|error| format!("{program} output was not UTF-8: {error}")) -} - -fn provenance() -> Result { - let root = project_root(); - Ok(json!({ - "gitSha": command_output("git", &["rev-parse", "HEAD"], &root)?, - "rustVersion": command_output("rustc", &["--version"], &root.join("src"))?, - "workingTreeDirty": !command_output("git", &["status", "--porcelain"], &root)?.is_empty(), - })) -} - -fn round_milliseconds(value: f64) -> f64 { - (value * 1_000.0).round() / 1_000.0 -} - -fn build_candidate( - profile: &str, - environment: &Environment, - measurement: &MeasurementConfig, - thresholds: &Thresholds, - summaries: &BTreeMap, - workloads: &[Workload], -) -> Result { - let workload_by_name: BTreeMap<_, _> = - workloads.iter().map(|workload| (workload.name.as_str(), workload)).collect(); - let cases = summaries - .iter() - .map(|(case, summary)| { - let (_, workload_name) = case.split_once('/').ok_or_else(|| format!("invalid case name {case}"))?; - let workload = - workload_by_name.get(workload_name).ok_or_else(|| format!("missing workload {workload_name}"))?; - Ok(( - case.clone(), - ExpectedCase { - init_and_first_ms: round_milliseconds(summary.init_and_first_ms), - warm_per_call_ms: round_milliseconds(summary.warm_per_call_ms), - peak_rss_bytes: summary.peak_rss_bytes.round(), - gated_metrics: workload_gates(workload, summary, thresholds), - aggregate_metrics: workload_aggregate_metrics(workload, summary, thresholds), - }, - )) - }) - .collect::>()?; - let mut expected_environment = environment.clone(); - expected_environment.enforce_machine_model = profile == "local-macos-arm64"; - expected_environment.enforce_cpu_model = profile.starts_with("github-ubuntu-x64-"); - let baseline = Baseline { - schema_version: SCHEMA_VERSION, - profile: profile.into(), - environment: expected_environment, - measurement: measurement.clone(), - thresholds: thresholds.clone(), - provenance: provenance()?, - cases, - }; - validate_baseline(&baseline)?; - Ok(baseline) -} - -fn write_json(path: &Path, value: &impl Serialize) -> Result<(), String> { - if let Some(parent) = path.parent() { - fs::create_dir_all(parent).map_err(|error| format!("could not create {}: {error}", parent.display()))?; - } - let bytes = serde_json::to_vec_pretty(value).map_err(|error| format!("JSON serialization failed: {error}"))?; - fs::write(path, [bytes, b"\n".to_vec()].concat()) - .map_err(|error| format!("could not write {}: {error}", path.display())) -} - -fn render_markdown( - baseline: &Baseline, - revision: &str, - outcome: &EvaluationOutcome, - failures: &[String], - sample_total: usize, - expected_file: &Path, -) -> Result { - let mut by_case: BTreeMap<&str, BTreeMap<&str, &MetricEvaluation>> = BTreeMap::new(); - for evaluation in &outcome.metrics { - by_case.entry(&evaluation.case).or_default().insert(&evaluation.metric, evaluation); - } - let mut lines = vec![ - "# Expected performance check".to_string(), - String::new(), - format!("Profile: `{}` ", baseline.profile), - format!("Revision: `{}` ", &revision[..revision.len().min(12)]), - format!("Samples per case: up to `{sample_total}`"), - String::new(), - "Per-case ratios remove the run-wide geometric-mean host-speed shift; aggregate ratios remain raw current/expected. `(info)` metrics are not gated." - .to_string(), - String::new(), - "| Engine / workload | Init + first | Warm / call | Peak RSS | Status |".to_string(), - "|---|---:|---:|---:|---|".to_string(), - ]; - for (case, evaluations) in by_case { - let mut cells = Vec::new(); - let mut failed = false; - for metric in Metric::ALL { - let evaluation = evaluations - .get(metric.key()) - .ok_or_else(|| format!("missing {} evaluation for {case}", metric.key()))?; - let suffix = if evaluation.status == EvaluationStatus::Info { " (info)" } else { "" }; - cells.push(format!("{:.3}x{suffix}", evaluation.normalized_ratio)); - failed |= matches!(evaluation.status, EvaluationStatus::Regression | EvaluationStatus::Improvement); - } - lines.push(format!("| {case} | {} | {} |", cells.join(" | "), if failed { "FAIL" } else { "pass" })); - } - lines.extend([ - String::new(), - "## Aggregate ratios".into(), - String::new(), - "| Metric | Ratio | Allowed range | Cases | Status |".into(), - "|---|---:|---:|---:|---|".into(), - ]); - for evaluation in &outcome.aggregates { - let label = Metric::ALL - .into_iter() - .find(|metric| metric.key() == evaluation.metric) - .map(Metric::label) - .unwrap_or(&evaluation.metric); - lines.push(format!( - "| {label} | {:.3}x | {:.3}x–{:.3}x | {} | {:?} |", - evaluation.ratio, evaluation.lower_bound, evaluation.upper_bound, evaluation.case_count, evaluation.status - )); - } - lines.extend([String::new(), "## Result".into(), String::new()]); - if failures.is_empty() { - lines.push("✅ Performance is within the checked-in two-sided expectations.".into()); - } else { - lines.extend(failures.iter().map(|failure| format!("* ❌ {failure}"))); - lines.extend([ - String::new(), - "A ready-to-review `performance-candidate-baseline.json` is included in the artifact.".into(), - "Regressions require a code fix. For a confirmed improvement, replace the expected file with the candidate." - .into(), - String::new(), - "```bash".into(), - format!( - "cargo run --locked --release -p performance-harness -- update --expected {}", - expected_file.display() - ), - "```".into(), - ]); - } - Ok(lines.join("\n") + "\n") -} - -fn prepare_run(output_dir: &Path) -> Result<(PathBuf, Vec), String> { - fs::create_dir_all(output_dir).map_err(|error| format!("could not create {}: {error}", output_dir.display()))?; - let fixtures = generate_fixtures(&output_dir.join("fixtures"))?; - let workloads = workload_matrix(&fixtures)?; - let executable = env::current_exe().map_err(|error| format!("could not locate performance harness: {error}"))?; - Ok((executable, workloads)) -} - -fn render_missing_profile_markdown( - profile: &str, - cpu_model: &str, - sample_total: usize, - expected_file: &Path, - diagnostic_errors: &[String], -) -> String { - let mut lines = vec![ - "# Expected performance check".to_string(), - String::new(), - format!("Profile: `{profile}` "), - format!("CPU model: `{cpu_model}` "), - format!("Calibration samples per case: `{sample_total}`"), - String::new(), - "## Result".into(), - String::new(), - format!("* ❌ No checked-in hard baseline exists for CPU model `{cpu_model}`."), - ]; - lines.extend(diagnostic_errors.iter().map(|failure| format!("* ❌ {failure}"))); - lines.extend([ - String::new(), - "The check remains failed. A CPU-enforced `performance-candidate-baseline.json` is included in the artifact." - .into(), - format!("Validate repeated hosted-runner measurements before adding it as `{}`.", expected_file.display()), - ]); - lines.join("\n") + "\n" -} - -fn run_missing_github_profile( - expected_file: &Path, - environment: &Environment, - output_dir: &Path, -) -> Result { - let profile = expected_file - .file_stem() - .and_then(|value| value.to_str()) - .ok_or_else(|| format!("invalid GitHub performance profile path {}", expected_file.display()))?; - let cpu_model = - environment.cpu_model.as_deref().ok_or_else(|| "GitHub runner CPU model could not be detected".to_string())?; - let (measurement, thresholds) = default_profile(profile)?; - let calibration_sample_count = measurement.sample_count + measurement.confirmation_sample_count; - let (executable, workloads) = prepare_run(output_dir)?; - let mut measurements = BTreeMap::new(); - collect_measurements(&executable, &workloads, calibration_sample_count, &measurement, &mut measurements, None)?; - let summaries = summarize_measurements(&measurements)?; - let diagnostic_errors = diagnostic_failures(&measurements)?; - let mut candidate = build_candidate(profile, environment, &measurement, &thresholds, &summaries, &workloads)?; - let provenance = candidate - .provenance - .as_object_mut() - .ok_or_else(|| "candidate baseline provenance must be an object".to_string())?; - provenance.insert( - "calibration".into(), - json!({ - "purpose": "missingGitHubCpuProfile", - "samplesPerCase": calibration_sample_count, - }), - ); - write_json(&output_dir.join("performance-candidate-baseline.json"), &candidate)?; - - let mut failures = vec![format!("no checked-in hard performance baseline exists for CPU model {cpu_model:?}")]; - failures.extend(diagnostic_errors.clone()); - let revision = command_output("git", &["rev-parse", "HEAD"], &project_root())?; - let sample_counts = measurements.iter().map(|(case, samples)| (case.clone(), samples.len())).collect(); - let evaluations: Vec = Vec::new(); - let aggregate_evaluations: Vec = Vec::new(); - let results = Results { - schema_version: SCHEMA_VERSION, - profile, - revision, - expected_file: expected_file.display().to_string(), - environment, - sample_counts, - summaries: &summaries, - evaluations: &evaluations, - aggregate_evaluations: &aggregate_evaluations, - failures: &failures, - measurements: &measurements, - }; - write_json(&output_dir.join("performance-results.json"), &results)?; - let markdown = render_missing_profile_markdown( - profile, - cpu_model, - calibration_sample_count, - expected_file, - &diagnostic_errors, - ); - fs::write(output_dir.join("performance-results.md"), &markdown) - .map_err(|error| format!("could not write performance markdown: {error}"))?; - print!("{markdown}"); - for failure in &failures { - println!("::error::{failure}"); - } - Ok(false) -} - -pub fn run_default_check(environment: &Environment, output_dir: &Path) -> Result { - let expected_file = default_expected_path(environment)?; - if environment.context == "github-actions" && !expected_file.is_file() { - return run_missing_github_profile(&expected_file, environment, output_dir); - } - run_check(&expected_file, output_dir) -} - -pub fn run_check(expected_file: &Path, output_dir: &Path) -> Result { - let baseline = load_baseline(expected_file)?; - let environment = detect_environment(); - validate_environment(&baseline.environment, &environment)?; - let (executable, workloads) = prepare_run(output_dir)?; - validate_case_set(&baseline, &workloads)?; - let mut measurements = BTreeMap::new(); - collect_measurements( - &executable, - &workloads, - baseline.measurement.sample_count, - &baseline.measurement, - &mut measurements, - None, - )?; - let mut summaries = summarize_measurements(&measurements)?; - let mut diagnostic_errors = diagnostic_failures(&measurements)?; - let mut outcome = evaluate_performance(&baseline, &summaries)?; - if diagnostic_errors.is_empty() && !outcome.failures.is_empty() { - eprintln!( - "Confirming apparent performance change in {} case(s) with {} additional samples", - outcome.confirmation_cases.len(), - baseline.measurement.confirmation_sample_count - ); - collect_measurements( - &executable, - &workloads, - baseline.measurement.confirmation_sample_count, - &baseline.measurement, - &mut measurements, - Some(&outcome.confirmation_cases), - )?; - summaries = summarize_measurements(&measurements)?; - diagnostic_errors = diagnostic_failures(&measurements)?; - outcome = evaluate_performance(&baseline, &summaries)?; - } - let candidate = build_candidate( - &baseline.profile, - &environment, - &baseline.measurement, - &baseline.thresholds, - &summaries, - &workloads, - )?; - write_json(&output_dir.join("performance-candidate-baseline.json"), &candidate)?; - let mut failures = diagnostic_errors; - failures.extend(outcome.failures.clone()); - let revision = command_output("git", &["rev-parse", "HEAD"], &project_root())?; - let sample_counts = measurements.iter().map(|(case, samples)| (case.clone(), samples.len())).collect(); - let results = Results { - schema_version: SCHEMA_VERSION, - profile: &baseline.profile, - revision: revision.clone(), - expected_file: expected_file.display().to_string(), - environment: &environment, - sample_counts, - summaries: &summaries, - evaluations: &outcome.metrics, - aggregate_evaluations: &outcome.aggregates, - failures: &failures, - measurements: &measurements, - }; - write_json(&output_dir.join("performance-results.json"), &results)?; - let sample_total = measurements.values().map(Vec::len).max().unwrap_or(0); - let markdown = render_markdown(&baseline, &revision, &outcome, &failures, sample_total, expected_file)?; - fs::write(output_dir.join("performance-results.md"), &markdown) - .map_err(|error| format!("could not write performance markdown: {error}"))?; - print!("{markdown}"); - for failure in &failures { - println!("::error::{failure}"); - } - Ok(failures.is_empty()) -} - -pub fn run_update(expected_file: &Path, profile: Option<&str>, output_dir: &Path) -> Result<(), String> { - let environment = detect_environment(); - let (profile, measurement, thresholds) = if expected_file.exists() { - let baseline = load_baseline(expected_file)?; - validate_environment(&baseline.environment, &environment)?; - (baseline.profile, baseline.measurement, baseline.thresholds) - } else { - let profile = profile - .map(str::to_string) - .or_else(|| expected_file.file_stem().and_then(|value| value.to_str()).map(str::to_string)) - .ok_or_else(|| "--profile is required to create this baseline".to_string())?; - let (measurement, thresholds) = default_profile(&profile)?; - (profile, measurement, thresholds) - }; - let (executable, workloads) = prepare_run(output_dir)?; - let mut measurements = BTreeMap::new(); - collect_measurements(&executable, &workloads, measurement.sample_count, &measurement, &mut measurements, None)?; - let diagnostic_errors = diagnostic_failures(&measurements)?; - if !diagnostic_errors.is_empty() { - return Err(diagnostic_errors.join("; ")); - } - let summaries = summarize_measurements(&measurements)?; - let candidate = build_candidate(&profile, &environment, &measurement, &thresholds, &summaries, &workloads)?; - write_json(expected_file, &candidate)?; - write_json(&output_dir.join("performance-candidate-baseline.json"), &candidate)?; - println!("Updated expected performance: {}", expected_file.display()); - println!("Measured {} cases with {} samples each", summaries.len(), measurement.sample_count); - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::worker::{FirstValidation, WarmMeasurement}; - - fn threshold( - regression: f64, - improvement: f64, - aggregate_regression: f64, - aggregate_improvement: f64, - ) -> MetricThreshold { - MetricThreshold { - minimum_expected: 0.0, - regression_factor: regression, - improvement_factor: improvement, - aggregate_regression_factor: aggregate_regression, - aggregate_improvement_factor: aggregate_improvement, - } - } - - fn synthetic_baseline() -> Baseline { - let thresholds = Thresholds { - init_and_first_ms: threshold(1.15, 0.85, 1.10, 0.90), - warm_per_call_ms: threshold(1.15, 0.85, 1.10, 0.90), - peak_rss_bytes: threshold(1.03, 0.97, 1.01, 0.99), - }; - let gated_metrics: Vec = Metric::ALL.into_iter().map(|metric| metric.key().to_string()).collect(); - let expected = ExpectedCase { - init_and_first_ms: 100.0, - warm_per_call_ms: 10.0, - peak_rss_bytes: 100_000_000.0, - gated_metrics: gated_metrics.clone(), - aggregate_metrics: gated_metrics, - }; - Baseline { - schema_version: SCHEMA_VERSION, - profile: "test".into(), - environment: Environment { - context: "local".into(), - system: "Test".into(), - architecture: "test".into(), - machine_model: None, - cpu_model: None, - logical_cpu_count: Some(1), - page_size_bytes: Some(4096), - enforce_machine_model: false, - enforce_cpu_model: false, - }, - measurement: MeasurementConfig { - sample_count: 5, - confirmation_sample_count: 4, - discarded_launch_count: 1, - warmup_iterations: 2, - }, - thresholds, - provenance: json!({}), - cases: BTreeMap::from([("cel/one".into(), expected.clone()), ("rego/two".into(), expected)]), - } - } - - fn scaled_summaries(scale: f64) -> BTreeMap { - synthetic_baseline() - .cases - .into_iter() - .map(|(case, expected)| { - ( - case, - Summary { - init_and_first_ms: expected.init_and_first_ms * scale, - warm_per_call_ms: expected.warm_per_call_ms * scale, - peak_rss_bytes: expected.peak_rss_bytes * scale, - }, - ) - }) - .collect() - } - - #[test] - fn stable_performance_passes() { - let outcome = evaluate_performance(&synthetic_baseline(), &scaled_summaries(1.005)).expect("evaluation"); - assert!(outcome.failures.is_empty()); - assert!(outcome.confirmation_cases.is_empty()); - } - - #[test] - fn common_host_shift_is_normalized_and_broad_change_still_fails() { - let baseline = synthetic_baseline(); - let mut within_aggregate_band = scaled_summaries(1.0); - for summary in within_aggregate_band.values_mut() { - summary.init_and_first_ms *= 1.09; - summary.warm_per_call_ms *= 1.09; - } - let outcome = evaluate_performance(&baseline, &within_aggregate_band).expect("within aggregate band"); - assert!(outcome.failures.is_empty()); - assert!( - outcome - .metrics - .iter() - .filter(|evaluation| evaluation.metric != Metric::PeakRss.key()) - .all(|evaluation| (evaluation.normalized_ratio - 1.0).abs() < 1e-12) - ); - - let mut outside_aggregate_band = scaled_summaries(1.0); - for summary in outside_aggregate_band.values_mut() { - summary.init_and_first_ms *= 1.11; - summary.warm_per_call_ms *= 1.11; - } - let outcome = evaluate_performance(&baseline, &outside_aggregate_band).expect("outside aggregate band"); - assert!(outcome.failures.iter().any(|failure| failure.starts_with("aggregate Init + first"))); - assert!(outcome.failures.iter().any(|failure| failure.starts_with("aggregate Warm / call"))); - } - - #[test] - fn tightened_boundaries_are_two_sided_and_inclusive() { - assert_eq!(evaluation_status(1.15, 0.85, 1.15), EvaluationStatus::Pass); - assert_eq!(evaluation_status(0.85, 0.85, 1.15), EvaluationStatus::Pass); - assert_eq!(evaluation_status(1.151, 0.85, 1.15), EvaluationStatus::Regression); - assert_eq!(evaluation_status(0.849, 0.85, 1.15), EvaluationStatus::Improvement); - assert_eq!(evaluation_status(1.031, 0.97, 1.03), EvaluationStatus::Regression); - assert_eq!(evaluation_status(0.969, 0.97, 1.03), EvaluationStatus::Improvement); - } - - #[test] - fn expected_timing_precision_is_one_microsecond() { - assert_eq!(round_milliseconds(123.456_789), 123.457); - } - - #[test] - fn per_case_regression_and_improvement_fail() { - let baseline = synthetic_baseline(); - let mut regression = scaled_summaries(1.0); - regression.get_mut("cel/one").expect("case").warm_per_call_ms = 15.0; - let outcome = evaluate_performance(&baseline, ®ression).expect("regression evaluation"); - assert!(outcome.failures.iter().any(|failure| failure.contains("regressed"))); - assert!(outcome.confirmation_cases.contains("cel/one")); - - let mut improvement = scaled_summaries(1.0); - improvement.get_mut("rego/two").expect("case").init_and_first_ms = 60.0; - let outcome = evaluate_performance(&baseline, &improvement).expect("improvement evaluation"); - assert!(outcome.failures.iter().any(|failure| failure.contains("improved"))); - assert!(outcome.confirmation_cases.contains("rego/two")); - } - - #[test] - fn aggregate_regression_and_improvement_fail() { - let baseline = synthetic_baseline(); - let regression = evaluate_performance(&baseline, &scaled_summaries(1.30)).expect("regression"); - assert!(regression.aggregates.iter().any(|value| value.status == EvaluationStatus::Regression)); - assert_eq!(regression.confirmation_cases, baseline.cases.keys().cloned().collect()); - - let improvement = evaluate_performance(&baseline, &scaled_summaries(0.75)).expect("improvement"); - assert!(improvement.aggregates.iter().any(|value| value.status == EvaluationStatus::Improvement)); - assert_eq!(improvement.confirmation_cases, baseline.cases.keys().cloned().collect()); - } - - #[test] - fn median_rejects_an_outlier() { - assert_eq!(median(vec![1.0, 100.0, 3.0, 4.0, 5.0]).expect("median"), 4.0); - } - - #[test] - fn peak_rss_parses_linux_and_macos_output() { - assert_eq!(parse_peak_rss("Maximum resident set size (kbytes): 123").expect("linux"), 123 * 1024); - assert_eq!(parse_peak_rss("456 maximum resident set size").expect("macOS"), 456); - } - - #[test] - fn environment_mismatch_fails() { - let baseline = synthetic_baseline(); - let mut actual = baseline.environment.clone(); - actual.context = "github-actions".into(); - assert!(validate_environment(&baseline.environment, &actual).is_err()); - } - - #[test] - fn measurement_summary_uses_process_medians() { - fn measurement(value: f64) -> Measurement { - Measurement { - label: "one".into(), - engine: "cel".into(), - template_count: 1, - iterations: 1, - samples: 1, - schema_init_ms: value / 2.0, - engine_init_ms: value / 2.0, - init_total_ms: value, - first_validation: FirstValidation { - wall_ms: value, - internal_ms: value, - fingerprint: "same".into(), - status: json!("OK"), - }, - warm: WarmMeasurement { - total_ms: value, - per_call_total_ms: value, - wall_median_ms: value, - wall_p95_ms: value, - internal_median_ms: value, - model_median_ms: value, - schema_median_ms: value, - rule_median_ms: value, - finalize_median_ms: value, - }, - fingerprints: Vec::new(), - peak_rss_bytes: value as u64, - sample: 0, - gate_process_lifecycle: true, - } - } - let measurements = BTreeMap::from([( - "cel/one".into(), - vec![measurement(1.0), measurement(100.0), measurement(3.0), measurement(4.0), measurement(5.0)], - )]); - let summary = summarize_measurements(&measurements).expect("summary")["cel/one"]; - assert_eq!(summary.init_and_first_ms, 8.0); - assert_eq!(summary.warm_per_call_ms, 4.0); - assert_eq!(summary.peak_rss_bytes, 4.0); - } - - #[test] - fn checked_in_baselines_are_valid_and_cover_the_same_cases() { - let mut github_files: Vec = fs::read_dir(expected_directory()) - .expect("expected directory") - .map(|entry| entry.expect("expected entry").path()) - .filter(|path| { - path.file_name() - .and_then(|name| name.to_str()) - .is_some_and(|name| name.starts_with("github-ubuntu-x64-") && name.ends_with(".json")) - }) - .collect(); - github_files.sort(); - let github_baselines: Vec = - github_files.iter().map(|path| load_baseline(path).expect("GitHub baseline")).collect(); - assert!(github_baselines.len() >= 3); - let reference = github_baselines.first().expect("GitHub reference baseline"); - let macos = load_baseline(&expected_directory().join("local-macos-arm64.json")).expect("macOS baseline"); - assert_eq!(reference.cases.keys().collect::>(), macos.cases.keys().collect::>()); - assert_eq!(reference.cases.len(), 57); - fn assert_threshold( - threshold: &MetricThreshold, - regression: f64, - improvement: f64, - aggregate_regression: f64, - aggregate_improvement: f64, - ) { - assert_eq!(threshold.regression_factor, regression); - assert_eq!(threshold.improvement_factor, improvement); - assert_eq!(threshold.aggregate_regression_factor, aggregate_regression); - assert_eq!(threshold.aggregate_improvement_factor, aggregate_improvement); - } - for github in &github_baselines { - assert!(github.environment.enforce_cpu_model); - assert!(github.environment.cpu_model.is_some()); - assert_eq!(reference.cases.keys().collect::>(), github.cases.keys().collect::>()); - for (case, expected_reference) in &reference.cases { - let expected = &github.cases[case]; - assert_eq!(expected_reference.gated_metrics, expected.gated_metrics, "{case} gated metrics"); - assert_eq!( - expected_reference.aggregate_metrics, expected.aggregate_metrics, - "{case} aggregate metrics" - ); - } - assert_threshold(&github.thresholds.init_and_first_ms, 1.15, 0.85, 1.10, 0.90); - assert_threshold(&github.thresholds.warm_per_call_ms, 1.15, 0.85, 1.10, 0.90); - assert_threshold(&github.thresholds.peak_rss_bytes, 1.03, 0.97, 1.01, 0.99); - } - assert_threshold(&macos.thresholds.init_and_first_ms, 1.08, 0.92, 1.07, 0.93); - assert_threshold(&macos.thresholds.warm_per_call_ms, 1.08, 0.92, 1.06, 0.94); - assert_threshold(&macos.thresholds.peak_rss_bytes, 1.02, 0.98, 1.01, 0.99); - let fanout = &reference.cases["cel/security-cross-reference-fanout"]; - assert_eq!(fanout.gated_metrics, vec![Metric::PeakRss.key()]); - assert!(fanout.aggregate_metrics.contains(&Metric::InitAndFirst.key().to_string())); - assert!(fanout.aggregate_metrics.contains(&Metric::WarmPerCall.key().to_string())); - } - - #[test] - fn github_cpu_profiles_are_selected_and_enforced() { - let mut environment = synthetic_baseline().environment; - environment.context = "github-actions".into(); - environment.system = "Linux".into(); - environment.architecture = "x86_64".into(); - environment.cpu_model = Some("AMD EPYC 7763 64-Core Processor".into()); - assert!( - default_expected_file(&environment) - .expect("7763 profile") - .ends_with("github-ubuntu-x64-amd-epyc-7763.json") - ); - let baseline = load_baseline(&default_expected_file(&environment).expect("7763 path")).expect("7763 baseline"); - validate_environment(&baseline.environment, &environment).expect("matching CPU"); - - environment.cpu_model = Some("AMD EPYC 9V74 80-Core Processor".into()); - assert!( - default_expected_file(&environment) - .expect("9V74 profile") - .ends_with("github-ubuntu-x64-amd-epyc-9v74.json") - ); - assert!(validate_environment(&baseline.environment, &environment).is_err()); - - environment.cpu_model = Some("INTEL(R) XEON(R) PLATINUM 8573C".into()); - let intel_file = default_expected_file(&environment).expect("8573C profile"); - assert!(intel_file.ends_with("github-ubuntu-x64-intel-xeon-platinum-8573c.json")); - let intel_baseline = load_baseline(&intel_file).expect("8573C baseline"); - validate_environment(&intel_baseline.environment, &environment).expect("matching Intel CPU"); - - assert_eq!( - github_profile_name("Intel(R) Xeon(R) Platinum 8370C CPU @ 2.80GHz").expect("8370C profile"), - "github-ubuntu-x64-intel-xeon-platinum-8370c" - ); - assert_eq!( - github_profile_name("AMD EPYC 9V45 96-Core Processor").expect("9V45 profile"), - "github-ubuntu-x64-amd-epyc-9v45" - ); - assert!(default_profile("github-ubuntu-x64-amd-epyc-9v45").is_ok()); - - environment.cpu_model = Some("Future Cloud 1234 Processor".into()); - assert!( - default_expected_path(&environment) - .expect("future profile path") - .ends_with("github-ubuntu-x64-future-cloud-1234.json") - ); - assert!(default_expected_file(&environment).is_err()); - } -} diff --git a/src/performance-harness/src/comparison.rs b/src/performance-harness/src/comparison.rs new file mode 100644 index 00000000..b3d9c3b5 --- /dev/null +++ b/src/performance-harness/src/comparison.rs @@ -0,0 +1,685 @@ +use crate::measurement::{self, ENGINES, Environment, Metric, Workload}; +use crate::worker::Measurement; +use serde::Serialize; +use std::collections::{BTreeMap, BTreeSet}; +use std::fs; +use std::path::{Path, PathBuf}; + +const SCHEMA_VERSION: u32 = 1; +const PAIR_COUNT: usize = 5; +const CONFIRMATION_PAIR_COUNT: usize = 4; +const DISCARDED_LAUNCH_COUNT: usize = 1; +const WARMUP_ITERATIONS: usize = 2; +pub(crate) const DEFAULT_TITLE: &str = "Performance comparison against base"; + +/// Head/base ratio limits. Both sides run on the same host, so the limits only absorb residual launch noise and the +/// natural growth of bundled schema data rather than differences between runner hardware. +#[derive(Debug, Clone, Copy, Serialize)] +#[serde(rename_all = "camelCase")] +struct RegressionLimit { + case_factor: f64, + aggregate_factor: f64, +} + +fn regression_limit(metric: Metric) -> RegressionLimit { + match metric { + Metric::InitAndFirst | Metric::WarmPerCall => RegressionLimit { case_factor: 1.20, aggregate_factor: 1.08 }, + Metric::PeakRss => RegressionLimit { case_factor: 1.10, aggregate_factor: 1.05 }, + } +} + +/// Below these base values, process launch and allocator noise dominate the metric, so it is reported but not gated. +fn stability_floor(metric: Metric) -> f64 { + match metric { + Metric::InitAndFirst => 5.0, + Metric::WarmPerCall => 0.30, + Metric::PeakRss => 16.0 * 1024.0 * 1024.0, + } +} + +#[derive(Debug, Clone, Copy, Serialize, PartialEq, Eq)] +#[serde(rename_all = "lowercase")] +enum EvaluationStatus { + Info, + Pass, + Regression, + Improvement, +} + +fn median(mut values: Vec) -> Result { + if values.is_empty() || values.iter().any(|value| !value.is_finite()) { + return Err("median requires finite samples".into()); + } + values.sort_by(f64::total_cmp); + let middle = values.len() / 2; + if values.len().is_multiple_of(2) { Ok((values[middle - 1] + values[middle]) / 2.0) } else { Ok(values[middle]) } +} + +fn geometric_mean(values: &[f64]) -> Result { + if values.is_empty() || values.iter().any(|value| !value.is_finite() || *value <= 0.0) { + return Err("geometric mean requires finite positive ratios".into()); + } + Ok((values.iter().map(|value| value.ln()).sum::() / values.len() as f64).exp()) +} + +fn write_json(path: &Path, value: &impl Serialize) -> Result<(), String> { + if let Some(parent) = path.parent() { + fs::create_dir_all(parent).map_err(|error| format!("could not create {}: {error}", parent.display()))?; + } + let bytes = serde_json::to_vec_pretty(value).map_err(|error| format!("JSON serialization failed: {error}"))?; + fs::write(path, [bytes, b"\n".to_vec()].concat()) + .map_err(|error| format!("could not write {}: {error}", path.display())) +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Side { + Base, + Head, +} + +#[derive(Debug, Default, Serialize)] +struct PairedMeasurements { + base: Vec, + head: Vec, +} + +impl PairedMeasurements { + fn side_mut(&mut self, side: Side) -> &mut Vec { + match side { + Side::Base => &mut self.base, + Side::Head => &mut self.head, + } + } + + fn launch_pairs(&self, metric: Metric) -> Vec { + self.base + .iter() + .zip(&self.head) + .map(|(base, head)| LaunchPair { + base: metric.measurement_value(base), + head: metric.measurement_value(head), + }) + .collect() + } +} + +#[derive(Debug, Clone, Copy, PartialEq)] +struct LaunchPair { + base: f64, + head: f64, +} + +#[derive(Debug, Clone)] +struct CaseInput { + pairs: BTreeMap>, + aggregate_metrics: BTreeSet, + gated_metrics: BTreeSet, +} + +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "camelCase")] +struct CaseComparison { + case: String, + metric: &'static str, + base_median: f64, + head_median: f64, + ratio: f64, + limit: f64, + pair_count: usize, + status: EvaluationStatus, +} + +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "camelCase")] +struct AggregateComparison { + metric: &'static str, + ratio: f64, + limit: f64, + case_count: usize, + status: EvaluationStatus, +} + +#[derive(Debug)] +struct ComparisonOutcome { + cases: Vec, + aggregates: Vec, + failures: Vec, + confirmation_cases: BTreeSet, +} + +#[derive(Debug, Serialize)] +#[serde(rename_all = "camelCase")] +struct ComparisonResults<'a> { + schema_version: u32, + base_revision: &'a str, + head_revision: &'a str, + environment: &'a Environment, + limits: BTreeMap<&'static str, RegressionLimit>, + case_comparisons: &'a [CaseComparison], + aggregate_comparisons: &'a [AggregateComparison], + diagnostic_changes: &'a [String], + failures: &'a [String], + measurements: &'a BTreeMap, +} + +/// The ratio of one head launch to the base launch measured beside it; the median of these ratios cancels host speed +/// changes that affect both launches of a pair. +fn median_paired_ratio(pairs: &[LaunchPair]) -> Result { + if pairs.iter().any(|pair| !(pair.base.is_finite() && pair.base > 0.0 && pair.head.is_finite() && pair.head > 0.0)) + { + return Err("paired comparison requires finite positive measurements".into()); + } + median(pairs.iter().map(|pair| pair.head / pair.base).collect()) +} + +fn classify(ratio: f64, limit: f64) -> EvaluationStatus { + if ratio > limit { + EvaluationStatus::Regression + } else if ratio < 1.0 / limit { + EvaluationStatus::Improvement + } else { + EvaluationStatus::Pass + } +} + +fn evaluate_comparison(inputs: &BTreeMap) -> Result { + let mut cases = Vec::new(); + let mut failures = Vec::new(); + let mut confirmation_cases = BTreeSet::new(); + let mut aggregate_ratios: BTreeMap> = BTreeMap::new(); + for (case, input) in inputs { + for metric in Metric::ALL { + let pairs = input.pairs.get(&metric).ok_or_else(|| format!("{case} has no {} pairs", metric.key()))?; + let ratio = median_paired_ratio(pairs)?; + let limit = regression_limit(metric).case_factor; + let status = + if input.gated_metrics.contains(&metric) { classify(ratio, limit) } else { EvaluationStatus::Info }; + if input.aggregate_metrics.contains(&metric) { + aggregate_ratios.entry(metric).or_default().push((case.clone(), ratio)); + } + let base_median = median(pairs.iter().map(|pair| pair.base).collect())?; + let head_median = median(pairs.iter().map(|pair| pair.head).collect())?; + if status == EvaluationStatus::Regression { + failures.push(format!( + "{case}: {} regressed to {ratio:.3}x base ({} vs {}; limit {limit:.2}x)", + metric.label(), + metric.format_value(head_median), + metric.format_value(base_median) + )); + confirmation_cases.insert(case.clone()); + } + cases.push(CaseComparison { + case: case.clone(), + metric: metric.key(), + base_median, + head_median, + ratio, + limit, + pair_count: pairs.len(), + status, + }); + } + } + + let mut aggregates = Vec::new(); + for (metric, ratios) in aggregate_ratios { + let ratio = geometric_mean(&ratios.iter().map(|(_, ratio)| *ratio).collect::>())?; + let limit = regression_limit(metric).aggregate_factor; + let status = classify(ratio, limit); + if status == EvaluationStatus::Regression { + failures.push(format!( + "aggregate {} regressed to {ratio:.3}x base across {} cases (limit {limit:.2}x)", + metric.label(), + ratios.len() + )); + confirmation_cases.extend(ratios.iter().map(|(case, _)| case.clone())); + } + aggregates.push(AggregateComparison { metric: metric.key(), ratio, limit, case_count: ratios.len(), status }); + } + Ok(ComparisonOutcome { cases, aggregates, failures, confirmation_cases }) +} + +fn case_inputs( + measurements: &BTreeMap, + workloads: &[Workload], +) -> Result, String> { + let workloads_by_name: BTreeMap<_, _> = + workloads.iter().map(|workload| (workload.name.as_str(), workload)).collect(); + measurements + .iter() + .map(|(case, paired)| { + let (_, workload_name) = case.split_once('/').ok_or_else(|| format!("invalid case name {case}"))?; + let workload = + workloads_by_name.get(workload_name).ok_or_else(|| format!("missing workload {workload_name}"))?; + let pairs: BTreeMap<_, _> = + Metric::ALL.into_iter().map(|metric| (metric, paired.launch_pairs(metric))).collect(); + let base_medians = pairs + .iter() + .map(|(metric, pairs)| Ok((*metric, median(pairs.iter().map(|pair| pair.base).collect())?))) + .collect::, String>>()?; + let is_stable = + |metric: Metric| base_medians.get(&metric).is_some_and(|median| *median >= stability_floor(metric)); + Ok(( + case.clone(), + CaseInput { + aggregate_metrics: workload.aggregate_metrics(is_stable).into_iter().collect(), + gated_metrics: workload.gated_metrics(is_stable).into_iter().collect(), + pairs, + }, + )) + }) + .collect() +} + +struct Executables<'a> { + base: &'a Path, + head: &'a Path, +} + +impl Executables<'_> { + fn path(&self, side: Side) -> &Path { + match side { + Side::Base => self.base, + Side::Head => self.head, + } + } +} + +/// Alternates which side launches first in successive pairs so that monotonic host drift during a pair penalizes base +/// and head equally. +fn launch_order(pair_index: usize) -> [Side; 2] { + if pair_index.is_multiple_of(2) { [Side::Base, Side::Head] } else { [Side::Head, Side::Base] } +} + +fn collect_pairs( + executables: &Executables, + workloads: &[Workload], + pair_count: usize, + measurements: &mut BTreeMap, + selected_cases: Option<&BTreeSet>, +) -> Result<(), String> { + for engine in ENGINES { + for workload in workloads { + let case = format!("{engine}/{}", workload.name); + if selected_cases.is_some_and(|selected| !selected.contains(&case)) { + continue; + } + let paired = measurements.entry(case.clone()).or_default(); + if paired.base.is_empty() { + for discarded in 0..DISCARDED_LAUNCH_COUNT { + eprintln!("Discarding launch {}/{DISCARDED_LAUNCH_COUNT} for {case}", discarded + 1); + for side in launch_order(discarded) { + let sample = -((discarded + 1) as i32); + measurement::run_measurement( + executables.path(side), + engine, + workload, + WARMUP_ITERATIONS, + sample, + )?; + } + } + } + for _ in 0..pair_count { + let pair_index = paired.base.len(); + eprintln!("Measuring {case} pair {}", pair_index + 1); + for side in launch_order(pair_index) { + let measurement = measurement::run_measurement( + executables.path(side), + engine, + workload, + WARMUP_ITERATIONS, + pair_index as i32, + ) + .map_err(|error| format!("{side:?} measurement failed: {error}"))?; + paired.side_mut(side).push(measurement); + } + } + } + } + Ok(()) +} + +fn side_measurements( + measurements: &BTreeMap, + side: Side, +) -> BTreeMap> { + measurements + .iter() + .map(|(case, paired)| { + let samples = match side { + Side::Base => &paired.base, + Side::Head => &paired.head, + }; + (case.clone(), samples.clone()) + }) + .collect() +} + +fn diagnostic_stability_failures(measurements: &BTreeMap) -> Result, String> { + let mut failures = Vec::new(); + for side in [Side::Base, Side::Head] { + let label = format!("{side:?}").to_ascii_lowercase(); + failures.extend( + measurement::diagnostic_failures(&side_measurements(measurements, side))? + .into_iter() + .map(|failure| format!("{label} {failure}")), + ); + } + Ok(failures) +} + +/// Cases whose diagnostics differ between base and head. A rule or schema-data change is expected to change them, so +/// this is reported for context and never fails the comparison. +fn diagnostic_changes(measurements: &BTreeMap) -> Result, String> { + let mut changes = Vec::new(); + for (case, paired) in measurements { + let (Some(base), Some(head)) = (paired.base.first(), paired.head.first()) else { + return Err(format!("no paired samples collected for {case}")); + }; + if measurement::diagnostic_signature(base)? != measurement::diagnostic_signature(head)? { + changes.push(case.clone()); + } + } + Ok(changes) +} + +fn short_revision(revision: &str) -> &str { + &revision[..revision.len().min(12)] +} + +fn render_markdown( + title: &str, + base_revision: &str, + head_revision: &str, + environment: &Environment, + outcome: &ComparisonOutcome, + diagnostic_changes: &[String], + failures: &[String], +) -> Result { + let mut by_case: BTreeMap<&str, BTreeMap<&str, &CaseComparison>> = BTreeMap::new(); + for comparison in &outcome.cases { + by_case.entry(&comparison.case).or_default().insert(comparison.metric, comparison); + } + let pair_total = outcome.cases.iter().map(|comparison| comparison.pair_count).max().unwrap_or(0); + let mut lines = vec![ + format!("# {title}"), + String::new(), + format!("Base: `{}` ", short_revision(base_revision)), + format!("Head: `{}` ", short_revision(head_revision)), + format!("CPU: `{}` ", environment.cpu_model().unwrap_or("unknown")), + format!("Interleaved launch pairs per case: up to `{pair_total}`"), + String::new(), + "Ratios are head / base: the median of per-pair ratios from base and head launches run back to back on the \ + same host (pinned to one CPU where `taskset` is available). Above 1 is slower or larger. `(info)` metrics are below their stability floor and are not \ + gated per case." + .to_string(), + String::new(), + "| Engine / workload | Init + first | Warm / call | Peak RSS | Status |".to_string(), + "|---|---:|---:|---:|---|".to_string(), + ]; + for (case, comparisons) in by_case { + let mut cells = Vec::new(); + let mut regressed = false; + for metric in Metric::ALL { + let comparison = comparisons + .get(metric.key()) + .ok_or_else(|| format!("missing {} comparison for {case}", metric.key()))?; + let suffix = if comparison.status == EvaluationStatus::Info { " (info)" } else { "" }; + cells.push(format!("{:.3}x{suffix}", comparison.ratio)); + regressed |= comparison.status == EvaluationStatus::Regression; + } + lines.push(format!("| {case} | {} | {} |", cells.join(" | "), if regressed { "FAIL" } else { "pass" })); + } + lines.extend([ + String::new(), + "## Aggregate (geometric mean of per-case ratios)".into(), + String::new(), + "| Metric | Ratio | Limit | Cases | Status |".into(), + "|---|---:|---:|---:|---|".into(), + ]); + for aggregate in &outcome.aggregates { + let label = Metric::ALL + .into_iter() + .find(|metric| metric.key() == aggregate.metric) + .map(Metric::label) + .unwrap_or(aggregate.metric); + lines.push(format!( + "| {label} | {:.3}x | {:.2}x | {} | {:?} |", + aggregate.ratio, aggregate.limit, aggregate.case_count, aggregate.status + )); + } + if !diagnostic_changes.is_empty() { + lines.extend([ + String::new(), + format!( + "Diagnostics differ from base in {} case(s); this is expected when rules or schema data change and \ + is not gated.", + diagnostic_changes.len() + ), + ]); + } + lines.extend([String::new(), "## Result".into(), String::new()]); + if failures.is_empty() { + lines.push("✅ No performance regression beyond the head/base limits.".into()); + } else { + lines.extend(failures.iter().map(|failure| format!("* ❌ {failure}"))); + lines.extend([ + String::new(), + "Each reported regression was confirmed with additional interleaved pairs. Fix the regression, or record \ + why it is intentional in the review; later changes are compared against the merged revision." + .into(), + ]); + } + Ok(lines.join("\n") + "\n") +} + +pub fn run_compare( + base_executable: &Path, + base_revision: &str, + title: &str, + output_dir: &Path, +) -> Result { + let (head_executable, workloads) = measurement::prepare_run(output_dir)?; + let base_executable = canonical_executable(base_executable)?; + if base_executable == canonical_executable(&head_executable)? { + return Err("the base executable must be a separate build from the running head harness".into()); + } + let executables = Executables { base: &base_executable, head: &head_executable }; + let mut measurements = BTreeMap::new(); + collect_pairs(&executables, &workloads, PAIR_COUNT, &mut measurements, None)?; + let mut stability_failures = diagnostic_stability_failures(&measurements)?; + let mut outcome = evaluate_comparison(&case_inputs(&measurements, &workloads)?)?; + if stability_failures.is_empty() && !outcome.failures.is_empty() { + eprintln!( + "Confirming apparent regression in {} case(s) with {CONFIRMATION_PAIR_COUNT} additional pairs", + outcome.confirmation_cases.len() + ); + collect_pairs( + &executables, + &workloads, + CONFIRMATION_PAIR_COUNT, + &mut measurements, + Some(&outcome.confirmation_cases), + )?; + stability_failures = diagnostic_stability_failures(&measurements)?; + outcome = evaluate_comparison(&case_inputs(&measurements, &workloads)?)?; + } + let mut failures = stability_failures; + failures.extend(outcome.failures.clone()); + let changes = diagnostic_changes(&measurements)?; + let environment = measurement::detect_environment(); + let head_revision = measurement::command_output("git", &["rev-parse", "HEAD"], &measurement::project_root())?; + let results = ComparisonResults { + schema_version: SCHEMA_VERSION, + base_revision, + head_revision: &head_revision, + environment: &environment, + limits: Metric::ALL.into_iter().map(|metric| (metric.key(), regression_limit(metric))).collect(), + case_comparisons: &outcome.cases, + aggregate_comparisons: &outcome.aggregates, + diagnostic_changes: &changes, + failures: &failures, + measurements: &measurements, + }; + write_json(&output_dir.join("performance-comparison.json"), &results)?; + let markdown = render_markdown(title, base_revision, &head_revision, &environment, &outcome, &changes, &failures)?; + fs::write(output_dir.join("performance-comparison.md"), &markdown) + .map_err(|error| format!("could not write performance comparison markdown: {error}"))?; + print!("{markdown}"); + for failure in &failures { + println!("::error::{failure}"); + } + Ok(failures.is_empty()) +} + +fn canonical_executable(path: &Path) -> Result { + fs::canonicalize(path).map_err(|error| format!("executable {} is not accessible: {error}", path.display())) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn every_metric() -> BTreeSet { + Metric::ALL.into_iter().collect() + } + + fn case_input(base: [f64; 3], head_scale: [f64; 3], pair_count: usize) -> CaseInput { + let pairs = Metric::ALL + .into_iter() + .zip(base.into_iter().zip(head_scale)) + .map(|(metric, (base, scale))| { + let pairs = (0..pair_count) + .map(|index| { + let host_drift = 1.0 + index as f64 * 0.05; + LaunchPair { base: base * host_drift, head: base * scale * host_drift } + }) + .collect(); + (metric, pairs) + }) + .collect(); + CaseInput { pairs, aggregate_metrics: every_metric(), gated_metrics: every_metric() } + } + + fn uniform_inputs(head_scale: [f64; 3]) -> BTreeMap { + ["cel/one", "rego/two", "composite/three"] + .into_iter() + .map(|case| (case.to_string(), case_input([100.0, 10.0, 100_000_000.0], head_scale, 5))) + .collect() + } + + #[test] + fn identical_revisions_pass_despite_host_drift() { + let outcome = evaluate_comparison(&uniform_inputs([1.0, 1.0, 1.0])).expect("evaluation"); + assert!(outcome.failures.is_empty()); + assert!(outcome.cases.iter().all(|comparison| (comparison.ratio - 1.0).abs() < 1e-12)); + assert!(outcome.aggregates.iter().all(|aggregate| aggregate.status == EvaluationStatus::Pass)); + } + + #[test] + fn schema_data_growth_within_limits_passes() { + let outcome = evaluate_comparison(&uniform_inputs([1.06, 1.05, 1.04])).expect("evaluation"); + assert!(outcome.failures.is_empty(), "{:?}", outcome.failures); + } + + #[test] + fn broad_regression_fails_the_aggregate_and_confirms_every_case() { + let inputs = uniform_inputs([1.10, 1.0, 1.0]); + let outcome = evaluate_comparison(&inputs).expect("evaluation"); + assert!(outcome.failures.iter().any(|failure| failure.starts_with("aggregate Init + first regressed"))); + assert_eq!(outcome.confirmation_cases, inputs.keys().cloned().collect()); + } + + #[test] + fn single_case_regression_fails_only_that_case() { + let mut inputs: BTreeMap = (0..20) + .map(|index| (format!("cel/case-{index}"), case_input([100.0, 10.0, 100_000_000.0], [1.0, 1.0, 1.0], 5))) + .collect(); + inputs.insert("cel/one".into(), case_input([100.0, 10.0, 100_000_000.0], [1.0, 1.30, 1.0], 5)); + let outcome = evaluate_comparison(&inputs).expect("evaluation"); + assert_eq!(outcome.confirmation_cases, BTreeSet::from(["cel/one".to_string()])); + assert!(outcome.failures.iter().any(|failure| failure.starts_with("cel/one: Warm / call regressed"))); + assert!(!outcome.failures.iter().any(|failure| failure.starts_with("aggregate Warm"))); + } + + #[test] + fn memory_limits_are_tighter_than_timing_limits() { + let mut inputs = uniform_inputs([1.0, 1.0, 1.0]); + inputs.insert("rego/two".into(), case_input([100.0, 10.0, 100_000_000.0], [1.0, 1.0, 1.12], 5)); + let outcome = evaluate_comparison(&inputs).expect("evaluation"); + assert!(outcome.failures.iter().any(|failure| failure.starts_with("rego/two: Peak RSS regressed"))); + } + + #[test] + fn improvements_are_reported_but_never_fail() { + let outcome = evaluate_comparison(&uniform_inputs([0.5, 0.5, 0.5])).expect("evaluation"); + assert!(outcome.failures.is_empty()); + assert!(outcome.aggregates.iter().all(|aggregate| aggregate.status == EvaluationStatus::Improvement)); + } + + #[test] + fn ungated_metrics_are_informational_and_excluded_from_the_aggregate() { + let mut inputs = uniform_inputs([1.0, 1.0, 1.0]); + let mut noisy = case_input([100.0, 0.1, 100_000_000.0], [1.0, 3.0, 1.0], 5); + noisy.gated_metrics.remove(&Metric::WarmPerCall); + noisy.aggregate_metrics.remove(&Metric::WarmPerCall); + inputs.insert("cel/tiny".into(), noisy); + let outcome = evaluate_comparison(&inputs).expect("evaluation"); + assert!(outcome.failures.is_empty()); + let tiny_warm = outcome + .cases + .iter() + .find(|comparison| comparison.case == "cel/tiny" && comparison.metric == Metric::WarmPerCall.key()) + .expect("tiny warm comparison"); + assert_eq!(tiny_warm.status, EvaluationStatus::Info); + let warm_aggregate = + outcome.aggregates.iter().find(|aggregate| aggregate.metric == Metric::WarmPerCall.key()).expect("warm"); + assert_eq!(warm_aggregate.case_count, 3); + } + + #[test] + fn a_single_outlier_pair_does_not_fail() { + let mut input = case_input([100.0, 10.0, 100_000_000.0], [1.0, 1.0, 1.0], 5); + let warm = input.pairs.get_mut(&Metric::WarmPerCall).expect("warm pairs"); + warm[2].head *= 2.0; + let outcome = evaluate_comparison(&BTreeMap::from([("cel/one".to_string(), input)])).expect("evaluation"); + assert!(outcome.failures.is_empty()); + } + + #[test] + fn non_positive_measurements_are_rejected() { + let mut input = case_input([100.0, 10.0, 100_000_000.0], [1.0, 1.0, 1.0], 5); + input.pairs.get_mut(&Metric::PeakRss).expect("rss pairs")[0].base = 0.0; + assert!(evaluate_comparison(&BTreeMap::from([("cel/one".to_string(), input)])).is_err()); + } + + /// The workflow reads the whole file as the tag name and fetches `refs/tags/`, so the file must contain + /// nothing but one release tag in the form the release workflow creates. + #[test] + fn drift_anchor_is_one_release_tag() { + let anchor = fs::read_to_string(Path::new(env!("CARGO_MANIFEST_DIR")).join("drift-anchor.txt")) + .expect("drift anchor file"); + let tag = anchor.trim_end_matches('\n'); + assert!(!tag.contains(char::is_whitespace), "one tag and nothing else: {anchor:?}"); + let (version, prerelease) = tag.split_once('-').unwrap_or((tag, "")); + assert!(prerelease.is_empty() || prerelease == "beta", "unexpected release kind in {tag}"); + let components: Vec<&str> = version.split('.').collect(); + assert_eq!(components.len(), 3, "release tags are MAJOR.MINOR.PATCH: {tag}"); + assert!(components.iter().all(|component| component.parse::().is_ok()), "{tag}"); + } + + #[test] + fn median_rejects_an_outlier_and_averages_even_counts() { + assert_eq!(median(vec![1.0, 100.0, 3.0, 4.0, 5.0]).expect("odd"), 4.0); + assert_eq!(median(vec![1.0, 2.0, 3.0, 4.0]).expect("even"), 2.5); + assert!(median(Vec::new()).is_err()); + } + + #[test] + fn launch_order_alternates() { + assert_eq!(launch_order(0), [Side::Base, Side::Head]); + assert_eq!(launch_order(1), [Side::Head, Side::Base]); + assert_eq!(launch_order(2), [Side::Base, Side::Head]); + } +} diff --git a/src/performance-harness/src/main.rs b/src/performance-harness/src/main.rs index 7273d940..8f96c89e 100644 --- a/src/performance-harness/src/main.rs +++ b/src/performance-harness/src/main.rs @@ -1,43 +1,44 @@ -mod baseline; +mod comparison; +mod measurement; mod worker; use std::env; -use std::path::{Path, PathBuf}; - -fn project_root() -> PathBuf { - PathBuf::from(env!("CARGO_MANIFEST_DIR")) - .parent() - .and_then(Path::parent) - .map(Path::to_path_buf) - .unwrap_or_else(|| PathBuf::from(".")) -} +use std::path::PathBuf; fn usage() -> &'static str { - "usage:\n performance-harness measure