From 20dcfaea55c8fd9f82f299b8ce4c79942b369f85 Mon Sep 17 00:00:00 2001 From: Alan Szmyt Date: Fri, 25 Sep 2026 20:56:16 -0400 Subject: [PATCH] test(flow): prove provider acceptance scenarios Add the Flow #30 compatibility, artifact, provider-failure, and privacy matrix with deterministic receipts and PR coverage evidence. Roadmap-Step: FLO-Q03 --- .github/workflows/ci.yml | 15 +- Cargo.toml | 5 + README.md | 1 + ROADMAP.md | 8 + docs/integrations/acceptance-scenarios.md | 93 ++ docs/integrations/hermetic-provider-kit.md | 4 + docs/integrations/scenario-fixtures.md | 8 + tests/fixtures/acceptance-scenarios.v1.json | 995 ++++++++++++++++++++ tests/fixtures/privacy-provider.sh | 10 + tests/hermetic_provider_kit.rs | 3 + tests/scenario_matrix/artifacts.rs | 145 +++ tests/scenario_matrix/contracts.rs | 120 +++ tests/scenario_matrix/mod.rs | 523 ++++++++++ tests/scenario_matrix/privacy.rs | 83 ++ tests/scenario_matrix/resolution.rs | 152 +++ tools/run_acceptance_scenarios.py | 128 +++ tools/test_acceptance_report.py | 64 ++ 17 files changed, 2355 insertions(+), 2 deletions(-) create mode 100644 docs/integrations/acceptance-scenarios.md create mode 100644 tests/fixtures/acceptance-scenarios.v1.json create mode 100644 tests/fixtures/privacy-provider.sh create mode 100644 tests/scenario_matrix/artifacts.rs create mode 100644 tests/scenario_matrix/contracts.rs create mode 100644 tests/scenario_matrix/mod.rs create mode 100644 tests/scenario_matrix/privacy.rs create mode 100644 tests/scenario_matrix/resolution.rs create mode 100644 tools/run_acceptance_scenarios.py create mode 100644 tools/test_acceptance_report.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5688982..e558688 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -43,8 +43,16 @@ jobs: - name: Lint all targets run: cargo clippy --all-targets --all-features --locked -- -D warnings - - name: Test all targets - run: cargo test --all-targets --locked + - name: Test all targets and verify acceptance coverage + run: python3 tools/run_acceptance_scenarios.py --all-targets + + - name: Retain normalized acceptance evidence + uses: actions/upload-artifact@v4 + with: + name: acceptance-scenarios-${{ matrix.toolchain }} + path: target/acceptance-scenarios.v1.report.json + if-no-files-found: error + retention-days: 7 - name: Test documentation run: cargo test --doc --locked @@ -82,6 +90,9 @@ jobs: - name: Check deterministic scenario sources run: python3 tools/generate_scenario_sources.py --check + - name: Reject incomplete acceptance reports + run: python3 tools/test_acceptance_report.py + - name: Validate architecture specifications run: python3 .agents/specs/validate-specs.py diff --git a/Cargo.toml b/Cargo.toml index ad2b60b..2c2b761 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -26,6 +26,11 @@ nix = { version = "0.30.1", default-features = false, features = ["process", "si [dev-dependencies] +# Conformance tests copy and hash the exact executable repeatedly. Debug symbols +# are not part of the contract; omit them to keep the PR fixture budget bounded. +[profile.test] +debug = 0 + [lints.rust] unsafe_code = "forbid" diff --git a/README.md b/README.md index abe5fc2..df55b0c 100644 --- a/README.md +++ b/README.md @@ -143,6 +143,7 @@ or a sandbox. - [Process authority and isolation](docs/integrations/authority-isolation.md) - [Artifact binding contract](docs/integrations/artifact-bindings.md) - [Scenario fixture contract](docs/integrations/scenario-fixtures.md) +- [Executable acceptance matrix](docs/integrations/acceptance-scenarios.md) - [Hermetic provider kit](docs/integrations/hermetic-provider-kit.md) - [Versioned contracts](contracts/README.md) - [Roadmap](ROADMAP.md) diff --git a/ROADMAP.md b/ROADMAP.md index 7707e3b..cb66b0a 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -35,6 +35,14 @@ PR #60 merged on 2026-09-24 as completing the hermetic provider-kit closeout. The exact next Flow checkpoint is [#30](https://github.com/egohygiene/flow/issues/30). +The #30 review candidate adds the +[executable acceptance matrix](docs/integrations/acceptance-scenarios.md): +versioned deterministic recipes, exact typed outcomes, two fresh-root receipts +per case, PR budgets, and machine-readable coverage/gaps. Its scope ends at +single-execution acceptance and refusal. After this candidate merges and the +default-branch gate passes, #49 is next. FLO-Q03 remains active until the later +real released-provider adapters satisfy its two-adapter exit criterion. + ### Central Flow chain 1. [#30](https://github.com/egohygiene/flow/issues/30) — prove compatibility, diff --git a/docs/integrations/acceptance-scenarios.md b/docs/integrations/acceptance-scenarios.md new file mode 100644 index 0000000..18cae11 --- /dev/null +++ b/docs/integrations/acceptance-scenarios.md @@ -0,0 +1,93 @@ +# Acceptance scenario matrix + +## Design and scope + +Flow #30 proves one provider execution boundary using the closed scenario +contract and the finalized hermetic provider kit. Test recipes own fault +injection; production code continues to own resolution, process preflight, +transcript validation, filesystem observation, and final artifact acceptance. +No new scheduler, provider algorithm, or recovery state is introduced. + +The versioned, test-owned catalog in +`tests/fixtures/acceptance-scenarios.v1.json` assigns stable scenario IDs, +deterministic recipes, expected typed outcomes, PR-tier budgets, and explicit +coverage gaps. Its executor lives under `tests/scenario_matrix/` and reuses the +provider kit's exact package finalization and execution helpers. Catalog rows +are conformance recipes, not public runtime plans or authorization. + +Each row must produce exactly one normalized receipt from an actual public API +outcome. An unexpected error, missing row, duplicate row, or expected-outcome +mismatch fails the suite. Raw error strings and provider-authored free text +are excluded from receipts. Recovery instructions name the failed boundary. +The harness compares receipts across fresh roots and checks canaries before +printing portable JSON evidence. + +## Boundaries under test + +- Resolution: compatibility, missing or unavailable observations, stale and + future versions, exact pins, deterministic conflict, and pre-invocation + fallback policy. +- Contract and preflight: schema rejection, identity, configuration, + authorization, executable integrity, and malformed provider evidence. +- Artifact observation and acceptance: complete, partial, missing, extra, + changed, contradictory, corrupt, escaped, linked, and incorrectly typed + outputs; exit zero cannot authorize promotion. +- Evidence: distinct empty, unavailable, incomplete, unsupported, invalid, + and failed states; bounded operational evidence and portable privacy. + +## Running and interpreting the matrix + +Run after populating Cargo's locked dependency cache: + +```console +python3 tools/run_acceptance_scenarios.py +python3 tools/run_acceptance_scenarios.py --all-targets +``` + +The driver uses `cargo test --locked --offline` and writes +`target/acceptance-scenarios.v1.report.json`. It removes an older report before +starting, fails on any failed Rust test or missing/duplicate/unexpected receipt, +and records the catalog digest, tested source-file digests, Cargo version, +coverage counts, residual gaps, and actual normalized receipts. Failure logs +stay local; CI uploads only the successful portable report. The source digest +is checked before and after execution so an edited source tree cannot be +mistaken for the tested one. + +The PR catalog budgets 30 seconds per execution, two executions per case, +16 KiB per receipt, four immediate output entries, and 1 MiB of generated +regular-file output. The fixed recipes create only files and empty directories; +this output-count budget is not a general recursive filesystem quota. The +driver has a 600-second Cargo wait timeout and a 4 MiB post-capture test-log +acceptance limit; it is not a general descendant-process supervisor. Provider +stdout/stderr and process deadlines remain enforced by the existing locked kit +limits. Budget measurements exclude compilation from the per-case time, but +include compilation in the driver timeout. Memory, whole-filesystem limits, +and kernel network isolation remain explicit gaps. + +The test profile omits debug symbols because the exact provider executable is +copied and hashed repeatedly. This keeps generated package size and PR time +bounded without bypassing any package or executable digest check. Each receipt +still names the actual build-specific bytes. The `argv-environment` recipe also +uses a tiny first-party shell probe; its exact observed subjects are retained +in that receipt's `probe_subjects` field. + +`observed-empty` means Flow actually observed an empty directory; its receipt +does not claim an accepted provider execution. Resolution success similarly +does not claim execution or artifact acceptance. Only receipts with +`accepted: true` crossed `accept_artifacts` successfully. Transcript mutations +start from actual kit execution and re-enter the public host-neutral transcript +validator; the receipt identifies the validator outcome, not a second launch. + +## Claim limits + +The kit is synthetic. Real released providers, native format validators, +cryptographic publisher authentication, OS sandboxing, atomic filesystem +snapshots, descriptor-bound launch, and descendant containment are not proven. +Flow #49 owns durable state; #31 owns retry and resume. The scenario catalog +does not expand those checkpoints or claim that two synthetic fixtures are +two real provider adapters. + +Operational error objects and rejected provider evidence remain host-local +diagnostics and may contain private values. Only the explicitly allowlisted +normalized receipts are portable. Privacy canaries do not establish a generic +redactor for arbitrary provider-authored messages. diff --git a/docs/integrations/hermetic-provider-kit.md b/docs/integrations/hermetic-provider-kit.md index 7d7c7cc..be19878 100644 --- a/docs/integrations/hermetic-provider-kit.md +++ b/docs/integrations/hermetic-provider-kit.md @@ -1,5 +1,9 @@ # Hermetic orchestration provider kit +The issue #30 [acceptance matrix](acceptance-scenarios.md) consumes this kit +through test-owned recipes. It adds executable coverage receipts and explicit +gaps while retaining the package, process, and artifact boundaries below. + ## Purpose and checkpoint boundary The hermetic provider kit is Flow-owned conformance infrastructure for issue diff --git a/docs/integrations/scenario-fixtures.md b/docs/integrations/scenario-fixtures.md index 3d3caf4..e07506c 100644 --- a/docs/integrations/scenario-fixtures.md +++ b/docs/integrations/scenario-fixtures.md @@ -159,6 +159,14 @@ contract rather than an executor. ## Checked-in corpus +Issue #30 adds an [executable acceptance matrix](acceptance-scenarios.md) on +top of this contract. Its test-owned catalog reuses the terminal/evidence +vocabulary and public provider-kit boundaries, with two fresh-root executions +per recipe and a completeness-checked JSON report. It does not interpret an +arbitrary scenario manifest as an executable plan. Illustrative digests in the +original intent fixtures are never presented as the executed package identity; +runtime receipts retain the actual finalized kit digests. + The corpus includes a single-provider success example plus multi-provider, interrupted, observed-empty, and expected-unavailable fixtures. The unavailable fixture is a valid negative scenario; it is distinct from malformed manifest diff --git a/tests/fixtures/acceptance-scenarios.v1.json b/tests/fixtures/acceptance-scenarios.v1.json new file mode 100644 index 0000000..9620139 --- /dev/null +++ b/tests/fixtures/acceptance-scenarios.v1.json @@ -0,0 +1,995 @@ +{ + "schema_version": "flow.acceptance-scenario-catalog/v1", + "fixture_version": "1.0.0", + "tier": "pull-request", + "budget": { + "timeout_ms": 30000, + "max_receipt_bytes": 16384, + "max_artifact_bytes": 1048576, + "max_artifacts": 4, + "repetitions": 2 + }, + "known_gaps": [ + "Synthetic providers only; no released holon compatibility or domain-format validation.", + "Unix symlink-capable PR host required; no Windows conformance claim.", + "Memory and kernel network/filesystem isolation are not enforced by this trusted-unconfined kit.", + "No durable interruption, retry, checkpoint, or resume; owned by Flow #49 and #31.", + "No signature verification, authenticated publisher, atomic filesystem snapshot, descriptor-bound launch, or descendant containment.", + "Only normalized receipts are portable; rejected raw evidence and authority profiles may contain private values.", + "Free-form provenance is structurally checked; it is not an authoritative content digest." + ], + "scenarios": [ + { + "scenario_id": "scenario:acceptance-resolution-compatible", + "family": "resolution", + "recipe": "compatible", + "expected": { + "boundary": "resolution", + "code": "selected", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-missing-provider", + "family": "resolution", + "recipe": "missing-provider", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-missing-observation", + "family": "resolution", + "recipe": "missing-observation", + "expected": { + "boundary": "resolution", + "code": "blocked", + "terminal_state": "unavailable", + "evidence_state": "unavailable", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-unavailable", + "family": "resolution", + "recipe": "unavailable", + "expected": { + "boundary": "resolution", + "code": "blocked", + "terminal_state": "unavailable", + "evidence_state": "unavailable", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-stale-version", + "family": "resolution", + "recipe": "stale-version", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-future-version", + "family": "resolution", + "recipe": "future-version", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-publisher-mismatch", + "family": "resolution", + "recipe": "publisher-mismatch", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-integrity-mismatch", + "family": "resolution", + "recipe": "integrity-mismatch", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-future-flow", + "family": "resolution", + "recipe": "future-flow", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-future-contract", + "family": "resolution", + "recipe": "future-contract", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-missing-capability", + "family": "resolution", + "recipe": "missing-capability", + "expected": { + "boundary": "resolution", + "code": "no-compatible-provider", + "terminal_state": "unsupported", + "evidence_state": "unsupported", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-observation-mismatch", + "family": "resolution", + "recipe": "observation-mismatch", + "expected": { + "boundary": "resolution", + "code": "blocked", + "terminal_state": "unavailable", + "evidence_state": "unavailable", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-permission-denied", + "family": "resolution", + "recipe": "permission-denied", + "expected": { + "boundary": "resolution", + "code": "blocked", + "terminal_state": "unavailable", + "evidence_state": "unavailable", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-malformed-manifest", + "family": "resolution", + "recipe": "malformed-manifest", + "expected": { + "boundary": "resolution", + "code": "malformed", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-conflict", + "family": "resolution", + "recipe": "conflict", + "expected": { + "boundary": "resolution", + "code": "conflict", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-fallback-forbidden", + "family": "resolution", + "recipe": "fallback-forbidden", + "expected": { + "boundary": "resolution", + "code": "blocked", + "terminal_state": "unavailable", + "evidence_state": "unavailable", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-fallback-explicit-resume", + "family": "resolution", + "recipe": "fallback-explicit-resume", + "expected": { + "boundary": "resolution", + "code": "blocked", + "terminal_state": "unavailable", + "evidence_state": "unavailable", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-fallback-before-effects", + "family": "resolution", + "recipe": "fallback-before-effects", + "expected": { + "boundary": "resolution", + "code": "selected", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-resolution-fallback-after-invocation", + "family": "resolution", + "recipe": "fallback-after-invocation", + "expected": { + "boundary": "transcript", + "code": "nonzero-exit", + "terminal_state": "failed", + "evidence_state": "failed", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-manifest", + "family": "contract", + "recipe": "schema-manifest", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-lock", + "family": "contract", + "recipe": "schema-lock", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-bindings", + "family": "contract", + "recipe": "schema-bindings", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-scenario", + "family": "contract", + "recipe": "schema-scenario", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-invocation", + "family": "contract", + "recipe": "schema-invocation", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-subjects", + "family": "contract", + "recipe": "schema-subjects", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-result", + "family": "contract", + "recipe": "schema-result", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-schema-event", + "family": "contract", + "recipe": "schema-event", + "expected": { + "boundary": "contract", + "code": "unsupported-schema", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-executable-mismatch", + "family": "contract", + "recipe": "executable-mismatch", + "expected": { + "boundary": "preflight", + "code": "subject-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-invocation-configuration-schema", + "family": "contract", + "recipe": "invocation-configuration-schema", + "expected": { + "boundary": "preflight", + "code": "context-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-invocation-publisher", + "family": "contract", + "recipe": "invocation-publisher", + "expected": { + "boundary": "preflight", + "code": "context-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-invocation-capability", + "family": "contract", + "recipe": "invocation-capability", + "expected": { + "boundary": "preflight", + "code": "context-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-invocation-authorization", + "family": "contract", + "recipe": "invocation-authorization", + "expected": { + "boundary": "preflight", + "code": "context-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-invocation-output", + "family": "contract", + "recipe": "invocation-output", + "expected": { + "boundary": "preflight", + "code": "context-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-configuration", + "family": "contract", + "recipe": "result-configuration", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-authorization", + "family": "contract", + "recipe": "result-authorization", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-capability", + "family": "contract", + "recipe": "result-capability", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-version", + "family": "contract", + "recipe": "result-version", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-integrity", + "family": "contract", + "recipe": "result-integrity", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-provenance", + "family": "contract", + "recipe": "result-provenance", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-duplicate-output", + "family": "contract", + "recipe": "result-duplicate-output", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-contract-result-terminal-contradiction", + "family": "contract", + "recipe": "result-terminal-contradiction", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-absolute-path", + "family": "artifact", + "recipe": "absolute-path", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-parent-traversal", + "family": "artifact", + "recipe": "parent-traversal", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-nested-escape", + "family": "artifact", + "recipe": "nested-escape", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-windows-path", + "family": "artifact", + "recipe": "windows-path", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-duplicate-identity", + "family": "artifact", + "recipe": "duplicate-identity", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-duplicate-port", + "family": "artifact", + "recipe": "duplicate-port", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-duplicate-locator", + "family": "artifact", + "recipe": "duplicate-locator", + "expected": { + "boundary": "observation", + "code": "invalid-bindings", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-file-as-directory", + "family": "artifact", + "recipe": "file-as-directory", + "expected": { + "boundary": "observation", + "code": "kind-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-directory-as-file", + "family": "artifact", + "recipe": "directory-as-file", + "expected": { + "boundary": "observation", + "code": "kind-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-symlink", + "family": "artifact", + "recipe": "symlink", + "expected": { + "boundary": "observation", + "code": "symlink", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-parent-symlink", + "family": "artifact", + "recipe": "parent-symlink", + "expected": { + "boundary": "observation", + "code": "symlink", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-changed-input-before-observation", + "family": "artifact", + "recipe": "changed-input-before-observation", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-digest-conflict", + "family": "artifact", + "recipe": "digest-conflict", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-undeclared-output-type", + "family": "artifact", + "recipe": "undeclared-output-type", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-duplicate-event", + "family": "artifact", + "recipe": "duplicate-event", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-stale-invocation", + "family": "artifact", + "recipe": "stale-invocation", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-stale-binding", + "family": "artifact", + "recipe": "stale-binding", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-changed-output", + "family": "artifact", + "recipe": "changed-output", + "expected": { + "boundary": "acceptance", + "code": "observation-changed", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-changed-input-after-observation", + "family": "artifact", + "recipe": "changed-input-after-observation", + "expected": { + "boundary": "acceptance", + "code": "observation-changed", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-removed-output", + "family": "artifact", + "recipe": "removed-output", + "expected": { + "boundary": "acceptance", + "code": "reobservation-failed", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-artifact-accepted", + "family": "artifact", + "recipe": "accepted", + "expected": { + "boundary": "acceptance", + "code": "accepted", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": true + } + }, + { + "scenario_id": "scenario:acceptance-artifact-observed-empty", + "family": "artifact", + "recipe": "observed-empty", + "expected": { + "boundary": "observation", + "code": "observed-empty", + "terminal_state": "complete", + "evidence_state": "observed-empty", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-success", + "family": "provider", + "recipe": "success", + "expected": { + "boundary": "acceptance", + "code": "accepted", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": true + } + }, + { + "scenario_id": "scenario:acceptance-provider-warning", + "family": "provider", + "recipe": "warning", + "expected": { + "boundary": "acceptance", + "code": "accepted", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": true + } + }, + { + "scenario_id": "scenario:acceptance-provider-partial-result", + "family": "provider", + "recipe": "partial-result", + "expected": { + "boundary": "acceptance", + "code": "incomplete-result", + "terminal_state": "partial", + "evidence_state": "incomplete", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-partial-output", + "family": "provider", + "recipe": "partial-output", + "expected": { + "boundary": "acceptance", + "code": "incomplete-result", + "terminal_state": "partial", + "evidence_state": "incomplete", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-missing-output", + "family": "provider", + "recipe": "missing-output", + "expected": { + "boundary": "observation", + "code": "missing-artifact", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-extra-output", + "family": "provider", + "recipe": "extra-output", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-contradictory-artifact-evidence", + "family": "provider", + "recipe": "contradictory-artifact-evidence", + "expected": { + "boundary": "acceptance", + "code": "evidence-mismatch", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-corrupt-artifact-evidence", + "family": "provider", + "recipe": "corrupt-artifact-evidence", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-nonzero-after-success", + "family": "provider", + "recipe": "nonzero-after-success", + "expected": { + "boundary": "transcript", + "code": "nonzero-exit", + "terminal_state": "failed", + "evidence_state": "failed", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-stdout-overflow", + "family": "provider", + "recipe": "stdout-overflow", + "expected": { + "boundary": "transcript", + "code": "stdout-limit", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-stderr-overflow", + "family": "provider", + "recipe": "stderr-overflow", + "expected": { + "boundary": "transcript", + "code": "stderr-limit", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-invalid-event", + "family": "provider", + "recipe": "invalid-event", + "expected": { + "boundary": "transcript", + "code": "invalid-event", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-invalid-result", + "family": "provider", + "recipe": "invalid-result", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-provider-success-with-host-rejection", + "family": "provider", + "recipe": "success-with-host-rejection", + "expected": { + "boundary": "transcript", + "code": "host-rejection", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-privacy-operational-stream", + "family": "privacy", + "recipe": "operational-stream", + "expected": { + "boundary": "privacy", + "code": "canaries-absent", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-privacy-argv-environment", + "family": "privacy", + "recipe": "argv-environment", + "expected": { + "boundary": "privacy", + "code": "canaries-absent", + "terminal_state": "complete", + "evidence_state": "complete", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-privacy-event-diagnostic", + "family": "privacy", + "recipe": "event-diagnostic", + "expected": { + "boundary": "transcript", + "code": "invalid-event", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + }, + { + "scenario_id": "scenario:acceptance-privacy-result-diagnostic", + "family": "privacy", + "recipe": "result-diagnostic", + "expected": { + "boundary": "transcript", + "code": "invalid-result", + "terminal_state": "invalid", + "evidence_state": "invalid", + "accepted": false + } + } + ] +} diff --git a/tests/fixtures/privacy-provider.sh b/tests/fixtures/privacy-provider.sh new file mode 100644 index 0000000..2ef3072 --- /dev/null +++ b/tests/fixtures/privacy-provider.sh @@ -0,0 +1,10 @@ +#!/bin/sh +# Synthetic direct-child privacy probe. No external commands or source writes. +set -eu +[ "$#" -eq 1 ] || exit 81 +[ "$1" = "FLOW_ARGV_CANARY_7429" ] || exit 82 +[ "${HOME+x}" != x ] || exit 83 +[ "$FLOW_PRIVATE" = "FLOW_ENV_CANARY_1938 FLOW_SECRET_CANARY_5813 FLOW_PRIVATE_CANARY_2604" ] || exit 84 +IFS= read -r request || exit 85 +printf "%s" "$FLOW_PROVIDER_STDOUT" +printf "%s %s" "$1" "$FLOW_PRIVATE" >&2 diff --git a/tests/hermetic_provider_kit.rs b/tests/hermetic_provider_kit.rs index aed4c30..e4b1c11 100644 --- a/tests/hermetic_provider_kit.rs +++ b/tests/hermetic_provider_kit.rs @@ -1,5 +1,8 @@ mod common; +#[path = "scenario_matrix/mod.rs"] +mod scenario_matrix; + use std::collections::BTreeMap; use std::fs; use std::path::{Path, PathBuf}; diff --git a/tests/scenario_matrix/artifacts.rs b/tests/scenario_matrix/artifacts.rs new file mode 100644 index 0000000..88cab11 --- /dev/null +++ b/tests/scenario_matrix/artifacts.rs @@ -0,0 +1,145 @@ +use super::*; + +// Keep the closed mutation table beside its observation and promotion assertions. +#[allow(clippy::too_many_lines)] +pub(super) fn run(fixture: &KitFixture, recipe: &str) -> (Outcome, Value) { + let workspace = fixture.root.path().join(WORKSPACE_LOCATOR); + let mut bindings = fixture.bindings.clone(); + if recipe == "observed-empty" { + let path = workspace.join("outputs/empty"); + fs::create_dir(&path).unwrap(); + "outputs/empty".clone_into(&mut bindings.outputs[0].locator); + bindings.outputs[0].kind = ArtifactKind::Directory; + let observed = observe_artifacts(&workspace, &bindings).unwrap(); + let directory = observed + .evidence() + .artifacts + .iter() + .find(|a| a.kind == ArtifactKind::Directory) + .unwrap(); + assert!(directory.manifest.is_empty()); + assert_eq!(directory.size_bytes, 0); + let mut actual = outcome( + "observation", + "observed-empty", + ScenarioTerminalState::Complete, + ); + actual.evidence_state = ScenarioEvidenceState::ObservedEmpty; + return ( + actual, + json!({"entry_count": directory.manifest.len(), "digest": directory.digest}), + ); + } + let prepared = fixture.prepare_lifecycle(CAPABILITIES[0], "success", false); + let execution = execute(fixture, &prepared).unwrap(); + match recipe { + "absolute-path" => { + "/private/FLOW_PRIVATE_CANARY_2604".clone_into(&mut bindings.outputs[0].locator); + } + "parent-traversal" => { + "../FLOW_PRIVATE_CANARY_2604".clone_into(&mut bindings.outputs[0].locator); + } + "nested-escape" => { + "outputs/../../FLOW_PRIVATE_CANARY_2604".clone_into(&mut bindings.outputs[0].locator); + } + "windows-path" => { + "C:/private/FLOW_PRIVATE_CANARY_2604".clone_into(&mut bindings.outputs[0].locator); + } + "duplicate-identity" => bindings.outputs[0] + .artifact_id + .clone_from(&bindings.inputs[0].artifact_id), + "duplicate-port" => bindings.outputs[0] + .port + .clone_from(&bindings.inputs[0].port), + "duplicate-locator" => bindings.outputs[0] + .locator + .clone_from(&bindings.inputs[0].locator), + "file-as-directory" => bindings.outputs[0].kind = ArtifactKind::Directory, + "directory-as-file" => { + fs::remove_file(fixture.output_path()).unwrap(); + fs::create_dir(fixture.output_path()).unwrap(); + } + "symlink" => { + let private = fixture.root.path().join(CANARIES[3]); + fs::write(&private, CANARIES[3]).unwrap(); + fs::remove_file(fixture.output_path()).unwrap(); + symlink(&private, &fixture.output_path()); + } + "parent-symlink" => { + let private = fixture.root.path().join(CANARIES[3]); + fs::create_dir(&private).unwrap(); + fs::write(private.join("inspection-report.json"), CANARIES[3]).unwrap(); + fs::remove_dir_all(workspace.join("outputs")).unwrap(); + symlink(&private, &workspace.join("outputs")); + } + "changed-input-before-observation" => { + fs::write(workspace.join(INPUT_LOCATOR), b"changed input").unwrap(); + } + "digest-conflict" => bindings.inputs[0].expected_digest = "a".repeat(64), + "undeclared-output-type" => "text/plain".clone_into(&mut bindings.outputs[0].media_type), + "duplicate-event" + | "stale-invocation" + | "stale-binding" + | "changed-output" + | "removed-output" + | "changed-input-after-observation" + | "accepted" => {} + other => panic!("unknown artifact recipe: {other}"), + } + let observed = match observe_artifacts(&workspace, &bindings) { + Ok(observed) => observed, + Err(error) => { + return ( + observation_error(&error), + json!({"provider_outcome": execution.result().outcome, "provider_exit": 0}), + ); + } + }; + let mut invocation = prepared.invocation.clone(); + let execution = if recipe == "duplicate-event" { + let mut events = execution.events().to_vec(); + let mut duplicate = events[1].clone(); + "event:duplicate-artifact".clone_into(&mut duplicate.event_id); + duplicate.sequence = 2; + events.last_mut().unwrap().sequence = 3; + events.insert(2, duplicate); + validate_transcript(&prepared, &transcript(&events, execution.result()), &[]).unwrap() + } else { + execution + }; + match recipe { + "stale-invocation" => "invocation:stale".clone_into(&mut invocation.invocation_id), + "stale-binding" => "bindings:stale".clone_into(&mut bindings.binding_set_id), + "changed-output" => fs::write(fixture.output_path(), b"changed output").unwrap(), + "removed-output" => fs::remove_file(fixture.output_path()).unwrap(), + "changed-input-after-observation" => { + fs::write(workspace.join(INPUT_LOCATOR), b"changed input").unwrap(); + } + _ => {} + } + match accept_artifacts( + prepared.resolved(), + &invocation, + &execution, + &bindings, + &observed, + ) { + Ok(accepted) => accepted_evidence(&accepted), + Err(error) => ( + acceptance_error(&error), + json!({"provider_outcome": execution.result().outcome, "provider_exit": 0, "observed_artifacts": observed.evidence().artifacts.len()}), + ), + } +} + +#[cfg(unix)] +fn symlink(target: &Path, link: &Path) { + std::os::unix::fs::symlink(target, link).unwrap(); +} + +#[cfg(not(unix))] +fn symlink(_target: &Path, _link: &Path) { + panic!( + "the acceptance matrix requires a Unix symlink-capable host; do not count a skipped case as passing" + ); +} diff --git a/tests/scenario_matrix/contracts.rs b/tests/scenario_matrix/contracts.rs new file mode 100644 index 0000000..87c526b --- /dev/null +++ b/tests/scenario_matrix/contracts.rs @@ -0,0 +1,120 @@ +use super::*; + +pub(super) fn run(fixture: &KitFixture, recipe: &str) -> (Outcome, Value) { + if let Some(document) = recipe.strip_prefix("schema-") { + return schema_case(fixture, document); + } + let mut prepared = fixture.prepare_lifecycle(CAPABILITIES[0], "success", false); + if recipe == "executable-mismatch" { + fs::write( + fixture + .root + .path() + .join(PACKAGE_LOCATOR) + .join(&fixture.executable_locator), + b"changed executable", + ) + .unwrap(); + let error = execute(fixture, &prepared).unwrap_err(); + assert!(matches!( + error, + ProcessRunnerError::SubjectObservation { .. } + )); + assert!(!fixture.output_path().exists()); + return ( + invalid("preflight", "subject-mismatch"), + json!({"invoked": false}), + ); + } + if let Some(field) = recipe.strip_prefix("invocation-") { + let invocation = &mut prepared.invocation; + match field { + "configuration-schema" => { + "flow.other-configuration/v1".clone_into(&mut invocation.configuration.schema_id); + } + "publisher" => "org.example.other".clone_into(&mut invocation.extension.publisher_id), + "capability" => "flow/transform-fixture".clone_into(&mut invocation.capability_id), + "authorization" => { + "authorization:other".clone_into(&mut invocation.authorization.authorization_id); + } + "output" => invocation.expected_output_types = vec!["text/plain".to_owned()], + _ => panic!("unknown invocation recipe: {field}"), + } + let error = Orchestrator::encode_process_request( + prepared.resolution.resolved().unwrap(), + invocation, + &prepared.subject_lock, + &prepared.subjects, + &prepared.authority, + ) + .unwrap_err(); + assert!(!fixture.output_path().exists()); + return (execution_error(&error), json!({"invoked": false})); + } + let execution = execute(fixture, &prepared).unwrap(); + let mut events = execution.events().to_vec(); + let mut result = execution.result().clone(); + match recipe { + "result-configuration" => result.configuration_digest = "a".repeat(64), + "result-authorization" => "authorization:other".clone_into(&mut result.authorization_id), + "result-capability" => "flow/transform-fixture".clone_into(&mut result.capability_id), + "result-version" => "9.0.0".clone_into(&mut result.extension_version), + "result-integrity" => result.extension_integrity = "b".repeat(64), + "result-provenance" => result.provenance[0].value = String::new(), + "result-duplicate-output" => result + .produced_artifacts + .push(result.produced_artifacts[0].clone()), + "result-terminal-contradiction" => events.last_mut().unwrap().state = EventState::Failed, + other => panic!("unknown contract recipe: {other}"), + } + let error = validate_transcript(&prepared, &transcript(&events, &result), &[]).unwrap_err(); + ( + execution_error(&error), + json!({"provider_outcome": execution.result().outcome, "accepted_execution": false}), + ) +} + +fn schema_case(fixture: &KitFixture, document: &str) -> (Outcome, Value) { + // Baselines pass first; the only mutation is an unsupported schema version. + macro_rules! reject_version { + ($value:expr) => {{ + let mut value = $value; + value.validate().unwrap(); + value.schema_version = "flow.unsupported/v99".to_owned(); + value.validate().unwrap_err(); + }}; + } + match document { + "manifest" => reject_version!(fixture.manifest.clone()), + "lock" => reject_version!(fixture.lock.clone()), + "bindings" => reject_version!(fixture.bindings.clone()), + "scenario" => reject_version!( + serde_json::from_str::(include_str!( + "../../contracts/examples/scenario-manifest.v1.example.json" + )) + .unwrap() + ), + "invocation" | "subjects" => { + let prepared = fixture.prepare_lifecycle(CAPABILITIES[0], "success", false); + if document == "invocation" { + reject_version!(prepared.invocation); + } else { + reject_version!(prepared.subject_lock); + } + } + "result" | "event" => { + let prepared = fixture.prepare_lifecycle(CAPABILITIES[0], "success", false); + let execution = execute(fixture, &prepared).unwrap(); + if document == "result" { + reject_version!(execution.result().clone()); + } else { + reject_version!(execution.events()[0].clone()); + } + } + _ => panic!("unknown schema document: {document}"), + } + ( + invalid("contract", "unsupported-schema"), + json!({"document": document, "baseline_valid": true}), + ) +} diff --git a/tests/scenario_matrix/mod.rs b/tests/scenario_matrix/mod.rs new file mode 100644 index 0000000..3cc0e47 --- /dev/null +++ b/tests/scenario_matrix/mod.rs @@ -0,0 +1,523 @@ +//! Test-owned recipes over public Flow boundaries, not a production scheduler. + +use super::*; +use flow::{ + Orchestrator, ProcessCompletion, ProcessTranscript, ScenarioEvidenceState, ScenarioManifest, + ScenarioTerminalState, +}; +use serde::Deserialize; +use serde_json::{Value, json}; +use std::collections::BTreeSet; +use std::time::Instant; + +mod artifacts; +mod contracts; +mod privacy; +mod resolution; + +const CATALOG: &str = include_str!("../fixtures/acceptance-scenarios.v1.json"); +const CANARIES: [&str; 4] = [ + "FLOW_ARGV_CANARY_7429", + "FLOW_ENV_CANARY_1938", + "FLOW_SECRET_CANARY_5813", + "FLOW_PRIVATE_CANARY_2604", +]; + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct Catalog { + schema_version: String, + fixture_version: String, + tier: String, + budget: Budget, + known_gaps: Vec, + scenarios: Vec, +} + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct Budget { + timeout_ms: u64, + max_receipt_bytes: usize, + max_artifact_bytes: u64, + max_artifacts: usize, + repetitions: usize, +} + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct Case { + scenario_id: String, + family: String, + recipe: String, + expected: Outcome, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +struct Outcome { + boundary: String, + code: String, + terminal_state: ScenarioTerminalState, + evidence_state: ScenarioEvidenceState, + accepted: bool, +} + +#[derive(Debug, Eq, PartialEq, Serialize)] +struct Receipt { + schema_version: &'static str, + scenario_id: String, + recipe_digest: String, + fixture_identity: Value, + outcome: Outcome, + recovery: &'static str, + evidence: Value, +} + +fn outcome(boundary: &str, code: &str, state: ScenarioTerminalState) -> Outcome { + let evidence_state = match state { + ScenarioTerminalState::Complete => ScenarioEvidenceState::Complete, + ScenarioTerminalState::Partial | ScenarioTerminalState::Interrupted => { + ScenarioEvidenceState::Incomplete + } + ScenarioTerminalState::Unavailable => ScenarioEvidenceState::Unavailable, + ScenarioTerminalState::Unsupported => ScenarioEvidenceState::Unsupported, + ScenarioTerminalState::Invalid => ScenarioEvidenceState::Invalid, + ScenarioTerminalState::Failed => ScenarioEvidenceState::Failed, + }; + Outcome { + boundary: boundary.to_owned(), + code: code.to_owned(), + terminal_state: state, + evidence_state, + accepted: false, + } +} + +fn invalid(boundary: &str, code: &str) -> Outcome { + outcome(boundary, code, ScenarioTerminalState::Invalid) +} + +fn recovery(boundary: &str) -> &'static str { + match boundary { + "resolution" => { + "Inspect pinned versions, availability, and operator selection policy before invocation." + } + "contract" => "Repair the named versioned document before resolving or executing it.", + "preflight" => { + "Rebuild matching invocation, subject, and authority evidence before launch." + } + "transcript" => { + "Inspect the provider protocol; retain failure and do not switch providers after invocation." + } + "observation" => "Repair root-relative declarations or candidate files and observe again.", + "acceptance" => { + "Preserve the rejected candidate; validate complete, current, exactly declared evidence." + } + "privacy" => "Keep raw operational evidence local; export only the allowlisted receipt.", + other => panic!("unhandled recovery boundary: {other}"), + } +} + +#[test] +#[allow(clippy::too_many_lines)] // One visible receipt lifecycle for each catalog row. +fn executable_acceptance_matrix() { + let catalog: Catalog = serde_json::from_str(CATALOG).unwrap(); + assert_eq!( + catalog.schema_version, + "flow.acceptance-scenario-catalog/v1" + ); + assert_eq!(catalog.fixture_version, "1.0.0"); + assert_eq!(catalog.tier, "pull-request"); + assert!(!catalog.known_gaps.is_empty()); + assert_eq!(catalog.budget.repetitions, 2); + assert!(catalog.budget.timeout_ms > 0); + let (provider_bytes, executable_name) = provider_binary_snapshot(); + let mut ids = BTreeSet::new(); + let mut recipes = BTreeSet::new(); + let mut states = BTreeSet::new(); + for case in &catalog.scenarios { + assert!(ids.insert(&case.scenario_id), "duplicate scenario ID"); + assert!( + recipes.insert((&case.family, &case.recipe)), + "duplicate recipe" + ); + assert!(case.scenario_id.starts_with("scenario:acceptance-")); + validate_expectation(case); + let mut previous = None; + for _ in 0..catalog.budget.repetitions { + let started = Instant::now(); + let fixture = KitFixture::new(CAPABILITIES[0], &provider_bytes, &executable_name); + let original_input = fs::read( + fixture + .root + .path() + .join(WORKSPACE_LOCATOR) + .join(INPUT_LOCATOR), + ) + .unwrap(); + let (actual, evidence) = match case.family.as_str() { + "resolution" => resolution::run(&fixture, &case.recipe), + "contract" => contracts::run(&fixture, &case.recipe), + "artifact" => artifacts::run(&fixture, &case.recipe), + "provider" => provider_case(&fixture, &case.recipe), + "privacy" => privacy_case(&fixture, &case.recipe), + other => panic!("unknown scenario family: {other}"), + }; + assert_eq!(actual, case.expected, "{}", case.scenario_id); + // Some artifact recipes intentionally alter the input to prove refusal. + if !case.recipe.starts_with("changed-input") { + assert_eq!( + fs::read( + fixture + .root + .path() + .join(WORKSPACE_LOCATOR) + .join(INPUT_LOCATOR) + ) + .unwrap(), + original_input + ); + } + check_artifact_budget(&fixture, &catalog.budget); + let receipt = Receipt { + schema_version: "flow.acceptance-scenario-receipt/v1", + scenario_id: case.scenario_id.clone(), + recipe_digest: digest_json( + &json!({"version": catalog.fixture_version, "family": case.family, "recipe": case.recipe}), + ), + fixture_identity: json!({"package_digest": fixture.package_digest, "executable_digest": fixture.executable_digest, "manifest_digest": digest_json(&fixture.manifest), "input_digest": digest_bytes(INPUT_BYTES)}), + recovery: recovery(&actual.boundary), + outcome: actual, + evidence, + }; + let encoded = serde_json::to_string(&receipt).unwrap(); + assert!(encoded.len() <= catalog.budget.max_receipt_bytes); + assert_private_values_absent(&encoded, fixture.root.path()); + assert!( + started.elapsed().as_millis() <= u128::from(catalog.budget.timeout_ms), + "scenario budget exceeded: {}", + case.scenario_id + ); + if let Some(previous) = &previous { + assert_eq!(&receipt, previous, "fresh-root drift: {}", case.scenario_id); + } + previous = Some(receipt); + } + let receipt = previous.unwrap(); + states.insert(serde_json::to_string(&receipt.outcome.evidence_state).unwrap()); + println!( + "FLOW_ACCEPTANCE_RECEIPT={}", + serde_json::to_string(&receipt).unwrap() + ); + } + for state in [ + "complete", + "observed-empty", + "unavailable", + "incomplete", + "unsupported", + "invalid", + "failed", + ] { + assert!( + states.contains(&format!("\"{state}\"")), + "missing distinct evidence state {state}" + ); + } +} + +fn validate_expectation(case: &Case) { + // Reuse the accepted scenario contract for the terminal/evidence vocabulary. + // This is an intent check, not a claim that its illustrative package ran. + let mut manifest: ScenarioManifest = serde_json::from_str(include_str!( + "../../contracts/examples/scenario-manifest.v1.example.json" + )) + .unwrap(); + manifest.scenario_id.clone_from(&case.scenario_id); + manifest.fixture_id = case.scenario_id.replace("scenario:", "fixture:"); + manifest.expectation.terminal_state = case.expected.terminal_state; + manifest.expectation.evidence_state = case.expected.evidence_state; + manifest.expectation.expected_artifacts.clear(); + manifest.validate().unwrap(); +} + +fn check_artifact_budget(fixture: &KitFixture, budget: &Budget) { + let root = fixture.root.path().join(WORKSPACE_LOCATOR).join("outputs"); + if fs::symlink_metadata(&root) + .unwrap() + .file_type() + .is_symlink() + { + // The observer has rejected this recipe; never follow its target here. + return; + } + let entries = fs::read_dir(root) + .unwrap() + .map(Result::unwrap) + .collect::>(); + assert!(entries.len() <= budget.max_artifacts); + let bytes: u64 = entries + .iter() + .map(|entry| { + let metadata = fs::symlink_metadata(entry.path()).unwrap(); + if metadata.is_file() { + metadata.len() + } else { + 0 + } + }) + .sum(); + assert!(bytes <= budget.max_artifact_bytes); +} + +fn assert_private_values_absent(value: &str, root: &Path) { + for canary in CANARIES { + assert!(!value.contains(canary), "portable evidence leaked a canary"); + } + assert!( + !value.contains(root.to_str().unwrap()), + "portable evidence leaked the host root" + ); +} + +fn execute( + fixture: &KitFixture, + prepared: &PreparedLifecycleRun, +) -> Result { + LocalProcessRunner::run( + fixture.root.path(), + prepared.resolved(), + &prepared.invocation, + &prepared.subject_lock, + &prepared.authority, + &NoSecrets, + &mut Vec::new(), + ) +} + +fn execution_error(error: &ExecutionError) -> Outcome { + match error { + ExecutionError::InvalidInvocation { .. } => invalid("preflight", "invalid-invocation"), + ExecutionError::Preflight { .. } => invalid("preflight", "context-mismatch"), + ExecutionError::InvalidEvent { .. } => invalid("transcript", "invalid-event"), + ExecutionError::InvalidResult { .. } => invalid("transcript", "invalid-result"), + ExecutionError::ProcessProtocol { .. } => invalid("transcript", "invalid-protocol"), + ExecutionError::EventSink { .. } => invalid("transcript", "host-rejection"), + ExecutionError::ProcessExit { .. } => { + outcome("transcript", "nonzero-exit", ScenarioTerminalState::Failed) + } + ExecutionError::ProcessOutputLimit { + stream, + limit, + observed, + } => { + assert_eq!(*observed, limit + 1); + invalid( + "transcript", + match stream { + ProcessStream::Stdout => "stdout-limit", + ProcessStream::Stderr => "stderr-limit", + }, + ) + } + other => panic!("unclassified execution failure: {other:?}"), + } +} + +fn acceptance_error(error: &ArtifactAcceptanceError) -> Outcome { + match error { + ArtifactAcceptanceError::Mismatch { message } => { + if message == "artifact acceptance requires a complete produced or reused result" { + outcome( + "acceptance", + "incomplete-result", + ScenarioTerminalState::Partial, + ) + } else { + invalid("acceptance", "evidence-mismatch") + } + } + ArtifactAcceptanceError::ObservationChanged => invalid("acceptance", "observation-changed"), + ArtifactAcceptanceError::Reobservation { .. } => { + invalid("acceptance", "reobservation-failed") + } + other => panic!("unclassified acceptance failure: {other:?}"), + } +} + +fn observation_error(error: &ArtifactObservationError) -> Outcome { + let code = match error { + ArtifactObservationError::InvalidBindings { .. } => "invalid-bindings", + ArtifactObservationError::Missing { .. } => "missing-artifact", + ArtifactObservationError::KindMismatch { .. } => "kind-mismatch", + ArtifactObservationError::Symlink { .. } | ArtifactObservationError::RootSymlink { .. } => { + "symlink" + } + ArtifactObservationError::PathEscape { .. } => "path-escape", + other => panic!("unclassified observation failure: {other:?}"), + }; + invalid("observation", code) +} + +fn accepted_evidence(accepted: &AcceptedArtifactSet) -> (Outcome, Value) { + let mut result = outcome("acceptance", "accepted", ScenarioTerminalState::Complete); + result.accepted = true; + ( + result, + json!({"inputs": accepted.inputs(), "outputs": accepted.outputs()}), + ) +} + +fn transcript(events: &[flow::ExtensionEvent], result: &flow::ExtensionResult) -> Vec { + let mut bytes = Vec::new(); + for event in events { + serde_json::to_writer(&mut bytes, event).unwrap(); + bytes.push(b'\n'); + } + serde_json::to_writer(&mut bytes, result).unwrap(); + bytes.push(b'\n'); + bytes +} + +fn validate_transcript( + prepared: &PreparedLifecycleRun, + bytes: &[u8], + stderr: &[u8], +) -> Result { + Orchestrator::validate_process_transcript( + prepared.resolved(), + &prepared.invocation, + &prepared.subject_lock, + &prepared.subjects, + &prepared.authority, + ProcessTranscript::new(ProcessCompletion::Exited { code: Some(0) }, bytes, stderr), + &mut Vec::new(), + ) +} + +fn provider_case(fixture: &KitFixture, recipe: &str) -> (Outcome, Value) { + let prepared = fixture.prepare_lifecycle(CAPABILITIES[0], recipe, false); + let execution = if recipe == "success-with-host-rejection" { + LocalProcessRunner::run( + fixture.root.path(), + prepared.resolved(), + &prepared.invocation, + &prepared.subject_lock, + &prepared.authority, + &NoSecrets, + &mut |_: &flow::ExtensionEvent| Err(EventSinkError::new(CANARIES[3])), + ) + } else { + execute(fixture, &prepared) + }; + fixture.assert_subjects_unchanged(&prepared); + match execution { + Err(ProcessRunnerError::Validation { source }) => ( + execution_error(&source), + json!({"retained_event_count": source.events().len(), "output_exists": fixture.output_path().exists()}), + ), + Err(other) => panic!("unexpected runner failure: {other:?}"), + Ok(execution) => { + let observed = match observe_artifacts( + &fixture.root.path().join(WORKSPACE_LOCATOR), + &fixture.bindings, + ) { + Ok(observed) => observed, + Err(error) => { + return ( + observation_error(&error), + json!({"provider_outcome": execution.result().outcome}), + ); + } + }; + match accept_artifacts( + prepared.resolved(), + &prepared.invocation, + &execution, + &fixture.bindings, + &observed, + ) { + Ok(accepted) => accepted_evidence(&accepted), + Err(error) => ( + acceptance_error(&error), + json!({"provider_outcome": execution.result().outcome, "partial_result": execution.result().partial_result}), + ), + } + } + } +} + +fn privacy_case(fixture: &KitFixture, recipe: &str) -> (Outcome, Value) { + if recipe == "argv-environment" { + return privacy::argv_environment(); + } + let prepared = fixture.prepare_lifecycle(CAPABILITIES[0], "success", false); + let execution = execute(fixture, &prepared).unwrap(); + let mut result = execution.result().clone(); + let mut events = execution.events().to_vec(); + match recipe { + "operational-stream" => { + let private = CANARIES.join(" "); + let accepted = + validate_transcript(&prepared, &transcript(&events, &result), private.as_bytes()) + .unwrap(); + assert_eq!(accepted, execution); + assert_private_values_absent(&format!("{accepted:?}"), fixture.root.path()); + assert!(!format!("{:?}", flow::SecretValue::new(CANARIES[2])).contains(CANARIES[2])); + } + "event-diagnostic" | "result-diagnostic" => { + let diagnostic = flow::Diagnostic { + severity: Severity::Error, + code: "canary.private".to_owned(), + message: CANARIES.join(" "), + redacted: false, + }; + if recipe == "event-diagnostic" { + events[0].diagnostics.push(diagnostic); + } else { + result.diagnostics.push(diagnostic); + } + let error = + validate_transcript(&prepared, &transcript(&events, &result), &[]).unwrap_err(); + return ( + execution_error(&error), + json!({"raw_evidence_retained_locally": true}), + ); + } + "argv-environment" => { + let mut profile = process_authority_profile( + prepared.resolved(), + &prepared.invocation, + &prepared.subject_lock, + &prepared.subjects, + ProcessIsolation::TrustedUnconfined, + ); + profile.requested.argv = vec![CANARIES[0].to_owned()]; + profile.granted.argv = profile.requested.argv.clone(); + let enforcement = process_enforcement_evidence(&profile, &prepared.subjects); + let authority = authorize_process( + prepared.resolved(), + &prepared.invocation, + &prepared.subject_lock, + &prepared.subjects, + &profile, + &enforcement, + ) + .unwrap(); + // Authority profiles are host-owned inputs and retain literal argv. + // They must never be copied wholesale into a portable result. + assert!(authority.profile().requested.argv[0].contains(CANARIES[0])); + assert_private_values_absent(&format!("{execution:?}"), fixture.root.path()); + } + other => panic!("unknown privacy recipe {other}"), + } + ( + outcome( + "privacy", + "canaries-absent", + ScenarioTerminalState::Complete, + ), + json!({"canary_count": CANARIES.len(), "execution_outcome": result.outcome}), + ) +} diff --git a/tests/scenario_matrix/privacy.rs b/tests/scenario_matrix/privacy.rs new file mode 100644 index 0000000..838edc8 --- /dev/null +++ b/tests/scenario_matrix/privacy.rs @@ -0,0 +1,83 @@ +use super::*; +use flow::ExtensionPort; + +#[cfg(unix)] +pub(super) fn argv_environment() -> (Outcome, Value) { + let mut permissions = common::manifest().requested_permissions; + permissions.environment_read = + vec!["FLOW_PRIVATE".to_owned(), "FLOW_PROVIDER_STDOUT".to_owned()]; + let fixture = common::process_runner_fixture( + include_bytes!("../fixtures/privacy-provider.sh"), + permissions.clone(), + permissions, + ); + let resolution = fixture.catalog.resolve(&fixture.request); + let resolved = resolution.resolved().unwrap(); + let mut invocation = common::invocation(resolved); + invocation.secret_handles = vec![ + "secret:env.flow_private".to_owned(), + "secret:env.flow_provider_stdout".to_owned(), + ]; + let subjects = observe_execution_subjects( + fixture.root.path(), + resolved, + &invocation, + &fixture.subject_lock, + ) + .unwrap(); + let mut profile = process_authority_profile( + resolved, + &invocation, + &fixture.subject_lock, + &subjects, + ProcessIsolation::TrustedUnconfined, + ); + profile.requested.argv = vec![CANARIES[0].to_owned()]; + profile.granted.argv = profile.requested.argv.clone(); + let enforcement = process_enforcement_evidence(&profile, &subjects); + let authority = authorize_process( + resolved, + &invocation, + &fixture.subject_lock, + &subjects, + &profile, + &enforcement, + ) + .unwrap(); + let provider = flow::HermeticExtension::new(flow::PortIdentity::from_resolved(resolved)); + let mut events = Vec::new(); + let expected = provider.invoke(&invocation, &mut events).unwrap(); + let stdout = String::from_utf8(transcript(&events, &expected)).unwrap(); + let secrets = |handle: &str| match handle { + "secret:env.flow_provider_stdout" => Some(flow::SecretValue::new(stdout.clone())), + "secret:env.flow_private" => Some(flow::SecretValue::new(CANARIES[1..].join(" "))), + _ => panic!("runner requested an unauthorized secret handle"), + }; + let execution = LocalProcessRunner::run( + fixture.root.path(), + resolved, + &invocation, + &fixture.subject_lock, + &authority, + &secrets, + &mut Vec::new(), + ) + .unwrap(); + assert_eq!(execution.events(), events); + assert_eq!(execution.result(), &expected); + assert_private_values_absent(&format!("{execution:?}"), fixture.root.path()); + assert!(!format!("{:?}", flow::SecretValue::new(CANARIES[2])).contains(CANARIES[2])); + ( + outcome( + "privacy", + "canaries-absent", + ScenarioTerminalState::Complete, + ), + json!({"canary_count": CANARIES.len(), "execution_outcome": execution.result().outcome, "probe_subjects": subjects.evidence()}), + ) +} + +#[cfg(not(unix))] +pub(super) fn argv_environment() -> (Outcome, Value) { + panic!("the privacy launch probe requires the declared Unix PR host"); +} diff --git a/tests/scenario_matrix/resolution.rs b/tests/scenario_matrix/resolution.rs new file mode 100644 index 0000000..b28469b --- /dev/null +++ b/tests/scenario_matrix/resolution.rs @@ -0,0 +1,152 @@ +use super::*; + +pub(super) fn run(fixture: &KitFixture, recipe: &str) -> (Outcome, Value) { + if recipe.starts_with("fallback-") || recipe == "conflict" { + return policy_case(fixture, recipe); + } + let mut manifest = fixture.manifest.clone(); + let mut lock = fixture.lock.clone(); + let mut observed = common::observation(&manifest, true); + let mut manifests = Vec::new(); + let mut observations = Vec::new(); + let mut capability = CAPABILITIES[0].capability_id; + match recipe { + "compatible" | "missing-provider" | "missing-observation" => {} + "unavailable" => observed.available = false, + "stale-version" => "0.0.9".clone_into(&mut lock.extensions[0].version), + "future-version" => "9.0.0".clone_into(&mut lock.extensions[0].version), + "publisher-mismatch" => { + "org.example.other".clone_into(&mut lock.extensions[0].publisher_id); + } + "integrity-mismatch" => lock.extensions[0].integrity.value = "a".repeat(64), + "observation-mismatch" => observed.integrity.value = "b".repeat(64), + "future-flow" => ">=9.0.0".clone_into(&mut manifest.compatibility.flow_version_requirement), + "future-contract" => manifest + .compatibility + .contract_families + .push("flow.extension-result/v2".to_owned()), + "missing-capability" => capability = "flow/missing-fixture", + "permission-denied" => lock.extensions[0] + .granted_permissions + .filesystem_write + .clear(), + "malformed-manifest" => { + "flow.extension-manifest/v99".clone_into(&mut manifest.schema_version); + } + other => panic!("unknown resolution recipe: {other}"), + } + if recipe != "missing-provider" { + manifests.push(manifest); + } + if recipe != "missing-observation" && recipe != "missing-provider" { + observations.push(observed); + } + let catalog = ExtensionCatalog::inspect(manifests, lock, observations).unwrap(); + let result = catalog.resolve(&ResolutionRequest::new( + "acceptance-resolution", + capability, + Domain::Flow, + "hermetic-process", + ExecutionModeKind::Process, + )); + resolution_evidence(&result) +} + +fn resolution_evidence(result: &ResolutionOutcome) -> (Outcome, Value) { + result.evidence().validate().unwrap(); + let (code, state) = match result.evidence().result { + ResolutionResult::Selected => { + assert!(result.resolved().is_some()); + ("selected", ScenarioTerminalState::Complete) + } + ResolutionResult::Blocked => ("blocked", ScenarioTerminalState::Unavailable), + ResolutionResult::NoCompatibleProvider => { + ("no-compatible-provider", ScenarioTerminalState::Unsupported) + } + ResolutionResult::Malformed => ("malformed", ScenarioTerminalState::Invalid), + ResolutionResult::Conflict => ("conflict", ScenarioTerminalState::Invalid), + }; + if code != "selected" { + assert!(result.resolved().is_none()); + } + ( + outcome("resolution", code, state), + serde_json::to_value(result.evidence()).unwrap(), + ) +} + +fn policy_case(fixture: &KitFixture, recipe: &str) -> (Outcome, Value) { + let primary = fixture.manifest.clone(); + let mut secondary = primary.clone(); + "org.egohygiene.synthetic-fallback".clone_into(&mut secondary.extension_id); + let mut lock = fixture.lock.clone(); + let mut second_lock = lock.extensions[0].clone(); + second_lock.extension_id.clone_from(&secondary.extension_id); + second_lock.precedence = if recipe == "conflict" { 100 } else { 50 }; + lock.extensions.push(second_lock); + let policy = match recipe { + "fallback-forbidden" | "conflict" => FallbackPolicy::Forbidden, + "fallback-explicit-resume" => FallbackPolicy::ExplicitResume, + "fallback-before-effects" | "fallback-after-invocation" => FallbackPolicy::BeforeEffects, + _ => unreachable!(), + }; + for item in &mut lock.capability_resolution { + item.ordered_extensions.push(secondary.extension_id.clone()); + item.fallback = policy; + } + let available = matches!(recipe, "conflict" | "fallback-after-invocation"); + let observations = vec![ + common::observation(&primary, available), + common::observation(&secondary, true), + ]; + let request = ResolutionRequest::new( + "acceptance-policy", + CAPABILITIES[0].capability_id, + Domain::Flow, + "hermetic-process", + ExecutionModeKind::Process, + ); + let forward = ExtensionCatalog::inspect( + [primary.clone(), secondary.clone()], + lock.clone(), + observations.clone(), + ) + .unwrap() + .resolve(&request); + let reverse = + ExtensionCatalog::inspect([secondary, primary], lock, observations.into_iter().rev()) + .unwrap() + .resolve(&request); + assert_eq!(forward.evidence(), reverse.evidence()); + if recipe == "fallback-before-effects" { + assert_eq!( + forward.resolved().unwrap().extension_id(), + "org.egohygiene.synthetic-fallback" + ); + assert_eq!(forward.evidence().fallback_order.len(), 2); + } + if recipe == "fallback-after-invocation" { + // Exercise a selected before-effects policy against a real failing process. + let mut prepared = + fixture.prepare_lifecycle(CAPABILITIES[0], "nonzero-after-success", false); + prepared.resolution = forward; + assert_eq!( + prepared.resolved().fallback_policy(), + FallbackPolicy::BeforeEffects + ); + let error = execute(fixture, &prepared).unwrap_err(); + assert!( + fixture.output_path().is_file(), + "the selected provider actually ran" + ); + if let ProcessRunnerError::Validation { source } = error { + // The executor receives one resolved token, never a catalog to reselect. + return ( + execution_error(&source), + json!({"selected_provider": prepared.resolved().extension_id(), "fallback_policy": "before-effects", "selected_process_created_output": true}), + ); + } + panic!("unexpected runner failure: {error:?}"); + } + resolution_evidence(&forward) +} diff --git a/tools/run_acceptance_scenarios.py b/tools/run_acceptance_scenarios.py new file mode 100644 index 0000000..e2ed284 --- /dev/null +++ b/tools/run_acceptance_scenarios.py @@ -0,0 +1,128 @@ +#!/usr/bin/env python3 +"""Execute Flow #30 recipes and retain only checked, normalized portable receipts. + +This is a test driver, not a Flow runtime. Rust owns assertions and public API +execution; this driver checks completeness and writes a bounded coverage report. +""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys + +ROOT = Path(__file__).resolve().parents[1] +CATALOG = ROOT / "tests/fixtures/acceptance-scenarios.v1.json" +MARKER = "FLOW_ACCEPTANCE_RECEIPT=" +FAMILIES = {"resolution", "contract", "artifact", "provider", "privacy"} + + +def digest(data): + return hashlib.sha256(data).hexdigest() + + +def source_identity(): + """Pin the tested recipe, provider, contracts, and implementation bytes.""" + paths = {"Cargo.toml", "Cargo.lock", "LICENSE", "tests/hermetic_provider_kit.rs", + "tools/run_acceptance_scenarios.py"} + for directory in ["src", "contracts", "tests/scenario_matrix", "tests/fixtures", "tests/common"]: + paths.update(str(path.relative_to(ROOT)) for path in (ROOT / directory).rglob("*") + if path.is_file() and "__pycache__" not in path.parts) + entries = {path: digest((ROOT / path).read_bytes()) for path in sorted(paths)} + return {"algorithm": "sha256", "files": entries, + "digest": digest(json.dumps(entries, sort_keys=True, separators=(",", ":")).encode())} + + +def verify_receipts(catalog, output): + scenarios = catalog["scenarios"] + expected = {case["scenario_id"]: case for case in scenarios} + if len(expected) != len(scenarios) or not scenarios: + raise ValueError("empty catalog or duplicate scenario ID") + if {case["family"] for case in scenarios} != FAMILIES: + raise ValueError("missing or unknown scenario family") + receipts = {} + for line in output.splitlines(): + if MARKER not in line: + continue + encoded = line.split(MARKER, 1)[1] + if len(encoded.encode()) > catalog["budget"]["max_receipt_bytes"]: + raise ValueError("receipt exceeds catalog budget") + receipt = json.loads(encoded) + identifier = receipt["scenario_id"] + if identifier not in expected or identifier in receipts: + raise ValueError("unknown or duplicate receipt") + if receipt["schema_version"] != "flow.acceptance-scenario-receipt/v1": + raise ValueError("unsupported receipt version") + if receipt["outcome"] != expected[identifier]["expected"]: + raise ValueError(f"unexpected outcome: {identifier}") + if not receipt["recovery"] or not receipt["fixture_identity"]: + raise ValueError("receipt lacks recovery or fixture identity") + receipts[identifier] = receipt + if set(receipts) != set(expected): + raise ValueError(f"missing {len(set(expected) - set(receipts))} scenario receipts") + return [receipts[key] for key in sorted(receipts)] + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, default=Path("target/acceptance-scenarios.v1.report.json")) + parser.add_argument("--all-targets", action="store_true", help="Also execute the rest of the required Rust test suite.") + parser.add_argument("--cargo", default="cargo", help="Cargo executable; dependencies must already be cached.") + args = parser.parse_args() + report_path = args.output if args.output.is_absolute() else ROOT / args.output + # A failed run must never leave an older successful report at this target. + report_path.parent.mkdir(parents=True, exist_ok=True) + report_path.unlink(missing_ok=True) + report_path.with_suffix(".failure.log").unlink(missing_ok=True) + catalog = json.loads(CATALOG.read_text()) + if catalog["schema_version"] != "flow.acceptance-scenario-catalog/v1": + raise ValueError("unsupported catalog version") + if os.name != "posix": + raise ValueError("the PR matrix requires a Unix symlink-capable host") + before = source_identity() + command = [args.cargo, "test", "--locked", "--offline"] + command += ["--all-targets"] if args.all_targets else ["--test", "hermetic_provider_kit", "scenario_matrix::"] + command += ["--", "--nocapture"] + result = subprocess.run(command, cwd=ROOT, text=True, capture_output=True, timeout=600, check=False) + if result.returncode: + # Retained locally for diagnosis. Raw test/provider text is not portable. + log = report_path.with_suffix(".failure.log") + log.write_text((result.stdout + result.stderr)[-4_194_304:]) + raise ValueError(f"Rust tests failed; inspect local diagnostics at {log}") + if len(result.stdout.encode()) + len(result.stderr.encode()) > 4_194_304: + raise ValueError("test output exceeds the 4 MiB driver budget") + receipts = verify_receipts(catalog, result.stdout) + after = source_identity() + if before != after: + changed = sorted(path for path in before["files"].keys() | after["files"].keys() + if before["files"].get(path) != after["files"].get(path)) + raise ValueError(f"source bytes changed during validation: {', '.join(changed)}") + report = { + "schema_version": "flow.acceptance-scenario-report/v1", + "catalog_digest": digest(CATALOG.read_bytes()), + "source_identity": before, + "toolchain": subprocess.check_output([args.cargo, "--version"], text=True).strip(), + "tier": catalog["tier"], + "status": "passed", + "scenario_count": len(receipts), + "executions_per_scenario": catalog["budget"]["repetitions"], + "coverage": {family: sum(case["family"] == family for case in catalog["scenarios"]) + for family in sorted(FAMILIES)}, + "budgets": catalog["budget"], + "known_gaps": catalog["known_gaps"], + "receipts": receipts, + } + temporary = report_path.with_suffix(".tmp") + temporary.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + temporary.replace(report_path) + print(f"PASS: {len(receipts)} acceptance scenarios, two fresh roots each; {report_path}") + + +if __name__ == "__main__": + try: + main() + except (ValueError, OSError, subprocess.SubprocessError, KeyError) as error: + print(f"FAIL: {error}", file=sys.stderr) + sys.exit(1) diff --git a/tools/test_acceptance_report.py b/tools/test_acceptance_report.py new file mode 100644 index 0000000..47639a3 --- /dev/null +++ b/tools/test_acceptance_report.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python3 +"""Regression checks for incomplete or misleading conformance reports.""" + +import copy +import json +import unittest + +from run_acceptance_scenarios import CATALOG, MARKER, verify_receipts + + +class ReceiptCompletenessTests(unittest.TestCase): + def setUp(self): + self.catalog = json.loads(CATALOG.read_text()) + self.receipts = [ + {"schema_version": "flow.acceptance-scenario-receipt/v1", + "scenario_id": case["scenario_id"], "outcome": case["expected"], + "recovery": "Synthetic checker fixture only.", + "fixture_identity": {"package_digest": "a" * 64}} + for case in self.catalog["scenarios"] + ] + + def encode(self, receipts): + return "\n".join(MARKER + json.dumps(receipt) for receipt in receipts) + + def test_complete_out_of_order_receipts_are_sorted(self): + result = verify_receipts(self.catalog, self.encode(reversed(self.receipts))) + self.assertEqual([r["scenario_id"] for r in result], + sorted(r["scenario_id"] for r in self.receipts)) + + def test_missing_duplicate_and_unknown_receipts_cannot_claim_completion(self): + unknown = copy.deepcopy(self.receipts[0]) + unknown["scenario_id"] = "scenario:unknown" + for receipts in [[], self.receipts[:-1], self.receipts + [self.receipts[0]], + self.receipts + [unknown]]: + with self.subTest(count=len(receipts)), self.assertRaises(ValueError): + verify_receipts(self.catalog, self.encode(receipts)) + + def test_false_acceptance_unknown_version_and_missing_evidence_fail(self): + for mutate in [ + lambda r: r["outcome"].update(accepted=True), + lambda r: r.update(schema_version="flow.acceptance-scenario-receipt/v99"), + lambda r: r.update(fixture_identity={}), + lambda r: r.update(recovery=""), + ]: + receipts = copy.deepcopy(self.receipts) + mutate(receipts[0]) + with self.assertRaises(ValueError): + verify_receipts(self.catalog, self.encode(receipts)) + + def test_duplicate_catalog_ids_and_missing_families_fail(self): + for scenarios in [self.catalog["scenarios"] + [self.catalog["scenarios"][0]], + [c for c in self.catalog["scenarios"] if c["family"] != "privacy"]]: + catalog = dict(self.catalog, scenarios=scenarios) + with self.assertRaises(ValueError): + verify_receipts(catalog, self.encode(self.receipts)) + + def test_unbounded_receipt_is_rejected(self): + self.receipts[0]["recovery"] = "x" * (self.catalog["budget"]["max_receipt_bytes"] + 1) + with self.assertRaises(ValueError): + verify_receipts(self.catalog, self.encode(self.receipts)) + + +if __name__ == "__main__": + unittest.main()