From 8d5f22e65ed0eb03a355547fe2a8fc70dd75a2d5 Mon Sep 17 00:00:00 2001 From: Alan Szmyt Date: Fri, 25 Sep 2026 22:21:34 -0400 Subject: [PATCH 1/2] feat(flow): persist durable prepared execution state Record versioned intent, authority, accepted checkpoints, and explicit recovery choices in locked immutable local snapshots. Refuse stale evidence and unsafe reopen; prove interruption boundaries and portable locking in CI. Closes #49 Roadmap-Step: FLO-Q03 --- .github/workflows/ci.yml | 27 + Cargo.lock | 33 + Cargo.toml | 1 + README.md | 8 +- ROADMAP.md | 25 +- contracts/README.md | 21 +- contracts/contract-set.v1.json | 69 +- .../examples/run-artifact.v1.example.json | 14 + .../examples/run-authority.v1.example.json | 10 + .../examples/run-checkpoint.v1.example.json | 66 ++ contracts/examples/run-plan.v1.example.json | 61 ++ .../examples/run-recovery.v1.example.json | 10 + .../examples/run-snapshot.v1.example.json | 82 +++ contracts/examples/run-state.v1.example.json | 78 +++ .../examples/run-validation.v1.example.json | 10 + .../fixtures/state/completed.v1.fixture.json | 154 +++++ .../state/run-artifact.v2.invalid.json | 14 + .../state/run-authority.v2.invalid.json | 10 + .../state/run-checkpoint.v2.invalid.json | 66 ++ .../fixtures/state/run-plan.v2.invalid.json | 61 ++ .../state/run-recovery.v2.invalid.json | 10 + .../state/run-snapshot.v2.invalid.json | 82 +++ .../fixtures/state/run-state.v2.invalid.json | 78 +++ .../state/run-validation.v2.invalid.json | 10 + contracts/schemas/run-artifact.v1.schema.json | 26 + .../schemas/run-authority.v1.schema.json | 52 ++ .../schemas/run-checkpoint.v1.schema.json | 50 ++ contracts/schemas/run-plan.v1.schema.json | 147 +++++ contracts/schemas/run-recovery.v1.schema.json | 56 ++ contracts/schemas/run-snapshot.v1.schema.json | 24 + contracts/schemas/run-state.v1.schema.json | 117 ++++ .../schemas/run-validation.v1.schema.json | 51 ++ docs/architecture/foundation/ARCHITECTURE.md | 36 +- docs/architecture/governance/DECISIONS.md | 7 +- .../decisions/ADR-0011-durable-run-state.md | 118 ++++ docs/integrations/durable-state.md | 149 +++++ docs/integrations/process-runner.md | 4 + src/lib.rs | 5 +- src/state/execution.rs | 559 ++++++++++++++++ src/state/mod.rs | 130 ++++ src/state/model.rs | 582 +++++++++++++++++ src/state/store.rs | 452 +++++++++++++ src/state/store_tests.rs | 52 ++ tests/durable_execution/mod.rs | 596 ++++++++++++++++++ tests/durable_state.rs | 369 +++++++++++ tests/hermetic_provider_kit.rs | 3 + tools/run_acceptance_scenarios.py | 3 +- tools/test_durable_contracts.py | 52 ++ tools/validate_contracts.py | 39 +- 49 files changed, 4651 insertions(+), 28 deletions(-) create mode 100644 contracts/examples/run-artifact.v1.example.json create mode 100644 contracts/examples/run-authority.v1.example.json create mode 100644 contracts/examples/run-checkpoint.v1.example.json create mode 100644 contracts/examples/run-plan.v1.example.json create mode 100644 contracts/examples/run-recovery.v1.example.json create mode 100644 contracts/examples/run-snapshot.v1.example.json create mode 100644 contracts/examples/run-state.v1.example.json create mode 100644 contracts/examples/run-validation.v1.example.json create mode 100644 contracts/fixtures/state/completed.v1.fixture.json create mode 100644 contracts/fixtures/state/run-artifact.v2.invalid.json create mode 100644 contracts/fixtures/state/run-authority.v2.invalid.json create mode 100644 contracts/fixtures/state/run-checkpoint.v2.invalid.json create mode 100644 contracts/fixtures/state/run-plan.v2.invalid.json create mode 100644 contracts/fixtures/state/run-recovery.v2.invalid.json create mode 100644 contracts/fixtures/state/run-snapshot.v2.invalid.json create mode 100644 contracts/fixtures/state/run-state.v2.invalid.json create mode 100644 contracts/fixtures/state/run-validation.v2.invalid.json create mode 100644 contracts/schemas/run-artifact.v1.schema.json create mode 100644 contracts/schemas/run-authority.v1.schema.json create mode 100644 contracts/schemas/run-checkpoint.v1.schema.json create mode 100644 contracts/schemas/run-plan.v1.schema.json create mode 100644 contracts/schemas/run-recovery.v1.schema.json create mode 100644 contracts/schemas/run-snapshot.v1.schema.json create mode 100644 contracts/schemas/run-state.v1.schema.json create mode 100644 contracts/schemas/run-validation.v1.schema.json create mode 100644 docs/architecture/governance/decisions/ADR-0011-durable-run-state.md create mode 100644 docs/integrations/durable-state.md create mode 100644 src/state/execution.rs create mode 100644 src/state/mod.rs create mode 100644 src/state/model.rs create mode 100644 src/state/store.rs create mode 100644 src/state/store_tests.rs create mode 100644 tests/durable_execution/mod.rs create mode 100644 tests/durable_state.rs create mode 100644 tools/test_durable_contracts.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e558688..81b1a49 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -93,6 +93,9 @@ jobs: - name: Reject incomplete acceptance reports run: python3 tools/test_acceptance_report.py + - name: Check durable contract references and closed shapes + run: python3 tools/test_durable_contracts.py + - name: Validate architecture specifications run: python3 .agents/specs/validate-specs.py @@ -101,3 +104,27 @@ jobs: - name: Validate repository agents run: python3 .agents/agents/validate-agents.py + + durable-portability: + name: Durable state ${{ matrix.os }} + runs-on: ${{ matrix.os }} + timeout-minutes: 10 + strategy: + fail-fast: false + matrix: + os: [macos-latest, windows-latest] + steps: + - name: Check out repository + uses: actions/checkout@v4 + + - name: Install stable Rust + uses: dtolnay/rust-toolchain@stable + + - name: Cache Cargo data + uses: Swatinem/rust-cache@v2 + + - name: Verify atomic commit interruption boundaries + run: cargo test --lib state::store::tests --locked + + - name: Verify portable state, process crashes, and workspace locking + run: cargo test --test durable_state --locked diff --git a/Cargo.lock b/Cargo.lock index 4874f75..1909d98 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -62,6 +62,7 @@ dependencies = [ name = "flow" version = "0.1.0" dependencies = [ + "fs2", "nix", "semver", "serde", @@ -70,6 +71,16 @@ dependencies = [ "thiserror", ] +[[package]] +name = "fs2" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9564fc758e15025b46aa6643b1b77d047d1a56a1aea6e01002ac0c7026876213" +dependencies = [ + "libc", + "winapi", +] + [[package]] name = "generic-array" version = "0.14.7" @@ -237,6 +248,28 @@ version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" +[[package]] +name = "winapi" +version = "0.3.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419" +dependencies = [ + "winapi-i686-pc-windows-gnu", + "winapi-x86_64-pc-windows-gnu", +] + +[[package]] +name = "winapi-i686-pc-windows-gnu" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6" + +[[package]] +name = "winapi-x86_64-pc-windows-gnu" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" + [[package]] name = "zmij" version = "1.0.23" diff --git a/Cargo.toml b/Cargo.toml index 2c2b761..706f3ac 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -15,6 +15,7 @@ test = false bench = false [dependencies] +fs2 = "0.4.3" semver = "1.0.26" serde = { version = "1.0.219", features = ["derive"] } serde_json = "1.0.140" diff --git a/README.md b/README.md index df55b0c..9e3fb06 100644 --- a/README.md +++ b/README.md @@ -27,7 +27,10 @@ subjects; require exact correlated process authority/isolation evidence; accept explicitly bound artifacts; and validate a closed scenario manifest for synthetic orchestration fixtures. It does not copy holon source or claim a product orchestrator, CLI, real provider adapter, operating-system sandbox, -scenario runner, durable run state, or resume support. +general scenario executor, automatic recovery scheduler, or public resume CLI. +Prepared process execution now has [durable run state](docs/integrations/durable-state.md) +with immutable plans, atomic snapshots, verified checkpoints, status inspection, +and explicit recovery decisions. ## Executable checkpoint @@ -58,6 +61,8 @@ binary is a synthetic conformance provider; it is not a public Flow CLI: - artifact bindings map immutable input and candidate-output IDs to portable root-relative locators; Flow observes file/directory bytes beneath one root and returns an accepted set only after exact host/provider correlation; +- `RunStore` persists prepared intent and authority before launch, records accepted + checkpoints after validation, and reopens with typed stale/corrupt-state refusal; - `flow.scenario-manifest/v1` pins synthetic inputs, providers, topology, expectations, resource budgets, coverage gaps, and cross-language canonical digests without defining plans or runs; and @@ -75,6 +80,7 @@ cargo run --example hermetic_extension --locked cargo run --example hermetic_process_transport --locked cargo run --example scenario_manifest --locked cargo test --test hermetic_provider_kit --locked +cargo test --test durable_state --locked ``` `EventSink` is a fallible, authoritative execution observer, not a best-effort diff --git a/ROADMAP.md b/ROADMAP.md index cb66b0a..776865c 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -30,18 +30,19 @@ supersedes: [] > older active-checkpoint and queue text below where they conflict. Re-query > live issue, release, and CI state before starting a branch. -PR #60 merged on 2026-09-24 as -[`c653c3667dd1879bd7009f83a4906ab6ae9ba832`](https://github.com/egohygiene/flow/commit/c653c3667dd1879bd7009f83a4906ab6ae9ba832), -completing the hermetic provider-kit closeout. The exact next Flow checkpoint is -[#30](https://github.com/egohygiene/flow/issues/30). - -The #30 review candidate adds the -[executable acceptance matrix](docs/integrations/acceptance-scenarios.md): -versioned deterministic recipes, exact typed outcomes, two fresh-root receipts -per case, PR budgets, and machine-readable coverage/gaps. Its scope ends at -single-execution acceptance and refusal. After this candidate merges and the -default-branch gate passes, #49 is next. FLO-Q03 remains active until the later -real released-provider adapters satisfy its two-adapter exit criterion. +#30 is complete: [PR #61](https://github.com/egohygiene/flow/pull/61) merged as +`dfb16b347975e3292dc2928c8c48463f51dcf6d1`, with green +[default-branch CI](https://github.com/egohygiene/flow/actions/runs/36209682152). +Its [acceptance matrix](docs/integrations/acceptance-scenarios.md) proves 81 +scenarios twice on Rust 1.85 and stable. + +The #49 review candidate adds [durable prepared execution](docs/integrations/durable-state.md): +versioned plans and state, atomic immutable snapshots, workspace locking, +accepted checkpoints, fresh resume eligibility, and explicit recovery decisions. +After this candidate merges and the default-branch gate passes, #31 is next. +FLO-Q03 remains active until real released-provider adapters satisfy its +remaining exit criteria. [#62](https://github.com/egohygiene/flow/issues/62) is +later documentation visualization work and does not block this sequence. ### Central Flow chain diff --git a/contracts/README.md b/contracts/README.md index 1611ac6..8db5cf2 100644 --- a/contracts/README.md +++ b/contracts/README.md @@ -1,7 +1,7 @@ # Flow contract set This directory contains Flow-owned suite interchange contracts. The initial -contract set is version `0.6.0`, status `provisional`, in the v1 compatibility +contract set is version `0.7.0`, status `provisional`, in the v1 compatibility family. Provisional means versioned and testable, not stable for production. | Contract | Purpose | @@ -22,6 +22,14 @@ family. Provisional means versioned and testable, not stable for production. | `flow.extension-result/v1` | partial/final outcomes, failures, validation, provenance, and explanation | | `flow.extension-resolution/v1` | deterministic selection, rejection, conflict, and fallback evidence | | `flow.scenario-manifest/v1` | stable scenario identity, immutable fixture topology, typed expectations, execution budgets, and bounded coverage claims | +| `flow.run-plan/v1` | immutable prepared execution intent and exact step context | +| `flow.run-state/v1` | durable step state, decisions, checkpoints, and predecessor identity | +| `flow.run-checkpoint/v1` | accepted artifacts and validation bound to plan, context, and attempt | +| `flow.run-artifact/v1` | input/output role and retained host observation | +| `flow.run-authority/v1` | launch grant or explicit operator denial | +| `flow.run-validation/v1` | accepted evidence and validator implementation identities | +| `flow.run-recovery/v1` | explicit retry/abandon decision and uncertainty acknowledgement | +| `flow.run-snapshot/v1` | atomic-storage envelope with state integrity digest | `contract-set.v1.json` is the machine-readable index. Schemas live in `schemas/`; deterministic examples live in `examples/`; extension compatibility @@ -79,6 +87,13 @@ The Rust and Python implementations independently reproduce the checked-in SHA-256 catalog. CI validates drift but never regenerates or rewrites canonical fixtures. +The durable state contracts are described in +[`docs/integrations/durable-state.md`](../docs/integrations/durable-state.md). +Their schemas reuse repository-local schema references; the validator resolves +these offline and never retrieves arbitrary URLs. The Rust state/store validators +also enforce context, digest, and transition invariants. Examples under +`fixtures/state/` are synthetic data, not current execution attestations. + `compatibility.flow_version_requirement` is parsed with Rust's `semver` `VersionReq` grammar and must use comma-separated comparators. For example, `>=0.1.0, <0.2.0` is a bounded range; `>=0.1.0 <0.2.0` is invalid. Invalid or @@ -100,7 +115,9 @@ package/executable observer, authority preflight, and bounded `trusted-unconfined` local runner are not real provider adapters. They do not prove publisher authenticity, signatures, transparency, domain-output validity, operating-system sandboxing, authenticated host evidence, -descriptor-bound launch, descendant containment, checkpoints, or resume. +descriptor-bound launch, or descendant containment. The durable coordinator adds +Flow-owned checkpoints and resume eligibility; provider-native checkpoint restore, +automatic recovery scheduling, and real-provider lifecycle proof remain later work. For this checkpoint, the caller owns configuration canonicalization and digest generation plus authorization issuance, authorization ID, and grants digest. diff --git a/contracts/contract-set.v1.json b/contracts/contract-set.v1.json index 5931560..d865957 100644 --- a/contracts/contract-set.v1.json +++ b/contracts/contract-set.v1.json @@ -1,6 +1,6 @@ { "schema_version": "flow.contract-set/v1", - "contract_set_version": "0.6.0", + "contract_set_version": "0.7.0", "status": "provisional", "contracts": [ { @@ -124,6 +124,73 @@ "fixtures/scenarios/mutable-reference.v1.invalid.json", "fixtures/scenarios/unknown-version.v1.invalid.json" ] + }, + { + "id": "flow.run-plan/v1", + "schema": "schemas/run-plan.v1.schema.json", + "example": "examples/run-plan.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-plan.v2.invalid.json" + ] + }, + { + "id": "flow.run-artifact/v1", + "schema": "schemas/run-artifact.v1.schema.json", + "example": "examples/run-artifact.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-artifact.v2.invalid.json" + ] + }, + { + "id": "flow.run-validation/v1", + "schema": "schemas/run-validation.v1.schema.json", + "example": "examples/run-validation.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-validation.v2.invalid.json" + ] + }, + { + "id": "flow.run-checkpoint/v1", + "schema": "schemas/run-checkpoint.v1.schema.json", + "example": "examples/run-checkpoint.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-checkpoint.v2.invalid.json" + ] + }, + { + "id": "flow.run-authority/v1", + "schema": "schemas/run-authority.v1.schema.json", + "example": "examples/run-authority.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-authority.v2.invalid.json" + ] + }, + { + "id": "flow.run-recovery/v1", + "schema": "schemas/run-recovery.v1.schema.json", + "example": "examples/run-recovery.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-recovery.v2.invalid.json" + ] + }, + { + "id": "flow.run-state/v1", + "schema": "schemas/run-state.v1.schema.json", + "example": "examples/run-state.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-state.v2.invalid.json" + ], + "fixtures": [ + "fixtures/state/completed.v1.fixture.json" + ] + }, + { + "id": "flow.run-snapshot/v1", + "schema": "schemas/run-snapshot.v1.schema.json", + "example": "examples/run-snapshot.v1.example.json", + "invalid_examples": [ + "fixtures/state/run-snapshot.v2.invalid.json" + ] } ] } diff --git a/contracts/examples/run-artifact.v1.example.json b/contracts/examples/run-artifact.v1.example.json new file mode 100644 index 0000000..afcc97b --- /dev/null +++ b/contracts/examples/run-artifact.v1.example.json @@ -0,0 +1,14 @@ +{ + "schema_version": "flow.run-artifact/v1", + "role": "input", + "observation": { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "digest": "2222222222222222222222222222222222222222222222222222222222222222", + "size_bytes": 128, + "manifest": [] + } +} diff --git a/contracts/examples/run-authority.v1.example.json b/contracts/examples/run-authority.v1.example.json new file mode 100644 index 0000000..1b09140 --- /dev/null +++ b/contracts/examples/run-authority.v1.example.json @@ -0,0 +1,10 @@ +{ + "schema_version": "flow.run-authority/v1", + "step_id": "step:inspect", + "attempt": 1, + "granted": true, + "authorization_id": "authorization:durable-example", + "profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777" +} diff --git a/contracts/examples/run-checkpoint.v1.example.json b/contracts/examples/run-checkpoint.v1.example.json new file mode 100644 index 0000000..d3e896f --- /dev/null +++ b/contracts/examples/run-checkpoint.v1.example.json @@ -0,0 +1,66 @@ +{ + "schema_version": "flow.run-checkpoint/v1", + "checkpoint_id": "checkpoint:example", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "context_digest": "216e6c91a49e420820ba87e6add8824c06cc279bf413f67529f276afde1f2049", + "attempt": 1, + "artifacts": [ + { + "schema_version": "flow.run-artifact/v1", + "role": "input", + "observation": { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "digest": "2222222222222222222222222222222222222222222222222222222222222222", + "size_bytes": 128, + "manifest": [] + } + }, + { + "schema_version": "flow.run-artifact/v1", + "role": "output", + "observation": { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle", + "digest": "51495603e74938fc7b5f86c3ae1d01ba4f9aaccf29dfe8ad26ea49bbd1ca864a", + "size_bytes": 384, + "manifest": [ + { + "locator": "summary.json", + "kind": "file", + "digest": "5555555555555555555555555555555555555555555555555555555555555555", + "size_bytes": 256 + }, + { + "locator": "évidence", + "kind": "directory", + "digest": "1e1b8003aaea7b8f60b4481e20c3ad61e627d84d54d2d459109d41b3b01799c2", + "size_bytes": 128 + }, + { + "locator": "évidence/details.json", + "kind": "file", + "digest": "6666666666666666666666666666666666666666666666666666666666666666", + "size_bytes": 128 + } + ] + } + } + ], + "validation": { + "schema_version": "flow.run-validation/v1", + "profile": "flow.accept-artifacts/v1", + "implementation_version": "0.1.0", + "implementation_digest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "platform": "linux", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "execution_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "artifacts_digest": "b054d38b4f2ca943473fab2217e6386c642b3e4bce46e3b43ef07bcfabae6022" + } +} diff --git a/contracts/examples/run-plan.v1.example.json b/contracts/examples/run-plan.v1.example.json new file mode 100644 index 0000000..c21fc37 --- /dev/null +++ b/contracts/examples/run-plan.v1.example.json @@ -0,0 +1,61 @@ +{ + "schema_version": "flow.run-plan/v1", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] +} diff --git a/contracts/examples/run-recovery.v1.example.json b/contracts/examples/run-recovery.v1.example.json new file mode 100644 index 0000000..607ed35 --- /dev/null +++ b/contracts/examples/run-recovery.v1.example.json @@ -0,0 +1,10 @@ +{ + "schema_version": "flow.run-recovery/v1", + "decision_id": "decision:retry-example", + "step_id": "step:inspect", + "attempt": 1, + "action": "retry", + "authorization_id": "authorization:durable-example", + "profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "acknowledged_uncertain_effects": true +} diff --git a/contracts/examples/run-snapshot.v1.example.json b/contracts/examples/run-snapshot.v1.example.json new file mode 100644 index 0000000..8bb354f --- /dev/null +++ b/contracts/examples/run-snapshot.v1.example.json @@ -0,0 +1,82 @@ +{ + "schema_version": "flow.run-snapshot/v1", + "state_digest": "8796604491ccdbab553b73b319608fbbc824da503645927e810b7577a40442d3", + "state": { + "schema_version": "flow.run-state/v1", + "sequence": 0, + "previous_digest": "", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "plan": { + "schema_version": "flow.run-plan/v1", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] + }, + "steps": [ + { + "step_id": "step:inspect", + "attempt": 0, + "status": "pending", + "checkpoint": null, + "failure": null + } + ], + "authority_decisions": [], + "recovery_decisions": [] + } +} diff --git a/contracts/examples/run-state.v1.example.json b/contracts/examples/run-state.v1.example.json new file mode 100644 index 0000000..420d933 --- /dev/null +++ b/contracts/examples/run-state.v1.example.json @@ -0,0 +1,78 @@ +{ + "schema_version": "flow.run-state/v1", + "sequence": 0, + "previous_digest": "", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "plan": { + "schema_version": "flow.run-plan/v1", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] + }, + "steps": [ + { + "step_id": "step:inspect", + "attempt": 0, + "status": "pending", + "checkpoint": null, + "failure": null + } + ], + "authority_decisions": [], + "recovery_decisions": [] +} diff --git a/contracts/examples/run-validation.v1.example.json b/contracts/examples/run-validation.v1.example.json new file mode 100644 index 0000000..cf06a61 --- /dev/null +++ b/contracts/examples/run-validation.v1.example.json @@ -0,0 +1,10 @@ +{ + "schema_version": "flow.run-validation/v1", + "profile": "flow.accept-artifacts/v1", + "implementation_version": "0.1.0", + "implementation_digest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "platform": "linux", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "execution_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "artifacts_digest": "b054d38b4f2ca943473fab2217e6386c642b3e4bce46e3b43ef07bcfabae6022" +} diff --git a/contracts/fixtures/state/completed.v1.fixture.json b/contracts/fixtures/state/completed.v1.fixture.json new file mode 100644 index 0000000..acfbc5e --- /dev/null +++ b/contracts/fixtures/state/completed.v1.fixture.json @@ -0,0 +1,154 @@ +{ + "schema_version": "flow.run-state/v1", + "sequence": 2, + "previous_digest": "cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "plan": { + "schema_version": "flow.run-plan/v1", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] + }, + "steps": [ + { + "step_id": "step:inspect", + "attempt": 1, + "status": "succeeded", + "checkpoint": { + "schema_version": "flow.run-checkpoint/v1", + "checkpoint_id": "checkpoint:example", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "context_digest": "216e6c91a49e420820ba87e6add8824c06cc279bf413f67529f276afde1f2049", + "attempt": 1, + "artifacts": [ + { + "schema_version": "flow.run-artifact/v1", + "role": "input", + "observation": { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "digest": "2222222222222222222222222222222222222222222222222222222222222222", + "size_bytes": 128, + "manifest": [] + } + }, + { + "schema_version": "flow.run-artifact/v1", + "role": "output", + "observation": { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle", + "digest": "51495603e74938fc7b5f86c3ae1d01ba4f9aaccf29dfe8ad26ea49bbd1ca864a", + "size_bytes": 384, + "manifest": [ + { + "locator": "summary.json", + "kind": "file", + "digest": "5555555555555555555555555555555555555555555555555555555555555555", + "size_bytes": 256 + }, + { + "locator": "évidence", + "kind": "directory", + "digest": "1e1b8003aaea7b8f60b4481e20c3ad61e627d84d54d2d459109d41b3b01799c2", + "size_bytes": 128 + }, + { + "locator": "évidence/details.json", + "kind": "file", + "digest": "6666666666666666666666666666666666666666666666666666666666666666", + "size_bytes": 128 + } + ] + } + } + ], + "validation": { + "schema_version": "flow.run-validation/v1", + "profile": "flow.accept-artifacts/v1", + "implementation_version": "0.1.0", + "implementation_digest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "platform": "linux", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "execution_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "artifacts_digest": "b054d38b4f2ca943473fab2217e6386c642b3e4bce46e3b43ef07bcfabae6022" + } + }, + "failure": null + } + ], + "authority_decisions": [ + { + "schema_version": "flow.run-authority/v1", + "step_id": "step:inspect", + "attempt": 1, + "granted": true, + "authorization_id": "authorization:durable-example", + "profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777" + } + ], + "recovery_decisions": [] +} diff --git a/contracts/fixtures/state/run-artifact.v2.invalid.json b/contracts/fixtures/state/run-artifact.v2.invalid.json new file mode 100644 index 0000000..1e7f3bb --- /dev/null +++ b/contracts/fixtures/state/run-artifact.v2.invalid.json @@ -0,0 +1,14 @@ +{ + "schema_version": "flow.run-artifact/v2", + "role": "input", + "observation": { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "digest": "2222222222222222222222222222222222222222222222222222222222222222", + "size_bytes": 128, + "manifest": [] + } +} diff --git a/contracts/fixtures/state/run-authority.v2.invalid.json b/contracts/fixtures/state/run-authority.v2.invalid.json new file mode 100644 index 0000000..1a0de43 --- /dev/null +++ b/contracts/fixtures/state/run-authority.v2.invalid.json @@ -0,0 +1,10 @@ +{ + "schema_version": "flow.run-authority/v2", + "step_id": "step:inspect", + "attempt": 1, + "granted": true, + "authorization_id": "authorization:durable-example", + "profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777" +} diff --git a/contracts/fixtures/state/run-checkpoint.v2.invalid.json b/contracts/fixtures/state/run-checkpoint.v2.invalid.json new file mode 100644 index 0000000..14237c5 --- /dev/null +++ b/contracts/fixtures/state/run-checkpoint.v2.invalid.json @@ -0,0 +1,66 @@ +{ + "schema_version": "flow.run-checkpoint/v2", + "checkpoint_id": "checkpoint:example", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "context_digest": "216e6c91a49e420820ba87e6add8824c06cc279bf413f67529f276afde1f2049", + "attempt": 1, + "artifacts": [ + { + "schema_version": "flow.run-artifact/v1", + "role": "input", + "observation": { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "digest": "2222222222222222222222222222222222222222222222222222222222222222", + "size_bytes": 128, + "manifest": [] + } + }, + { + "schema_version": "flow.run-artifact/v1", + "role": "output", + "observation": { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle", + "digest": "51495603e74938fc7b5f86c3ae1d01ba4f9aaccf29dfe8ad26ea49bbd1ca864a", + "size_bytes": 384, + "manifest": [ + { + "locator": "summary.json", + "kind": "file", + "digest": "5555555555555555555555555555555555555555555555555555555555555555", + "size_bytes": 256 + }, + { + "locator": "évidence", + "kind": "directory", + "digest": "1e1b8003aaea7b8f60b4481e20c3ad61e627d84d54d2d459109d41b3b01799c2", + "size_bytes": 128 + }, + { + "locator": "évidence/details.json", + "kind": "file", + "digest": "6666666666666666666666666666666666666666666666666666666666666666", + "size_bytes": 128 + } + ] + } + } + ], + "validation": { + "schema_version": "flow.run-validation/v1", + "profile": "flow.accept-artifacts/v1", + "implementation_version": "0.1.0", + "implementation_digest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "platform": "linux", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "execution_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "artifacts_digest": "b054d38b4f2ca943473fab2217e6386c642b3e4bce46e3b43ef07bcfabae6022" + } +} diff --git a/contracts/fixtures/state/run-plan.v2.invalid.json b/contracts/fixtures/state/run-plan.v2.invalid.json new file mode 100644 index 0000000..9fcf957 --- /dev/null +++ b/contracts/fixtures/state/run-plan.v2.invalid.json @@ -0,0 +1,61 @@ +{ + "schema_version": "flow.run-plan/v2", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] +} diff --git a/contracts/fixtures/state/run-recovery.v2.invalid.json b/contracts/fixtures/state/run-recovery.v2.invalid.json new file mode 100644 index 0000000..c99a634 --- /dev/null +++ b/contracts/fixtures/state/run-recovery.v2.invalid.json @@ -0,0 +1,10 @@ +{ + "schema_version": "flow.run-recovery/v2", + "decision_id": "decision:retry-example", + "step_id": "step:inspect", + "attempt": 1, + "action": "retry", + "authorization_id": "authorization:durable-example", + "profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "acknowledged_uncertain_effects": true +} diff --git a/contracts/fixtures/state/run-snapshot.v2.invalid.json b/contracts/fixtures/state/run-snapshot.v2.invalid.json new file mode 100644 index 0000000..018ff56 --- /dev/null +++ b/contracts/fixtures/state/run-snapshot.v2.invalid.json @@ -0,0 +1,82 @@ +{ + "schema_version": "flow.run-snapshot/v2", + "state_digest": "8796604491ccdbab553b73b319608fbbc824da503645927e810b7577a40442d3", + "state": { + "schema_version": "flow.run-state/v1", + "sequence": 0, + "previous_digest": "", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "plan": { + "schema_version": "flow.run-plan/v1", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] + }, + "steps": [ + { + "step_id": "step:inspect", + "attempt": 0, + "status": "pending", + "checkpoint": null, + "failure": null + } + ], + "authority_decisions": [], + "recovery_decisions": [] + } +} diff --git a/contracts/fixtures/state/run-state.v2.invalid.json b/contracts/fixtures/state/run-state.v2.invalid.json new file mode 100644 index 0000000..52f72c7 --- /dev/null +++ b/contracts/fixtures/state/run-state.v2.invalid.json @@ -0,0 +1,78 @@ +{ + "schema_version": "flow.run-state/v2", + "sequence": 0, + "previous_digest": "", + "plan_digest": "0e91d3a2af624fd222d6cf2aa1b005b8d9b3ad2f3e9951cde35b2e9e99b102d1", + "plan": { + "schema_version": "flow.run-plan/v1", + "plan_id": "plan:durable-example", + "run_id": "run:durable-example", + "steps": [ + { + "step_id": "step:inspect", + "depends_on": [], + "context": { + "run_id": "run:durable-example", + "invocation_id": "invocation:durable-example", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "provider": { + "extension_id": "org.egohygiene.synthetic-adapters", + "version": "0.1.0", + "publisher_id": "org.egohygiene", + "integrity": "1111111111111111111111111111111111111111111111111111111111111111" + }, + "interface": { + "kind": "process", + "name": "synthetic-process", + "protocol": "flow.extension-invocation/v1" + }, + "capability_id": "optiflow/inspect-collection", + "capability_digest": "2222222222222222222222222222222222222222222222222222222222222222", + "configuration_schema": "synthetic.optiflow-inspect/v1", + "configuration_digest": "3333333333333333333333333333333333333333333333333333333333333333", + "configuration_values_digest": "44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a", + "subject_lock_digest": "4444444444444444444444444444444444444444444444444444444444444444", + "authorization_id": "authorization:durable-example", + "authority_profile_digest": "5555555555555555555555555555555555555555555555555555555555555555", + "enforcement_evidence_digest": "6666666666666666666666666666666666666666666666666666666666666666", + "grants_digest": "7777777777777777777777777777777777777777777777777777777777777777", + "bindings": { + "schema_version": "flow.artifact-bindings/v1", + "binding_set_id": "bindings:synthetic-optiflow-scan", + "digest_algorithm": "sha256", + "inputs": [ + { + "artifact_id": "artifact:synthetic-collection", + "port": "port:collection", + "media_type": "application/vnd.flow.collection-reference+json", + "kind": "file", + "locator": "inputs/source collection.json", + "expected_digest": "2222222222222222222222222222222222222222222222222222222222222222" + } + ], + "outputs": [ + { + "artifact_id": "artifact:synthetic-collection-evidence", + "port": "port:evidence", + "media_type": "application/vnd.optiflow.collection-evidence+json", + "kind": "directory", + "locator": "outputs/evidence bundle" + } + ] + } + } + } + ] + }, + "steps": [ + { + "step_id": "step:inspect", + "attempt": 0, + "status": "pending", + "checkpoint": null, + "failure": null + } + ], + "authority_decisions": [], + "recovery_decisions": [] +} diff --git a/contracts/fixtures/state/run-validation.v2.invalid.json b/contracts/fixtures/state/run-validation.v2.invalid.json new file mode 100644 index 0000000..3f2fd51 --- /dev/null +++ b/contracts/fixtures/state/run-validation.v2.invalid.json @@ -0,0 +1,10 @@ +{ + "schema_version": "flow.run-validation/v2", + "profile": "flow.accept-artifacts/v1", + "implementation_version": "0.1.0", + "implementation_digest": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "platform": "linux", + "invocation_digest": "1111111111111111111111111111111111111111111111111111111111111111", + "execution_digest": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "artifacts_digest": "b054d38b4f2ca943473fab2217e6386c642b3e4bce46e3b43ef07bcfabae6022" +} diff --git a/contracts/schemas/run-artifact.v1.schema.json b/contracts/schemas/run-artifact.v1.schema.json new file mode 100644 index 0000000..fcdb451 --- /dev/null +++ b/contracts/schemas/run-artifact.v1.schema.json @@ -0,0 +1,26 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-artifact.v1.schema.json", + "title": "Flow run artifact v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "role", + "observation" + ], + "properties": { + "schema_version": { + "const": "flow.run-artifact/v1" + }, + "role": { + "enum": [ + "input", + "output" + ] + }, + "observation": { + "$ref": "artifact-observations.v1.schema.json#/properties/artifacts/items" + } + } +} diff --git a/contracts/schemas/run-authority.v1.schema.json b/contracts/schemas/run-authority.v1.schema.json new file mode 100644 index 0000000..717b309 --- /dev/null +++ b/contracts/schemas/run-authority.v1.schema.json @@ -0,0 +1,52 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-authority.v1.schema.json", + "title": "Flow run authority v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "step_id", + "attempt", + "granted", + "authorization_id", + "profile_digest", + "enforcement_digest", + "grants_digest" + ], + "properties": { + "schema_version": { + "const": "flow.run-authority/v1" + }, + "step_id": { + "type": "string", + "maxLength": 160, + "pattern": "^step:[a-z0-9][a-z0-9._-]*$" + }, + "attempt": { + "type": "integer", + "minimum": 0, + "maximum": 4095 + }, + "granted": { + "type": "boolean" + }, + "authorization_id": { + "type": "string", + "maxLength": 160, + "pattern": "^authorization:[a-z0-9][a-z0-9._-]*$" + }, + "profile_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "enforcement_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "grants_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + } + } +} diff --git a/contracts/schemas/run-checkpoint.v1.schema.json b/contracts/schemas/run-checkpoint.v1.schema.json new file mode 100644 index 0000000..0a7181e --- /dev/null +++ b/contracts/schemas/run-checkpoint.v1.schema.json @@ -0,0 +1,50 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-checkpoint.v1.schema.json", + "title": "Flow run checkpoint v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "checkpoint_id", + "plan_digest", + "context_digest", + "attempt", + "artifacts", + "validation" + ], + "properties": { + "schema_version": { + "const": "flow.run-checkpoint/v1" + }, + "checkpoint_id": { + "type": "string", + "maxLength": 160, + "pattern": "^checkpoint:[a-z0-9][a-z0-9._-]*$" + }, + "plan_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "context_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "attempt": { + "type": "integer", + "minimum": 1, + "maximum": 4095 + }, + "artifacts": { + "type": "array", + "items": { + "$ref": "run-artifact.v1.schema.json" + }, + "maxItems": 4096, + "minItems": 1 + }, + "validation": { + "$ref": "run-validation.v1.schema.json" + } + } +} diff --git a/contracts/schemas/run-plan.v1.schema.json b/contracts/schemas/run-plan.v1.schema.json new file mode 100644 index 0000000..67ded36 --- /dev/null +++ b/contracts/schemas/run-plan.v1.schema.json @@ -0,0 +1,147 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-plan.v1.schema.json", + "title": "Flow run plan v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "plan_id", + "run_id", + "steps" + ], + "properties": { + "schema_version": { + "const": "flow.run-plan/v1" + }, + "plan_id": { + "type": "string", + "maxLength": 160, + "pattern": "^plan:[a-z0-9][a-z0-9._-]*$" + }, + "run_id": { + "type": "string", + "maxLength": 160, + "pattern": "^run:[a-z0-9][a-z0-9._-]*$" + }, + "steps": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "step_id", + "depends_on", + "context" + ], + "properties": { + "step_id": { + "type": "string", + "maxLength": 160, + "pattern": "^step:[a-z0-9][a-z0-9._-]*$" + }, + "depends_on": { + "type": "array", + "items": { + "type": "string", + "maxLength": 160, + "pattern": "^step:[a-z0-9][a-z0-9._-]*$" + }, + "maxItems": 256 + }, + "context": { + "type": "object", + "additionalProperties": false, + "required": [ + "run_id", + "invocation_id", + "invocation_digest", + "provider", + "interface", + "capability_id", + "capability_digest", + "configuration_schema", + "configuration_digest", + "configuration_values_digest", + "subject_lock_digest", + "authorization_id", + "authority_profile_digest", + "enforcement_evidence_digest", + "grants_digest", + "bindings" + ], + "properties": { + "run_id": { + "type": "string", + "maxLength": 160, + "pattern": "^run:[a-z0-9][a-z0-9._-]*$" + }, + "invocation_id": { + "type": "string", + "maxLength": 160, + "pattern": "^invocation:[a-z0-9][a-z0-9._-]*$" + }, + "invocation_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "provider": { + "$ref": "extension-invocation.v1.schema.json#/properties/extension" + }, + "interface": { + "$ref": "extension-invocation.v1.schema.json#/properties/interface" + }, + "capability_id": { + "type": "string", + "pattern": "^[a-z][a-z0-9-]*/[a-z][a-z0-9-]*$" + }, + "capability_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "configuration_schema": { + "type": "string", + "minLength": 1, + "maxLength": 256 + }, + "configuration_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "configuration_values_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "subject_lock_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "authorization_id": { + "type": "string", + "maxLength": 160, + "pattern": "^authorization:[a-z0-9][a-z0-9._-]*$" + }, + "authority_profile_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "enforcement_evidence_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "grants_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "bindings": { + "$ref": "artifact-bindings.v1.schema.json" + } + } + } + } + }, + "maxItems": 256, + "minItems": 1 + } + } +} diff --git a/contracts/schemas/run-recovery.v1.schema.json b/contracts/schemas/run-recovery.v1.schema.json new file mode 100644 index 0000000..b28fe48 --- /dev/null +++ b/contracts/schemas/run-recovery.v1.schema.json @@ -0,0 +1,56 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-recovery.v1.schema.json", + "title": "Flow run recovery v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "decision_id", + "step_id", + "attempt", + "action", + "authorization_id", + "profile_digest", + "acknowledged_uncertain_effects" + ], + "properties": { + "schema_version": { + "const": "flow.run-recovery/v1" + }, + "decision_id": { + "type": "string", + "maxLength": 160, + "pattern": "^decision:[a-z0-9][a-z0-9._-]*$" + }, + "step_id": { + "type": "string", + "maxLength": 160, + "pattern": "^step:[a-z0-9][a-z0-9._-]*$" + }, + "attempt": { + "type": "integer", + "minimum": 0, + "maximum": 4095 + }, + "action": { + "enum": [ + "retry", + "abandon" + ] + }, + "authorization_id": { + "type": "string", + "maxLength": 160, + "pattern": "^authorization:[a-z0-9][a-z0-9._-]*$" + }, + "profile_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "acknowledged_uncertain_effects": { + "type": "boolean", + "const": true + } + } +} diff --git a/contracts/schemas/run-snapshot.v1.schema.json b/contracts/schemas/run-snapshot.v1.schema.json new file mode 100644 index 0000000..2b0d256 --- /dev/null +++ b/contracts/schemas/run-snapshot.v1.schema.json @@ -0,0 +1,24 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-snapshot.v1.schema.json", + "title": "Flow run snapshot v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "state_digest", + "state" + ], + "properties": { + "schema_version": { + "const": "flow.run-snapshot/v1" + }, + "state_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "state": { + "$ref": "run-state.v1.schema.json" + } + } +} diff --git a/contracts/schemas/run-state.v1.schema.json b/contracts/schemas/run-state.v1.schema.json new file mode 100644 index 0000000..f3ab83a --- /dev/null +++ b/contracts/schemas/run-state.v1.schema.json @@ -0,0 +1,117 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-state.v1.schema.json", + "title": "Flow run state v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "sequence", + "previous_digest", + "plan_digest", + "plan", + "steps", + "authority_decisions", + "recovery_decisions" + ], + "properties": { + "schema_version": { + "const": "flow.run-state/v1" + }, + "sequence": { + "type": "integer", + "minimum": 0, + "maximum": 4095 + }, + "previous_digest": { + "type": "string", + "pattern": "^([a-f0-9]{64})?$" + }, + "plan_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "plan": { + "$ref": "run-plan.v1.schema.json" + }, + "steps": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "step_id", + "attempt", + "status", + "checkpoint", + "failure" + ], + "properties": { + "step_id": { + "type": "string", + "maxLength": 160, + "pattern": "^step:[a-z0-9][a-z0-9._-]*$" + }, + "attempt": { + "type": "integer", + "minimum": 0, + "maximum": 4095 + }, + "status": { + "enum": [ + "pending", + "running", + "succeeded", + "failed", + "cancelled", + "denied", + "abandoned" + ] + }, + "checkpoint": { + "anyOf": [ + { + "$ref": "run-checkpoint.v1.schema.json" + }, + { + "type": "null" + } + ] + }, + "failure": { + "anyOf": [ + { + "enum": [ + "cancelled", + "timed-out", + "process", + "artifact-observation", + "artifact-acceptance" + ] + }, + { + "type": "null" + } + ] + } + } + }, + "maxItems": 256, + "minItems": 1 + }, + "authority_decisions": { + "type": "array", + "items": { + "$ref": "run-authority.v1.schema.json" + }, + "maxItems": 4096 + }, + "recovery_decisions": { + "type": "array", + "items": { + "$ref": "run-recovery.v1.schema.json" + }, + "maxItems": 4096 + } + } +} diff --git a/contracts/schemas/run-validation.v1.schema.json b/contracts/schemas/run-validation.v1.schema.json new file mode 100644 index 0000000..cd792cd --- /dev/null +++ b/contracts/schemas/run-validation.v1.schema.json @@ -0,0 +1,51 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://egohygiene.github.io/flow/contracts/run-validation.v1.schema.json", + "title": "Flow run validation v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "profile", + "implementation_version", + "implementation_digest", + "platform", + "invocation_digest", + "execution_digest", + "artifacts_digest" + ], + "properties": { + "schema_version": { + "const": "flow.run-validation/v1" + }, + "profile": { + "const": "flow.accept-artifacts/v1" + }, + "implementation_version": { + "type": "string", + "minLength": 1, + "maxLength": 256 + }, + "implementation_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "platform": { + "type": "string", + "minLength": 1, + "maxLength": 256 + }, + "invocation_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "execution_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + }, + "artifacts_digest": { + "type": "string", + "pattern": "^[a-f0-9]{64}$" + } + } +} diff --git a/docs/architecture/foundation/ARCHITECTURE.md b/docs/architecture/foundation/ARCHITECTURE.md index acc1e01..931c39a 100644 --- a/docs/architecture/foundation/ARCHITECTURE.md +++ b/docs/architecture/foundation/ARCHITECTURE.md @@ -3,12 +3,12 @@ schema: aether.architecture-document/v1 id: flow-architecture title: Flow Architecture kind: architecture-document -version: 0.11.0 +version: 0.12.0 status: draft owners: - egohygiene created: 2026-08-13 -updated: 2026-09-24 +updated: 2026-09-25 governed_by: - architecture-architecture depends_on: @@ -101,6 +101,30 @@ execution subjects, profile, and evidence agree. The evidence is caller- attested: correlation is implemented, while operating-system enforcement and evidence authentication are not. +### Durable execution state + +Flow owns an explicit run workspace and a versioned immutable plan. The durable +coordinator records authorization and execution intent before entering the local +runner, then records a checkpoint only after the existing transcript and artifact +acceptance gates succeed. In-memory seams remain available for conformance and +unit tests; supported persistent execution enters through the durable coordinator. + +Snapshots are immutable, ordered, digest-linked documents committed by a +same-directory rename under an exclusive workspace lock. Reopening validates the +whole retained sequence and never infers completion from output files. Interrupted +attempts require an explicit recovery decision. Reuse assessment compares the +plan, inputs, provider and capability identities, configuration, authority, +implementation profile, and freshly observed artifacts. Deserialized evidence +does not reconstruct opaque execution or authorization tokens. + +State contains typed decisions, identities, relative artifact bindings, and +allowlisted validation evidence. Secret values, configuration values, raw process +streams, and provider-authored messages remain outside the store. The workspace +is trusted local state, not a cryptographically authenticated or sandboxed store. +ADR-0011 owns the persistence, migration, and recovery rationale. Graph scheduling, +automatic retry, provider-native checkpoints, and downstream invalidation policy +remain later orchestration work. + ### External adapters Adapters isolate process invocation, version/capability probing, structured @@ -245,12 +269,14 @@ authority, local-process execution, artifact observation, and acceptance. The second stage consumes the exact accepted first-stage artifact in the same workspace, and two fresh roots must produce equal portable evidence and output bytes. This is conformance infrastructure, not a scenario-manifest executor or -production graph scheduler. Flow does not yet supply +production graph scheduler. Issue #49 adds the versioned durable coordinator, +immutable prepared plans, atomic snapshots, accepted checkpoints, fresh reuse +assessment, and explicit recovery decisions described above. Flow does not yet supply the public CLI, real holon adapters, signature or transparency verification, provider-native artifact validation, an atomic filesystem snapshot, an operating-system sandbox or authenticated enforcement evidence, -descriptor-bound execution, process-tree containment, durable run state, -checkpoints, retry, or resume. Structural units beyond these library seams +descriptor-bound execution, process-tree containment, automatic retry/resume, +or provider-native checkpoint restoration. Structural units beyond these library seams remain constraints for later adapter and orchestration work, not claims about current source layout. diff --git a/docs/architecture/governance/DECISIONS.md b/docs/architecture/governance/DECISIONS.md index fc0e0b7..2382a2d 100644 --- a/docs/architecture/governance/DECISIONS.md +++ b/docs/architecture/governance/DECISIONS.md @@ -3,12 +3,12 @@ schema: aether.architecture-document/v1 id: flow-decisions title: Flow Decisions kind: architecture-document -version: 0.8.0 +version: 0.9.0 status: draft owners: - egohygiene created: 2026-08-13 -updated: 2026-09-21 +updated: 2026-09-25 governed_by: - architecture-decisions depends_on: @@ -64,9 +64,12 @@ justifies separate ADRs. | [ADR-0008](decisions/ADR-0008-locked-execution-subjects.md) | Require exact locked package and executable subjects | Accepted | 2026-09-21 | None | Authenticity evidence, launch-time file binding, or transitive runtime identity changes the subject boundary | | [ADR-0009](decisions/ADR-0009-process-authority-isolation.md) | Bind process authority to explicit isolation evidence | Accepted | 2026-09-21 | None | A real sandbox, authenticated host evidence, or new authority dimension changes the preflight boundary | | [ADR-0010](decisions/ADR-0010-bounded-direct-process-supervision.md) | Bound direct provider launch and supervision | Accepted | 2026-09-21 | None | Sandbox enforcement, descriptor-bound launch, process-tree containment, or durable interruption changes the runner boundary | +| [ADR-0011](decisions/ADR-0011-durable-run-state.md) | Persist prepared intent and acceptance in immutable local snapshots | Proposed | Pending review | None | Migration, distributed writers, automatic recovery, authenticated state, or stronger durability changes the boundary | ## Active decisions +ADR-0011 is proposed with Flow #49 and remains subject to maintainer review. + The indexed ADRs are authoritative. Summaries in other documents must link back to them rather than recreate rationale. diff --git a/docs/architecture/governance/decisions/ADR-0011-durable-run-state.md b/docs/architecture/governance/decisions/ADR-0011-durable-run-state.md new file mode 100644 index 0000000..6d89ba4 --- /dev/null +++ b/docs/architecture/governance/decisions/ADR-0011-durable-run-state.md @@ -0,0 +1,118 @@ +--- +schema: aether.architecture-decision/v1 +id: adr-0011 +title: Persist prepared execution intent and acceptance in immutable local snapshots +kind: architecture-decision +status: proposed +accepted: null +owners: + - egohygiene +scope: + - flow +governed_by: + - architecture-decisions +supersedes: [] +superseded_by: [] +related: + - flow-architecture + - flow-roadmap + - adr-0007 + - adr-0009 + - adr-0010 +--- + +# ADR-0011 — Persist prepared execution intent and acceptance + +## Context and authority + +Flow #49 requires durable plans, execution intent, checkpoints, authority, +validation, and recovery decisions before #31 can prove lifecycle behavior. +The existing library gates produce opaque tokens but lose their history when +the host exits. A file left by a provider cannot resolve that uncertainty. +The maintainer authorized implementation on 2026-09-25; this decision is proposed +for review with the implementation PR and has no claimed acceptance date yet. + +## Decision + +Use closed v1 records and an explicit local workspace. Persist immutable prepared +plans containing ordered dependencies and exact context digests. Record launch +authority and a running attempt before entering the existing process runner. +Record success only after transcript validation and fresh artifact acceptance. + +Commit numbered immutable JSON snapshots by writing and syncing a pending file, +then renaming it within the same directory. Link each state to its predecessor's +digest. Hold an OS-backed exclusive lock for the store's lifetime, including +provider execution. Never delete the lock file or guess that another writer died +from a PID or elapsed time. A process exit releases the OS lock. + +Reopen checks the complete retained history and every state transition. A pending +file, missing middle snapshot, unsupported version, malformed record, or identity +mismatch is a typed refusal. No automatic truncation, repair, or fallback occurs. +An error after rename can mean the new snapshot committed; the old handle becomes +unusable and the caller must reopen to learn the durable result. + +An interrupted running record remains uncertain. Matching current identities and +fresh observations support eligibility assessment; they do not recreate opaque +authorization or acceptance tokens. Retry requires an explicit recorded decision, +fresh matching authority, and acknowledgement of uncertain external effects. It +returns the step to pending without launching it. Successful work cannot be +re-executed through the same step. Whole-graph scheduling, downstream invalidation, +provider-native checkpoint restoration, and automatic retry policy remain later work. + +## Rationale and alternatives + +Immutable snapshots keep each reviewed state and make interrupted writes visible. +They avoid overwriting an open destination, which simplifies portable rename +behavior. A single overwritten JSON file was considered but would retain less +diagnostic history. SQLite transactions would also provide a sound persistence +boundary, but would add a database engine and migration surface to this bounded +local document store. The initial implementation uses the small `fs2` dependency +for advisory locks on Unix and Windows. + +## Assumptions and trade-offs + +- The caller controls a local filesystem with working advisory locks and atomic + same-directory rename. Network filesystems and hostile concurrent writers are + outside the v1 guarantee. +- Unix syncs both files and directories; Windows flushes files and claims process- + crash consistency, not power-loss durability. Hardware durability is not proven. +- Full-history validation and repeated snapshots favor inspectability over scale. + Step, record, snapshot-count, and total-history budgets bound the initial store. +- Digests detect accidental corruption and mismatched context. They do not + authenticate the operator, attest a running binary, detect an intentionally + removed tail without an external anchor, or contain an unconfined provider. +- A validator-source and dependency-recipe digest makes source changes visible. + It is not a claim about every downstream build flag or dynamically linked byte. +- Configuration values, secrets, source bytes, raw streams, and provider messages + are omitted. Identifiers, relative locators, and content digests remain sensitive + local metadata; hashing is not encryption. + +## Migration, rollback, and retention + +Breaking schema or interpretation changes require a new major identifier. Never +reinterpret unknown fields or downgrade records in place. A future explicit +migrator must preserve the original quiescent workspace, write a separate target, +record the source identity and migration decision, and revalidate compatibility +and authority. No migrator is shipped in v1. + +Rollback restores a preserved compatible workspace only after an operator resolves +work performed after that snapshot. Restoring older state never grants permission +to repeat effects. Incompatible binaries refuse the newer state. Retain all +snapshots until the operator archives or deletes the entire quiescent workspace; +there is no automatic pruning, repair, artifact deletion, or secret retention. + +## Validation and review triggers + +Validate real process success/failure, fresh reopen, stale identities and bytes, +cancellation, interrupted acceptance, explicit retry, and preserved sources. +Exercise partial writes and failures before/after rename, abrupt process exit, +cross-process lock contention, corruption, schema refusal, and privacy canaries. +Run portable store tests on Linux, macOS, and Windows. Revisit the decision for +distributed writers, automatic recovery, authenticated state, a new durable +backend, schema migration, or stronger filesystem/process containment. + +## Related artifacts + +Flow #11, #30, #49, and #31; `flow-architecture`; +[`durable-state.md`](../../../integrations/durable-state.md); and the +`flow.run-*/v1` contract family. diff --git a/docs/integrations/durable-state.md b/docs/integrations/durable-state.md new file mode 100644 index 0000000..25c76cc --- /dev/null +++ b/docs/integrations/durable-state.md @@ -0,0 +1,149 @@ +# Durable prepared execution + +Flow #49 adds a library API for persistent execution of fully prepared process +steps. It builds on the exact subject, authority, transcript, and artifact gates. +There is no product CLI or graph scheduler in this checkpoint. #31 owns the +broader lifecycle scenario matrix; #53 owns the supported CLI. + +## Public entry points + +1. Construct a `ProcessStepContext` with the current resolution, invocation, + subject lock, opaque authorization, artifact bindings, and explicit execution + and artifact roots. Roots and tokens remain caller-local. +2. Call `PlannedStep::prepare`, then assemble a `RunPlan` with stable `plan:`, + `run:`, and `step:` identities and sorted dependencies on preceding steps. + Plans contain fully pinned expected inputs; unresolved future outputs require + later planning support. +3. Call `RunStore::create` with a new workspace path. Its parent must exist. +4. Call `execute` for a ready step with explicit secrets, cancellation, and event + adapters. The store persists intent before launch, then an accepted checkpoint + or a bounded typed failure. It holds its exclusive workspace lock throughout. +5. Drop the handle and use `RunStore::open` after restart. `state()` exposes the + recorded status without writing files or launching a process. +6. Use `assess` with the complete expected plan and fresh runtime context. + `Reusable` requires matching intent, validator identity, and fresh input/output + observations. It is eligibility evidence, not a reconstructed acceptance token. + +`RunState::new` remains an in-memory model. `Orchestrator` and `LocalProcessRunner` +remain useful low-level seams for unit tests and controlled integrations. Callers +requiring restart semantics must use `RunStore::execute` rather than treating a +low-level runner result as a persisted run. + +## Records and identity + +| Contract | Meaning | +| --- | --- | +| `flow.run-plan/v1` | Immutable prepared steps, dependency order, and exact context | +| `flow.run-state/v1` | Complete run state, numbered history, decisions, and checkpoints | +| `flow.run-checkpoint/v1` | Accepted step bound to plan, context, and attempt | +| `flow.run-artifact/v1` | Input/output role and existing host observation record | +| `flow.run-authority/v1` | Granted launch authority or explicit operator denial | +| `flow.run-validation/v1` | Accepted evidence digests and validation implementation identity | +| `flow.run-recovery/v1` | Explicit retry/abandon decision and uncertainty acknowledgement | +| `flow.run-snapshot/v1` | Storage envelope and state integrity digest | + +Every checkpoint binds the complete plan digest and its step context digest. +That context includes provider identity/version/integrity, capability definition, +process interface, exact subject lock, configuration schema and claimed digest, +an independently computed digest of actual configuration values, invocation, +authority/enforcement/grant digests, bindings, and expected input digests. + +Validation records hash accepted execution evidence and the retained artifact +records without storing provider messages. The implementation identity includes +the validation sources and locked dependency recipe compiled into the library, +crate version, and operating-system name. It is reproducibility metadata, not +publisher authentication or a digest of the final application binary. Changed +identity requires explicit future compatibility/migration work. + +Objects use closed fields. Canonical identity serializes recursively ordered +object keys as compact UTF-8 JSON; arrays retain their declared order. The Rust +semantic validators additionally enforce identity and state relationships that +JSON Schema alone does not express. See `contracts/fixtures/state/` for examples. + +## Commit and restart behavior + +A workspace contains `workspace.lock`, numbered immutable `*.json` snapshots, +and, only while a commit is unfinished, `snapshot.pending`. Each snapshot records +the previous state's digest. Flow syncs the pending file before renaming it and +never replaces an existing numbered snapshot. Unix also syncs parent directories. +Windows v1 guarantees cover process crashes; directory/power-loss durability is +not claimed. Use a local filesystem with working locks and atomic rename. + +The lock is advisory and OS-backed. A concurrent open returns `Busy`; a killed +writer releases the lock. Do not delete or replace `workspace.lock` to bypass +contention. A failed commit poisons that handle where commit progress is uncertain; +reopen to inspect whether the new snapshot actually committed. + +Reopen checks the full retained sequence. Pending bytes, corruption, unsupported +versions, identity conflicts, unsafe entries, and sequence gaps produce typed +refusals. Flow preserves the directory for inspection instead of guessing that an +older snapshot is safe. A checksum is not authentication: a hostile writer can +rewrite history, and removing a complete tail is not detectable without an +external anchor. + +| Recorded condition | Meaning after reopen | +| --- | --- | +| Pending | Ready only when dependencies, current context, and input observations match | +| Running | Host completion is uncertain; explicit recovery is required | +| Succeeded | Eligible for reuse only after current evidence is checked | +| Failed, cancelled, denied | Inspect the typed outcome; explicit recovery is required | +| Abandoned | Operator stopped this step; no automatic execution | + +`cancel_pending` and `deny_pending` record decisions before provider code runs. +Normal timeout/cancellation retains the runner's direct-child cleanup semantics. +A host crash cannot promise that an unconfined provider or its descendants died. + +`decide_recovery` requires matching fresh context, authority, a stable decision ID, +and acknowledgement of uncertain effects. Retry records a pending step without +launching it. Operators must inspect and preserve/quarantine residual outputs and +resolve surviving processes before an explicit new attempt. Flow never deletes +or overwrites those artifacts automatically. A succeeded step cannot be retried +through this API. Automatic reuse scheduling, targeted downstream invalidation, +provider-native checkpoint restoration, and exactly-once external effects are +outside v1. + +## Privacy, budgets, and retention + +No configuration values, secret values/handles, source contents, raw process +streams, absolute workspace roots, or provider-authored free text are stored. +The allowlisted identities and relative locators can still be private metadata. +Digests are not encryption, and caller-supplied identifiers are not generically +redacted. Unix creates workspace directories with mode 0700 and files with 0600; +Windows inherits the selected parent's ACL. + +V1 limits a plan to 256 steps, each step to 4096 artifact records, a snapshot to +8 MiB, a run to 4096 snapshots, and retained snapshot bytes to 64 MiB. Exceeding +a budget is a refusal, never silent pruning. Large artifact directory manifests +may exceed the state budget. + +Keep the entire history until explicitly archived or removed. Stop writers and +resolve uncertain provider effects first. Archive the whole workspace, not a +single middle snapshot. Cleanup of state never implies deletion of source or +output artifacts. No automatic retention task or garbage collector is installed. + +## Migration and rollback + +Unknown versions are refused. Breaking interpretation changes use new major +identifiers; v1 has no in-place migration or downgrade. A future explicit migrator +must preserve the original quiescent workspace, record its exact identity, produce +a separate destination, and revalidate compatibility and authority. Rollback uses +a preserved compatible workspace only after an operator resolves effects that +occurred since it was saved. Restoring a backup never authorizes repeated work. +The rationale and review triggers live in +[ADR-0011](../architecture/governance/decisions/ADR-0011-durable-run-state.md). + +## Verification + +```console +cargo test --lib state::store::tests --locked +cargo test --test durable_state --locked +cargo test --test hermetic_provider_kit durable_execution --locked +python3 tools/validate_contracts.py +``` + +The portable store suite covers restart, abrupt process exit, concurrent opens, +schema/corruption refusal, partial writes, and immutable history. Unit fault +injection covers failures after pending-file creation, after file sync, and after +rename. Real provider tests exercise the durable success/failure path, interrupted +acceptance, stale evidence, explicit recovery, and privacy. macOS/Windows CI runs +the portable store suite; Linux also runs the full provider matrix. diff --git a/docs/integrations/process-runner.md b/docs/integrations/process-runner.md index 226a608..1a763b9 100644 --- a/docs/integrations/process-runner.md +++ b/docs/integrations/process-runner.md @@ -1,5 +1,9 @@ # Bounded local process runner +For execution that must survive host restarts, use the +[durable coordinator](durable-state.md). This document describes its underlying +in-memory process boundary. + ## Purpose and scope `LocalProcessRunner` turns Flow's validated process contracts into one real diff --git a/src/lib.rs b/src/lib.rs index 3fc85e8..5fc2def 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -7,7 +7,8 @@ //! authority/isolation preflight. Its local runner can launch the exact //! trusted-unconfined executable with bounded transport, timeout/cancellation //! control, and direct-child reaping; sandboxing, descendant containment, and -//! durable recovery remain separate work. +//! automatic graph recovery remain separate work. The durable coordinator stores +//! prepared intent, acceptance, and explicit recovery decisions in a local workspace. pub mod artifacts; pub mod authority; @@ -19,6 +20,7 @@ pub mod hermetic; pub mod process; pub mod runner; pub mod scenario; +pub mod state; pub use artifacts::*; pub use authority::*; @@ -41,3 +43,4 @@ pub use runner::{ ProcessRunnerError, ProcessWorker, SecretResolver, SecretValue, }; pub use scenario::*; +pub use state::*; diff --git a/src/state/execution.rs b/src/state/execution.rs new file mode 100644 index 0000000..49f3167 --- /dev/null +++ b/src/state/execution.rs @@ -0,0 +1,559 @@ +use std::path::Path; + +use sha2::{Digest, Sha256}; +use thiserror::Error; + +use crate::{ + AcceptedArtifactSet, ArtifactAcceptanceError, ArtifactBindingSet, ArtifactObservationError, + AuthorizedProcess, CancellationSignal, EventSink, ExecutionSubjectLock, ExtensionInvocation, + LocalProcessRunner, Orchestrator, ProcessIsolation, ProcessRunnerError, ResolvedExtension, + SecretResolver, accept_artifacts, observe_artifacts, observe_execution_subjects, +}; + +use super::{ + CheckpointContext, PlannedStep, RUN_ARTIFACT_V1, RUN_AUTHORITY_V1, RUN_CHECKPOINT_V1, + RUN_RECOVERY_V1, RUN_VALIDATION_PROFILE, RUN_VALIDATION_V1, RunArtifactRecord, RunArtifactRole, + RunAuthorityDecision, RunCheckpoint, RunFailure, RunPlan, RunRecoveryAction, + RunRecoveryDecision, RunState, RunStepStatus, RunStore, RunValidationEvidence, StateBoundary, + StateError, digest, require, +}; + +/// Fresh caller-owned context. Absolute roots and authorization tokens are never serialized. +pub struct ProcessStepContext<'a> { + pub execution_root: &'a Path, + pub artifact_root: &'a Path, + pub resolved: &'a ResolvedExtension, + pub invocation: &'a ExtensionInvocation, + pub subjects: &'a ExecutionSubjectLock, + pub authority: &'a AuthorizedProcess, + pub bindings: &'a ArtifactBindingSet, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum ResumeEligibility { + Ready, + Reusable, + ApprovalRequired, + DependencyBlocked, + Abandoned, +} + +/// An explicit caller decision. It does not itself grant process authority. +pub struct RecoveryApproval { + pub decision_id: String, + pub action: RunRecoveryAction, + pub acknowledge_uncertain_effects: bool, +} + +#[derive(Debug, Error)] +pub enum DurableExecutionError { + #[error("durable coordination failed: {0}")] + State(#[from] StateError), + #[error("cancelled before launch")] + CancelledBeforeLaunch, + #[error("provider execution failed")] + Process(#[source] ProcessRunnerError), + #[error("artifact observation failed")] + Observation(#[source] ArtifactObservationError), + #[error("artifact acceptance failed")] + Acceptance(#[source] ArtifactAcceptanceError), +} + +impl PlannedStep { + /// Prepare immutable intent through the existing subject and authority preflight. + /// + /// Input bytes are checked at execution/assessment time, allowing prepared + /// dependencies whose expected input identities are already known. + /// + /// # Errors + /// Rejects invalid bindings, process identity, subject bytes, or authority. + pub fn prepare( + step_id: String, + depends_on: Vec, + context: &ProcessStepContext<'_>, + ) -> Result { + context.preflight()?; + Ok(Self { + step_id, + depends_on, + context: context.identity()?, + }) + } +} + +impl ProcessStepContext<'_> { + fn identity(&self) -> Result { + let invocation = self.invocation; + Ok(CheckpointContext { + run_id: invocation.run_id.clone(), + invocation_id: invocation.invocation_id.clone(), + invocation_digest: digest(invocation)?, + provider: invocation.extension.clone(), + interface: invocation.interface.clone(), + capability_id: invocation.capability_id.clone(), + capability_digest: digest(self.resolved.capability())?, + configuration_schema: invocation.configuration.schema_id.clone(), + configuration_digest: invocation.configuration.digest.clone(), + configuration_values_digest: digest(&invocation.configuration.values)?, + subject_lock_digest: digest(self.subjects)?, + authorization_id: invocation.authorization.authorization_id.clone(), + authority_profile_digest: self.authority.profile_digest().to_owned(), + enforcement_evidence_digest: self.authority.evidence_digest().to_owned(), + grants_digest: invocation.authorization.grants_digest.clone(), + bindings: self.bindings.clone(), + }) + } + + fn preflight(&self) -> Result<(), StateError> { + self.bindings.validate().map_err(|_| StateError::Invalid { + rule: "artifact bindings", + })?; + let declared: std::collections::BTreeSet<_> = self + .invocation + .input_artifacts + .iter() + .map(|a| (&a.artifact_id, &a.digest)) + .collect(); + let bound: std::collections::BTreeSet<_> = self + .bindings + .inputs + .iter() + .map(|a| (&a.artifact_id, &a.expected_digest)) + .collect(); + require(declared == bound, "invocation input bindings")?; + let subjects = observe_execution_subjects( + self.execution_root, + self.resolved, + self.invocation, + self.subjects, + ) + .map_err(|_| StateError::Stale { + boundary: StateBoundary::Provider, + })?; + Orchestrator::encode_process_request( + self.resolved, + self.invocation, + self.subjects, + &subjects, + self.authority, + ) + .map_err(|_| StateError::Stale { + boundary: StateBoundary::Authority, + })?; + require( + self.authority.isolation() == ProcessIsolation::TrustedUnconfined, + "supported durable runner isolation", + ) + } + + fn check_inputs(&self) -> Result<(), StateError> { + if self.bindings.inputs.is_empty() { + return Ok(()); + } + let bindings = ArtifactBindingSet { + outputs: Vec::new(), + ..self.bindings.clone() + }; + let observed = + observe_artifacts(self.artifact_root, &bindings).map_err(|_| StateError::Stale { + boundary: StateBoundary::Inputs, + })?; + for (binding, observation) in bindings.inputs.iter().zip(&observed.evidence().artifacts) { + if binding.expected_digest != observation.digest { + return Err(StateError::Stale { + boundary: StateBoundary::Inputs, + }); + } + } + Ok(()) + } +} + +impl RunStore { + /// Assess current evidence against the caller's expected complete plan. + /// + /// `Reusable` is an eligibility decision about recorded acceptance, not a + /// reconstructed `AcceptedArtifactSet` or permission to launch a provider. + /// + /// # Errors + /// Refuses changed plans, configuration, inputs, subjects, authority, validation + /// implementations, and output evidence. This method writes no state. + pub fn assess( + &self, + expected_plan: &RunPlan, + step_id: &str, + context: &ProcessStepContext<'_>, + ) -> Result { + self.ensure_usable()?; + if expected_plan.digest()? != self.state.plan_digest { + return Err(StateError::Stale { + boundary: StateBoundary::Plan, + }); + } + let index = self.state.step_index(step_id)?; + self.check_context(index, context)?; + let planned = &self.state.plan.steps[index]; + if planned.depends_on.iter().any(|id| { + self.state + .steps + .iter() + .any(|s| &s.step_id == id && s.status != RunStepStatus::Succeeded) + }) { + return Ok(ResumeEligibility::DependencyBlocked); + } + let step = &self.state.steps[index]; + match step.status { + RunStepStatus::Pending => Ok(ResumeEligibility::Ready), + RunStepStatus::Succeeded => { + let checkpoint = step.checkpoint.as_ref().ok_or(StateError::Corrupt)?; + let validation = &checkpoint.validation; + if validation.implementation_version != env!("CARGO_PKG_VERSION") + || validation.implementation_digest != validation_implementation_digest() + || validation.platform != std::env::consts::OS + || validation.profile != RUN_VALIDATION_PROFILE + { + return Err(StateError::Stale { + boundary: StateBoundary::Validation, + }); + } + let current = + observe_artifacts(context.artifact_root, context.bindings).map_err(|_| { + StateError::Stale { + boundary: StateBoundary::Artifacts, + } + })?; + let saved: Vec<_> = checkpoint + .artifacts + .iter() + .map(|a| a.observation.clone()) + .collect(); + let mut current = current.into_evidence().artifacts; + current.sort_by(|a, b| a.artifact_id.cmp(&b.artifact_id)); + if saved != current { + return Err(StateError::Stale { + boundary: StateBoundary::Artifacts, + }); + } + Ok(ResumeEligibility::Reusable) + } + RunStepStatus::Abandoned => Ok(ResumeEligibility::Abandoned), + RunStepStatus::Running + | RunStepStatus::Failed + | RunStepStatus::Cancelled + | RunStepStatus::Denied => Ok(ResumeEligibility::ApprovalRequired), + } + } + + fn check_context( + &self, + index: usize, + context: &ProcessStepContext<'_>, + ) -> Result<(), StateError> { + let saved = &self.state.plan.steps[index].context; + let current = context.identity()?; + let checks = [ + ( + saved.provider == current.provider + && saved.subject_lock_digest == current.subject_lock_digest, + StateBoundary::Provider, + ), + ( + saved.capability_id == current.capability_id + && saved.capability_digest == current.capability_digest + && saved.interface == current.interface, + StateBoundary::Capability, + ), + ( + saved.configuration_schema == current.configuration_schema + && saved.configuration_digest == current.configuration_digest + && saved.configuration_values_digest == current.configuration_values_digest, + StateBoundary::Configuration, + ), + ( + saved.authority_profile_digest == current.authority_profile_digest + && saved.enforcement_evidence_digest == current.enforcement_evidence_digest + && saved.authorization_id == current.authorization_id + && saved.grants_digest == current.grants_digest, + StateBoundary::Authority, + ), + (saved.bindings == current.bindings, StateBoundary::Bindings), + (saved == ¤t, StateBoundary::Invocation), + ]; + for (matches, boundary) in checks { + if !matches { + return Err(StateError::Stale { boundary }); + } + } + context.preflight()?; + context.check_inputs() + } + + /// Persist launch intent, execute one prepared step, and persist accepted evidence. + /// + /// The store remains locked throughout the attempt. Any failure to commit the + /// terminal result is an error; the prior running record remains uncertain. + /// Existing completed steps are never executed again by this method. + /// + /// # Errors + /// Returns typed coordination, cancellation, process, observation, or acceptance + /// failures. Raw provider errors remain caller-local and are never serialized. + #[allow(clippy::too_many_lines)] + pub fn execute( + &mut self, + step_id: &str, + context: &ProcessStepContext<'_>, + secrets: &dyn SecretResolver, + cancellation: &dyn CancellationSignal, + events: &mut dyn EventSink, + ) -> Result { + if self.assess(&self.state.plan, step_id, context)? != ResumeEligibility::Ready { + return Err(StateError::Transition.into()); + } + if cancellation.is_cancelled() { + self.cancel_pending(step_id)?; + return Err(DurableExecutionError::CancelledBeforeLaunch); + } + let index = self.state.step_index(step_id)?; + let mut next = self.state.clone(); + next.steps[index].attempt += 1; + next.steps[index].status = RunStepStatus::Running; + next.authority_decisions + .push(authority_decision(&next, index, true)); + self.commit(next)?; + + if cancellation.is_cancelled() { + self.finish_failure(index, RunFailure::Cancelled)?; + return Err(DurableExecutionError::CancelledBeforeLaunch); + } + + let execution = match LocalProcessRunner::run_with_cancellation( + context.execution_root, + context.resolved, + context.invocation, + context.subjects, + context.authority, + secrets, + cancellation, + events, + ) { + Ok(value) => value, + Err(error) => { + let failure = match &error { + ProcessRunnerError::Cancelled { .. } => RunFailure::Cancelled, + ProcessRunnerError::TimedOut { .. } => RunFailure::TimedOut, + _ => RunFailure::Process, + }; + self.finish_failure(index, failure)?; + return Err(DurableExecutionError::Process(error)); + } + }; + let observed = match observe_artifacts(context.artifact_root, context.bindings) { + Ok(value) => value, + Err(error) => { + self.finish_failure(index, RunFailure::ArtifactObservation)?; + return Err(DurableExecutionError::Observation(error)); + } + }; + let accepted = match accept_artifacts( + context.resolved, + context.invocation, + &execution, + context.bindings, + &observed, + ) { + Ok(value) => value, + Err(error) => { + self.finish_failure(index, RunFailure::ArtifactAcceptance)?; + return Err(DurableExecutionError::Acceptance(error)); + } + }; + let mut artifacts: Vec<_> = accepted + .inputs() + .iter() + .map(|a| (RunArtifactRole::Input, a)) + .chain( + accepted + .outputs() + .iter() + .map(|a| (RunArtifactRole::Output, a)), + ) + .map(|(role, observation)| RunArtifactRecord { + schema_version: RUN_ARTIFACT_V1.to_owned(), + role, + observation: observation.clone(), + }) + .collect(); + artifacts.sort_by(|a, b| a.observation.artifact_id.cmp(&b.observation.artifact_id)); + let attempt = self.state.steps[index].attempt; + let context_digest = digest(&self.state.plan.steps[index].context)?; + let checkpoint_id = format!( + "checkpoint:{}", + digest(&(&self.state.plan_digest, step_id, attempt, &context_digest))? + ); + let validation = RunValidationEvidence { + schema_version: RUN_VALIDATION_V1.to_owned(), + profile: RUN_VALIDATION_PROFILE.to_owned(), + implementation_version: env!("CARGO_PKG_VERSION").to_owned(), + implementation_digest: validation_implementation_digest(), + platform: std::env::consts::OS.to_owned(), + invocation_digest: self.state.plan.steps[index] + .context + .invocation_digest + .clone(), + execution_digest: digest(&(execution.events(), execution.result()))?, + artifacts_digest: digest(&artifacts)?, + }; + let checkpoint = RunCheckpoint { + schema_version: RUN_CHECKPOINT_V1.to_owned(), + checkpoint_id, + plan_digest: self.state.plan_digest.clone(), + context_digest, + attempt, + artifacts, + validation, + }; + let mut next = self.state.clone(); + next.steps[index].status = RunStepStatus::Succeeded; + next.steps[index].checkpoint = Some(checkpoint); + self.commit(next)?; + Ok(accepted) + } + + fn finish_failure(&mut self, index: usize, failure: RunFailure) -> Result<(), StateError> { + let mut next = self.state.clone(); + next.steps[index].status = if failure == RunFailure::Cancelled { + RunStepStatus::Cancelled + } else { + RunStepStatus::Failed + }; + next.steps[index].failure = Some(failure); + self.commit(next) + } + + /// Cancel a pending step without entering provider code. + /// + /// # Errors + /// Refuses non-pending steps or failed persistence. + pub fn cancel_pending(&mut self, step_id: &str) -> Result<(), StateError> { + let index = self.state.step_index(step_id)?; + require( + self.state.steps[index].status == RunStepStatus::Pending, + "pending cancellation", + )?; + let mut next = self.state.clone(); + next.steps[index].status = RunStepStatus::Cancelled; + next.steps[index].failure = Some(RunFailure::Cancelled); + self.commit(next) + } + + /// Record an operator denial without granting authority or entering provider code. + /// + /// # Errors + /// Refuses non-pending steps or failed persistence. + pub fn deny_pending(&mut self, step_id: &str) -> Result<(), StateError> { + let index = self.state.step_index(step_id)?; + require( + self.state.steps[index].status == RunStepStatus::Pending, + "pending denial", + )?; + let mut next = self.state.clone(); + next.steps[index].status = RunStepStatus::Denied; + next.authority_decisions + .push(authority_decision(&next, index, false)); + self.commit(next) + } + + /// Record an explicit recovery choice using fresh matching authority. + /// + /// Retry only returns the step to pending. No process is launched. The caller + /// acknowledges that an interrupted unconfined provider may still have effects + /// or surviving descendants and must resolve those before retrying. + /// + /// # Errors + /// Refuses stale evidence, completed work, missing acknowledgement, duplicate + /// decision identities, or failed persistence. + pub fn decide_recovery( + &mut self, + step_id: &str, + context: &ProcessStepContext<'_>, + approval: RecoveryApproval, + ) -> Result<(), StateError> { + let index = self.state.step_index(step_id)?; + self.check_context(index, context)?; + if !matches!( + self.state.steps[index].status, + RunStepStatus::Running + | RunStepStatus::Failed + | RunStepStatus::Cancelled + | RunStepStatus::Denied + ) { + return Err(StateError::Transition); + } + require( + approval.acknowledge_uncertain_effects, + "explicit recovery acknowledgement", + )?; + let mut next = self.state.clone(); + let identity = &next.plan.steps[index].context; + next.recovery_decisions.push(RunRecoveryDecision { + schema_version: RUN_RECOVERY_V1.to_owned(), + decision_id: approval.decision_id, + step_id: step_id.to_owned(), + attempt: next.steps[index].attempt, + action: approval.action, + authorization_id: identity.authorization_id.clone(), + profile_digest: identity.authority_profile_digest.clone(), + acknowledged_uncertain_effects: true, + }); + next.steps[index].status = if approval.action == RunRecoveryAction::Retry { + RunStepStatus::Pending + } else { + RunStepStatus::Abandoned + }; + next.steps[index].failure = None; + self.commit(next) + } +} + +fn authority_decision(state: &RunState, index: usize, granted: bool) -> RunAuthorityDecision { + let identity = &state.plan.steps[index].context; + RunAuthorityDecision { + schema_version: RUN_AUTHORITY_V1.to_owned(), + step_id: state.steps[index].step_id.clone(), + attempt: state.steps[index].attempt, + granted, + authorization_id: identity.authorization_id.clone(), + profile_digest: identity.authority_profile_digest.clone(), + enforcement_digest: identity.enforcement_evidence_digest.clone(), + grants_digest: identity.grants_digest.clone(), + } +} + +/// Identity of the validation sources and locked dependency recipe compiled into Flow. +/// This is reproducibility metadata, not executable or publisher authentication. +#[must_use] +pub fn validation_implementation_digest() -> String { + let sources: &[&[u8]] = &[ + include_bytes!("../lib.rs"), + include_bytes!("../contracts.rs"), + include_bytes!("../catalog.rs"), + include_bytes!("../execution.rs"), + include_bytes!("../process.rs"), + include_bytes!("../runner.rs"), + include_bytes!("../artifacts.rs"), + include_bytes!("../authority.rs"), + include_bytes!("../execution_subjects.rs"), + include_bytes!("mod.rs"), + include_bytes!("model.rs"), + include_bytes!("store.rs"), + include_bytes!("execution.rs"), + include_bytes!("../../Cargo.lock"), + ]; + let mut hasher = Sha256::new(); + hasher.update(b"flow.run-validator-sources/v1\0"); + for source in sources { + hasher.update((source.len() as u64).to_le_bytes()); + hasher.update(source); + } + format!("{:x}", hasher.finalize()) +} diff --git a/src/state/mod.rs b/src/state/mod.rs new file mode 100644 index 0000000..5d3fc30 --- /dev/null +++ b/src/state/mod.rs @@ -0,0 +1,130 @@ +//! Versioned local run state and durable coordination of prepared process steps. +//! +//! The store records intent before launch and acceptance after host validation. +//! Reopening never launches a provider or recreates opaque trust tokens. See +//! `docs/integrations/durable-state.md` for storage and recovery limits. + +mod execution; +mod model; +mod store; + +pub use execution::{ + DurableExecutionError, ProcessStepContext, RecoveryApproval, ResumeEligibility, + validation_implementation_digest, +}; +pub use model::*; +pub use store::RunStore; + +use serde::Serialize; +use serde_json::Value; +use sha2::{Digest, Sha256}; +use thiserror::Error; + +/// A precise boundary whose current evidence disagrees with saved intent. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum StateBoundary { + Plan, + Invocation, + Provider, + Capability, + Configuration, + Authority, + Bindings, + Inputs, + Artifacts, + Validation, +} + +/// Store failures never carry raw provider messages or source bytes. +#[derive(Debug, Error)] +pub enum StateError { + #[error("the run workspace is already open")] + Busy, + #[error("the run workspace already exists")] + AlreadyExists, + #[error("the run workspace contains an incomplete write; preserve it for recovery")] + IncompleteWrite, + #[error("the run workspace contains an unexpected or unsafe filesystem entry")] + UnsafeEntry, + #[error("the durable record exceeds its resource budget")] + Limit, + #[error("unsupported durable record schema")] + UnsupportedSchema, + #[error("malformed durable record")] + Malformed, + #[error("durable record identity or sequence is corrupt")] + Corrupt, + #[error("invalid durable state: {rule}")] + Invalid { rule: &'static str }, + #[error("current evidence disagrees with durable state at {boundary:?}")] + Stale { boundary: StateBoundary }, + #[error("step is not ready for execution or the requested transition")] + Transition, + #[error("step identifier does not exist in this plan")] + UnknownStep, + #[error("the store must be reopened after a failed commit")] + ReopenRequired, + #[error("workspace I/O failed during {operation}")] + Io { + operation: &'static str, + #[source] + source: std::io::Error, + }, +} + +fn require(condition: bool, rule: &'static str) -> Result<(), StateError> { + if condition { + Ok(()) + } else { + Err(StateError::Invalid { rule }) + } +} + +fn schema(actual: &str, expected: &str) -> Result<(), StateError> { + if actual == expected { + Ok(()) + } else { + Err(StateError::UnsupportedSchema) + } +} + +fn valid_digest(value: &str) -> bool { + value.len() == 64 + && value + .bytes() + .all(|b| b.is_ascii_digit() || (b'a'..=b'f').contains(&b)) +} + +fn valid_id(value: &str, prefix: &str) -> bool { + value.len() <= 160 && crate::artifacts::has_prefixed_id(value, prefix) +} + +fn canonical_bytes(value: &impl Serialize) -> Result, StateError> { + fn ordered(value: Value) -> Value { + match value { + Value::Object(map) => { + let sorted: std::collections::BTreeMap<_, _> = map.into_iter().collect(); + Value::Object( + sorted + .into_iter() + .map(|(key, value)| (key, ordered(value))) + .collect(), + ) + } + Value::Array(values) => Value::Array(values.into_iter().map(ordered).collect()), + other => other, + } + } + serde_json::to_vec(&ordered( + serde_json::to_value(value).map_err(|_| StateError::Malformed)?, + )) + .map_err(|_| StateError::Malformed) +} + +fn digest(value: &impl Serialize) -> Result { + Ok(format!("{:x}", Sha256::digest(canonical_bytes(value)?))) +} + +fn io_error(operation: &'static str, source: std::io::Error) -> StateError { + StateError::Io { operation, source } +} diff --git a/src/state/model.rs b/src/state/model.rs new file mode 100644 index 0000000..21a34d1 --- /dev/null +++ b/src/state/model.rs @@ -0,0 +1,582 @@ +use std::collections::BTreeSet; + +use serde::{Deserialize, Serialize}; + +use super::{StateError, digest, require, schema, valid_digest, valid_id}; +use crate::{ + ArtifactBindingSet, HostArtifactObservation, InvocationExtension, InvocationInterface, +}; + +pub const RUN_PLAN_V1: &str = "flow.run-plan/v1"; +pub const RUN_STATE_V1: &str = "flow.run-state/v1"; +pub const RUN_CHECKPOINT_V1: &str = "flow.run-checkpoint/v1"; +pub const RUN_ARTIFACT_V1: &str = "flow.run-artifact/v1"; +pub const RUN_AUTHORITY_V1: &str = "flow.run-authority/v1"; +pub const RUN_VALIDATION_V1: &str = "flow.run-validation/v1"; +pub const RUN_RECOVERY_V1: &str = "flow.run-recovery/v1"; +pub const RUN_SNAPSHOT_V1: &str = "flow.run-snapshot/v1"; +pub const RUN_VALIDATION_PROFILE: &str = "flow.accept-artifacts/v1"; +pub const MAX_RUN_STEPS: usize = 256; +pub const MAX_RUN_ARTIFACTS: usize = 4096; +pub const MAX_RUN_RECORD_BYTES: u64 = 8 * 1024 * 1024; +pub const MAX_RUN_SNAPSHOTS: u64 = 4096; + +/// Immutable prepared intent. Values and secret material are represented by digests. +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunPlan { + pub schema_version: String, + pub plan_id: String, + pub run_id: String, + pub steps: Vec, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct PlannedStep { + pub step_id: String, + pub depends_on: Vec, + pub context: CheckpointContext, +} + +/// Complete identity needed to assess a saved step against a current invocation. +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct CheckpointContext { + pub run_id: String, + pub invocation_id: String, + pub invocation_digest: String, + pub provider: InvocationExtension, + pub interface: InvocationInterface, + pub capability_id: String, + pub capability_digest: String, + pub configuration_schema: String, + pub configuration_digest: String, + pub configuration_values_digest: String, + pub subject_lock_digest: String, + pub authorization_id: String, + pub authority_profile_digest: String, + pub enforcement_evidence_digest: String, + pub grants_digest: String, + pub bindings: ArtifactBindingSet, +} + +/// A complete state snapshot. Public records are inspectable data, not trust tokens. +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunState { + pub schema_version: String, + pub sequence: u64, + pub previous_digest: String, + pub plan_digest: String, + pub plan: RunPlan, + pub steps: Vec, + pub authority_decisions: Vec, + pub recovery_decisions: Vec, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunStepState { + pub step_id: String, + pub attempt: u64, + pub status: RunStepStatus, + pub checkpoint: Option, + pub failure: Option, +} + +#[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum RunStepStatus { + Pending, + Running, + Succeeded, + Failed, + Cancelled, + Denied, + Abandoned, +} + +/// A running record observed after reopen is uncertain, never automatically retried. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum RunInspectionStatus { + Prepared, + Succeeded, + RecoveryRequired, + Stopped, +} + +#[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum RunFailure { + Cancelled, + TimedOut, + Process, + ArtifactObservation, + ArtifactAcceptance, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunCheckpoint { + pub schema_version: String, + pub checkpoint_id: String, + pub plan_digest: String, + pub context_digest: String, + pub attempt: u64, + pub artifacts: Vec, + pub validation: RunValidationEvidence, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunArtifactRecord { + pub schema_version: String, + pub role: RunArtifactRole, + pub observation: HostArtifactObservation, +} + +#[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum RunArtifactRole { + Input, + Output, +} + +/// Digests of accepted evidence only; provider free text is deliberately absent. +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunValidationEvidence { + pub schema_version: String, + pub profile: String, + pub implementation_version: String, + pub implementation_digest: String, + pub platform: String, + pub invocation_digest: String, + pub execution_digest: String, + pub artifacts_digest: String, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunAuthorityDecision { + pub schema_version: String, + pub step_id: String, + pub attempt: u64, + pub granted: bool, + pub authorization_id: String, + pub profile_digest: String, + pub enforcement_digest: String, + pub grants_digest: String, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields)] +pub struct RunRecoveryDecision { + pub schema_version: String, + pub decision_id: String, + pub step_id: String, + pub attempt: u64, + pub action: RunRecoveryAction, + pub authorization_id: String, + pub profile_digest: String, + pub acknowledged_uncertain_effects: bool, +} + +#[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum RunRecoveryAction { + Retry, + Abandon, +} + +impl RunPlan { + /// Validate a bounded, ordered graph of fully pinned process steps. + /// + /// # Errors + /// Rejects unknown versions, malformed identities, or invalid dependencies. + pub fn validate(&self) -> Result<(), StateError> { + schema(&self.schema_version, RUN_PLAN_V1)?; + require(valid_id(&self.plan_id, "plan:"), "plan identity")?; + require(valid_id(&self.run_id, "run:"), "run identity")?; + require( + !self.steps.is_empty() && self.steps.len() <= MAX_RUN_STEPS, + "step budget", + )?; + let mut seen = BTreeSet::new(); + let mut invocations = BTreeSet::new(); + let mut outputs = BTreeSet::new(); + for step in &self.steps { + require(valid_id(&step.step_id, "step:"), "step identity")?; + require( + step.depends_on.windows(2).all(|p| p[0] < p[1]), + "sorted unique dependencies", + )?; + require( + step.depends_on.iter().all(|id| seen.contains(id)), + "dependencies precede step", + )?; + require(seen.insert(step.step_id.clone()), "unique steps")?; + require( + invocations.insert(&step.context.invocation_id), + "unique invocations", + )?; + step.context.validate()?; + require(step.context.run_id == self.run_id, "step run identity")?; + for output in &step.context.bindings.outputs { + require( + outputs.insert(&output.artifact_id), + "unique output artifact identities", + )?; + } + } + Ok(()) + } + + /// Deterministic digest of the complete immutable plan. + /// + /// # Errors + /// Rejects an invalid plan or an unencodable record. + pub fn digest(&self) -> Result { + self.validate()?; + digest(self) + } +} + +impl CheckpointContext { + fn validate(&self) -> Result<(), StateError> { + require(valid_id(&self.run_id, "run:"), "context run identity")?; + require( + valid_id(&self.invocation_id, "invocation:"), + "invocation identity", + )?; + require( + valid_id(&self.authorization_id, "authorization:"), + "authorization identity", + )?; + require( + crate::contracts::is_extension_id(&self.provider.extension_id), + "provider identity", + )?; + require( + crate::contracts::is_publisher_id(&self.provider.publisher_id), + "publisher identity", + )?; + crate::contracts::validate_strict_version("provider version", &self.provider.version) + .map_err(|_| StateError::Invalid { + rule: "provider version", + })?; + require( + crate::contracts::is_capability_id(&self.capability_id), + "capability identity", + )?; + require( + self.interface.kind == crate::ExecutionModeKind::Process, + "process interface", + )?; + require( + !self.interface.name.is_empty() && !self.interface.protocol.is_empty(), + "interface identity", + )?; + require( + !self.configuration_schema.is_empty() && self.configuration_schema.len() <= 256, + "configuration schema", + )?; + for value in [ + &self.invocation_digest, + &self.provider.integrity, + &self.capability_digest, + &self.configuration_digest, + &self.configuration_values_digest, + &self.subject_lock_digest, + &self.authority_profile_digest, + &self.enforcement_evidence_digest, + &self.grants_digest, + ] { + require(valid_digest(value), "sha256 identity")?; + } + require( + self.bindings.inputs.len() + self.bindings.outputs.len() <= MAX_RUN_ARTIFACTS, + "artifact budget", + )?; + self.bindings.validate().map_err(|_| StateError::Invalid { + rule: "artifact bindings", + }) + } +} + +impl RunState { + /// Construct the initial in-memory record without touching the filesystem. + /// + /// # Errors + /// Rejects invalid prepared intent. + pub fn new(plan: RunPlan) -> Result { + let plan_digest = plan.digest()?; + let steps = plan + .steps + .iter() + .map(|step| RunStepState { + step_id: step.step_id.clone(), + attempt: 0, + status: RunStepStatus::Pending, + checkpoint: None, + failure: None, + }) + .collect(); + Ok(Self { + schema_version: RUN_STATE_V1.to_owned(), + sequence: 0, + previous_digest: String::new(), + plan_digest, + plan, + steps, + authority_decisions: Vec::new(), + recovery_decisions: Vec::new(), + }) + } + + #[must_use] + pub fn inspection_status(&self) -> RunInspectionStatus { + if self + .steps + .iter() + .all(|step| step.status == RunStepStatus::Succeeded) + { + RunInspectionStatus::Succeeded + } else if self + .steps + .iter() + .any(|step| step.status == RunStepStatus::Running) + { + RunInspectionStatus::RecoveryRequired + } else if self.steps.iter().any(|step| { + matches!( + step.status, + RunStepStatus::Failed + | RunStepStatus::Cancelled + | RunStepStatus::Denied + | RunStepStatus::Abandoned + ) + }) { + RunInspectionStatus::Stopped + } else { + RunInspectionStatus::Prepared + } + } + + /// Validate structure and correlation without recreating trusted runtime evidence. + /// + /// # Errors + /// Rejects unsupported schemas or inconsistent persisted identities and states. + pub fn validate(&self) -> Result<(), StateError> { + schema(&self.schema_version, RUN_STATE_V1)?; + require(self.sequence < MAX_RUN_SNAPSHOTS, "snapshot budget")?; + require( + if self.sequence == 0 { + self.previous_digest.is_empty() + } else { + valid_digest(&self.previous_digest) + }, + "snapshot predecessor", + )?; + require(self.plan.digest()? == self.plan_digest, "plan digest")?; + require(self.steps.len() == self.plan.steps.len(), "step inventory")?; + require( + self.authority_decisions.len() as u64 <= MAX_RUN_SNAPSHOTS + && self.recovery_decisions.len() as u64 <= MAX_RUN_SNAPSHOTS, + "decision budget", + )?; + require( + self.steps + .iter() + .filter(|s| s.status == RunStepStatus::Running) + .count() + <= 1, + "single active step", + )?; + for (step, planned) in self.steps.iter().zip(&self.plan.steps) { + require(step.step_id == planned.step_id, "step order")?; + require(step.attempt <= self.sequence, "attempt sequence")?; + require( + (step.status == RunStepStatus::Succeeded) == step.checkpoint.is_some(), + "completion requires checkpoint", + )?; + require( + matches!( + step.status, + RunStepStatus::Failed | RunStepStatus::Cancelled + ) == step.failure.is_some(), + "failure classification", + )?; + require( + (step.status == RunStepStatus::Cancelled) + == (step.failure == Some(RunFailure::Cancelled)), + "cancellation classification", + )?; + if matches!( + step.status, + RunStepStatus::Running | RunStepStatus::Succeeded | RunStepStatus::Failed + ) { + require(step.attempt > 0, "started attempt")?; + } + if let Some(checkpoint) = &step.checkpoint { + checkpoint.validate(&self.plan_digest, planned, step.attempt)?; + } + if step.attempt > 0 { + require( + self.authority_decisions.iter().any(|decision| { + decision.step_id == step.step_id + && decision.attempt == step.attempt + && decision.granted + }), + "recorded launch authority", + )?; + } + } + self.validate_decisions() + } + + fn validate_decisions(&self) -> Result<(), StateError> { + let mut authority_keys = BTreeSet::new(); + for decision in &self.authority_decisions { + schema(&decision.schema_version, RUN_AUTHORITY_V1)?; + let index = self.step_index(&decision.step_id)?; + let context = &self.plan.steps[index].context; + require( + !decision.granted || authority_keys.insert((&decision.step_id, decision.attempt)), + "unique launch authority", + )?; + require( + decision.attempt <= self.steps[index].attempt, + "authority attempt", + )?; + require( + decision.authorization_id == context.authorization_id + && decision.profile_digest == context.authority_profile_digest + && decision.enforcement_digest == context.enforcement_evidence_digest + && decision.grants_digest == context.grants_digest, + "authority identity", + )?; + } + let mut recovery_ids = BTreeSet::new(); + for decision in &self.recovery_decisions { + schema(&decision.schema_version, RUN_RECOVERY_V1)?; + let index = self.step_index(&decision.step_id)?; + let context = &self.plan.steps[index].context; + require( + valid_id(&decision.decision_id, "decision:") + && recovery_ids.insert(&decision.decision_id), + "recovery identity", + )?; + require( + decision.attempt <= self.steps[index].attempt, + "recovery attempt", + )?; + require( + decision.authorization_id == context.authorization_id + && decision.profile_digest == context.authority_profile_digest, + "recovery authority", + )?; + require( + decision.acknowledged_uncertain_effects, + "explicit recovery acknowledgement", + )?; + } + Ok(()) + } + + pub(super) fn step_index(&self, id: &str) -> Result { + self.steps + .iter() + .position(|step| step.step_id == id) + .ok_or(StateError::UnknownStep) + } +} + +impl RunCheckpoint { + fn validate( + &self, + plan_digest: &str, + planned: &PlannedStep, + attempt: u64, + ) -> Result<(), StateError> { + schema(&self.schema_version, RUN_CHECKPOINT_V1)?; + require( + valid_id(&self.checkpoint_id, "checkpoint:"), + "checkpoint identity", + )?; + require( + self.plan_digest == plan_digest + && self.context_digest == digest(&planned.context)? + && self.attempt == attempt, + "checkpoint context", + )?; + schema(&self.validation.schema_version, RUN_VALIDATION_V1)?; + require( + self.validation.profile == RUN_VALIDATION_PROFILE, + "validation profile", + )?; + require( + !self.validation.implementation_version.is_empty() + && self.validation.implementation_version.len() <= 256, + "validation implementation", + )?; + require( + valid_digest(&self.validation.implementation_digest) + && !self.validation.platform.is_empty() + && self.validation.platform.len() <= 256, + "validation implementation identity", + )?; + require( + self.validation.invocation_digest == planned.context.invocation_digest, + "validation invocation", + )?; + require( + valid_digest(&self.validation.execution_digest) + && self.validation.artifacts_digest == digest(&self.artifacts)?, + "validation evidence identity", + )?; + let bindings = &planned.context.bindings; + require( + self.artifacts.len() == bindings.inputs.len() + bindings.outputs.len(), + "checkpoint artifacts", + )?; + let observations = crate::HostArtifactObservationSet { + schema_version: crate::ARTIFACT_OBSERVATIONS_V1.to_owned(), + binding_set_id: bindings.binding_set_id.clone(), + digest_algorithm: crate::SHA256.to_owned(), + directory_manifest_profile: crate::DIRECTORY_MANIFEST_V1.to_owned(), + artifacts: self + .artifacts + .iter() + .map(|a| a.observation.clone()) + .collect(), + }; + observations.validate().map_err(|_| StateError::Invalid { + rule: "checkpoint observations", + })?; + for artifact in &self.artifacts { + schema(&artifact.schema_version, RUN_ARTIFACT_V1)?; + let observation = &artifact.observation; + let matches = match artifact.role { + RunArtifactRole::Input => bindings.inputs.iter().any(|b| { + b.artifact_id == observation.artifact_id + && b.port == observation.port + && b.locator == observation.locator + && b.kind == observation.kind + && b.media_type == observation.media_type + && b.expected_digest == observation.digest + }), + RunArtifactRole::Output => bindings.outputs.iter().any(|b| { + b.artifact_id == observation.artifact_id + && b.port == observation.port + && b.locator == observation.locator + && b.kind == observation.kind + && b.media_type == observation.media_type + }), + }; + require(matches, "checkpoint artifact binding")?; + } + Ok(()) + } +} diff --git a/src/state/store.rs b/src/state/store.rs new file mode 100644 index 0000000..7104745 --- /dev/null +++ b/src/state/store.rs @@ -0,0 +1,452 @@ +use std::fs::{self, File, OpenOptions}; +use std::io::{Read, Write}; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use super::{ + MAX_RUN_RECORD_BYTES, MAX_RUN_SNAPSHOTS, RUN_SNAPSHOT_V1, RunPlan, RunRecoveryAction, RunState, + RunStepStatus, StateError, canonical_bytes, digest, io_error, require, schema, +}; + +const LOCK_FILE: &str = "workspace.lock"; +const PENDING_FILE: &str = "snapshot.pending"; +const MAX_HISTORY_BYTES: u64 = 64 * 1024 * 1024; + +#[cfg(test)] +#[derive(Clone, Copy, Eq, PartialEq)] +enum CommitFault { + Created, + Synced, + Renamed, +} + +#[derive(Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +struct Snapshot { + schema_version: String, + state_digest: String, + state: RunState, +} + +/// One exclusively held local workspace. Dropping the handle releases its OS lock. +/// The lock file is never deleted; a killed process cannot leave a stale lock claim. +pub struct RunStore { + root: PathBuf, + _lock: File, + pub(super) state: RunState, + poisoned: bool, + #[cfg(test)] + fault: Option, +} + +impl std::fmt::Debug for RunStore { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("RunStore") + .field("run_id", &self.state.plan.run_id) + .field("sequence", &self.state.sequence) + .field("poisoned", &self.poisoned) + .finish_non_exhaustive() + } +} + +impl RunStore { + /// Create a new private workspace without replacing an existing directory. + /// + /// # Errors + /// Rejects invalid plans, existing paths, failed locking, or failed persistence. + pub fn create(workspace: &Path, plan: RunPlan) -> Result { + let state = RunState::new(plan)?; + state.validate()?; + let mut builder = fs::DirBuilder::new(); + builder.recursive(false); + #[cfg(unix)] + { + use std::os::unix::fs::DirBuilderExt; + builder.mode(0o700); + } + builder.create(workspace).map_err(|error| { + if error.kind() == std::io::ErrorKind::AlreadyExists { + StateError::AlreadyExists + } else { + io_error("create workspace", error) + } + })?; + let root = checked_root(workspace)?; + if let Some(parent) = root.parent() { + sync_directory(parent)?; + } + let lock = lock_workspace(&root, true)?; + let store = Self { + root, + _lock: lock, + state, + poisoned: false, + #[cfg(test)] + fault: None, + }; + store.write_snapshot(&store.state)?; + Ok(store) + } + + /// Exclusively reopen and validate every retained snapshot. This never executes work. + /// + /// # Errors + /// Rejects concurrent opens, partial writes, unsupported versions, corrupt history, + /// unsafe entries, and failed I/O. It never falls back to an older valid snapshot. + pub fn open(workspace: &Path) -> Result { + let root = checked_root(workspace)?; + let lock = lock_workspace(&root, false)?; + let (state, _) = load_history(&root)?; + Ok(Self { + root, + _lock: lock, + state, + poisoned: false, + #[cfg(test)] + fault: None, + }) + } + + #[must_use] + pub const fn state(&self) -> &RunState { + &self.state + } + + pub(super) fn ensure_usable(&self) -> Result<(), StateError> { + if self.poisoned { + Err(StateError::ReopenRequired) + } else { + Ok(()) + } + } + + pub(super) fn commit(&mut self, mut next: RunState) -> Result<(), StateError> { + if self.poisoned { + return Err(StateError::ReopenRequired); + } + next.sequence = self + .state + .sequence + .checked_add(1) + .ok_or(StateError::Limit)?; + next.previous_digest = digest(&self.state)?; + next.validate()?; + validate_transition(&self.state, &next)?; + let (on_disk, bytes) = match load_history(&self.root) { + Ok(history) => history, + Err(error) => { + self.poisoned = true; + return Err(error); + } + }; + if on_disk != self.state { + self.poisoned = true; + return Err(StateError::Corrupt); + } + if bytes + canonical_bytes(&next)?.len() as u64 + 256 > MAX_HISTORY_BYTES { + return Err(StateError::Limit); + } + if let Err(error) = self.write_snapshot(&next) { + self.poisoned = true; + return Err(error); + } + self.state = next; + Ok(()) + } + + fn write_snapshot(&self, state: &RunState) -> Result<(), StateError> { + let snapshot = Snapshot { + schema_version: RUN_SNAPSHOT_V1.to_owned(), + state_digest: digest(state)?, + state: state.clone(), + }; + let bytes = canonical_bytes(&snapshot)?; + if bytes.len() as u64 > MAX_RUN_RECORD_BYTES { + return Err(StateError::Limit); + } + let pending = self.root.join(PENDING_FILE); + let mut options = private_options(); + let mut file = options + .write(true) + .create_new(true) + .open(&pending) + .map_err(|error| io_error("create pending snapshot", error))?; + #[cfg(test)] + self.inject_fault(CommitFault::Created)?; + file.write_all(&bytes) + .map_err(|error| io_error("write snapshot", error))?; + file.sync_all() + .map_err(|error| io_error("sync snapshot", error))?; + drop(file); + #[cfg(test)] + self.inject_fault(CommitFault::Synced)?; + let destination = self.root.join(snapshot_name(state.sequence)); + if destination + .try_exists() + .map_err(|error| io_error("check snapshot destination", error))? + { + return Err(StateError::Corrupt); + } + fs::rename(pending, destination).map_err(|error| io_error("commit snapshot", error))?; + #[cfg(test)] + self.inject_fault(CommitFault::Renamed)?; + sync_directory(&self.root) + } + + #[cfg(test)] + fn inject_fault(&self, phase: CommitFault) -> Result<(), StateError> { + if self.fault == Some(phase) { + Err(io_error( + "injected commit interruption", + std::io::Error::other("fixture"), + )) + } else { + Ok(()) + } + } +} + +#[cfg(test)] +#[path = "store_tests.rs"] +mod tests; + +fn checked_root(workspace: &Path) -> Result { + let metadata = + fs::symlink_metadata(workspace).map_err(|error| io_error("inspect workspace", error))?; + if !metadata.is_dir() || metadata.file_type().is_symlink() { + return Err(StateError::UnsafeEntry); + } + workspace + .canonicalize() + .map_err(|error| io_error("resolve workspace", error)) +} + +fn private_options() -> OpenOptions { + let mut options = OpenOptions::new(); + options.write(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + options.mode(0o600); + } + options +} + +fn lock_workspace(root: &Path, create: bool) -> Result { + let path = root.join(LOCK_FILE); + if !create { + check_regular(&path)?; + } + let mut options = private_options(); + options.read(true).write(true); + if create { + options.create_new(true); + } + let file = options + .open(path) + .map_err(|error| io_error("open workspace lock", error))?; + if create { + file.sync_all() + .map_err(|error| io_error("sync workspace lock", error))?; + } + fs2::FileExt::try_lock_exclusive(&file).map_err(|error| { + if error.kind() == std::io::ErrorKind::WouldBlock + || error.raw_os_error() == fs2::lock_contended_error().raw_os_error() + { + StateError::Busy + } else { + io_error("lock workspace", error) + } + })?; + Ok(file) +} + +fn snapshot_name(sequence: u64) -> String { + format!("{sequence:020}.json") +} + +fn check_regular(path: &Path) -> Result { + let metadata = + fs::symlink_metadata(path).map_err(|error| io_error("inspect state entry", error))?; + if !metadata.is_file() || metadata.file_type().is_symlink() { + return Err(StateError::UnsafeEntry); + } + Ok(metadata.len()) +} + +fn load_history(root: &Path) -> Result<(RunState, u64), StateError> { + let mut paths = Vec::new(); + let mut bytes = 0_u64; + for entry in fs::read_dir(root).map_err(|error| io_error("enumerate state", error))? { + let entry = entry.map_err(|error| io_error("read state entry", error))?; + let name = entry.file_name(); + if name == PENDING_FILE { + return Err(StateError::IncompleteWrite); + } + let length = check_regular(&entry.path())?; + if name == LOCK_FILE { + require(length == 0, "empty lock file")?; + continue; + } + if length > MAX_RUN_RECORD_BYTES { + return Err(StateError::Limit); + } + bytes = bytes.checked_add(length).ok_or(StateError::Limit)?; + if bytes > MAX_HISTORY_BYTES || paths.len() as u64 >= MAX_RUN_SNAPSHOTS { + return Err(StateError::Limit); + } + paths.push(entry.path()); + } + paths.sort(); + let mut previous: Option = None; + for (index, path) in paths.iter().enumerate() { + if path.file_name().and_then(|name| name.to_str()) != Some(&snapshot_name(index as u64)) { + return Err(StateError::Corrupt); + } + let mut encoded = Vec::new(); + File::open(path) + .map_err(|error| io_error("open snapshot", error))? + .take(MAX_RUN_RECORD_BYTES + 1) + .read_to_end(&mut encoded) + .map_err(|error| io_error("read snapshot", error))?; + if encoded.len() as u64 > MAX_RUN_RECORD_BYTES { + return Err(StateError::Limit); + } + let header: serde_json::Value = + serde_json::from_slice(&encoded).map_err(|_| StateError::Malformed)?; + schema( + header + .get("schema_version") + .and_then(serde_json::Value::as_str) + .unwrap_or(""), + RUN_SNAPSHOT_V1, + )?; + let snapshot: Snapshot = + serde_json::from_slice(&encoded).map_err(|_| StateError::Malformed)?; + snapshot.state.validate()?; + if snapshot.state.sequence != index as u64 + || snapshot.state_digest != digest(&snapshot.state)? + { + return Err(StateError::Corrupt); + } + if let Some(old) = &previous { + if snapshot.state.previous_digest != digest(old)? { + return Err(StateError::Corrupt); + } + validate_transition(old, &snapshot.state)?; + } else if snapshot.state != RunState::new(snapshot.state.plan.clone())? { + return Err(StateError::Corrupt); + } + previous = Some(snapshot.state); + } + previous + .map(|state| (state, bytes)) + .ok_or(StateError::IncompleteWrite) +} + +fn validate_transition(old: &RunState, next: &RunState) -> Result<(), StateError> { + require( + next.sequence == old.sequence + 1 && next.plan == old.plan, + "immutable plan and ordered history", + )?; + require( + next.authority_decisions + .starts_with(&old.authority_decisions) + && next.recovery_decisions.starts_with(&old.recovery_decisions), + "append-only decisions", + )?; + let changed: Vec<_> = old + .steps + .iter() + .zip(&next.steps) + .filter(|(a, b)| a != b) + .collect(); + require(changed.len() == 1, "one state transition per snapshot")?; + let (before, after) = changed[0]; + let authority = &next.authority_decisions[old.authority_decisions.len()..]; + let recovery = &next.recovery_decisions[old.recovery_decisions.len()..]; + match (before.status, after.status) { + (RunStepStatus::Pending, RunStepStatus::Running) => { + require( + after.attempt == before.attempt + 1 && authority.len() == 1 && recovery.is_empty(), + "launch transition", + )?; + require( + authority[0].granted + && authority[0].step_id == after.step_id + && authority[0].attempt == after.attempt, + "launch decision", + )?; + let index = next.step_index(&after.step_id)?; + require( + next.plan.steps[index].depends_on.iter().all(|id| { + old.steps + .iter() + .any(|s| &s.step_id == id && s.status == RunStepStatus::Succeeded) + }), + "completed dependencies", + )?; + } + (RunStepStatus::Pending, RunStepStatus::Denied) => { + require( + after.attempt == before.attempt && authority.len() == 1 && recovery.is_empty(), + "denial transition", + )?; + require( + !authority[0].granted + && authority[0].step_id == after.step_id + && authority[0].attempt == after.attempt, + "denial decision", + )?; + } + (RunStepStatus::Pending, RunStepStatus::Cancelled) + | ( + RunStepStatus::Running, + RunStepStatus::Succeeded | RunStepStatus::Failed | RunStepStatus::Cancelled, + ) => { + require( + after.attempt == before.attempt && authority.is_empty() && recovery.is_empty(), + "terminal transition", + )?; + } + ( + RunStepStatus::Running + | RunStepStatus::Failed + | RunStepStatus::Cancelled + | RunStepStatus::Denied, + RunStepStatus::Pending | RunStepStatus::Abandoned, + ) => { + require( + after.attempt == before.attempt && authority.is_empty() && recovery.len() == 1, + "explicit recovery transition", + )?; + let decision = &recovery[0]; + require( + decision.step_id == after.step_id && decision.attempt == after.attempt, + "recovery attempt identity", + )?; + require( + (decision.action == RunRecoveryAction::Retry) + == (after.status == RunStepStatus::Pending), + "recovery action", + )?; + } + _ => return Err(StateError::Transition), + } + Ok(()) +} + +#[cfg(unix)] +fn sync_directory(path: &Path) -> Result<(), StateError> { + File::open(path) + .and_then(|file| file.sync_all()) + .map_err(|error| io_error("sync state directory", error)) +} + +// Windows lacks a portable directory fsync in std. Files are flushed before an +// atomic rename; the v1 Windows guarantee covers process crashes, not power loss. +#[cfg(not(unix))] +fn sync_directory(_path: &Path) -> Result<(), StateError> { + Ok(()) +} diff --git a/src/state/store_tests.rs b/src/state/store_tests.rs new file mode 100644 index 0000000..c839dd8 --- /dev/null +++ b/src/state/store_tests.rs @@ -0,0 +1,52 @@ +use super::{CommitFault, RunStore}; +use crate::{RunPlan, RunStepStatus, StateError}; +use std::fs; + +#[test] +fn interruptions_at_each_commit_boundary_preserve_an_honest_reopen_result() { + let parent = std::env::temp_dir().join(format!("flow-atomic-commit-{}", std::process::id())); + fs::create_dir(&parent).unwrap(); + for (index, phase) in [ + CommitFault::Created, + CommitFault::Synced, + CommitFault::Renamed, + ] + .into_iter() + .enumerate() + { + let workspace = parent.join(index.to_string()); + let plan: RunPlan = serde_json::from_str(include_str!( + "../../contracts/examples/run-plan.v1.example.json" + )) + .unwrap(); + let mut store = RunStore::create(&workspace, plan).unwrap(); + let original = fs::read(workspace.join("00000000000000000000.json")).unwrap(); + store.fault = Some(phase); + assert!(matches!( + store.cancel_pending("step:inspect"), + Err(StateError::Io { .. }) + )); + assert_eq!(store.state().steps[0].status, RunStepStatus::Pending); + assert!(matches!( + store.cancel_pending("step:inspect"), + Err(StateError::ReopenRequired) + )); + drop(store); + assert_eq!( + fs::read(workspace.join("00000000000000000000.json")).unwrap(), + original + ); + if phase == CommitFault::Renamed { + let reopened = RunStore::open(&workspace).unwrap(); + assert_eq!(reopened.state().steps[0].status, RunStepStatus::Cancelled); + assert_eq!(reopened.state().sequence, 1); + } else { + assert!(matches!( + RunStore::open(&workspace), + Err(StateError::IncompleteWrite) + )); + assert!(workspace.join("snapshot.pending").is_file()); + } + } + fs::remove_dir_all(parent).unwrap(); +} diff --git a/tests/durable_execution/mod.rs b/tests/durable_execution/mod.rs new file mode 100644 index 0000000..60d2212 --- /dev/null +++ b/tests/durable_execution/mod.rs @@ -0,0 +1,596 @@ +use super::{ + CAPABILITIES, INPUT_LOCATOR, KitFixture, NoSecrets, PreparedLifecycleRun, WORKSPACE_LOCATOR, + provider_binary_snapshot, +}; +use flow::{ + DurableExecutionError, EventSinkError, ExtensionEvent, NeverCancelled, PlannedStep, + ProcessStepContext, RUN_PLAN_V1, RecoveryApproval, ResumeEligibility, RunFailure, + RunInspectionStatus, RunPlan, RunRecoveryAction, RunState, RunStepStatus, RunStore, + StateBoundary, StateError, +}; +use std::fs; +use std::path::PathBuf; + +struct Harness { + kit: KitFixture, + prepared: PreparedLifecycleRun, + artifacts: PathBuf, +} + +impl Harness { + fn new(mode: &str) -> Self { + let (bytes, name) = provider_binary_snapshot(); + let kit = KitFixture::new(CAPABILITIES[0], &bytes, &name); + let prepared = kit.prepare_lifecycle(CAPABILITIES[0], mode, mode == "await-interruption"); + let artifacts = kit.root.path().join(WORKSPACE_LOCATOR); + Self { + kit, + prepared, + artifacts, + } + } + + fn context(&self) -> ProcessStepContext<'_> { + ProcessStepContext { + execution_root: self.kit.root.path(), + artifact_root: &self.artifacts, + resolved: self.prepared.resolved(), + invocation: &self.prepared.invocation, + subjects: &self.prepared.subject_lock, + authority: &self.prepared.authority, + bindings: &self.kit.bindings, + } + } + + fn plan(&self) -> RunPlan { + RunPlan { + schema_version: RUN_PLAN_V1.to_owned(), + plan_id: "plan:durable-kit".to_owned(), + run_id: self.prepared.invocation.run_id.clone(), + steps: vec![ + PlannedStep::prepare("step:inspect".to_owned(), Vec::new(), &self.context()) + .unwrap(), + ], + } + } + + fn workspace(&self) -> PathBuf { + self.kit.root.path().join("durable state") + } +} + +fn snapshots(path: &std::path::Path) -> Vec> { + let mut paths: Vec<_> = fs::read_dir(path) + .unwrap() + .map(|e| e.unwrap().path()) + .collect(); + paths.sort(); + paths.into_iter().map(|p| fs::read(p).unwrap()).collect() +} + +#[test] +fn durable_success_records_intent_before_events_and_reopens_without_reexecution() { + let fixture = Harness::new("success"); + let plan = fixture.plan(); + let mut store = RunStore::create(&fixture.workspace(), plan.clone()).unwrap(); + let source = fs::read(fixture.artifacts.join(INPUT_LOCATOR)).unwrap(); + let mut event_count = 0; + let mut sink = |_: &ExtensionEvent| { + event_count += 1; + let snapshot: serde_json::Value = serde_json::from_slice( + &fs::read(fixture.workspace().join("00000000000000000001.json")).unwrap(), + ) + .unwrap(); + let running: RunState = serde_json::from_value(snapshot["state"].clone()).unwrap(); + assert_eq!(running.steps[0].status, RunStepStatus::Running); + assert!(running.steps[0].checkpoint.is_none()); + assert!(running.authority_decisions[0].granted); + Ok::<(), EventSinkError>(()) + }; + let accepted = store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut sink, + ) + .unwrap(); + assert!(event_count > 0); + assert_eq!( + store.state().inspection_status(), + RunInspectionStatus::Succeeded + ); + assert_eq!(store.state().sequence, 2); + assert_eq!( + store.state().steps[0] + .checkpoint + .as_ref() + .unwrap() + .artifacts + .len(), + accepted.inputs().len() + accepted.outputs().len() + ); + let recorded = store.state().clone(); + drop(store); + let mut reopened = RunStore::open(&fixture.workspace()).unwrap(); + assert_eq!(reopened.state(), &recorded); + assert_eq!( + reopened + .assess(&plan, "step:inspect", &fixture.context()) + .unwrap(), + ResumeEligibility::Reusable + ); + let before = snapshots(&fixture.workspace()); + let mut events = Vec::new(); + assert!(matches!( + reopened.execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut events + ), + Err(DurableExecutionError::State(StateError::Transition)) + )); + assert!(events.is_empty()); + assert_eq!(snapshots(&fixture.workspace()), before); + assert_eq!( + fs::read(fixture.artifacts.join(INPUT_LOCATOR)).unwrap(), + source + ); +} + +#[test] +fn durable_resume_refuses_changed_plan_configuration_provider_inputs_and_artifacts() { + let fixture = Harness::new("success"); + let plan = fixture.plan(); + let mut store = RunStore::create(&fixture.workspace(), plan.clone()).unwrap(); + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new(), + ) + .unwrap(); + drop(store); + let store = RunStore::open(&fixture.workspace()).unwrap(); + let mut changed_plan = plan.clone(); + changed_plan.plan_id = "plan:changed".to_owned(); + assert!(matches!( + store.assess(&changed_plan, "step:inspect", &fixture.context()), + Err(StateError::Stale { + boundary: StateBoundary::Plan + }) + )); + for (field, boundary) in [ + ("configuration", StateBoundary::Configuration), + ("provider", StateBoundary::Provider), + ("capability", StateBoundary::Capability), + ("authority", StateBoundary::Authority), + ("invocation", StateBoundary::Invocation), + ] { + let mut invocation = fixture.prepared.invocation.clone(); + match field { + "configuration" => { + invocation + .configuration + .values + .insert("unhashed-value".to_owned(), serde_json::json!("changed")); + } + "provider" => "0.2.0".clone_into(&mut invocation.extension.version), + "capability" => "flow/other-fixture".clone_into(&mut invocation.capability_id), + "authority" => invocation.authorization.grants_digest = "a".repeat(64), + "invocation" => "invocation:changed".clone_into(&mut invocation.invocation_id), + _ => unreachable!(), + } + let context = ProcessStepContext { + invocation: &invocation, + ..fixture.context() + }; + assert!( + matches!(store.assess(&plan, "step:inspect", &context), Err(StateError::Stale { boundary: observed }) if observed == boundary), + "{field}" + ); + } + let input = fixture.artifacts.join(INPUT_LOCATOR); + let original = fs::read(&input).unwrap(); + fs::write(&input, b"changed input").unwrap(); + assert!(matches!( + store.assess(&plan, "step:inspect", &fixture.context()), + Err(StateError::Stale { + boundary: StateBoundary::Inputs + }) + )); + fs::write(&input, original).unwrap(); + fs::write(fixture.kit.output_path(), b"changed output").unwrap(); + assert!(matches!( + store.assess(&plan, "step:inspect", &fixture.context()), + Err(StateError::Stale { + boundary: StateBoundary::Artifacts + }) + )); + fs::remove_file(fixture.kit.output_path()).unwrap(); + assert!(matches!( + store.assess(&plan, "step:inspect", &fixture.context()), + Err(StateError::Stale { + boundary: StateBoundary::Artifacts + }) + )); +} + +#[test] +fn durable_input_drift_and_prestart_cancellation_prevent_launch() { + let fixture = Harness::new("success"); + let mut store = RunStore::create(&fixture.workspace(), fixture.plan()).unwrap(); + fs::write( + fixture.artifacts.join(INPUT_LOCATOR), + b"changed before launch", + ) + .unwrap(); + assert!(matches!( + store.execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new() + ), + Err(DurableExecutionError::State(StateError::Stale { + boundary: StateBoundary::Inputs + })) + )); + assert_eq!(store.state().sequence, 0); + assert!(!fixture.kit.output_path().exists()); + + let cancelled = Harness::new("success"); + let mut store = RunStore::create(&cancelled.workspace(), cancelled.plan()).unwrap(); + assert!(matches!( + store.execute( + "step:inspect", + &cancelled.context(), + &NoSecrets, + &|| true, + &mut Vec::new() + ), + Err(DurableExecutionError::CancelledBeforeLaunch) + )); + assert!(!cancelled.kit.output_path().exists()); + assert_eq!(store.state().steps[0].attempt, 0); + assert!(store.state().authority_decisions.is_empty()); +} + +#[test] +fn durable_cancellation_observed_after_intent_commit_still_prevents_launch() { + use std::cell::Cell; + let fixture = Harness::new("success"); + let mut store = RunStore::create(&fixture.workspace(), fixture.plan()).unwrap(); + let checks = Cell::new(0); + let cancellation = || { + checks.set(checks.get() + 1); + checks.get() >= 2 + }; + assert!(matches!( + store.execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &cancellation, + &mut Vec::new() + ), + Err(DurableExecutionError::CancelledBeforeLaunch) + )); + assert!(!fixture.kit.output_path().exists()); + drop(store); + let store = RunStore::open(&fixture.workspace()).unwrap(); + assert_eq!(store.state().steps[0].attempt, 1); + assert_eq!(store.state().steps[0].status, RunStepStatus::Cancelled); +} + +#[test] +fn durable_changed_validator_identity_is_inspectable_but_not_reusable() { + let fixture = Harness::new("success"); + let plan = fixture.plan(); + let mut store = RunStore::create(&fixture.workspace(), plan.clone()).unwrap(); + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new(), + ) + .unwrap(); + drop(store); + let snapshot = fixture.workspace().join("00000000000000000002.json"); + let mut value: serde_json::Value = + serde_json::from_slice(&fs::read(&snapshot).unwrap()).unwrap(); + value["state"]["steps"][0]["checkpoint"]["validation"]["implementation_digest"] = + "0".repeat(64).into(); + value["state_digest"] = super::digest_json(&value["state"]).into(); + fs::write(snapshot, serde_json::to_vec(&value).unwrap()).unwrap(); + let store = RunStore::open(&fixture.workspace()).unwrap(); + assert_eq!(store.state().steps[0].status, RunStepStatus::Succeeded); + assert!(matches!( + store.assess(&plan, "step:inspect", &fixture.context()), + Err(StateError::Stale { + boundary: StateBoundary::Validation + }) + )); +} + +#[test] +fn durable_dependency_status_blocks_launch_and_abandonment_survives_reopen() { + let fixture = Harness::new("success"); + let mut plan = fixture.plan(); + let mut predecessor = plan.steps[0].clone(); + predecessor.step_id = "step:predecessor".to_owned(); + predecessor.context.invocation_id = "invocation:predecessor".to_owned(); + predecessor.context.invocation_digest = "9".repeat(64); + predecessor.context.bindings.outputs[0].artifact_id = "artifact:predecessor".to_owned(); + plan.steps[0].depends_on = vec![predecessor.step_id.clone()]; + plan.steps.insert(0, predecessor); + let mut store = RunStore::create(&fixture.workspace(), plan.clone()).unwrap(); + assert_eq!( + store + .assess(&plan, "step:inspect", &fixture.context()) + .unwrap(), + ResumeEligibility::DependencyBlocked + ); + assert!( + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new() + ) + .is_err() + ); + assert!(!fixture.kit.output_path().exists()); + store.cancel_pending("step:inspect").unwrap(); + store + .decide_recovery( + "step:inspect", + &fixture.context(), + RecoveryApproval { + decision_id: "decision:abandon".to_owned(), + action: RunRecoveryAction::Abandon, + acknowledge_uncertain_effects: true, + }, + ) + .unwrap(); + drop(store); + let store = RunStore::open(&fixture.workspace()).unwrap(); + assert_eq!(store.state().steps[1].status, RunStepStatus::Abandoned); + assert_eq!( + store.state().recovery_decisions[0].action, + RunRecoveryAction::Abandon + ); +} + +#[test] +fn durable_provider_and_artifact_failures_survive_restart_without_checkpoints() { + for (mode, expected) in [ + ("nonzero-after-success", RunFailure::Process), + ("missing-output", RunFailure::ArtifactObservation), + ( + "contradictory-artifact-evidence", + RunFailure::ArtifactAcceptance, + ), + ] { + let fixture = Harness::new(mode); + let mut store = RunStore::create(&fixture.workspace(), fixture.plan()).unwrap(); + assert!( + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new() + ) + .is_err() + ); + drop(store); + let store = RunStore::open(&fixture.workspace()).unwrap(); + assert_eq!( + store.state().steps[0].status, + RunStepStatus::Failed, + "{mode}" + ); + assert_eq!(store.state().steps[0].failure, Some(expected), "{mode}"); + assert!(store.state().steps[0].checkpoint.is_none()); + assert_eq!( + store + .assess(&fixture.plan(), "step:inspect", &fixture.context()) + .unwrap(), + ResumeEligibility::ApprovalRequired + ); + } +} + +#[test] +fn durable_interruption_leaves_uncertain_intent_and_requires_explicit_retry() { + let fixture = Harness::new("success"); + let plan = fixture.plan(); + let mut store = RunStore::create(&fixture.workspace(), plan.clone()).unwrap(); + let interrupted = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + store.execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut |_: &ExtensionEvent| -> Result<(), EventSinkError> { + panic!("simulate host interruption after provider completion"); + }, + ) + })); + assert!(interrupted.is_err()); + drop(store); + let mut store = RunStore::open(&fixture.workspace()).unwrap(); + assert!(fixture.kit.output_path().exists()); + assert_eq!( + store.state().inspection_status(), + RunInspectionStatus::RecoveryRequired + ); + assert_eq!( + store + .assess(&plan, "step:inspect", &fixture.context()) + .unwrap(), + ResumeEligibility::ApprovalRequired + ); + assert!( + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new() + ) + .is_err() + ); + let before = snapshots(&fixture.workspace()); + assert!( + store + .decide_recovery( + "step:inspect", + &fixture.context(), + RecoveryApproval { + decision_id: "decision:missing-ack".to_owned(), + action: RunRecoveryAction::Retry, + acknowledge_uncertain_effects: false, + } + ) + .is_err() + ); + assert_eq!(snapshots(&fixture.workspace()), before); + store + .decide_recovery( + "step:inspect", + &fixture.context(), + RecoveryApproval { + decision_id: "decision:explicit-retry".to_owned(), + action: RunRecoveryAction::Retry, + acknowledge_uncertain_effects: true, + }, + ) + .unwrap(); + assert_eq!( + store + .assess(&plan, "step:inspect", &fixture.context()) + .unwrap(), + ResumeEligibility::Ready + ); + assert_eq!(store.state().steps[0].attempt, 1); + // Explicit operator cleanup preserves the uncertain output before retry; + // the store itself never overwrites or deletes provider artifacts. + fs::rename( + fixture.kit.output_path(), + fixture.kit.root.path().join("quarantined-output.json"), + ) + .unwrap(); + drop(store); + let mut store = RunStore::open(&fixture.workspace()).unwrap(); + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new(), + ) + .unwrap(); + assert_eq!(store.state().steps[0].attempt, 2); + assert_eq!(store.state().recovery_decisions.len(), 1); + assert_eq!(store.state().authority_decisions.len(), 2); +} + +#[test] +fn durable_terminal_commit_failure_never_returns_accepted_artifacts() { + let fixture = Harness::new("success"); + let mut store = RunStore::create(&fixture.workspace(), fixture.plan()).unwrap(); + let mut sink = |_: &ExtensionEvent| { + fs::write( + fixture.workspace().join("snapshot.pending"), + b"simulated incomplete write", + ) + .unwrap(); + Ok::<(), EventSinkError>(()) + }; + assert!(matches!( + store.execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut sink + ), + Err(DurableExecutionError::State(StateError::IncompleteWrite)) + )); + assert_eq!(store.state().steps[0].status, RunStepStatus::Running); + assert!(store.state().steps[0].checkpoint.is_none()); +} + +#[cfg(unix)] +#[test] +fn durable_cancellation_and_timeout_are_distinct_persisted_failures() { + for cancel in [true, false] { + let fixture = Harness::new("await-interruption"); + let mut store = RunStore::create(&fixture.workspace(), fixture.plan()).unwrap(); + let signal = || cancel && fixture.kit.lifecycle_control_path().is_file(); + assert!( + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &signal, + &mut Vec::new() + ) + .is_err() + ); + drop(store); + let store = RunStore::open(&fixture.workspace()).unwrap(); + assert_eq!( + store.state().steps[0].failure, + Some(if cancel { + RunFailure::Cancelled + } else { + RunFailure::TimedOut + }) + ); + assert!(store.state().steps[0].checkpoint.is_none()); + super::assert_recorded_process_reaped(&fixture.kit.lifecycle_control_path()); + } +} + +#[test] +fn durable_store_omits_configuration_values_and_provider_diagnostics() { + let mut fixture = Harness::new("success"); + fixture.prepared.invocation.configuration.values.insert( + "seed".to_owned(), + serde_json::json!("FLOW_STATE_PRIVATE_CANARY_4951"), + ); + fixture.prepared.invocation.configuration.digest = + super::digest_json(&fixture.prepared.invocation.configuration.values); + let plan = fixture.plan(); + let mut store = RunStore::create(&fixture.workspace(), plan).unwrap(); + store + .execute( + "step:inspect", + &fixture.context(), + &NoSecrets, + &NeverCancelled, + &mut Vec::new(), + ) + .unwrap(); + let encoded = snapshots(&fixture.workspace()).concat(); + let text = String::from_utf8(encoded).unwrap(); + assert!(!text.contains("FLOW_STATE_PRIVATE_CANARY_4951")); + assert!(!text.contains(&fixture.kit.root.path().to_string_lossy().to_string())); +} diff --git a/tests/durable_state.rs b/tests/durable_state.rs new file mode 100644 index 0000000..7f27267 --- /dev/null +++ b/tests/durable_state.rs @@ -0,0 +1,369 @@ +use std::fs; +use std::io::{BufRead, BufReader, Write}; +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Mutex, MutexGuard}; + +use flow::{RunInspectionStatus, RunPlan, RunState, RunStepStatus, RunStore, StateError}; +use serde_json::Value; + +static NEXT_ROOT: AtomicU64 = AtomicU64::new(0); +// Avoid inheriting unrelated test-owned locks during subprocess fork/exec. +// The explicit contention tests still overlap two opens of the same workspace. +static FILESYSTEM_TEST: Mutex<()> = Mutex::new(()); + +struct Root { + path: PathBuf, + _guard: MutexGuard<'static, ()>, +} + +impl Root { + fn new() -> Self { + let guard = FILESYSTEM_TEST + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let id = NEXT_ROOT.fetch_add(1, Ordering::SeqCst); + let path = std::env::temp_dir().join(format!( + "flow durable state 説明 {} {id}", + std::process::id() + )); + fs::create_dir(&path).unwrap(); + Self { + path, + _guard: guard, + } + } + + fn workspace(&self) -> PathBuf { + self.path.join("run state") + } +} + +impl Drop for Root { + fn drop(&mut self) { + let _result = fs::remove_dir_all(&self.path); + } +} + +fn plan() -> RunPlan { + serde_json::from_str(include_str!( + "../contracts/examples/run-plan.v1.example.json" + )) + .unwrap() +} + +fn state_files(root: &Path) -> Vec<(String, Vec)> { + let mut files: Vec<_> = fs::read_dir(root) + .unwrap() + .map(|entry| { + let entry = entry.unwrap(); + ( + entry.file_name().into_string().unwrap(), + fs::read(entry.path()).unwrap(), + ) + }) + .collect(); + files.sort_by(|a, b| a.0.cmp(&b.0)); + files +} + +#[test] +fn deterministic_reopen_and_cancellation_preserve_intent_and_old_snapshots() { + let root = Root::new(); + let initial = RunState::new(plan()).unwrap(); + let mut store = RunStore::create(&root.workspace(), plan()).unwrap(); + assert_eq!(store.state(), &initial); + let before = state_files(&root.workspace()); + store.cancel_pending("step:inspect").unwrap(); + assert_eq!(store.state().steps[0].attempt, 0); + assert_eq!(store.state().steps[0].status, RunStepStatus::Cancelled); + assert_eq!( + store.state().inspection_status(), + RunInspectionStatus::Stopped + ); + assert!(store.state().authority_decisions.is_empty()); + let expected = store.state().clone(); + drop(store); + let reopened = RunStore::open(&root.workspace()).unwrap(); + assert_eq!(reopened.state(), &expected); + assert!( + before + .iter() + .all(|item| state_files(&root.workspace()).contains(item)) + ); + drop(reopened); + let unchanged = state_files(&root.workspace()); + drop(RunStore::open(&root.workspace()).unwrap()); + assert_eq!( + state_files(&root.workspace()), + unchanged, + "inspection must not rewrite history" + ); +} + +#[test] +fn same_process_concurrent_open_is_refused_and_drop_releases_lock() { + let root = Root::new(); + let store = RunStore::create(&root.workspace(), plan()).unwrap(); + assert!(matches!( + RunStore::open(&root.workspace()), + Err(StateError::Busy) + )); + assert!(matches!( + RunStore::create(&root.workspace(), plan()), + Err(StateError::AlreadyExists) + )); + drop(store); + drop(RunStore::open(&root.workspace()).unwrap()); +} + +#[test] +fn denial_is_durable_and_never_records_granted_authority() { + let root = Root::new(); + let mut store = RunStore::create(&root.workspace(), plan()).unwrap(); + store.deny_pending("step:inspect").unwrap(); + drop(store); + let store = RunStore::open(&root.workspace()).unwrap(); + assert_eq!(store.state().steps[0].status, RunStepStatus::Denied); + assert_eq!(store.state().authority_decisions.len(), 1); + assert!(!store.state().authority_decisions[0].granted); + assert_eq!(store.state().steps[0].attempt, 0); +} + +#[test] +fn oversized_records_are_refused_before_deserialization() { + let root = Root::new(); + drop(RunStore::create(&root.workspace(), plan()).unwrap()); + fs::OpenOptions::new() + .write(true) + .open(root.workspace().join("00000000000000000000.json")) + .unwrap() + .set_len(flow::MAX_RUN_RECORD_BYTES + 1) + .unwrap(); + assert!(matches!( + RunStore::open(&root.workspace()), + Err(StateError::Limit) + )); +} + +#[test] +fn another_process_holds_the_lock_and_crash_releases_it() { + let root = Root::new(); + drop(RunStore::create(&root.workspace(), plan()).unwrap()); + let mut child = child_command(&root.workspace(), "hold") + .stdout(Stdio::piped()) + .stdin(Stdio::piped()) + .spawn() + .unwrap(); + let reader = BufReader::new(child.stdout.take().unwrap()); + let ready = reader + .lines() + .map(Result::unwrap) + .any(|line| line == "FLOW_STATE_CHILD_READY"); + assert!(ready); + assert!(matches!( + RunStore::open(&root.workspace()), + Err(StateError::Busy) + )); + child.kill().unwrap(); + let _status = child.wait().unwrap(); + let reopened = RunStore::open(&root.workspace()).unwrap(); + assert_eq!(reopened.state().sequence, 0); +} + +#[test] +fn abrupt_process_exit_retains_committed_cancellation_without_running_destructors() { + let root = Root::new(); + drop(RunStore::create(&root.workspace(), plan()).unwrap()); + let output = child_command(&root.workspace(), "cancel-and-exit") + .output() + .unwrap(); + assert_eq!(output.status.code(), Some(73)); + let store = RunStore::open(&root.workspace()).unwrap(); + assert_eq!(store.state().sequence, 1); + assert_eq!(store.state().steps[0].status, RunStepStatus::Cancelled); +} + +fn child_command(workspace: &Path, mode: &str) -> Command { + let mut command = Command::new(std::env::current_exe().unwrap()); + command + .args(["--exact", "workspace_child", "--ignored", "--nocapture"]) + .env("FLOW_STATE_TEST_WORKSPACE", workspace) + .env("FLOW_STATE_TEST_MODE", mode); + command +} + +#[test] +#[ignore = "subprocess entrypoint invoked by parent tests"] +fn workspace_child() { + let path = PathBuf::from(std::env::var_os("FLOW_STATE_TEST_WORKSPACE").unwrap()); + let mut store = RunStore::open(&path).unwrap(); + match std::env::var("FLOW_STATE_TEST_MODE").unwrap().as_str() { + "hold" => { + println!("FLOW_STATE_CHILD_READY"); + std::io::stdout().flush().unwrap(); + let mut line = String::new(); + std::io::stdin().read_line(&mut line).unwrap(); + } + "cancel-and-exit" => { + store.cancel_pending("step:inspect").unwrap(); + std::process::exit(73); + } + mode => panic!("unknown test mode {mode}"), + } +} + +#[test] +fn partial_write_is_reported_without_removing_or_ignoring_evidence() { + let root = Root::new(); + drop(RunStore::create(&root.workspace(), plan()).unwrap()); + fs::write(root.workspace().join("snapshot.pending"), b"{partial").unwrap(); + let before = state_files(&root.workspace()); + let opened = RunStore::open(&root.workspace()); + assert!( + matches!(opened, Err(StateError::IncompleteWrite)), + "{opened:?}" + ); + assert_eq!(state_files(&root.workspace()), before); +} + +#[test] +fn malformed_or_corrupt_latest_snapshot_never_falls_back_to_previous_success() { + for mutation in [ + "truncated", + "checksum", + "identity", + "predecessor", + "unknown-field", + "future-schema", + ] { + let root = Root::new(); + let mut store = RunStore::create(&root.workspace(), plan()).unwrap(); + store.cancel_pending("step:inspect").unwrap(); + drop(store); + let path = root.workspace().join("00000000000000000001.json"); + let mut value: Value = serde_json::from_slice(&fs::read(&path).unwrap()).unwrap(); + match mutation { + "truncated" => { + fs::write(&path, b"{").unwrap(); + } + "checksum" => { + value["state_digest"] = "0".repeat(64).into(); + } + "identity" => { + value["state"]["plan"]["run_id"] = "run:other".into(); + } + "predecessor" => { + value["state"]["previous_digest"] = "0".repeat(64).into(); + } + "unknown-field" => { + value["unexpected"] = true.into(); + } + "future-schema" => { + value["schema_version"] = "flow.run-snapshot/v2".into(); + } + _ => unreachable!(), + } + if mutation != "truncated" { + fs::write(&path, serde_json::to_vec(&value).unwrap()).unwrap(); + } + let before = state_files(&root.workspace()); + let error = RunStore::open(&root.workspace()).unwrap_err(); + if mutation == "future-schema" { + assert!(matches!(error, StateError::UnsupportedSchema)); + } + assert_eq!( + state_files(&root.workspace()), + before, + "{mutation} evidence must remain intact" + ); + } +} + +#[test] +fn sequence_gaps_and_unrecognized_entries_fail_closed() { + for name in ["00000000000000000002.json", "foreign.txt"] { + let root = Root::new(); + drop(RunStore::create(&root.workspace(), plan()).unwrap()); + fs::write(root.workspace().join(name), b"{}").unwrap(); + assert!(matches!( + RunStore::open(&root.workspace()), + Err(StateError::Corrupt) + )); + } +} + +#[test] +fn plan_validation_rejects_cycles_duplicates_and_identity_drift() { + let valid = plan(); + let mut mutated = valid.clone(); + let own_id = mutated.steps[0].step_id.clone(); + mutated.steps[0].depends_on.push(own_id); + assert!(mutated.validate().is_err()); + mutated = valid.clone(); + mutated.steps.push(mutated.steps[0].clone()); + assert!(mutated.validate().is_err()); + mutated = valid.clone(); + mutated.run_id = "run:other".to_owned(); + assert!(mutated.validate().is_err()); + mutated = valid; + mutated.schema_version = "flow.run-plan/v2".to_owned(); + assert!(matches!( + mutated.validate(), + Err(StateError::UnsupportedSchema) + )); +} + +#[test] +fn records_reject_unknown_fields_and_false_completion() { + let mut value = serde_json::to_value(RunState::new(plan()).unwrap()).unwrap(); + value["steps"][0]["status"] = "succeeded".into(); + let state: RunState = serde_json::from_value(value.clone()).unwrap(); + assert!(state.validate().is_err()); + value["steps"][0]["secret"] = "PRIVATE_CANARY".into(); + assert!(serde_json::from_value::(value).is_err()); + let mut state: RunState = serde_json::from_str(include_str!( + "../contracts/fixtures/state/completed.v1.fixture.json" + )) + .unwrap(); + state.validate().unwrap(); + state.steps[0].checkpoint.as_mut().unwrap().context_digest = "0".repeat(64); + assert!(state.validate().is_err()); +} + +#[cfg(unix)] +#[test] +fn symlink_state_and_workspace_are_rejected_and_permissions_are_private() { + use std::os::unix::fs::{PermissionsExt, symlink}; + let root = Root::new(); + drop(RunStore::create(&root.workspace(), plan()).unwrap()); + assert_eq!( + fs::metadata(root.workspace()).unwrap().permissions().mode() & 0o777, + 0o700 + ); + for (name, _) in state_files(&root.workspace()) { + assert_eq!( + fs::metadata(root.workspace().join(name)) + .unwrap() + .permissions() + .mode() + & 0o777, + 0o600 + ); + } + let alias = root.path.join("alias"); + symlink(root.workspace(), &alias).unwrap(); + assert!(matches!( + RunStore::open(&alias), + Err(StateError::UnsafeEntry) + )); + let snapshot = root.workspace().join("00000000000000000000.json"); + let outside = root.path.join("outside.json"); + fs::rename(&snapshot, &outside).unwrap(); + symlink(&outside, &snapshot).unwrap(); + assert!(matches!( + RunStore::open(&root.workspace()), + Err(StateError::UnsafeEntry) + )); +} diff --git a/tests/hermetic_provider_kit.rs b/tests/hermetic_provider_kit.rs index e4b1c11..714b3d4 100644 --- a/tests/hermetic_provider_kit.rs +++ b/tests/hermetic_provider_kit.rs @@ -3,6 +3,9 @@ mod common; #[path = "scenario_matrix/mod.rs"] mod scenario_matrix; +#[path = "durable_execution/mod.rs"] +mod durable_execution; + use std::collections::BTreeMap; use std::fs; use std::path::{Path, PathBuf}; diff --git a/tools/run_acceptance_scenarios.py b/tools/run_acceptance_scenarios.py index e2ed284..744c085 100644 --- a/tools/run_acceptance_scenarios.py +++ b/tools/run_acceptance_scenarios.py @@ -27,7 +27,8 @@ def source_identity(): """Pin the tested recipe, provider, contracts, and implementation bytes.""" paths = {"Cargo.toml", "Cargo.lock", "LICENSE", "tests/hermetic_provider_kit.rs", "tools/run_acceptance_scenarios.py"} - for directory in ["src", "contracts", "tests/scenario_matrix", "tests/fixtures", "tests/common"]: + paths.add("tests/durable_state.rs") + for directory in ["src", "contracts", "tests/scenario_matrix", "tests/durable_execution", "tests/fixtures", "tests/common"]: paths.update(str(path.relative_to(ROOT)) for path in (ROOT / directory).rglob("*") if path.is_file() and "__pycache__" not in path.parts) entries = {path: digest((ROOT / path).read_bytes()) for path in sorted(paths)} diff --git a/tools/test_durable_contracts.py b/tools/test_durable_contracts.py new file mode 100644 index 0000000..48a415a --- /dev/null +++ b/tools/test_durable_contracts.py @@ -0,0 +1,52 @@ +#!/usr/bin/env python3 +"""Regression checks for nested durable schemas and offline reference resolution.""" + +import copy +import unittest + +from validate_contracts import CONTRACTS, load_object, validate_instance + + +class DurableContracts(unittest.TestCase): + def validate(self, name, value): + errors = [] + schema = load_object(CONTRACTS / "schemas" / f"{name}.v1.schema.json") + validate_instance(value, schema, name, errors) + return errors + + def test_nested_completed_example_and_null_pending_fields(self): + for path in ["examples/run-state.v1.example.json", "fixtures/state/completed.v1.fixture.json"]: + self.assertEqual(self.validate("run-state", load_object(CONTRACTS / path)), []) + + def test_nested_future_version_unknown_field_and_bad_optional_value(self): + original = load_object(CONTRACTS / "fixtures/state/completed.v1.fixture.json") + for mutation in ["version", "private", "optional"]: + value = copy.deepcopy(original) + if mutation == "version": + value["steps"][0]["checkpoint"]["schema_version"] = "flow.run-checkpoint/v2" + elif mutation == "private": + value["plan"]["steps"][0]["context"]["configuration_values"] = {"secret": "canary"} + else: + value["steps"][0]["failure"] = {} + self.assertTrue(self.validate("run-state", value), mutation) + + def test_recovery_requires_boolean_acknowledgement(self): + value = load_object(CONTRACTS / "examples/run-recovery.v1.example.json") + for acknowledgement in [False, 1, "true", None]: + value["acknowledged_uncertain_effects"] = acknowledgement + self.assertTrue(self.validate("run-recovery", value)) + + def test_references_never_read_remote_or_escaping_paths(self): + for reference in ["https://example.com/schema.json", "../outside.schema.json", "/tmp/schema.json", "missing.schema.json"]: + errors = [] + validate_instance({}, {"$ref": reference}, "fixture", errors) + self.assertTrue(errors, reference) + + def test_external_reference_preserves_closed_nested_objects(self): + value = load_object(CONTRACTS / "examples/run-artifact.v1.example.json") + value["observation"]["source_bytes"] = "private" + self.assertTrue(self.validate("run-artifact", value)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/validate_contracts.py b/tools/validate_contracts.py index 345bf8d..15f333c 100644 --- a/tools/validate_contracts.py +++ b/tools/validate_contracts.py @@ -139,17 +139,39 @@ def validate_instance( if root_schema is None: root_schema = schema + alternatives = schema.get("anyOf") + if isinstance(alternatives, list): + matches = False + for alternative in alternatives: + alternative_errors: list[str] = [] + validate_instance(instance, alternative, location, alternative_errors, root_schema) + matches = matches or not alternative_errors + require(matches, f"{location}: does not match any permitted shape", errors) + reference = schema.get("$ref") if isinstance(reference, str): target: Any = root_schema - if reference.startswith("#/"): - for raw_segment in reference[2:].split("/"): + fragment = reference + if not reference.startswith("#"): + filename, _, pointer = reference.partition("#") + if re.fullmatch(r"[a-z][a-z0-9.-]*\.schema\.json", filename) is None: + errors.append(f"{location}: unsupported schema reference {reference}") + return + path = CONTRACTS / "schemas" / filename + if not path.is_file(): + errors.append(f"{location}: missing local schema reference {reference}") + return + root_schema = load_object(path) + target = root_schema + fragment = "#" + pointer if pointer else "#" + if fragment.startswith("#/"): + for raw_segment in fragment[2:].split("/"): segment = raw_segment.replace("~1", "/").replace("~0", "~") if not isinstance(target, dict) or segment not in target: errors.append(f"{location}: unresolved schema reference {reference}") return target = target[segment] - else: + elif fragment != "#": errors.append(f"{location}: unsupported schema reference {reference}") return if not isinstance(target, dict): @@ -164,6 +186,7 @@ def validate_instance( "string": lambda value: isinstance(value, str), "integer": lambda value: isinstance(value, int) and not isinstance(value, bool), "boolean": lambda value: isinstance(value, bool), + "null": lambda value: value is None, } if expected_type in type_checks and not type_checks[expected_type](instance): errors.append(f"{location}: expected {expected_type}") @@ -176,6 +199,8 @@ def validate_instance( if isinstance(instance, str): if "minLength" in schema: require(len(instance) >= schema["minLength"], f"{location}: string is too short", errors) + if "maxLength" in schema: + require(len(instance) <= schema["maxLength"], f"{location}: string is too long", errors) if "pattern" in schema: require(bool(re.search(schema["pattern"], instance)), f"{location}: does not match pattern", errors) if isinstance(instance, int) and not isinstance(instance, bool) and "minimum" in schema: @@ -1737,6 +1762,14 @@ def main() -> int: rejected_invalid_instances += 1 expected = { + "flow.run-plan/v1", + "flow.run-artifact/v1", + "flow.run-validation/v1", + "flow.run-checkpoint/v1", + "flow.run-authority/v1", + "flow.run-recovery/v1", + "flow.run-state/v1", + "flow.run-snapshot/v1", "flow.artifact-bindings/v1", "flow.artifact-observations/v1", "flow.artifact/v1", From 948eb4dfeb17e701bae61978bd6d593d28ee85e8 Mon Sep 17 00:00:00 2001 From: Alan Szmyt Date: Fri, 25 Sep 2026 22:24:58 -0400 Subject: [PATCH 2/2] test(flow): respect Windows workspace lock semantics Inspect the empty coordination file through metadata instead of reading it through a second handle while Windows holds a byte-range lock. Select stable explicitly for the macOS and Windows portability jobs. Roadmap-Step: FLO-Q03 --- .github/workflows/ci.yml | 2 ++ tests/durable_state.rs | 13 +++++++++---- 2 files changed, 11 insertions(+), 4 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 81b1a49..40e2280 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -109,6 +109,8 @@ jobs: name: Durable state ${{ matrix.os }} runs-on: ${{ matrix.os }} timeout-minutes: 10 + env: + RUSTUP_TOOLCHAIN: stable strategy: fail-fast: false matrix: diff --git a/tests/durable_state.rs b/tests/durable_state.rs index 7f27267..16a46c1 100644 --- a/tests/durable_state.rs +++ b/tests/durable_state.rs @@ -58,10 +58,15 @@ fn state_files(root: &Path) -> Vec<(String, Vec)> { .unwrap() .map(|entry| { let entry = entry.unwrap(); - ( - entry.file_name().into_string().unwrap(), - fs::read(entry.path()).unwrap(), - ) + let bytes = if entry.file_name() == "workspace.lock" { + // Windows byte-range locks also forbid reads through other handles. + // The empty coordination file has no persisted payload to compare. + assert_eq!(entry.metadata().unwrap().len(), 0); + Vec::new() + } else { + fs::read(entry.path()).unwrap() + }; + (entry.file_name().into_string().unwrap(), bytes) }) .collect(); files.sort_by(|a, b| a.0.cmp(&b.0));