From a96d9cc59a729dea367a3c20da2e56b3f64dd98d Mon Sep 17 00:00:00 2001 From: justinhelmer <1403438+justinhelmer@users.noreply.github.com> Date: Thu, 24 Sep 2026 16:44:34 +0000 Subject: [PATCH] feat(ship): recover original units without replacement work Preserve original durable identity, publication ownership, budgets, and idempotency while a separate Workflow resumes only the proven findings or review checkpoint. Fail closed on moved or ambiguous evidence and prevent terminal recovery replies from creating replacement pipelines. --- deploy/cloudflare-memory/runLedger.test.ts | 58 +- deploy/cloudflare-memory/worker.ts | 52 +- deploy/cloudflare/coordinator.test.ts | 6 +- deploy/cloudflare/coordinator.ts | 19 +- deploy/cloudflare/worker.ts | 5 +- ...-and-a-pipeline-idles-instead-of-ending.md | 36 + docs/reference/specs/agent-ship.md | 4 + docs/reference/specs/http-ingress.md | 3 +- docs/reference/specs/run-history.md | 1 + docs/reference/specs/thread-admission.md | 5 +- src/channels/adminCoordinator.test.ts | 1390 ++++++++++++++++- src/channels/adminCoordinator.ts | 823 +++++++++- src/core/coordinator/contract.test.ts | 47 + src/core/coordinator/contract.ts | 141 +- src/core/coordinator/driver.test.ts | 227 +++ src/core/coordinator/driver.ts | 232 ++- src/core/coordinator/instanceStore.ts | 25 + src/core/coordinator/instancesClient.test.ts | 14 + src/core/coordinator/instancesClient.ts | 8 +- src/core/dispatch/reattach.test.ts | 34 +- src/core/dispatch/reattach.ts | 8 + src/core/dispatch/reply.test.ts | 3 + src/core/dispatch/run.ts | 4 + src/core/dispatch/ship.test.ts | 74 +- src/core/dispatch/ship.ts | 54 + src/core/dispatcher.test.ts | 195 +++ src/core/dispatcher.ts | 186 ++- src/core/refusal.ts | 1 + src/core/runEvents.ts | 2 + src/core/runnerOwnership.test.ts | 35 + src/core/runnerOwnership.ts | 14 +- src/core/ship/coordinator.test.ts | 33 + src/core/ship/coordinator.ts | 106 +- src/core/trace/durationAllowlist.json | 1 + src/index.ts | 22 +- 35 files changed, 3753 insertions(+), 115 deletions(-) diff --git a/deploy/cloudflare-memory/runLedger.test.ts b/deploy/cloudflare-memory/runLedger.test.ts index 9eb95befa..b83925ce6 100644 --- a/deploy/cloudflare-memory/runLedger.test.ts +++ b/deploy/cloudflare-memory/runLedger.test.ts @@ -701,11 +701,25 @@ describe("run ledger — the coordinator's event and the key (items 47–48)", ( expect(sent).toEqual([]); // The interrupted close written outside `finish`: one send, the record's status on it. expect( - await post("/runs/put", { storeKey: key, record: { ...childRecord("p1", "slack:C1:3.0", "interrupted") } }), + await post("/runs/put", { + storeKey: key, + record: { + ...childRecord("p1", "slack:C1:3.0", "interrupted"), + events: [ + { + type: "coordinator_tag", + parentInstanceId: "ship_acme_api_1", + transportWorkflowId: "recovery-review-1", + at: 1, + }, + ], + eventCount: 1, + }, + }), ).toMatchObject({ status: 200, data: { ok: true, stored: true } }); expect(sent).toEqual([ { - instance: "ship_acme_api_1", + instance: "recovery-review-1", type: "run-finished-p1", payload: expect.objectContaining({ runId: "p1", status: "interrupted", parentInstanceId: "ship_acme_api_1" }), }, @@ -1105,6 +1119,46 @@ describe("run ledger — the coordinator's unit rows (item 50)", () => { }); }); + it("lists only active recovery rows in SQL, ignores terminal history, and fails closed on malformed candidate JSON", async () => { + const key = storeKey(); + const active = unit("U12", { + pr: { number: 7, url: "https://github.com/acme/api/pull/7" }, + recovery: { + kind: "review", + round: 2, + expectedHeadSha: "a".repeat(40), + remainingMs: 60_000, + claimedAt: 2_000, + step: "U12/recovery/2/review", + reviewRunId: "review-1", + previousEnding: { kind: "aborted", report: "recoverable", at: 1_000 }, + workflowId: "recovery-review-1", + deadlineAt: 62_000, + reviewKey: `${INSTANCE_ID}:U12/1/review`, + }, + }); + const terminal = unit("U13", { + ending: { kind: "merged", report: "done", at: 3_000 }, + }); + expect((await post("/runs/coordinator/units/put", { storeKey: key, units: [active, terminal] })).status).toBe(200); + const stub = env.RUNS.get(env.RUNS.idFromName(key)); + + await expect(runInDurableObject(stub, (inst: RunHistoryDO) => inst.listActiveRecoveries())).resolves.toEqual([ + active, + ]); + + await runInDurableObject(stub, async (_inst, state) => { + state.storage.sql.exec( + `INSERT INTO coordinator_units (instance_id, unit, json, updated_at) VALUES (?, ?, ?, ?)`, + INSTANCE_ID, + "U14", + `{"recovery":`, + 4_000, + ); + }); + await expect(runInDurableObject(stub, (inst: RunHistoryDO) => inst.listActiveRecoveries())).rejects.toThrow(); + }); + it("the validating wake boundary accepts a first-segment resume without inventing a renewal segment row", async () => { const key = storeKey(); const row = unit("U12", { diff --git a/deploy/cloudflare-memory/worker.ts b/deploy/cloudflare-memory/worker.ts index e4c87f8d4..176c4c795 100644 --- a/deploy/cloudflare-memory/worker.ts +++ b/deploy/cloudflare-memory/worker.ts @@ -117,6 +117,7 @@ import { sendRunFinished, STEP_NAME_PATTERN, UNIT_PATTERN, + unitOfIdempotencyKey, type CoordinatorInstance, type CoordinatorUnit, type RunFinishedSend, @@ -2771,6 +2772,37 @@ export class RunHistoryDO extends DurableObject { .map((r) => JSON.parse(r.json) as CoordinatorUnit); } + async listActiveRecoveries(): Promise { + return this.sql + .exec<{ json: string }>( + `SELECT json FROM coordinator_units + WHERE json_valid(json) = 0 OR json_type(json, '$.recovery') IS NOT NULL + ORDER BY rowid`, + ) + .toArray() + .map((r) => { + const unit: unknown = JSON.parse(r.json); + if (!isCoordinatorUnit(unit) || unit.recovery === undefined) + throw new Error("active recovery index contains a malformed coordinator unit"); + return unit; + }); + } + + private recoveryTransport(parentInstanceId: string, idempotencyKey: string | undefined): string | undefined { + const unit = idempotencyKey === undefined ? undefined : unitOfIdempotencyKey(idempotencyKey); + if (unit === undefined) return undefined; + const row = this.sql + .exec<{ json: string }>( + `SELECT json FROM coordinator_units WHERE instance_id = ? AND unit = ?`, + parentInstanceId, + unit, + ) + .toArray()[0]; + if (row === undefined) return undefined; + const parsed = JSON.parse(row.json) as CoordinatorUnit; + return parsed.recovery?.workflowId; + } + // ---- the thread events of a unit-owned thread (record 0051's reply-as-event rule) -------------- /** The next sequence assigned in one transaction, the per-event cap applied @@ -3025,9 +3057,11 @@ export class RunHistoryDO extends DurableObject { // tells the waiting parent the child resumed — best effort, beside the // bot's own announcement; a duplicate is consumed and re-armed, harmless. if (req.meta.restartOf !== undefined && req.meta.parentInstanceId !== undefined) { + const recoveryTransport = this.recoveryTransport(req.meta.parentInstanceId, req.meta.idempotencyKey); const sent = await sendChildSignal(this.env.SHIP_COORDINATOR, { runId: req.runId, parentInstanceId: req.meta.parentInstanceId, + ...(recoveryTransport !== undefined ? { transportWorkflowId: recoveryTransport } : {}), kind: "resumed", reason: `restarted from run ${req.meta.restartOf}`, at: now, @@ -3318,7 +3352,14 @@ export class RunHistoryDO extends DurableObject { if ((await this.ctx.storage.getAlarm()) === null) await this.ctx.storage.setAlarm(systemClock() + RUN_SWEEP_INTERVAL_MS); await this.refreshSessionBytes(record.session?.key); - const event = await sendRunFinished(this.env.SHIP_COORDINATOR, record); + const transportWorkflowId = + record.parentInstanceId === undefined + ? undefined + : this.recoveryTransport(record.parentInstanceId, record.idempotencyKey); + const event = await sendRunFinished(this.env.SHIP_COORDINATOR, { + ...record, + ...(transportWorkflowId !== undefined ? { transportWorkflowId } : {}), + }); if (event.kind === "failed") console.warn(`[runs/finish] ${runId} → ${event.type} not delivered to ${event.instance}: ${event.reason}`); return { ...out, event: event.kind }; @@ -3654,7 +3695,11 @@ export class RunHistoryDO extends DurableObject { // `finishedAt` equals `startedAt`); the parent confirms by `read-record` // before it acts, so a duplicate send is harmless. if (result.stored && record.parentInstanceId !== undefined && record.finishedAt > record.startedAt) { - const event = await sendRunFinished(this.env.SHIP_COORDINATOR, record); + const transportWorkflowId = record.events.find((event) => event.type === "coordinator_tag")?.transportWorkflowId; + const event = await sendRunFinished(this.env.SHIP_COORDINATOR, { + ...record, + ...(transportWorkflowId !== undefined ? { transportWorkflowId } : {}), + }); if (event.kind === "failed") console.warn(`[runs/put] ${record.id} → ${event.type} not delivered to ${event.instance}: ${event.reason}`); } @@ -5220,6 +5265,7 @@ const LEDGER_ROUTES = new Set([ "/runs/coordinator/stop", "/runs/coordinator/units/put", "/runs/coordinator/units/claim-legacy-continuation", + "/runs/coordinator/units/list-active-recoveries", "/runs/coordinator/units/list", "/runs/coordinator/events/append", "/runs/coordinator/events/list", @@ -5935,6 +5981,8 @@ async function handleLedger(pathname: string, body: unknown, env: Env): Promise< return json({ error: "instanceId must be a Workflow instance id" }, 400); return json({ units: await stub.listUnits(b.instanceId) }); } + if (pathname === "/runs/coordinator/units/list-active-recoveries") + return json({ units: await stub.listActiveRecoveries() }); if (pathname === "/runs/coordinator/wake") { if (!isCoordinatorUnit(b.unit)) return json({ error: "unit must be a coordinator unit row" }, 400); if (typeof b.waitId !== "string" || !STEP_NAME_PATTERN.test(b.waitId)) diff --git a/deploy/cloudflare/coordinator.test.ts b/deploy/cloudflare/coordinator.test.ts index c7234e595..8a36b1b00 100644 --- a/deploy/cloudflare/coordinator.test.ts +++ b/deploy/cloudflare/coordinator.test.ts @@ -81,7 +81,11 @@ describe("the coordinator holds no credential", () => { ); expect(source).not.toMatch(/https?:\/\/(?!switchboard-keepalive\.internal)/); }); - it("runs the plan runner's driver and nothing of its own: `run()` is one `runPlan` over the platform's step and the container bot", () => { + it("selects the recovery driver only from typed recovery params and otherwise runs the plan driver", () => { + expect(source).toContain('if (event.payload.kind === "recover-original-unit")'); + expect(source).toMatch( + /return runOriginalUnitRecovery\(workflowSteps\(step\), containerBot\(this\.env\), event\.instanceId, params\);/, + ); expect(source).toMatch(/return runPlan\(workflowSteps\(step\), containerBot\(this\.env\), event\.instanceId\);/); }); }); diff --git a/deploy/cloudflare/coordinator.ts b/deploy/cloudflare/coordinator.ts index 03ff37c82..32077dcde 100644 --- a/deploy/cloudflare/coordinator.ts +++ b/deploy/cloudflare/coordinator.ts @@ -22,7 +22,9 @@ import { NonRetryableError } from "cloudflare:workflows"; import { COORDINATOR_IDENTITY, COORDINATOR_STEP_PATH_PREFIX } from "../../src/core/coordinator/contract.ts"; import { readBotAnswer, + runOriginalUnitRecovery, runPlan, + type OriginalUnitRecoveryParams, type CoordinatorBot, type PlanRunSummary, type StepRunner, @@ -38,7 +40,17 @@ export type CoordinatorEnv = Pick; +export type ShipCoordinatorParams = Record | OriginalUnitRecoveryParams; + +function originalUnitRecoveryParams(value: ShipCoordinatorParams): OriginalUnitRecoveryParams | undefined { + if ( + value.kind !== "recover-original-unit" || + typeof value.parentInstanceId !== "string" || + typeof value.unit !== "string" + ) + return undefined; + return { kind: value.kind, parentInstanceId: value.parentInstanceId, unit: value.unit }; +} /** The bot behind the container binding: the reply as the wire carried it. * The transport's failures throw for the step's retry; the door's own refusal @@ -81,6 +93,11 @@ function workflowSteps(step: WorkflowStep): StepRunner { export class ShipCoordinator extends WorkflowEntrypoint { async run(event: Readonly>, step: WorkflowStep): Promise { + if (event.payload.kind === "recover-original-unit") { + const params = originalUnitRecoveryParams(event.payload); + if (params === undefined) throw new NonRetryableError("the original-unit recovery params are malformed"); + return runOriginalUnitRecovery(workflowSteps(step), containerBot(this.env), event.instanceId, params); + } return runPlan(workflowSteps(step), containerBot(this.env), event.instanceId); } } diff --git a/deploy/cloudflare/worker.ts b/deploy/cloudflare/worker.ts index d33b13330..ddb5393a2 100644 --- a/deploy/cloudflare/worker.ts +++ b/deploy/cloudflare/worker.ts @@ -325,7 +325,10 @@ async function handleCoordinatorInstances(request: Request, env: Env): Promise:/…` idempotency namespace are inputs, not values a recovery caller may supply. The latest completed review boundary must itself be `request_changes` or `no_verdict`: a later review start or approval makes the older boundary ambiguous or stale. A posted `request_changes` verdict resumes that round's findings step; a `no_verdict` ending may start one read-only review. An externally moved head is a refusal, not an adoption path. **Amended in re-evaluation on 2026-09-24:** a terminal `step_threw` exactly at that round's `findings/pr-check` may instead prove that the original owned findings child already completed and pushed the current head before the publication check failed. This preserves the accepted reasons—the same unit, owner, budget and namespace produced the head—and narrows the next action to read-only re-review. It requires one exact completed coding record, the original findings key, the original unit thread, a pushed-head receipt for the original ref, ordering after the authorizing review and before the terminal ending, and full fresh PR/ref/base/head agreement; missing, duplicate or contradictory evidence remains a moved-head refusal. + +The commit boundary is one store-owned compare-and-replace of the exact terminal row into an explicit recovery claim, paired with transfer of the exact publication owner before any child admission. Replay reads the claim and starts no second child; a stale row, rival or unknown owner, exhausted lease or rounds, incomplete or contradictory lifecycle, missing or ambiguous child evidence, review-post failure, or any identity/binding mismatch starts nothing. A failure after provisional ownership restores the byte-identical prior row and owner; cleanup failure is surfaced and the old publication binding is never silently replaced. Legacy rows remain readable elsewhere, but a row carrying both `idle` and `ending` is uninterpretable for this transition and is refused. + +The original Workflow is terminal and cannot be reopened, while its ordinary re-issue creates a new coordinator attempt and budget. The claim therefore creates a separate physical `recovery-` Workflow as an execution checkpoint only; it does not resume the terminal backing execution. Its first step revalidates the durable claim, publication owner and unchanged GitHub head; actual child admission repeats the owner, fixed-head and absolute-deadline checks because Workflow steps are cached. Every bot call, child tag and idempotency key after that still names the original instance and unit. The checkpoint's rebase-attempt and asynchronous-review-restart counters begin at zero because they name steps only inside its distinct `/recovery/` namespace; the durable review-round number remains the lifetime round gate. A definite Workflow-create failure restores the prior row and owner. A create whose result or status cannot be read keeps the claim because the Workflow may already exist, so retry meets the same Workflow id instead of spending again. A terminal duplicate restores the prior ending with a consumed-evidence receipt rather than stranding or replaying the claim. After a bot restart, the durable claim reconstructs only its exact process-local publication owner before revalidation; a rival owner still refuses it. Replies nudge the recovery Workflow transport id while child identity remains the original instance and unit. + +**One hard-case trace.** An original `U12` ended after review round one posted `request_changes`, with 60 minutes left and head `7777777777777777777777777777777777777777`. The original requester invokes `agent:ship recover unit :U12` in the original thread. Admission reads the original instance and row, selects only that segment's exact `:U12[/sN]/1/review[/aN]` record, verifies its posted findings and the still-open pull request at that exact head, reserves the publication owner, and replaces the terminal row with a claim naming `recovery-` and the lease's absolute deadline. The Workflow create times out after the platform may have accepted it, so admission leaves the claim and owner intact. A retry uses the same Workflow id and receives `duplicate`; the Workflow's first step reloads the claim, reconstructs the same owner after a bot restart, and rechecks the unchanged head. Actual spawn admission repeats those checks and refuses if scheduling delay spent the lease. It starts `:U12/recovery/1/findings`; only that completed child may atomically advance the publication binding to its recorded push before `:U12/recovery/2/review` starts. The compatibility variant begins one step later: the terminal row names `U12/1/findings/pr-check`, the exact original findings record proves its completed push to the current head, and admission atomically updates the publication binding while claiming recovery directly at round two's read-only review. It never runs unit-start, branch or round zero. Terminal `unit-end` compare-and-replaces the claim with the new ending plus a consumed-evidence receipt and releases the owner; a settlement refusal is retried and cannot be reported as completion. This trace proves that an ambiguous transport result, cached step and process restart create neither a sibling unit nor a second child. + +**Difficulty map for this amendment.** + +1. **The claim-to-Workflow boundary** is most likely to be wrong: ambiguous create results, exact rollback and process restart must preserve one owner and one execution. +2. **Stage reconstruction** must choose exactly one current-round owned review record and refuse every stale, foreign, unposted or moved-head candidate. +3. **The remaining budgets** must be provable from the durable row and enforced at actual admission. A full segment carries its start time; a stopped-segment wake lacks the resume timestamp and a cost-capped terminal row lacks cumulative spend, so both refuse rather than minting time or money. The claim stores an absolute deadline so scheduling delay cannot renew the lease. Legacy terminal rows do not retain a complete lifetime coding/review/waiting split, so a recovery report labels those counters as checkpoint-local and explicitly excludes earlier activity instead of presenting zeroed counters as unit history. + +**Why not dispatch the child directly from the Slack request?** Deterministic child idempotency can deduplicate the first spawn, but the Slack request cannot durably drive findings reconciliation, re-review and terminal settlement after it returns or the bot restarts. The separately named Workflow owns that multi-step continuation while every child remains in the original unit namespace. + +This amendment changes none of the record's ordinary idle, warm, routing or re-issue behavior. It adds a privileged repair boundary for a named original unit, only while the remote head is unchanged or is the one exact head already pushed by that original unit's completed findings child before its terminal publication check failed. External moved-head recovery, lifecycle normalization for all writers, and replacement of the existing legacy decoders remain separate decisions. The acceptance reasons still hold: the unit, not prose, owns continuation; one durable transition admits one next child; and the original clock and grants constrain the work rather than being renewed by recovery. + +### Validation criteria added + +| Criterion | Proof | +|---|---| +| Unchanged-head `request_changes` recovery commits the original row and publication owner once, then starts the original round's findings child with the original parent and idempotency namespace | `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::unchanged-head request_changes CAS-claims the original row and owner, then admits a Workflow carrying only the original unit identity`; `src/core/coordinator/driver.test.ts::runOriginalUnitRecovery::continues request_changes through the original findings namespace and re-review without a start, branch, or generated unit` | +| Unchanged-head `no_verdict` recovery commits once, then starts one read-only review as the original unit within its remaining lease | `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::unchanged-head no_verdict admits one named Workflow and replay reuses its durable claim without charging or dispatching`; `src/core/coordinator/driver.test.ts::runOriginalUnitRecovery::continues no_verdict with one read-only review in the original namespace` | +| An original findings child that completed and pushed before its terminal `findings/pr-check` failure advances the same publication binding once and resumes directly at read-only re-review; any external, missing or ambiguous moved head still refuses | `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::reconstructs a failed original findings pr-check from its completed owned push and resumes at read-only re-review`; `src/core/coordinator/driver.test.ts::runOriginalUnitRecovery::continues a completed original findings checkpoint directly with its read-only re-review` | +| Missing, unreadable, ambiguous, foreign or mismatched evidence; moved head; lifecycle ambiguity; stale CAS; ownership loss; exhausted rounds or lease; review-post failure; and rollback failure all refuse before child admission without changing the prior binding | `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::*` | +| Restart, replay and concurrent retry neither double-charge the remaining budget nor start two children, and a later writer still passes the ordinary publication gate | `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::keeps an ambiguous Workflow admission claimed so retry can meet the same Workflow id without a second budget charge`; `::reconstructs the exact process-local publication owner from a durable claim after restart`; `::terminal settlement clears the durable claim by CAS and releases the original publication owner`; `src/core/coordinator/driver.test.ts::runOriginalUnitRecovery::*` | diff --git a/docs/reference/specs/agent-ship.md b/docs/reference/specs/agent-ship.md index aa1f2a22b..c3cc60fb3 100644 --- a/docs/reference/specs/agent-ship.md +++ b/docs/reference/specs/agent-ship.md @@ -71,6 +71,7 @@ The coding → review → fix loop to LGTM as one [pipeline](../vocabulary.md#pi | 2: compound gate — allowed ship but denied coding → refused naming coding; denied repo → refused; nothing handed to the runner either way | `[unit]` `src/core/dispatcher.test.ts::agent:ship (the hand-off to the plan runner)::permission: user allowed ship but not coding…`, `::allowed all three agents but denied the target repo…`, `src/core/dispatch/ship.test.ts::runShipBranch — the agent:ship fork hands every admitted request to the plan runner::refused at the preflight…` | | 9: auto-merge is the pull request's own fact — parsed from `auto_merge`, named at entry in the hand-off's reply and at the approved head by the `merge_ready` report reading the facts fresh; no repository-level check remains and a failed repository lookup refuses nothing | `[unit]` `src/execution/githubPulls.test.ts::githubPulls::fetchPullRequestFacts (ship entry checks)::autoMergeEnabled is true for a non-null auto_merge…`; `src/core/dispatcher.test.ts::agent:ship (the hand-off to the plan runner)::an adopted person-authored PR behaves identically…`, `::a failed repository lookup refuses nothing…`; `src/core/ship/coordinator.test.ts::the unit pipeline — every ending the ship pipeline has, on step returns::merge-ready under…`; `src/core/coordinator/driver.test.ts::the plan runner's driver — a resume at review (agent-ship item 10)::a task row carrying a resume re-reads the pull request…` | | 3, 16: a task the preflight admits is handed as a generated one-unit plan instance named by the task and the thread — the `U1` row on the deterministic `plan//u1` branch off the default branch, no resume, the reply naming the runner, the card ✅, no workspace attached and no model turn in the bot; a repo-shaped `repoCtx.ref` never becomes the base | `[unit]` `src/core/dispatcher.test.ts::agent:ship (the hand-off to the plan runner)::a task the preflight admits is handed to the runner…`, `::a repo-shaped repoCtx.ref never becomes the pipeline base…` | +| 9, 10: every publication ownership claim and reservation fails closed while a complete periodic ownership rebuild is in flight. A runner rebase and a legacy continuation return the named retryable `publication_ownership_unknown`; neither invokes the resolver, starts replacement work, weakens a foreign-owner refusal nor mutates the durable row | `[unit]` `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/merge — the runner's squash of a unit's pull request (item 9)::the runner's rebase returns the named transient while periodic ownership recovery is in flight`; `src/core/dispatcher.test.ts::a unit-owned thread (record 0051's reply-as-event and gone-instance rules)::a legacy continuation returns publication_ownership_unknown when periodic ownership recovery starts before reservation` | | 9, 10: exact continuation head and existing-PR publication — every future `merge_ready` ending persists the final reviewed full head as `lastPush`; adopt/resume requires an open same-repository PR with explicit head/base refs, an existing base and a full head SHA; the hand-off persists repository, PR, refs, expected head, publication ref and owner, while first discovery persists the same authority for later coding rounds; a legacy `merge_ready` reply recovers only one head jointly attested by that same unit's completed coding and posted approving review records, rechecks requester, exact live PR/ref/base/head and sole ownership, then takes one exclusive ownership token after agent/profile gates and before admission; only the hand-off's final start gate conditionally replaces the exact legacy row with `lastPush` plus the complete binding before reissue, then the exact token and current owner transfer sole ownership to the durable new attempt before Workflow create; an intervening owner, transfer failure, create refusal or later gate restores only the exact legacy row and conditionally releases only the reservation or transferred owner, while a losing rollback CAS is surfaced without clobbering newer state; absent, partial, ambiguous, moved, foreign, closed, wrong-base, wrong-requester or changed-owner evidence starts nothing; spawn refuses a missing binding or changed/unknown sole owner; the run compares request, workspace, durable and fresh PR facts before opening the harness; pushes and every salvage path require the exact destination plus an atomic full-SHA lease, so stale, moved, foreign, missing, mismatched, deleted, unavailable and concurrent-movement states leave work unpublished and create no alternate ref | `[unit]` `src/core/ship/preflight.test.ts::shipPreflight — the base ref existence check before the pipeline branch is cut (agent-ship item 10, issues 1827 and 2161)::an existing pull request whose bound base is missing fails closed instead of publishing against a fallback`, `::a pull request's own base that exists leaves the adopt unchanged; a PR without its own base fails before a lookup`; `src/core/coordinator/handOff.test.ts::handOffToCoordinator — the ship request as a plan runner instance (item 16)::a resume at review (agent-ship item 10) is the one generated unit with the pull request on its row and the entry's branch — the pull request's own head — so the runner opens it at the review round; the reply names the pull request and that no coding round runs first`; `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/spawn — the child as the parent record's requester (item 9)::an existing-PR coding spawn carries the durable publication binding only for its sole owner and fails closed after ownership changes`, `src/channels/adminCoordinator.test.ts::pr-check keeps the machine's adopted pull request authoritative before branch discovery (agent-ship items 9, 10 and 12)::persists the first complete same-repository PR read as future coding rounds' exact publication authority`; `src/core/existingPrPublication.test.ts::verifyExistingPrPublication — exact existing-PR publication binding::*`; `src/core/dispatch/runLoop.test.ts::the pi harness — every preset's runs, in the run's container::the gate's push rules follow the thread: a pull-request thread's or a unit child's bound branch is the run's own — the one push target, its base protected; a plain thread's binding is the protected base`, `src/core/dispatch/runLoop.test.ts::runLoop — the model turn and everything that rides on it::a blocked existing-PR publication keeps the local checkpoint attached and renders truthful partial status without an alternate push`; `src/core/harness/pi/toolRules.test.ts::judgeToolCall — an existing-PR publication fence::*`; `src/core/coordinator/driver.test.ts::the plan runner's driver — the Workflow body over the step runner (item 9)::a review that requests changes is followed by findings and re-review at the moved head…`; `src/core/dispatcher.test.ts::a unit-owned thread (record 0051's reply-as-event and gone-instance rules)::a legacy merge-ready row recovers one exact approved child head for the same unit, persists the full binding before reissue, and starts no substitute run`, `::a real reissue transfers the recovered pull request to the new attempt, whose coding spawn passes the publication-owner gate`, `::two concurrent legacy recoveries conditionally replace the same row once, so the losing caller fails closed without reissue`, `::an ownership claim that lands during the durable compare makes the legacy continuation lose without overwriting ownership or starting work`, `::an ownership change whose exact rollback loses a CAS race surfaces the failure without clobbering newer state or starting work`, `::a later ship authorization refusal leaves a legacy row unrepaired and ownership unclaimed`, `::legacy merge-ready recovery*`; `src/core/coordinator/handOff.test.ts::handOffToCoordinator — the ship request as a plan runner instance (item 16)::a hand-off refusal before the final start gate never reserves a legacy transition`, `::a provisional legacy transition transfers to the durable new attempt before create and rolls back when creation refuses`, `::a refused ownership transfer aborts before the Workflow can start`; `src/core/runnerOwnership.test.ts::RunnerOwnershipFence::an exclusive reservation transfers only from its exact token and current owner to the new runner`, `::a failed reissued start releases only the transferred owner and never a successor`, `::an intervening runner cannot displace an existing owner, and a stale release cannot erase it` | | 10: entry checks — a user-named open PR with no new task → a resume at review for ANY author, with the pull request, head and url on the task row and the PR's own base; a new task in a thread carrying an open PR → the generated plan adopts it (branch the PR's head, base its own, the reply names the PR and no new branch), a person's PR identically; a seeded plan request keeps the graph's branches; a fork head refused on adopt and resume; a closed or unfetchable adopt/resume target refused | `[unit]` `src/core/dispatcher.test.ts::agent:ship (the hand-off to the plan runner)::thread with user-named open PR and no new task text → a resume at review…`, `::a resume prefers the PR's OWN base ref…`, `::a bare reference to a person's open PR resumes at review…`, `::thread PR open + new task text → the generated plan ADOPTS the pull request…`, `::an adopted person-authored PR behaves identically…`, `::a seeded plan request in the same pull-request thread stays seeded…`, `::a fork-head PR is refused on adopt and on resume…`; `src/core/ship/preflight.test.ts::shipPreflight — the entry cases (agent-ship item 10) and the auto-merge fact (item 9)::*`; `src/core/dispatch/ship.test.ts::runShipBranch — the agent:ship fork hands every admitted request to the plan runner::a resume at review: the open pull request…` | | 10: entry checks — a foreign PR cited in new task text falls through to a task off the default branch on the task's own branch — even when its facts fetch fails — and drops the PR-derived ref (`refFromPr`) and a unit branch of ship's own the thread stayed bound at (`plan//` → the default branch; a person's own typed ref still wins); an INHERITED unreachable PR or a BARE in-message reference stays fail-closed; the resolver flags in-message PRs (`prFromMessage`) and PR-head refs (`refFromPr`) | `[unit]` `src/core/dispatcher.test.ts::agent:ship (the hand-off to the plan runner)::new task text citing a human-authored open PR does NOT bind it…`, `::in-message cited PR + new task text + FAILING prFacts fetch…`, `::inherited unreachable PR (prUnpostable) + new task text…`, `::in-message cited PR + NO task text + failing prFacts fetch…`, `src/core/repoContext.test.ts::…PR-source flags for ship…::…prFromMessage…`, `src/core/ship/preflight.test.ts::shipPreflight — the entry cases (agent-ship item 10) and the auto-merge fact (item 9)::a unit branch of ship's own is never a fresh task's base…` | @@ -140,6 +141,9 @@ The coding → review → fix loop to LGTM as one [pipeline](../vocabulary.md#pi | Runner (item 15): the plan graph — every unit heading in order with the dependencies its bullet names (lists, `to` ranges, `none`), its slug and its branch; `unitSlug`, `unitBranch` and `parsePlanBranch` refusing every other shape; `planIdOf` from the file's name or a throw naming the path; `planInstanceId` names the plan; `parseShipPlanRequest` reads `plan [units …]` and nothing else | `[unit]` `src/core/ship/coordinator.test.ts::the plan graph — units, their dependencies, their branches::*` | | Runner (item 15): the cursor — ready units in the plan's order once their in-play dependencies are done; a selection narrowing the plan with a dependency outside it counted satisfied; a failed unit blocking its dependents transitively while every other ready unit stays in play; an unknown unit, a unit started twice or before it is ready, and a cycle refused by name | `[unit]` `src/core/ship/coordinator.test.ts::the plan cursor — ready units in dependency order, a failure blocking its dependents::*` | | Runner (item 15): every ending on step returns — merge-ready on a plan branch through the merge step and off one for a person's merge; the findings round trip: `request_changes` enters the findings step (a coding spawn briefed with the review run under the review round's index, never a `fix` round), then the re-review briefed with the review run and the coding run that answered it, the declined disposition on the report; the findings step's `busy` answer waiting a chunk under the unit's clock and asking the same step again, and a `busy` answer under the reserve ending the unit `wall_clock_cap` with no coding run started; the round cap with its declined/unaddressed split; the wall-clock cap with the children's budgets clipped and the reservation; a stop naming its mode, honored around a posted verdict and never over one; the aborts (no pull request at round 0 with the child's words leading, a branch not created, a failed child, no new head unless every finding was declined); no verdict; an approval that did not post; a resume at review skipping round 0 | `[unit]` `src/core/ship/coordinator.test.ts::the unit pipeline — every ending the ship pipeline has, on step returns::*` | +| 16: `agent:ship recover unit :` is the only explicit original-unit recovery grammar. It runs before ordinary preflight and generated-plan hand-off, passes the resolved requester and thread to the deterministic recovery boundary, and reports the one durable recovery Workflow; a create whose commit cannot be determined reports the retained checkpoint as indeterminate rather than refused. Malformed recovery-shaped text and a refused recovery start no generated plan | `[unit]` `src/core/dispatch/ship.test.ts::runShipBranch — the agent:ship fork hands every admitted request to the plan runner::recognizes only the explicit recover-unit grammar with a valid original unit key`, `::refuses recovery-shaped malformed text instead of handing it to a replacement generated plan`, `::runs explicit recovery before ordinary preflight and hands the resolved requester and thread to the deterministic coordinator operation`, `::reports an indeterminate Workflow start with the retained checkpoint instead of calling it refused`, `::surfaces a deterministic recovery refusal and never falls through to generated-plan hand-off` | +| 16: original-unit recovery reconstructs only the claimed original row. A posted `request_changes` continues through findings and re-review under `:/recovery/...`; one uniquely attributable completed findings child is consumed whether or not it changed the head, and only its proven push may advance the durable publication head before re-review. A terminal original `findings/pr-check` may resume directly at read-only re-review only when that record proves the current head, exact original key/ref/thread and ordering; every other moved head refuses. `no_verdict` starts one read-only review. Neither path runs unit-start, branch, round zero, or a generated unit; actual child admission and post-attach setup recheck owner, exact PR/ref/base/head and the absolute end of the exact remaining lease for coding and review. The durable recovery boundary is carried across resume and restart, constrains dispatch and hard-stops the child. The physical recovery Workflow is only a new checkpoint, never a claimed resumption of the terminal Workflow: lifetime review rounds and enforceable remaining limits carry forward, checkpoint-local attempt counters use its distinct step namespace, cost-capped rows whose cumulative spend is unavailable refuse, and elapsed-category reports explicitly exclude unavailable earlier activity. Lifecycle and restart-death events use the recovery Workflow only as transport while retaining original identity; boot ownership reads active recovery claims independently of terminal parents. Every recovery mutation names the exact recovery Workflow; ambiguous CAS responses are reconciled before cleanup. Human-only and draft outcomes settle as typed terminal holds, queued or later replies never start replacement work, failed reply persistence tells the sender, and terminal settlement records the evidence as consumed and replays without releasing a successor owner | `[unit]` `src/core/coordinator/driver.test.ts::runOriginalUnitRecovery::*`; `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::*`; `src/core/ship/coordinator.test.ts::original-unit recovery accounting::labels elapsed categories as checkpoint-local instead of claiming the legacy unit's missing lifetime history`; `src/core/coordinator/contract.test.ts::sendRunFinished — the event a terminal record sends::uses a recovery checkpoint only as transport while the payload retains the original instance`; `src/core/runnerOwnership.test.ts::recoverRunnerOwnedPulls::rebuilds an active recovery claim even though its original parent Workflow is terminal` | +| 16: the durable ownership rebuild asks SQLite only for rows that may carry an active recovery. Terminal valid history is excluded before application parsing; malformed rows are retained as fail-closed candidates, so corruption cannot be mistaken for “unowned” while valid active rows remain recoverable on the next complete pass | `[unit]` `deploy/cloudflare-memory/runLedger.test.ts::run ledger — the coordinator's unit rows (item 50)::lists only active recovery rows in SQL, ignores terminal history, and fails closed on malformed candidate JSON`; `src/core/runnerOwnership.test.ts::RunnerOwnershipFence::refuses sweep ownership reads until a complete live-run listing rebuilds durable ownership` | | Runner (items 10 and 15): the recover path rewrites before it opens — the identity rewrite runs with the instance's repo, base, branch and requester over an empty start state before the create, and an unreadable rewrite is `github_unavailable` (asked again), never an open over unverified identities | `[unit]` `src/channels/adminCoordinator.test.ts::pr-check recover — the answer says why nothing was recovered, and GitHub being down is never none::the recover path runs the identity rewrite before it opens…` | | Runner (item 15): round zero's dead-child recover — a `failed` child's abort repeats the bot's reason for recovering nothing (`no_commits`, `no_base`, or neither claim on a bare `none`); the bot answers `none` with `no_commits` on GitHub's empty-branch 422 and `no_base` without a create, any other create failure is `github_unavailable`, and the driver keeps `unrecovered` through the parse. A findings run must complete: interrupted or failed findings work stops without a review even when it recorded an attempted push, and its report claims no unverified remote save. | `[unit]` `src/core/ship/coordinator.test.ts::the unit pipeline — every ending the ship pipeline has, on step returns::aborts: a round 0 that opened no pull request…`, `src/core/ship/coordinator.test.ts::the unit pipeline — the event, the timeout and the confirmation (the durable half)::a findings run that did not complete remains resumable…`, `src/channels/adminCoordinator.test.ts::pr-check recover — the answer says why nothing was recovered, and GitHub being down is never none::*`, `src/core/coordinator/driver.test.ts::the plan runner's driver — the Workflow body over the step runner (item 9)::a failed coding child whose recover pr-check answers none with a reason…` | | Runner (item 15): the wait is sliced — every wait one chunk, never the child's whole budget; a live answer waits the next chunk; the slices before the budget plus the margin sum to exactly that, the last the remainder; past it an overdue child is asked about every chunk, never in a zero-length wait; a finished record ends the wait as the event would; a busy wait is a chunk | `[unit]` `src/core/ship/coordinator.test.ts::the unit pipeline — the event, the timeout and the confirmation (the durable half)::the wait is sliced…` | diff --git a/docs/reference/specs/http-ingress.md b/docs/reference/specs/http-ingress.md index 7c46a83d6..412a68069 100644 --- a/docs/reference/specs/http-ingress.md +++ b/docs/reference/specs/http-ingress.md @@ -63,7 +63,8 @@ HTTP is single-shot request/response, unlike Slack's long-lived threads. The end | 9: `findMergedPrByHead` asks for the closed pull requests heading the branch, newest first, answers the latest merged one with its number, url, merge commit and when, skips a closed-unmerged row (a test merge commit is not a merge) and a merged row without a well-formed merge commit, null with none, throws on a failed lookup and before any fetch without a credential | `[unit]` `src/execution/githubPulls.test.ts::githubPulls::findMergedPrByHead asks for the closed pull requests heading the branch…`, `::throws when no credential is available, before any fetch` | | 9: the driver on a unit already merged — the pre-check under `/pr-check` answering merged ends the unit with no branch and no child, the ending told with the pull request as already merged and when, its dependent starting and the plan completing; a merge landing during round 0 ends the unit merged at the round's pr-check with round 0 noted completed and no review child; a merged answer without its merge commit is unreadable and fails the instance at once | `[unit]` `src/core/coordinator/driver.test.ts::the plan runner's driver — a unit whose pull request already merged (item 9)::a unit merged before the attempt…`, `::a merge that lands during round 0…` | | 9: the shim's instance route — the body it accepts, the bot's authorization answer read fail-closed, the wire shape of created / duplicate / failed and its read back; the status route's path, its wire shape both ways, and absence as the engine's `instance.not_found` code only — a failure whose text merely says not found is a failure | `[unit]` `src/core/coordinator/instancesRoute.test.ts::*` | -| 9: the Workflow's boundary — the binding's class is declared in `coordinator.ts` over the narrowed env and re-exported by `worker.ts`, same-script under the shim's own name; the module imports the entry type-only, names no forwarded secret, fetches only through the container binding with the coordinator bearer from the token map, and its `run()` is one `runPlan` over the platform's step and the container bot; the state Worker's template binds the same class across scripts by the bot's script name | `[unit]` `deploy/cloudflare/coordinator.test.ts::*` | +| 9: the Workflow's boundary — the binding's class is declared in `coordinator.ts` over the narrowed env and re-exported by `worker.ts`, same-script under the shim's own name; the module imports the entry type-only, names no forwarded secret, fetches only through the container binding with the coordinator bearer from the token map, and its `run()` selects `runOriginalUnitRecovery` only from typed recovery parameters and otherwise runs `runPlan`; the state Worker's template binds the same class across scripts by the bot's script name | `[unit]` `deploy/cloudflare/coordinator.test.ts::*`; `src/core/coordinator/instancesClient.test.ts::createInstanceViaShim — the bot's request for a coordinator instance::forwards typed Workflow parameters unchanged` | +| 9: `recover-unit` claims one ended original row from its durable instance/unit, exact segment-and-attempt review record, publication binding and fresh unchanged GitHub head; the Workflow and each actual child admission revalidate the same claim, owner, fixed head and absolute lease deadline. Definite create failure or a moved head during Workflow revalidation restores the exact prior row and releases only the matching owner; an unanswered create stays claimed so retry meets the same Workflow id; a terminal duplicate records the evidence consumed and restores the ending. Only an attributable completed findings child may CAS-advance the publication head; requester/thread mismatch, evidence ambiguity, stopped-segment or cost-cap ambiguity, exhausted budgets, stale CAS, ownership loss and rollback/cleanup loss refuse before child admission; terminal settlement is acknowledged before completion | `[unit]` `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::*`; `src/core/coordinator/driver.test.ts::runOriginalUnitRecovery::*` | | 9: the plan runner's own steps — `plan` answers the units, caps, base, repository and budgets; `unit-start` opens the unit's thread through the requesting thread's channel and no review thread, finds the board issue, writes the row, answers the same thread twice, runs a one-unit plan's unit in the requesting thread with no thread opened — the choice keyed on the unit count ([agent-ship.md](agent-ship.md) item 16), a task's generated plan and a one-unit checked-in selection alike — refuses without a channel or on a failed open; `branch` creates from the base and answers a failure as `ok: false`; `spawn` composes a contract brief and a findings brief into the child's turn and options, dispatches a review brief into the unit's thread, refuses a brief before the unit started and one it cannot compose, and refuses a malformed body; `read-record` answers the pull request, the verdict, the reviewed head, whether the verdict stands — from the child's recorded post without asking GitHub (true at the reviewed head on the unit's pull request; false with the recorded reason for a skip), a recorded post at another head, verdict or pull request left to GitHub, and GitHub asked up to three times a pause apart when the record is silent (false for another author, another head or another verdict after the last look; absent when GitHub stays silent) — the dispositions and the handoff; `pr-check` remembers the pull request on the row, a merged one too; `round` appends the boundary and redraws the card; `unit-end` writes the ending and posts the thread's copy of the report (the full report from an older driver, nothing for an empty copy while still answering told); `finish` writes the parent's record from the rows, closes the card and tells the requesting thread | `[unit]` `src/channels/adminCoordinator.test.ts::the plan runner's steps — plan, unit-start, branch, round, unit-end, finish (item 9)::*` | | 9: the briefs composed bot-side — the contract from the plan, the specs and the rules at the base ref with the board issue, a task unit's from the ship request; the review turn with the prior rounds' records on a re-review, the dispositions matched and the dropped ids noted; the findings message from the review run's findings and final words; an unreadable plan, a missing run or a brief of an unknown kind throws by name | `[unit]` `src/core/coordinator/briefs.test.ts::contractFor — the unit's contract from the repository at the base ref::*`, `::composeChild — the child a brief names::*` | | 9: `merge` — every guard green squashes at exactly the approved head with the title as the commit and answers the squash's sha; a coordinator bearer without `plan:merge` is refused naming the grant and GitHub is never asked; a task instance, a branch of another shape or another plan waits for a person and the release pull request is refused by name; a closed pull request, one heading another branch, a head that moved, a review by another author or another verdict are each refused naming what is off; red checks refuse naming the runs, running or unreported checks answer `pending`; GitHub's refusal of the squash is answered in its words; the facts, the reviews, the checks or the merge call unreadable are a passing `502`; a malformed body is `400`, an unknown instance or unit `404`; `unit-end` leaves the ending on the board issue as the runner's handoff and a failed comment never fails the step | `[unit]` `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/merge — the runner's squash of a unit's pull request (item 9)::*` | diff --git a/docs/reference/specs/run-history.md b/docs/reference/specs/run-history.md index 010ab8cbc..c407b2ac6 100644 --- a/docs/reference/specs/run-history.md +++ b/docs/reference/specs/run-history.md @@ -149,6 +149,7 @@ Every [run](../vocabulary.md#run) becomes a durable record — identity, timing, | 49: the record's own surfaces — the plan, the caps, the card, the run id and the label are accepted when shaped, also after a JSON round-trip, and a malformed one is refused | `[unit]` `src/core/coordinator/contract.test.ts::isCoordinatorInstance — the parent ship record::accepts the plan, the caps, the card, the run id and the label when present…` | | 50: a unit row's idle — why, at, renewalsLeft, wakes and the optional continuation facts accepted; a missing why, a negative count, a malformed spendUsd or handoff refused; `unit-end` on an `idle` ending writes it with `wakes: 0` and no `ending`, the report still reaches the thread and the card line reads `idle · `; a `why` past `IDLE_WHY_MAX` is the route's 400, the body's `headSha` lands as `lastPush`, and a later real ending drops the idle; the unit's readable facts carry `{why, at, renewalsLeft, wakes}` and a row without one carries none | `[unit]` `src/core/coordinator/contract.test.ts::isCoordinatorUnit — one unit's row::the idle on a row (record 0051, run-history item 50)…`, `src/channels/adminCoordinator.test.ts::the plan runner's steps — plan, unit-start, branch, round, unit-end, finish (item 9)::unit-end with an idle ending writes idle {why, at, renewalsLeft, from, runId, spendUsd, handoff, wakes: 0} and no ending…`, `src/core/unitRuns.test.ts::unitFactsOf — the idle on the facts::*` | | 50: a merge-ready row's durable head — `unit-end` writes the final reviewed full head as `lastPush`; a normal reply repairs a legacy row only from one exact same-unit coding-and-approval head, accepting the coding record's final head or its durable exact-unit-branch `pushed[]` head but never a same-thread standalone run, after fresh requester, open same-repository PR/ref/base/head and sole-owner agreement, then conditionally replaces only the exact unchanged row with `lastPush` and the full publication binding before reissue; the reservation transfers only from its exact token/current owner to the durable new attempt before create, while any transfer/create failure restores only the exact legacy row and releases only its owner; a losing rollback CAS is surfaced without overwriting newer state; absent, unreadable, partial, ambiguous, moved, foreign, closed, wrong-branch, wrong-base, wrong-requester, changed-owner and stale evidence writes nothing and starts nothing | `[unit]` `src/channels/adminCoordinator.test.ts::the plan runner's steps — plan, unit-start, branch, round, unit-end, finish (item 9)::unit-end writes the ending and the pull request on the row…`; `src/core/coordinator/driver.test.ts::the plan runner's driver — the Workflow body over the step runner (item 9)::a review that requests changes is followed by findings and re-review at the moved head…`; `src/core/dispatcher.test.ts::a unit-owned thread (record 0051's reply-as-event and gone-instance rules)::a legacy merge-ready row recovers one exact approved child head for the same unit, persists the full binding before reissue, and starts no substitute run`, `::a real reissue transfers the recovered pull request to the new attempt, whose coding spawn passes the publication-owner gate`, `::two concurrent legacy recoveries conditionally replace the same row once, so the losing caller fails closed without reissue`, `::an ownership change whose exact rollback loses a CAS race surfaces the failure without clobbering newer state or starting work`, `::legacy merge-ready recovery*`; `src/core/coordinator/instanceStore.test.ts::InMemoryCoordinatorInstanceStore::claimLegacyContinuation replaces only the exact expected row, so one caller wins and stale state is never overwritten`, `::WorkerCoordinatorInstanceStore (over a state Worker double)::claimLegacyContinuation replaces only the exact expected row, so one caller wins and stale state is never overwritten`; `deploy/cloudflare-memory/runLedger.test.ts::run ledger — the coordinator's unit rows (item 50)::claim-legacy-continuation atomically replaces only the exact expected row, so a concurrent or stale caller cannot erase newer state` | +| 50: an original-unit recovery claim replaces one terminal `ending` and is mutually exclusive with `idle` and `ending`. It retains the stage, round, unchanged full head, exact remaining milliseconds and absolute deadline, claim time, original-unit step, exact authorizing review key/run and findings, byte-for-byte prior ending, and the separate Workflow id. Terminal settlement removes the claim by full-row CAS and writes a receipt that prevents the same evidence from being consumed twice | `[unit]` `src/core/coordinator/contract.test.ts::isCoordinatorUnit — one unit's row::an original-unit recovery claim retains its exact lease, evidence, previous ending and Workflow id, mutually exclusive with idle or ending`; `src/channels/adminCoordinator.test.ts::POST /admin/coordinator/recover-unit — unchanged-head original-unit recovery::*` | | 50: indexed wake answers retain every segment continuation fact and reject malformed keys or answers; the wake route stores one answer with its event marks, replays it without another write, and no-event answers do not increment the idle count; reopening segment one leaves the renewal-only `segments` list unchanged and passes the production Worker's row validator | `[unit]` `src/core/coordinator/contract.test.ts::isCoordinatorUnit — one unit's row::wake answers are keyed by the indexed wait…`, `src/channels/adminCoordinator.test.ts::the plan runner's steps — plan, unit-start, branch, round, unit-end, finish (item 9)::unit-wake*`, `deploy/cloudflare-memory/runLedger.test.ts::run ledger — the coordinator's unit rows (item 50)::the validating wake boundary accepts a first-segment resume…` | | 50: a unit row's segments — an index from two up with an optional sha and run id and a time accepted, a first-segment index, a missing time, a malformed sha or a non-array refused | `[unit]` `src/core/coordinator/contract.test.ts::isCoordinatorUnit — one unit's row::segments are the renewals the unit spent…` | | 50: a unit row's shape — a full row, its round-trip and a bare one accepted; `record` accepts exactly four digits and rejects another shape; a review thread with its key and an optional link accepted, one without its key, with a malformed link or as a bare string refused; a bad instance id, a missing unit, slug or branch, a non-array dependency list, a malformed pull request, round, ending or issue refused | `[unit]` `src/core/coordinator/contract.test.ts::isCoordinatorUnit — one unit's row::accepts a full row, its JSON round-trip and a bare one (the branch, the dependencies and no rounds)` | diff --git a/docs/reference/specs/thread-admission.md b/docs/reference/specs/thread-admission.md index 89cd35bf4..54b786cfc 100644 --- a/docs/reference/specs/thread-admission.md +++ b/docs/reference/specs/thread-admission.md @@ -20,7 +20,7 @@ Before this, a follow-up in a thread with a run in flight dispatched a second, f - **Under the operator the decision, not the slot, says what a reply is** ([routing-and-config.md](routing-and-config.md) item 29; the one-door plan's admission unit). With `routing.operator: on` the dispatcher executes the operator's decision before any fold. A plain reply's `steer` bind resolves its target from the thread's live owner at execution — the current child, never an ended run id copied from the transcript — while an explicitly typed `steer run …` keeps the id the person named and may target another thread. `steer.run`'s wired sender (`createSteerSender`) applies the steer owner rule ([authorization.md](authorization.md) item 16a) and the same live-agent allowlist, with no confirm-class hand-back; an ended target fails the command run once with `run ended at