From df9d1fb19f8038bb3a42a02d5cfb6449b55ac91a Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Mon, 21 Sep 2026 15:21:08 +0800 Subject: [PATCH] Warm-respawn a supervisor unit whose host was lost MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #1994 made a lost host's agent continuable, but only for a workflow agent.run node: the observer checkpoints the live transcript, the reconciler's abandon keeps the reference, and the node's own respawn consumes it. A supervisor-spawned unit was excluded — its envelope never opted in, so its retry restarted the conversation from zero even though the machinery to continue it was already there. The staging loop that every spawn wave, retry and resolve passes through now opts a unit in only when a later retry could consume the checkpoint: the unit carries a subtask id (resolver units carry none, and the retry lookup matches on it), and SupervisorBounds.CanRespawnAfterWave says the run can still afford to respawn it. The subtask-attempt lookup falls back to a TERMINAL row's checkpoint columns when the attempt left no result to read, which is the shape an abandon leaves. A Running row's checkpoint belongs to a session that may still be live, so it is never resumed. For "terminal and still holding a checkpoint" to mean an abandon, a deliberate cancel now releases the columns as completion does; only the reconciler's abandon and spool recovery keep them. The resume fold tells the two kinds of restored conversation apart: a capture finished with its tree intact, a checkpoint did not, so only the second carries the lost-host block and its provenance. Its tree sentence reads the resumed attempt's own pushed branch, not the effective clone ref, so a dependent unit is never told that its producer's branch is its own published work. The same change stops a producer's handoff ref from suppressing the honest-redo line on the ordinary path when the prior attempt pushed nothing itself. The billing stays honest: the graded roster keys by subtask and takes the latest, so one host loss spends one graded attempt, not two. --- .../Services/Agents/AgentRunExecutor.cs | 2 +- .../Services/Agents/AgentRunService.cs | 33 +- .../RealSupervisorActionExecutor.Spawn.cs | 58 ++- .../Services/Supervisor/SupervisorBounds.cs | 22 + .../Retention/ArtifactRetentionPolicy.cs | 7 +- .../Workflows/Nodes/Builtin/AgentCodeNode.cs | 3 +- .../CodeSpace.Messages/Agents/AgentTask.cs | 7 +- .../Agents/ResumableSession.cs | 13 +- .../Artifacts/ArtifactRetention.cs | 12 +- .../RealModelSupervisorWholeLoopE2ETests.cs | 7 + .../Workflows/SupervisorResolveFlowTests.cs | 23 + .../SupervisorRetryWorldStateFlowTests.cs | 468 +++++++++++++++++- .../Agents/AgentRetryCausesTests.cs | 8 +- .../Agents/SupervisorBoundsTests.cs | 44 ++ .../Agents/SupervisorDependencyGateTests.cs | 19 + .../SupervisorDependencyStagingTests.cs | 61 ++- .../Workflows/AgentCodeNodeTests.cs | 2 +- 17 files changed, 751 insertions(+), 38 deletions(-) diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs index add042785..292db89d9 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs @@ -5408,7 +5408,7 @@ private sealed record SessionCheckpointTick(Guid TeamId, IAgentHarness Harness, /// The checkpoint coordinates for a run whose envelope OPTED IN, else null — the one place the opt-in is read, so /// the produce side and the consume side cannot disagree about which runs are checkpointed. A run whose failed /// attempt nobody can retry writes nothing, which is what keeps the artifact store free of a per-minute - /// transcript copy for every benchmark cell, review child and supervisor unit on the fleet. + /// transcript copy for every benchmark cell, review child and supervisor unit that no retry could resume. /// private static SessionCheckpointTick? CheckpointTickFor(AgentTask task, Guid teamId, IAgentHarness harness, string? workingDirectory, AgentRunFacts facts) => task.CheckpointSessionTranscript ? new SessionCheckpointTick(teamId, harness, workingDirectory, facts) : null; diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs index 35bffde39..85f729eac 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs @@ -774,13 +774,19 @@ public async Task CancelRunningAsync(Guid runId, string reason, AgentRunAb if (snapshot is null || snapshot.Status != AgentRunStatus.Running) return false; + // 3c: a deliberate cancel is a CLEAN landing, so it releases the mid-run session checkpoint exactly as + // completion does. Nobody owes this run a continuation, and a kept reference would pin the artifact + // Referenced (terminal in the retention ledger) for good — and make a later retry of the same subtask read + // the cancel as a host loss. Only the reconciler's abandon keeps these columns. var cancelled = await _db.AgentRun .Where(r => r.Id == runId && r.Status == AgentRunStatus.Running && r.FenceEpoch == snapshot.FenceEpoch) .ExecuteUpdateAsync(s => s .SetProperty(r => r.Status, AgentRunStatus.Cancelled) .SetProperty(r => r.FenceEpoch, r => r.FenceEpoch + 1) .SetProperty(r => r.Error, reason) - .SetProperty(r => r.CompletedAt, (DateTimeOffset?)DateTimeOffset.UtcNow), cancellationToken) + .SetProperty(r => r.CompletedAt, (DateTimeOffset?)DateTimeOffset.UtcNow) + .SetProperty(r => r.SessionTranscriptCheckpointArtifactId, (Guid?)null) + .SetProperty(r => r.SessionTranscriptCheckpointAt, (DateTimeOffset?)null), cancellationToken) .ConfigureAwait(false); if (cancelled == 0) return false; @@ -956,7 +962,7 @@ await _db.AgentRun.AsNoTracking().SingleOrDefaultAsync(r => r.Id == runId, cance var candidates = await _db.AgentRun.AsNoTracking() .Where(a => a.TeamId == teamId && a.WorkflowRunId == supervisorRunId && a.SessionId != null) .OrderByDescending(a => a.CreatedDate).ThenByDescending(a => a.Id) - .Select(a => new { a.Id, a.SessionId, a.ResultJson, a.TaskJson }) + .Select(a => new { a.Id, a.Status, a.SessionId, a.ResultJson, a.TaskJson, a.SessionTranscriptCheckpointArtifactId, a.SessionTranscriptCheckpointAt }) .ToListAsync(cancellationToken).ConfigureAwait(false); // The most-recent RESUMABLE prior attempt of THIS subtask — skip a captured-but-transcript-less attempt so it @@ -966,11 +972,34 @@ await _db.AgentRun.AsNoTracking().SingleOrDefaultAsync(r => r.Id == runId, cance if (SubtaskIdOf(candidate.TaskJson) != subtaskId) continue; if (TryResumable(candidate.Id, candidate.SessionId, candidate.ResultJson) is { } resumable) return resumable; + + // 3c: a prior attempt whose HOST died has no result at all — the reconciler's abandon writes none — so + // the captured-transcript read above finds nothing for exactly the population a warm retry helps most. + // Its mid-run checkpoint is what survives, and a TERMINAL row still naming one is the signature of an + // abandon: every clean landing — completion or cancel — releases these two columns in its own terminal + // write; only an abandon keeps them. + if (TryResumableFromCheckpoint(candidate.Id, candidate.Status, candidate.SessionId, candidate.SessionTranscriptCheckpointArtifactId, candidate.SessionTranscriptCheckpointAt) is { } continued) return continued; } return null; } + /// + /// A prior attempt's MID-RUN checkpoint as a resumable session — the host-loss counterpart of + /// . Both halves are required for the same reason that one demands both: a session id + /// with no transcript resumes into "No conversation found", and a transcript nothing can address is not + /// resumable at all. + /// + /// And only from a TERMINAL row. The checkpoint columns are written while the attempt is Running, so a + /// row still Running holds the checkpoint of a session that may be live — a kill-wave is best-effort, and a + /// revived parent stops the orphan sweep from selecting it — and resuming it would fork a conversation that is + /// still being written. + /// + private static ResumableSession? TryResumableFromCheckpoint(Guid agentRunId, AgentRunStatus status, string? sessionId, Guid? checkpointArtifactId, DateTimeOffset? checkpointAt) => + AgentRunStateMachine.IsTerminal(status) && sessionId is { Length: > 0 } sid && checkpointArtifactId is { } artifactId && checkpointAt is { } at + ? new ResumableSession(agentRunId, sid, null, artifactId, at) + : null; + /// /// A prior agent run's (id, session id, result json) → its RESUMABLE session, or null when not resumable: no /// session id, no transcript (both-or-neither — a resume without one fails "No conversation found"), or a diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs b/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs index f261c3608..2b87c8f7e 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs @@ -161,6 +161,19 @@ internal static IReadOnlyList DependsOnFor(SupervisorPlannedSubtask? pla internal static DependencyStagingResult PreferPriorAttemptStaging(DependencyStagingResult priorAttemptStaging, DependencyStagingResult dependencyStaging) => priorAttemptStaging.Ref is not null ? priorAttemptStaging : dependencyStaging; + /// + /// 3c: whether a staged unit opts into the mid-run session checkpoint — only when a later retry could actually + /// consume it. A checkpoint costs a whole-file read and an artifact write a minute for as long as the unit runs, + /// and a host loss keeps the survivor for good, so a unit nobody can resume must not pay for one. + /// + /// Two exclusions. A unit with no subtask id can never be FOUND by the retry lookup, which matches on it: + /// that is every resolver unit, whose task is built without one, and a re-resolve stages a fresh resolver rather + /// than resuming the old one. And a unit the run can no longer afford to respawn has nobody to hand a restored + /// conversation to (). + /// + internal static bool CheckpointsSessionTranscript(AgentTask task, SupervisorTurnContext context, int waveSize) => + task.SubtaskId is { Length: > 0 } && SupervisorBounds.CanRespawnAfterWave(context, waveSize); + /// The subtask's target repository, resolved the SAME way will resolve it — a pure pre-computation so dependency staging can look up the right repo's manifest before the task itself is built. private static Guid? ResolveTargetRepositoryId(SupervisorAgentDispatch? spec, SupervisorTurnContext context) { @@ -433,7 +446,11 @@ private async Task ExecuteRetryAsync(SupervisorDecision dec var (escalatedTask, escalation) = await ApplyRetryEscalationAsync(builtTask, priorResult, context, cancellationToken).ConfigureAwait(false); - var task = ApplyRetryDisposition(escalatedTask, prior, priorResult, workspaceHasPriorWork: effectiveStaging.Ref is not null); + // The continuity sentence describes the resumed attempt's OWN git state, so it reads that attempt's own + // pushed branch — never the effective clone ref. For a dependent unit whose attempt pushed nothing, the + // effective ref is the PRODUCER's handoff branch: it says nothing about whether this attempt's work survived, + // and naming it would tell the agent another unit's branch is its own published work. + var task = ApplyRetryDisposition(escalatedTask, prior, priorResult, workspaceRef: priorAttemptStaging.Ref); if (AgentRetryCauses.Classify(priorResult?.Error) == AgentRetryCauses.GatewayFormatFault) _logger.LogWarning("Supervisor retry of subtask {SubtaskId}: the prior attempt died on a gateway FORMAT fault — retrying FRESH (a conversation replay re-triggers the fault) with extended thinking disabled ({EnvVar}=0)", retry.SubtaskId, AgentRetryCauses.MaxThinkingTokensEnvVar); @@ -513,20 +530,44 @@ private async Task ExecuteRetryAsync(SupervisorDecision dec /// World-state continuity (the prior branch tip staging) is decided elsewhere and stays UNCHANGED either way — /// the degrade drops the broken conversation, never the preserved work. /// - internal static AgentTask ApplyRetryDisposition(AgentTask task, ResumableSession? prior, SupervisorAgentResult? priorResult, bool workspaceHasPriorWork) + internal static AgentTask ApplyRetryDisposition(AgentTask task, ResumableSession? prior, SupervisorAgentResult? priorResult, string? workspaceRef) { if (AgentRetryCauses.Classify(priorResult?.Error) == AgentRetryCauses.GatewayFormatFault) return AgentRetryCauses.ApplyFormatFaultMitigation(task); - return prior is null ? task : ApplyResumeRecord(task, prior, workspaceHasPriorWork); + return prior is null ? task : ApplyResumeRecord(task, prior, workspaceRef); } - /// The pure fold of a resumable prior attempt onto the task: always stamps the session/transcript, and — ONLY when is false — appends the honest-redo line so the hint's truth value always matches the actual git state. Internal + static so the honesty branch is unit-pinned directly. - internal static AgentTask ApplyResumeRecord(AgentTask task, ResumableSession prior, bool workspaceHasPriorWork) + /// + /// The pure fold of a resumable prior attempt onto the task: always stamps the session + transcript, then says + /// what is true about the WORLD that conversation refers to. + /// + /// Two shapes, because the prior attempt ended two different ways. An attempt that FINISHED left its + /// workspace behind, so the only open question is whether it pushed a branch — + /// null means it did not, and the honest-redo line says so. An attempt whose + /// HOST died () left nothing but the conversation: its clone is + /// unreachable, so it owes the lost-host block instead, it carries its provenance onto the new run + /// (ResumedFromCheckpointAt for the permanent confinement record, ResumedFromAgentRunId for the + /// column), and its transcript ref is marked a CHECKPOINT so the executor degrades to a cold start rather than + /// failing the attempt when those bytes cannot be read. + /// + /// Internal + static so both honesty branches are unit-pinned directly. + /// + /// The branch the RESUMED attempt itself pushed, or null when it pushed none — never the effective clone ref, which for a dependent unit is its producer's handoff branch and says nothing about this attempt's work. + internal static AgentTask ApplyResumeRecord(AgentTask task, ResumableSession prior, string? workspaceRef) { var resumed = task with { ResumeFromSessionId = prior.SessionId, RestoredTranscript = prior.InlineTranscript, RestoredTranscriptArtifactId = prior.TranscriptArtifactId }; - return workspaceHasPriorWork ? resumed : resumed with { Goal = AgentRetryContinuity.WithHonestNoContinuityHint(resumed.Goal) }; + if (prior.CheckpointAt is { } checkpointAt) + return resumed with + { + RestoredTranscriptIsCheckpoint = true, + ResumedFromCheckpointAt = checkpointAt, + ResumedFromAgentRunId = prior.AgentRunId, + Goal = AgentRetryContinuity.WithLostHostHint(resumed.Goal, workspaceRef, treeOwed: task.RepositoryId is not null), + }; + + return workspaceRef is not null ? resumed : resumed with { Goal = AgentRetryContinuity.WithHonestNoContinuityHint(resumed.Goal) }; } /// @@ -775,9 +816,12 @@ private async Task StageAgentsAndParkAsync(IReadOnlyList<(A var reclaimed = k < orphans.Count; reclaimedAny |= reclaimed; + // 3c: the ONE staging seam every spawn wave, retry and resolve passes through, so the checkpoint opt-in is + // decided once here rather than at each verb's own task build — see CheckpointsSessionTranscript for who + // is excluded and why. var agentRunId = reclaimed ? orphans[k] - : await CreateResolvedAgentRunAsync(tasks[k].Task, tasks[k].Spec, context, cancellationToken).ConfigureAwait(false); + : await CreateResolvedAgentRunAsync(tasks[k].Task with { CheckpointSessionTranscript = CheckpointsSessionTranscript(tasks[k].Task, context, tasks.Count) }, tasks[k].Spec, context, cancellationToken).ConfigureAwait(false); StageAgentWait(context, k, agentRunId); agentRunIds.Add(agentRunId); diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs index c8c4a5c4e..9424be752 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs @@ -91,6 +91,28 @@ private static bool IsReauthorableRejection(SupervisorPriorDecision decision) => return null; } + /// + /// Whether the run could still afford to RESPAWN a unit this wave is about to stage — read from the same + /// total-spawn bound enforces, so the two can never disagree about what the run can + /// still do. + /// + /// A later retry costs exactly one spawn, and refuses it when + /// TotalSpawnedAgents + 1 would exceed the cap. By then this wave's own agents + /// are on the tape, so the room for it has to exist NOW: strictly less than the cap, not at it. + /// + /// The spawn cap and nothing else, deliberately. It is the one bound that says "no further agent can ever + /// be created" — whereas a run near its no-progress cap can still retry, because a wave that makes progress + /// resets that cadence. Gating on no-progress would leave the units most likely to be retried un-checkpointed, + /// which is the opposite of the point. The 3c consumer of this predicate pays a whole-file read and an artifact + /// write per minute, so it is worth asking whether anyone can consume the result. + /// + public static bool CanRespawnAfterWave(SupervisorTurnContext context, int waveSize) + { + ArgumentNullException.ThrowIfNull(context); + + return context.TotalSpawnedAgents + waveSize < (context.MaxTotalSpawns ?? SupervisorLane.DefaultMaxTotalSpawns); + } + /// How many agents the decision would spawn: a spawn fans out its subtaskIds; a retry is exactly one. Best-effort read — a malformed payload reads 0 (it stages nothing, so it can't breach a count bound). internal static int SpawnCount(SupervisorDecision decision) { diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs index c67552d6f..823fa0419 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs @@ -40,9 +40,10 @@ public static class ArtifactRetentionPolicy /// /// A mid-run session-transcript checkpoint. TWO HOURS, not seven days, and the short floor is the whole reason - /// the class exists: a run writes one of these per minute, each supersedes the last, and the run's terminal write - /// clears the column that references the survivor — so on the seven-day floor a single long run would hold every - /// superseded copy of a growing transcript for over a week. Two hours still sits far outside the window in which + /// the class exists: a run writes one of these per minute, each supersedes the last, and a clean landing + /// (completion or a deliberate cancel) clears the column that references the survivor — so on the seven-day floor + /// a single long run would hold every superseded copy of a growing transcript for over a week. An abandon keeps + /// the column on purpose, and that survivor is then Referenced for good; this floor does not collect it. Two hours still sits far outside the window in which /// the reference lands (the stamp is the next statement after the write) and far outside the window in which a /// continuation reads it (an abandon follows the host's death within one liveness window), so the floor costs /// nothing it protects. The quarantine stays the standard 24 h: the second, independent wait is unchanged. diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs index 54225485e..04577032c 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs @@ -506,7 +506,8 @@ private static AgentTask ApplyRespawnResumeHint(AgentTask task, JsonElement? pri /// base ref/pin byte-identical. /// /// The returned flag is whether the honest-redo line is OWED, and it follows the PRIMARY repo alone — - /// the same workspaceHasPriorWork: effectiveStaging.Ref is not null read the supervisor's retry uses. A + /// the same read the supervisor's retry makes of its prior attempt's own pushed branch + /// (workspaceRef: priorAttemptStaging.Ref), which is keyed on the target repository too. A /// multi-repo attempt whose primary push FAILED while a sibling's succeeded still repins that sibling, but its /// primary re-clones the default branch, so the resumed conversation must still be told its changes are not /// there: an OR across repos would suppress the line precisely where the agent's own repo lost its work. Nothing diff --git a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs index a830a0e2b..37d3cac9e 100644 --- a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs +++ b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs @@ -109,9 +109,10 @@ public sealed record AgentTask /// /// An opt-in rather than a default, because a checkpoint nobody will consume is pure waste: it costs a /// whole-file read and an artifact write per minute, per running agent, per worker. Only a producer whose failed - /// attempt can actually be RETRIED sets it — today that is agent.run for a node whose own retry policy - /// allows more than one attempt. The benchmark lanes (one attempt per cell by protocol), review children and - /// supervisor units leave it false. + /// attempt can actually be RETRIED sets it — today agent.run for a node whose own retry policy allows more + /// than one attempt, and a supervisor unit that carries a subtask id while its run can still afford to respawn + /// it. The benchmark lanes (one attempt per cell by protocol), review children and supervisor resolver units + /// leave it false. /// /// [JsonIgnore(WhenWritingDefault)] so an envelope that did not opt in adds nothing to task_json. /// diff --git a/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs b/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs index f5c07e221..a1303f8f0 100644 --- a/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs +++ b/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs @@ -10,4 +10,15 @@ namespace CodeSpace.Messages.Agents; /// off THIS id, never a separately-resolved "latest attempt" id, so a resume hint's honesty claim about git state /// always describes the SAME attempt whose conversation it restores. /// -public sealed record ResumableSession(Guid AgentRunId, string SessionId, string? InlineTranscript, Guid? TranscriptArtifactId); +/// +/// When the transcript this session restores was CHECKPOINTED mid-run, because its attempt's host died before it +/// could finish — null for the ordinary case, where the transcript was captured by an attempt that completed. +/// +/// The two are not interchangeable and the difference is what the consumer owes the agent. A captured +/// transcript comes from an attempt that ran to its end with its workspace intact, so the only open question is +/// whether a branch was pushed. A checkpoint comes from an attempt whose machine is gone: the conversation may +/// describe turns the checkpoint never saw and edits the new sandbox does not contain, so only this one owes the +/// lost-tree sentence — and only this one may degrade to a cold start when its bytes turn out to be unreadable, +/// since a captured ref that cannot be read is a real fault. +/// +public sealed record ResumableSession(Guid AgentRunId, string SessionId, string? InlineTranscript, Guid? TranscriptArtifactId, DateTimeOffset? CheckpointAt = null); diff --git a/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs b/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs index 72593072f..246849d84 100644 --- a/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs +++ b/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs @@ -35,11 +35,13 @@ public enum ArtifactRetentionClass /// A mid-run resumable session transcript, referenced only by /// agent_run.session_transcript_checkpoint_artifact_id. Its own class rather than /// because its lifetime is nothing like an event payload's: exactly one - /// checkpoint per run is ever useful (the newest), each one supersedes the last, and the run's terminal write - /// clears the column — so every intermediate is garbage within minutes and the survivor within hours. Sharing the - /// event class's seven-day floor would hold roughly a gigabyte of superseded transcript per long run for over a - /// week, and the growth gate that makes checkpoints worth taking is exactly what stops the content-addressed - /// store deduplicating them. + /// checkpoint per run is ever useful (the newest), each one supersedes the last, and a clean landing (completion + /// or a deliberate cancel) clears the column — so every intermediate is garbage within minutes and the survivor + /// within hours. An abandon-class ending (the reconciler's abandon, or its spool recovery) deliberately KEEPS the + /// column, because that survivor is what the retry resumes from, and a kept reference is Referenced for good. + /// Sharing the event class's seven-day floor would hold roughly a gigabyte of superseded transcript per long run + /// for over a week, and the growth gate that makes checkpoints worth taking is exactly what stops the + /// content-addressed store deduplicating them. /// SessionTranscriptCheckpoint = 5, } diff --git a/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs b/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs index df69d9a38..da409d5ff 100644 --- a/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs +++ b/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs @@ -1176,6 +1176,13 @@ public async Task The_real_model_reacts_to_a_failed_subtask_by_retrying() // gateway outage is non-gating LOUD infra; a no-secret config skips NOT-EVALUATED. The perpetual-failure scenario // force-STOPs cleanly on a bound (no-progress / total-spawn cap), never a run Failure, so a model that recovered // reads Drove from the ledger and is never mis-gated as a CodeFault. + // + // SCOPE, for the host-loss slice: the retries this arm drives are COLD by construction and must stay so. Each + // agent here reaches a real terminal exit with its tree intact, so the row carries no mid-run checkpoint and + // the respawn resumes nothing — the warm arm needs a machine that never came back, which no live wire can + // stage. That path is pinned deterministically instead (SupervisorRetryWorldStateFlowTests drives the REAL + // reconciler's abandon; SupervisorDependencyStagingTests pins the fold). This arm asserts nothing about the + // opt-in itself — the staging seam's own tests do. var baseUrl = Env(RealModelSupervisorDecisionFlowTests.BaseUrlEnvVar); var apiKey = Env(RealModelSupervisorDecisionFlowTests.ApiKeyEnvVar); var model = Env(RealModelSupervisorDecisionFlowTests.ModelIdEnvVar); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs index e7a81d964..75e58a1e0 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs @@ -64,6 +64,29 @@ public async Task Resolve_stages_one_resolver_agent_whose_goal_names_every_branc task.PushProducedBranch.ShouldBe(true, "the resolver MUST push its reconciled branch so a downstream PR-open has a head"); } + [Fact] + public async Task A_resolver_unit_never_opts_into_the_session_checkpoint() + { + // Resolve stages through the same seam as spawn and retry, but its unit carries no subtask id — and the retry + // lookup matches on one, so nothing could ever resume a resolver's checkpoint. It would pay a whole-file read + // and an artifact write a minute for nothing, and keep the survivor for good on a host loss. The run here has + // ample spawn headroom, so only the subtask-id gate can keep the flag off. + // MUTATION: drop the SubtaskId gate from CheckpointsSessionTranscript → the resolver opts in → red. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var runId = await SeedSupervisorRunAsync(teamId, userId); + + var context = ContextWith(runId, teamId, + repositoryId: Guid.NewGuid(), + spawn: SpawnWithBranches("codespace/agent/web", "codespace/agent/api"), + merge: ConflictedMerge("src/Shared.cs")); + + await ExecuteResolveAsync(context); + + var task = JsonSerializer.Deserialize((await StagedAgentRunsAsync(runId)).ShouldHaveSingleItem().TaskJson, AgentJson.Options)!; + task.SubtaskId.ShouldBeNull("precondition: a resolver's task is built without a subtask id"); + task.CheckpointSessionTranscript.ShouldBeFalse("no retry can find a unit without a subtask id, so none may pay for a checkpoint"); + } + [Theory] [InlineData("no-conflict")] // a clean merge on the tape → nothing to resolve [InlineData("no-repo")] // conflict present but no repository bound diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs index 11833fa8b..5c3cdd396 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs @@ -194,6 +194,422 @@ public async Task With_two_recorded_attempts_the_git_ref_always_matches_the_conv task.Goal.ShouldNotContain(AgentRetryContinuity.HonestNoContinuityHint, customMessage: "the resumed attempt's own branch IS preserved — asserting otherwise would be a lie"); } + // ── 3c: a unit whose HOST died resumes from its mid-run checkpoint ───────────── + + [Fact] + public async Task A_retry_of_a_unit_whose_host_died_resumes_from_its_checkpoint() + { + // The population #1994 excluded. A supervisor unit whose host is lost is abandoned by the reconciler with NO + // result at all, so the captured-transcript lookup every warm retry uses finds nothing. What survives is the + // checkpoint on the row, and a terminal row still naming one is the signature of that abandon: every clean + // landing — completion or cancel — releases those columns in its own terminal write. + // MUTATION: drop the TryResumableFromCheckpoint fall-through → the retry cold-starts with no provenance → red. + // MUTATION: drop the CheckpointAt branch in ApplyResumeRecord → the retry still resumes the checkpoint, but + // unmarked (an unreadable ref would then fail the attempt), with no provenance and the ordinary honest-redo + // line in place of the lost-host block → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var checkpointArtifactId = Guid.NewGuid(); + var lostAttemptRunId = await SeedHostLostAttemptAsync(teamId, repoId, runId, (checkpointArtifactId, "sess-lost-host")); + + var abandoned = await ReconcileUntilTerminalAsync(lostAttemptRunId); + + abandoned.Status.ShouldBe(AgentRunStatus.Failed, "the reconciler terminalizes a run whose host never came back"); + abandoned.SessionTranscriptCheckpointArtifactId.ShouldBe(checkpointArtifactId, "and the abandon KEEPS the checkpoint — releasing it is a clean landing's job, not the abandon's"); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", lostAttemptRunId)); + + var task = await ExecuteRetryAsync(context, "sb"); + + task.ResumeFromSessionId.ShouldBe("sess-lost-host", "the CLI is told WHICH conversation to resume — a transcript with no id names nothing"); + task.RestoredTranscriptArtifactId.ShouldBe(checkpointArtifactId, "the checkpoint rides as a REF the executor resolves just before invocation"); + task.RestoredTranscriptIsCheckpoint.ShouldBeTrue("best-effort bytes: unreadable must cost the conversation, never this retry attempt"); + task.ResumedFromCheckpointAt.ShouldNotBeNull("the launch stamps this onto the run's permanent confinement record"); + task.ResumedFromAgentRunId.ShouldBe(lostAttemptRunId, "which attempt took over from which is a column, not prose"); + task.Goal.ShouldContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive, "a restored conversation describes a machine that is gone, and the agent must be told"); + + using var verify = _fixture.BeginScope(); + var respawn = await verify.Resolve().AgentRun.AsNoTracking() + .Where(r => r.WorkflowRunId == runId && r.Id != lostAttemptRunId).OrderByDescending(r => r.CreatedDate).FirstAsync(); + + respawn.ResumedFromAgentRunId.ShouldBe(lostAttemptRunId, "and the task's provenance is promoted onto the row, like AgentDefinitionId"); + } + + [Fact] + public async Task A_retry_of_a_unit_lost_without_a_checkpoint_is_cold_and_claims_nothing() + { + // The same host loss from a unit that checkpointed nothing — it never opted in, or died before its first + // checkpoint. Production leaves such a row with no session id either: the checkpoint stamp is the only writer + // of a Running row's session id, and it writes both together. Nothing to restore, so the respawn cold-starts + // and claims neither a lost machine nor a restored conversation. + // A NEGATIVE CONTROL, shielded twice: the lookup never loads a row with no session id, and the guard would + // refuse it anyway. No single-line mutation reaches it — dropping the guard alone stays green here. + // MUTATION: drop the lookup's session-id filter AND TryResumableFromCheckpoint's guard → the retry claims a + // conversation that was never written → red. The guard on its own is pinned where production can reach it: + // the deliberate-cancel arm (its row keeps a session id) and the still-running arm. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var lostAttemptRunId = await SeedHostLostAttemptAsync(teamId, repoId, runId, checkpoint: null); + + (await ReconcileUntilTerminalAsync(lostAttemptRunId)).Status.ShouldBe(AgentRunStatus.Failed); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", lostAttemptRunId)); + + ShouldBeColdAndClaimNothing(await ExecuteRetryAsync(context, "sb"), "no checkpoint was taken, so there is no conversation to restore"); + } + + [Fact] + public async Task A_retry_of_a_unit_that_was_deliberately_cancelled_is_cold_and_claims_no_host_loss() + { + // A deliberate cancel (an operator's, or the parent-terminal sweep's) is a CLEAN landing: nobody owes the + // unit a continuation, so it releases its checkpoint exactly as completion does. Left in place, the next + // retry of the same subtask — after an operator's Continue revives the run — would read the cancel as a host + // loss and hand the agent three false claims (a lost machine, a checkpoint provenance stamp, and + // resumed_from_agent_run_id), and the reference would pin the artifact Referenced for good. + // MUTATION: drop the two checkpoint SetProperty calls from CancelRunningAsync → the cancelled row still names + // its checkpoint → red. + // MUTATION: relax TryResumableFromCheckpoint's guard to accept a bare session id (the cancel keeps it) → the + // retry resumes a conversation with no transcript behind it → red, whichever path the resume then takes. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var cancelledRunId = await SeedLiveAttemptAsync(teamId, repoId, runId, (Guid.NewGuid(), "sess-cancelled")); + + (await CancelAsync(cancelledRunId)).ShouldBeTrue("the attempt was Running at the epoch the cancel read"); + + using (var mid = _fixture.BeginScope()) + { + var cancelled = await mid.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == cancelledRunId); + cancelled.Status.ShouldBe(AgentRunStatus.Cancelled); + cancelled.SessionTranscriptCheckpointArtifactId.ShouldBeNull("a clean landing releases the checkpoint"); + cancelled.SessionTranscriptCheckpointAt.ShouldBeNull(); + cancelled.SessionId.ShouldBe("sess-cancelled", "the cancel releases the checkpoint, not the session id — which is what makes the guard reachable here"); + } + + (await FindResumableAsync(teamId, runId)).ShouldBeNull("a cancelled attempt left nothing a retry may resume"); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", cancelledRunId)); + + ShouldBeColdAndClaimNothing(await ExecuteRetryAsync(context, "sb"), "a deliberately cancelled attempt's machine was not lost, and it left no conversation to restore"); + } + + [Fact] + public async Task A_retry_never_resumes_a_prior_attempt_that_is_still_running() + { + // The checkpoint columns are written while an attempt is Running, so a row still Running holds the checkpoint + // of a session that may be live: a kill-wave is best-effort, and once a revived parent is Pending again the + // orphan sweep stops selecting it. Resuming it would fork a conversation that is still being written. + // MUTATION: drop the terminal-status guard from TryResumableFromCheckpoint → the retry resumes the live + // session → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var liveRunId = await SeedLiveAttemptAsync(teamId, repoId, runId, (Guid.NewGuid(), "sess-still-live")); + + (await FindResumableAsync(teamId, runId)).ShouldBeNull("a Running attempt's checkpoint belongs to a session that may still be live"); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", liveRunId)); + + ShouldBeColdAndClaimNothing(await ExecuteRetryAsync(context, "sb"), "nothing may resume a conversation another process is still writing"); + } + + // ── 3c: the tree sentence describes the resumed attempt's OWN git state ───────── + + [Fact] + public async Task A_dependent_unit_lost_before_it_pushed_is_told_its_work_is_gone_not_that_its_producers_branch_is_its_own() + { + // A dependent unit is cloned at its producer's handoff branch. When its own attempt lost its host before it + // pushed, that branch holds the PRODUCER's work and none of this attempt's, so the tree sentence must be the + // honest "nothing was preserved" — never "your previous attempt published ``". + // MUTATION: pass the effective clone ref (effectiveStaging.Ref) as the resumed attempt's workspaceRef → the + // goal names the producer's branch as this attempt's published work → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var producer = await RunProducerAsync(teamId, repoId, runId); + var lostAttemptRunId = await SeedHostLostAttemptAsync(teamId, repoId, runId, (Guid.NewGuid(), "sess-dependent")); + + (await ReconcileUntilTerminalAsync(lostAttemptRunId)).Status.ShouldBe(AgentRunStatus.Failed); + + var task = await ExecuteRetryAsync(DependentContext(runId, teamId, repoId, producer.Spawn, await FailedAttempt(teamId, "sb", lostAttemptRunId, sequence: 3)), "sb"); + + task.Workspace!.Repositories.Single().Ref.ShouldBe(producer.Branch, "staging is unchanged: a dependent unit still clones its producer's handoff"); + task.RestoredTranscriptIsCheckpoint.ShouldBeTrue(); + task.Goal.ShouldContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive); + task.Goal.ShouldContain(AgentRetryContinuity.HonestNoContinuityHint, Case.Sensitive, "the lost attempt pushed nothing, so none of its tree survived"); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPublishedBranchHint(producer.Branch), Case.Sensitive, "the producer's branch is not this attempt's published work"); + } + + [Fact] + public async Task A_dependent_unit_resumed_from_a_captured_transcript_is_told_its_own_changes_were_not_preserved() + { + // The same rule on the ordinary path. An attempt that finished without pushing left its changes in a + // workspace nobody will see again; the clone ref is still its producer's handoff, but that is not THIS + // attempt's work, so it must not suppress the honest-redo line — which the effective ref used to do. + // MUTATION: pass effectiveStaging.Ref as workspaceRef → the producer's branch suppresses the line → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var producer = await RunProducerAsync(teamId, repoId, runId); + var (failedAttemptRunId, _) = await RunFailingPriorAttemptAsync(teamId, repoId, runId, "exit 1"); + (await ManifestsAsync(failedAttemptRunId, teamId)).ShouldBeEmpty("precondition: the failed attempt pushed nothing of its own"); + + await StampResumableSessionAsync(failedAttemptRunId, "sess-captured", "the dependent attempt's conversation\n"); + + var task = await ExecuteRetryAsync(DependentContext(runId, teamId, repoId, producer.Spawn, await FailedAttempt(teamId, "sb", failedAttemptRunId, sequence: 3)), "sb"); + + task.ResumeFromSessionId.ShouldBe("sess-captured", "the captured conversation is still resumed"); + task.Workspace!.Repositories.Single().Ref.ShouldBe(producer.Branch, "staging is unchanged: a dependent unit still clones its producer's handoff"); + task.Goal.ShouldContain(AgentRetryContinuity.HonestNoContinuityHint, Case.Sensitive, "the restored conversation describes changes this workspace does not contain"); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive, "an attempt that finished did not lose its machine"); + } + + [Fact] + public async Task A_dependent_unit_that_pushed_its_own_branch_before_its_host_died_is_told_that_branch_is_here() + { + // The other side of the same rule: when the lost attempt DID push, its own branch outranks the producer's + // handoff as the clone ref, the published work is there, and the sentence names THAT branch. + // MUTATION: drop the resumed attempt's ref (pass workspaceRef: null) → the goal says nothing was preserved + // while the clone holds this attempt's own pushed work → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var producer = await RunProducerAsync(teamId, repoId, runId); + var (lostAttemptRunId, _) = await RunPriorAttemptAsync(teamId, repoId, runId, "printf 'own work\\n' > own.txt; echo edited"); + var ownBranch = (await SingleManifestAsync(lostAttemptRunId, teamId)).Branch.ShouldNotBeNull("precondition: the attempt published its own branch before its host died"); + + await LoseTheHostAfterPublishingAsync(lostAttemptRunId, (Guid.NewGuid(), "sess-published")); + + var task = await ExecuteRetryAsync(DependentContext(runId, teamId, repoId, producer.Spawn, await FailedAttempt(teamId, "sb", lostAttemptRunId, sequence: 3)), "sb"); + + task.Workspace!.Repositories.Single().Ref.ShouldBe(ownBranch, "the attempt's own pushed branch outranks its producer's handoff"); + task.RestoredTranscriptIsCheckpoint.ShouldBeTrue(); + task.Goal.ShouldContain(AgentRetryContinuity.LostHostPublishedBranchHint(ownBranch), Case.Sensitive, "the published work IS here, and the agent is told which branch holds it"); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPublishedBranchHint(producer.Branch), Case.Sensitive); + task.Goal.ShouldNotContain(AgentRetryContinuity.HonestNoContinuityHint, Case.Sensitive, "asserting its work is gone would be a lie"); + } + + // ── 3c: the checkpoint opt-in at the staging seam ───────────────────────────── + + [Theory] + [InlineData(2, 8, true)] // room for this retry AND another after it + [InlineData(7, 8, false)] // this retry lands ON the cap — nothing can follow, so nobody would read a checkpoint + public async Task A_staged_unit_checkpoints_only_while_the_run_could_still_respawn_it(int totalSpawned, int cap, bool checkpoints) + { + // The opt-in, pinned at the STAGING SEAM rather than only as a pure predicate: this is the one place every + // spawn wave, retry and resolve passes through, so it is where the envelope either carries the flag or not. + // MUTATION: set CheckpointSessionTranscript unconditionally (or never) at that seam → one arm reds. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: null) with { TotalSpawnedAgents = totalSpawned, MaxTotalSpawns = cap }; + + (await ExecuteRetryAsync(context, "sb")).CheckpointSessionTranscript.ShouldBe(checkpoints); + } + + [Theory] + [InlineData(5, 8, true)] // 5 + 2 = 7 < 8: a retry still fits after the whole wave + [InlineData(6, 8, false)] // 6 + 2 = 8: the WAVE lands on the cap, though one agent alone (6 + 1 = 7) would not + public async Task A_spawn_wave_checkpoints_only_when_a_retry_still_fits_after_the_whole_wave(int totalSpawned, int cap, bool checkpoints) + { + // The seam reads the WAVE's size, not one agent's: by the time any retry is decided, every agent this wave + // stages is already on the tape and counted. + // MUTATION: pass 1 instead of tasks.Count at the staging seam → the (6, 8) arm checkpoints both units → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var context = ContextWith(runId, teamId, repoId, plan: Plan(("sa", null), ("sb", null)), priorAttempt: null) with { TotalSpawnedAgents = totalSpawned, MaxTotalSpawns = cap }; + + var staged = await ExecuteSpawnAsync(context, "sa", "sb"); + + staged.Count.ShouldBe(2, "one wave, two units"); + staged.ShouldAllBe(t => t.CheckpointSessionTranscript == checkpoints); + } + + private static void ShouldBeColdAndClaimNothing(AgentTask task, string because) + { + task.ResumeFromSessionId.ShouldBeNull(because); + task.RestoredTranscript.ShouldBeNull(because); + task.RestoredTranscriptArtifactId.ShouldBeNull(because); + task.RestoredTranscriptIsCheckpoint.ShouldBeFalse(because); + task.ResumedFromCheckpointAt.ShouldBeNull(because); + task.ResumedFromAgentRunId.ShouldBeNull(because); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPreamble, customMessage: because); + task.Goal.ShouldNotContain(AgentRetryContinuity.HonestNoContinuityHint, customMessage: because); + } + + /// + /// A prior attempt of "sb" in the state a HOST LOSS leaves it: created through the REAL + /// (so its envelope, SubtaskId and columns are the production shape), claimed by a + /// worker that is now gone — its owner id and fence epoch still on the row, its lease lapsed — with a durable + /// handle minted on a host that never came back whose own wall clock has passed: the exact shape + /// AgentRunReconcilerService.DeferToTheMintingHostAsync stops deferring and abandons. A checkpoint, when + /// given, is stamped the way the observer's drain tick leaves it — artifact, time and session id together, the + /// only way production writes a Running row's session id. + /// + /// Left un-executed rather than run and then aged, and the reason bounds what these arms prove: the stale + /// sweep deliberately EXCLUDES a run carrying events inside (a streaming + /// agent whose lease merely lapsed must never be abandoned), and agent_run_event is append-only by + /// trigger — so a run that genuinely executed cannot be backdated into this shape. Such a run also published + /// nothing; the published-branch case is 's. + /// + private Task SeedHostLostAttemptAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId, (Guid ArtifactId, string SessionId)? checkpoint) => + SeedClaimedAttemptAsync(teamId, repositoryId, supervisorRunId, checkpoint, hostLost: true); + + /// A prior attempt of "sb" still RUNNING on a live worker — claimed, its lease fresh, a checkpoint on the row. What a deliberate cancel lands on, and what a best-effort kill-wave or a revived parent can leave behind while a retry is decided. + private Task SeedLiveAttemptAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId, (Guid ArtifactId, string SessionId) checkpoint) => + SeedClaimedAttemptAsync(teamId, repositoryId, supervisorRunId, checkpoint, hostLost: false); + + private async Task SeedClaimedAttemptAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId, (Guid ArtifactId, string SessionId)? checkpoint, bool hostLost) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + // The envelope a supervisor unit really carries: SubtaskId is what FindResumableSubtaskAttemptAsync matches + // on, and CheckpointSessionTranscript is the opt-in that let its executor checkpoint at all. + var created = await scope.Resolve().CreateAsync( + new AgentTask { Goal = "do sb", Harness = "scripted", Model = "test-model", RepositoryId = repositoryId, SubtaskId = "sb", CheckpointSessionTranscript = true }, + teamId, supervisorRunId, NodeId, iterationKey: "", cancellationToken: CancellationToken.None); + + var run = await db.AgentRun.SingleAsync(r => r.Id == created.Id); + var claimedAt = DateTimeOffset.UtcNow - (hostLost ? TimeSpan.FromMinutes(20) : TimeSpan.FromMinutes(2)); + + run.Status = AgentRunStatus.Running; + run.OwnerId = Guid.NewGuid(); + run.FenceEpoch = 1; + run.StartedAt = claimedAt; + run.HeartbeatAt = claimedAt; + run.LeaseExpiresAt = hostLost ? claimedAt + AgentRunLiveness.Window : DateTimeOffset.UtcNow + AgentRunLiveness.Window; + run.RunnerHandleJson = hostLost ? JsonSerializer.Serialize(ForeignHostHandle(), AgentJson.Options) : null; + run.SessionId = checkpoint?.SessionId; + run.SessionTranscriptCheckpointArtifactId = checkpoint?.ArtifactId; + run.SessionTranscriptCheckpointAt = checkpoint is null ? null : claimedAt + TimeSpan.FromMinutes(1); + await db.SaveChangesAsync(); + + return run.Id; + } + + /// A durable handle minted on a host that never came back, whose own wall clock has already passed — the reconciler cannot probe it from here and stops deferring to it. + private static SandboxHandle ForeignHostHandle() + { + var spoolDirectory = Path.Combine(Path.GetTempPath(), "cs-supervisor-host-loss-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(spoolDirectory); + + return new SandboxHandle { Kind = "local", ProcessId = 0x7FFFFFFF, LaunchHost = "a-host-that-never-came-back", SpoolDirectory = spoolDirectory, Deadline = DateTimeOffset.UtcNow.AddMinutes(-1) }; + } + + /// + /// Turn an attempt that really ran and PUBLISHED into the row a host loss leaves when the machine dies between + /// the executor's publish and its terminal write: the manifest and the pushed branch are already real, and the + /// abandon then writes Failed with no result while keeping the checkpoint. Written directly rather than through + /// the reconciler because a run that really executed carries fresh events, which the stale sweep skips, and + /// agent_run_event is append-only — the reconciler's own abandon is what + /// drives. + /// + private async Task LoseTheHostAfterPublishingAsync(Guid agentRunId, (Guid ArtifactId, string SessionId) checkpoint) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var run = await db.AgentRun.SingleAsync(r => r.Id == agentRunId); + + run.Status = AgentRunStatus.Failed; + run.ResultJson = null; + run.SessionId = checkpoint.SessionId; + run.SessionTranscriptCheckpointArtifactId = checkpoint.ArtifactId; + run.SessionTranscriptCheckpointAt = DateTimeOffset.UtcNow.AddMinutes(-2); + await db.SaveChangesAsync(); + } + + /// + /// Sweep with the REAL reconciler until the row this test owns is terminal, and return it. The stale sweep is + /// deployment-wide and takes the oldest lapsed leases first, at + /// a time, so rows other tests left behind can fill a sweep; each sweep terminalizes what it takes, so a bounded + /// number of them reaches ours. + /// + private async Task ReconcileUntilTerminalAsync(Guid agentRunId) + { + const int maxSweeps = 10; + + for (var sweep = 0; sweep < maxSweeps; sweep++) + { + using var scope = _fixture.BeginScope(); + await scope.Resolve().ReconcileAsync(CancellationToken.None); + + var row = await scope.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == agentRunId); + if (AgentRunStateMachine.IsTerminal(row.Status)) return row; + } + + throw new ShouldAssertException($"agent run {agentRunId} was still not terminal after {maxSweeps} reconciler sweeps. Check that its lease has lapsed, that it has no event inside AgentRunLiveness.Window, and that its handle's deadline has passed — the reconciler logs 'leaving agent run ... alone' when it defers."); + } + + private async Task CancelAsync(Guid agentRunId) + { + using var scope = _fixture.BeginScope(); + return await scope.Resolve().CancelRunningAsync(agentRunId, "Cancelled by an operator.", AgentRunAbandonCause.OperatorCancelled, CancellationToken.None); + } + + private async Task FindResumableAsync(Guid teamId, Guid supervisorRunId) + { + using var scope = _fixture.BeginScope(); + return await scope.Resolve().FindResumableSubtaskAttemptAsync(teamId, supervisorRunId, "sb", CancellationToken.None); + } + + /// Run the plan's producer "pa" for real and record it as a Succeeded spawn — the handoff a dependent "sb" is cloned at. Returns the producer's pushed branch and that spawn decision. + private async Task<(string Branch, SupervisorPriorDecision Spawn)> RunProducerAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId) + { + var (producerRunId, _) = await RunPriorAttemptAsync(teamId, repositoryId, supervisorRunId, "printf 'producer work\\n' > producer.txt; echo edited", subtaskId: "pa"); + var branch = (await SingleManifestAsync(producerRunId, teamId)).Branch.ShouldNotBeNull("precondition: the producer really pushed, so a dependent is cloned at its branch"); + + return (branch, await SucceededSpawn(teamId, "pa", producerRunId)); + } + // ─── Drive the real executor ────────────────────────────────────────────────── private async Task ExecuteRetryAsync(SupervisorTurnContext context, string subtaskId) @@ -213,6 +629,20 @@ private async Task ExecuteRetryAsync(SupervisorTurnContext context, s return JsonSerializer.Deserialize(run.TaskJson, AgentJson.Options)!; } + /// One spawn wave through the real executor, returning the envelope of every unit it staged. + private async Task> ExecuteSpawnAsync(SupervisorTurnContext context, params string[] subtaskIds) + { + using var scope = _fixture.BeginScope(); + + var payload = JsonSerializer.Serialize(new SupervisorSpawnPayload { SubtaskIds = subtaskIds }, AgentJson.Options); + + await scope.Resolve().ExecuteAsync(new SupervisorDecision { Kind = SupervisorDecisionKinds.Spawn, PayloadJson = payload }, context, CancellationToken.None); + + var runs = await scope.Resolve().AgentRun.AsNoTracking().Where(r => r.WorkflowRunId == context.SupervisorRunId && r.NodeId == NodeId).ToListAsync(); + + return runs.Select(r => JsonSerializer.Deserialize(r.TaskJson, AgentJson.Options)!).ToList(); + } + // ─── Context / decision-tape builders ───────────────────────────────────────── private static SupervisorTurnContext ContextWith(Guid runId, Guid teamId, Guid repositoryId, SupervisorPriorDecision plan, SupervisorPriorDecision? priorAttempt) => new() @@ -226,29 +656,51 @@ private async Task ExecuteRetryAsync(SupervisorTurnContext context, s AgentProfile = new CodeSpace.Messages.Dtos.Agents.SupervisorAgentProfile { RepositoryId = repositoryId }, }; - private static SupervisorPriorDecision Plan(string subtaskId) + /// The dependent shape: "sb" depends on the producer "pa", whose spawn and "sb"'s own recorded attempt follow the plan on the tape. + private static SupervisorTurnContext DependentContext(Guid runId, Guid teamId, Guid repositoryId, SupervisorPriorDecision producerSpawn, SupervisorPriorDecision dependentAttempt) + { + var plan = Plan(("pa", null), ("sb", new[] { "pa" })); + + return ContextWith(runId, teamId, repositoryId, plan, priorAttempt: null) with { PriorDecisions = new[] { plan, producerSpawn, dependentAttempt } }; + } + + private static SupervisorPriorDecision Plan(string subtaskId) => Plan((subtaskId, null)); + + private static SupervisorPriorDecision Plan(params (string Id, string[]? DependsOn)[] subtasks) { var payload = JsonSerializer.Serialize(new SupervisorPlanPayload { Goal = Goal, - Subtasks = new List { new() { Id = subtaskId, Title = subtaskId, Instruction = $"do {subtaskId}" } }, + Subtasks = subtasks.Select(s => new SupervisorPlannedSubtask { Id = s.Id, Title = s.Id, Instruction = $"do {s.Id}", DependsOn = s.DependsOn }).ToList(), }, AgentJson.Options); return new SupervisorPriorDecision { Id = Guid.NewGuid(), Sequence = 1, DecisionKind = SupervisorDecisionKinds.Plan, Status = SupervisorDecisionStatus.Succeeded, PayloadJson = payload, OutcomeJson = "{}" }; } /// A prior FAILED spawn recording (subtaskId, REAL agentRunId) — the positional subtaskIds[i] ↔ agentResults[i] shape reads to find "this subtask's latest attempt", UNFILTERED on success (the whole point of a retry). - private async Task FailedAttempt(Guid teamId, string subtaskId, Guid agentRunId) + private Task FailedAttempt(Guid teamId, string subtaskId, Guid agentRunId, long sequence = 2) => + RecordedSpawn(teamId, subtaskId, agentRunId, ("Failed", "acceptance failed"), sequence); + + /// A prior SUCCEEDED spawn of a producer — what dependency staging reads to hand its branch to a dependent. + private Task SucceededSpawn(Guid teamId, string subtaskId, Guid agentRunId) => + RecordedSpawn(teamId, subtaskId, agentRunId, ("Succeeded", null), sequence: 2); + + private async Task RecordedSpawn(Guid teamId, string subtaskId, Guid agentRunId, (string Status, string? Error) outcome, long sequence) { - using var scope = _fixture.BeginScope(); - var manifests = await scope.Resolve().ListForAgentRunAsync(agentRunId, teamId, CancellationToken.None); + var manifests = await ManifestsAsync(agentRunId, teamId); - var result = new SupervisorAgentResult { AgentRunId = agentRunId, Status = "Failed", Error = "acceptance failed", ProducedBranch = manifests.FirstOrDefault()?.Branch }; + var result = new SupervisorAgentResult { AgentRunId = agentRunId, Status = outcome.Status, Error = outcome.Error, ProducedBranch = manifests.FirstOrDefault()?.Branch }; var payload = JsonSerializer.Serialize(new SupervisorSpawnPayload { SubtaskIds = new[] { subtaskId } }, AgentJson.Options); - var outcome = JsonSerializer.Serialize(new { agentRunIds = new[] { agentRunId }, agentCount = 1, agentResults = new[] { result } }, AgentJson.Options); + var outcomeJson = JsonSerializer.Serialize(new { agentRunIds = new[] { agentRunId }, agentCount = 1, agentResults = new[] { result } }, AgentJson.Options); + + return new SupervisorPriorDecision { Id = Guid.NewGuid(), Sequence = sequence, DecisionKind = SupervisorDecisionKinds.Spawn, Status = SupervisorDecisionStatus.Succeeded, PayloadJson = payload, OutcomeJson = outcomeJson }; + } - return new SupervisorPriorDecision { Id = Guid.NewGuid(), Sequence = 2, DecisionKind = SupervisorDecisionKinds.Spawn, Status = SupervisorDecisionStatus.Succeeded, PayloadJson = payload, OutcomeJson = outcome }; + private async Task> ManifestsAsync(Guid agentRunId, Guid teamId) + { + using var scope = _fixture.BeginScope(); + return await scope.Resolve().ListForAgentRunAsync(agentRunId, teamId, CancellationToken.None); } /// A prior FAILED retry recording (subtaskId, REAL agentRunId), SEQUENCED AFTER — the decision-tape-literal "latest attempt" reads, independent of whether it is actually resumable. diff --git a/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs index 3d2bf0e02..b8393fefe 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs @@ -50,7 +50,7 @@ public void A_format_fault_retry_goes_fresh_with_thinking_disabled() var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); var result = Result("API Error: Content block is not a thinking block"); - var task = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, result, workspaceHasPriorWork: true); + var task = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, result, workspaceRef: "agent/prior"); task.ResumeFromSessionId.ShouldBeNull("a conversation replay re-triggers the format fault deterministically — the retry must NOT resume"); task.RestoredTranscript.ShouldBeNull(); @@ -63,11 +63,11 @@ public void An_ordinary_failure_keeps_todays_resume_semantics_byte_identically() { var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); - var resumed = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, Result("acceptance: ./check.sh exited 2"), workspaceHasPriorWork: true); + var resumed = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, Result("acceptance: ./check.sh exited 2"), workspaceRef: "agent/prior"); resumed.ResumeFromSessionId.ShouldBe("sess-1"); resumed.Environment.ContainsKey(AgentRetryCauses.MaxThinkingTokensEnvVar).ShouldBeFalse("no degrade on an ordinary failure"); - var cold = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior: null, Result("boom"), workspaceHasPriorWork: false); + var cold = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior: null, Result("boom"), workspaceRef: null); cold.ResumeFromSessionId.ShouldBeNull(); cold.Environment.ContainsKey(AgentRetryCauses.MaxThinkingTokensEnvVar).ShouldBeFalse(); } @@ -125,7 +125,7 @@ public async Task Every_retry_lane_repairs_a_format_fault_through_the_same_helpe const string liveError = "API Error: Content block is not a thinking block"; var supervisorTask = RealSupervisorActionExecutor.ApplyRetryDisposition( - Task_(), new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null), Result(liveError), workspaceHasPriorWork: true); + Task_(), new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null), Result(liveError), workspaceRef: "agent/prior"); var priorAttempt = JsonDocument.Parse($$""" {"status":"Failed","exitReason":"non-zero-exit","error":"{{liveError}}","sessionId":"sess-1","sessionTranscript":"transcript"} diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs index dd403b722..164cd0e94 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs @@ -274,6 +274,50 @@ public void The_terminal_reasons_are_pinned() SupervisorStopReasons.PlanInvalid.ShouldBe("plan structurally invalid"); } + // ─── 3c: may a unit this wave stages ever be respawned? ───────────────────────── + + [Theory] + [InlineData(0, 3, 10, true)] // plenty of room after the wave + [InlineData(6, 3, 10, true)] // 9 of 10 spent — one retry still fits + [InlineData(7, 3, 10, false)] // the wave lands exactly ON the cap: no retry can follow + [InlineData(8, 3, 10, false)] // the wave itself would breach it; PostDecision refuses the wave, and nothing follows either + [InlineData(0, 1, 1, false)] // a one-spawn run: its only agent can never be retried + public void A_unit_is_checkpointed_only_while_the_run_could_still_respawn_it(int alreadySpawned, int waveSize, int cap, bool expected) + { + // The 3c opt-in. A checkpoint costs a whole-file read and an artifact write a minute for as long as the unit + // runs, so it is paid for only where somebody could consume it — and the only bound that says "no further + // agent can EVER be created" is the total-spawn cap. A later retry costs exactly one spawn, and by then this + // wave's own agents are on the tape, so the room has to exist now: strictly less than the cap, not at it. + // MUTATION: use <= instead of < → the exactly-on-the-cap arm reds, and every unit of a spent run would pay + // for a checkpoint nobody can consume. + var context = Context(turn: 1, totalSpawned: alreadySpawned) with { MaxTotalSpawns = cap }; + + SupervisorBounds.CanRespawnAfterWave(context, waveSize).ShouldBe(expected); + } + + [Theory] + [InlineData(4)] // an operator-set cap, which rehydrate copies onto the context from the plan + [InlineData(null)] // a context carrying no cap (legacy): the predicate's fallback must be the plan's own default + public void The_respawn_headroom_reads_the_same_cap_the_spawn_bound_enforces(int? configuredCap) + { + // The two must never disagree about what the run can still do: if PostDecision would REFUSE a further retry, + // the unit must not have been checkpointed for one. Driven through both entry points on one context. + // MUTATION: `<=` for `<` → the explicit-cap arm reds; a fallback other than + // SupervisorLane.DefaultMaxTotalSpawns (or none) → the null arm reds, because the plan still enforces it. + var plan = SupervisorGoalPlan.From(new SupervisorGoalConfig { Goal = "g", MaxTotalSpawns = configuredCap }); + var cap = plan.MaxTotalSpawns; + var atCap = Context(turn: 1, totalSpawned: cap - 1) with { MaxTotalSpawns = configuredCap }; + + SupervisorBounds.CanRespawnAfterWave(atCap, waveSize: 1).ShouldBeFalse("the wave lands on the cap, so nothing can follow it"); + SupervisorBounds.PostDecision(atCap with { TotalSpawnedAgents = cap }, plan, Spawn("a")) + .ShouldBe(SupervisorStopReasons.TotalSpawnCapReached, "and the bound agrees: the retry that would have consumed the checkpoint is refused"); + + var withRoom = Context(turn: 1, totalSpawned: cap - 2) with { MaxTotalSpawns = configuredCap }; + + SupervisorBounds.CanRespawnAfterWave(withRoom, waveSize: 1).ShouldBeTrue(); + SupervisorBounds.PostDecision(withRoom with { TotalSpawnedAgents = cap - 1 }, plan, Spawn("a")).ShouldBeNull("the retry really is affordable — without this the predicate could be false-negative everywhere and still pass"); + } + // ─── Helpers ──────────────────────────────────────────────────────────────────── private static SupervisorTurnContext Context(int turn, int totalSpawned = 0, int noProgress = 0, decimal runSpend = 0m) => diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs index d7b3b5174..6ff3bfa52 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs @@ -227,6 +227,25 @@ public void A_plan_less_legacy_tape_keeps_its_result_fold() SupervisorDependencyGate.LatestAgentRunId(Context(spawn), "legacy").ShouldBe(legacyRunId); } + // ── LatestResultsBySubtask: the roster SupervisorAmendPrecondition.GradedAttempts grades from ── + + [Fact] + public void A_warm_respawn_of_a_lost_host_replaces_the_abandoned_attempt_rather_than_adding_a_second() + { + // The host-loss slice's billing question. A unit whose host died is abandoned with NO result at all and + // then respawned warm from its checkpoint — so the tape carries two entries for one subtask. The graded + // roster must keep exactly ONE, the respawn's, or a single host loss spends two of the run's graded + // attempts and the amend precondition grades a verdict the machine, not the work, produced. + // MUTATION: fold the tape in reverse (or keep the first write per subtask) → the abandoned attempt wins → red. + var tape = new[] { Spawn(("a", "Failed", null)), Retry(("a", "Succeeded", true)) }; + var respawnRunId = SupervisorOutcome.ReadAgentResults(tape[1].OutcomeJson).Single().AgentRunId; + + var graded = SupervisorDependencyGate.LatestResultsBySubtask(tape); + + graded.Count.ShouldBe(1, "one subtask owes one graded attempt no matter how many machines it burned through"); + graded["a"].AgentRunId.ShouldBe(respawnRunId, "the respawn's verdict is the subtask's verdict — the abandoned attempt was never graded by anything but the reconciler"); + } + // ── Frontier: the decider's guidance ─────────────────────────────────────────────── [Fact] diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs index 8da991a09..518f78b7d 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs @@ -417,7 +417,7 @@ public void A_workspace_pinned_to_prior_work_carries_no_honest_redo_hint() var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli" }; var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); - var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceHasPriorWork: true); + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef: "agent/prior"); resumed.ResumeFromSessionId.ShouldBe("sess-1"); resumed.RestoredTranscript.ShouldBe("transcript"); @@ -430,12 +430,69 @@ public void A_workspace_with_no_preserved_git_state_gets_the_honest_redo_hint() var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli" }; var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); - var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceHasPriorWork: false); + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef: null); resumed.ResumeFromSessionId.ShouldBe("sess-1", "the conversation is still restored"); resumed.Goal.ShouldBe($"do the thing\n\n{AgentRetryContinuity.HonestNoContinuityHint}", "the goal now HONESTLY says the git changes are NOT present, so the agent never trusts a restored conversation implying work it can't see"); } + // ── 3c: a unit whose HOST died resumes from its checkpoint, and owes a different sentence ── + + [Theory] + [InlineData("agent/prior", true, "published")] // the lost attempt pushed its own branch — that work IS in the fresh clone + [InlineData(null, true, "redo")] // it pushed nothing — none of its tree survived + [InlineData(null, false, "none")] // no repository at all — there was never a tree to lose + public void A_unit_resumed_from_a_host_loss_checkpoint_carries_its_provenance_and_the_lost_host_block(string? workspaceRef, bool hasRepository, string treeSentence) + { + // A checkpoint is not a captured transcript. The attempt that wrote it never finished: its machine is gone, + // so the conversation may describe turns the checkpoint never saw and edits the new sandbox does not + // contain — which is true even when a branch WAS pushed, because the unpublished remainder died with the + // host. The ordinary honest-redo line only covers "no branch to continue from", a smaller claim. The whole + // goal is asserted, so each arm pins exactly WHICH tree sentence follows the preamble. + // MUTATION: drop the CheckpointAt branch from ApplyResumeRecord → the task carries no provenance, is not + // marked a checkpoint (so an unreadable ref would FAIL the attempt instead of degrading), and the goal says + // only what an ordinary retry says → red. + // MUTATION: pass no branch to WithLostHostHint → the "published" arm reds; owe a tree regardless of the + // repository → the "none" arm reds. + var priorRunId = Guid.NewGuid(); + var checkpointAt = DateTimeOffset.UtcNow.AddMinutes(-2); + var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli", RepositoryId = hasRepository ? Guid.NewGuid() : null }; + var prior = new ResumableSession(priorRunId, "sess-lost", null, Guid.NewGuid(), checkpointAt); + + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef); + + resumed.ResumeFromSessionId.ShouldBe("sess-lost", "the CLI is told WHICH conversation to resume — a transcript with no id names nothing"); + resumed.RestoredTranscriptArtifactId.ShouldBe(prior.TranscriptArtifactId, "the checkpoint rides as a REF the executor resolves just before invocation"); + resumed.RestoredTranscriptIsCheckpoint.ShouldBeTrue("this ref is best-effort: unreadable must cost the conversation, never the attempt"); + resumed.ResumedFromCheckpointAt.ShouldBe(checkpointAt, "the launch stamps this onto the run's permanent confinement record"); + resumed.ResumedFromAgentRunId.ShouldBe(priorRunId, "which attempt took over from which is a column, not prose"); + resumed.Goal.ShouldBe($"do the thing\n\n{AgentRetryContinuity.LostHostPreamble}{ExpectedTreeSentence(treeSentence, workspaceRef)}", "the preamble always, then exactly the one sentence that is true about the tree"); + } + + private static string ExpectedTreeSentence(string treeSentence, string? workspaceRef) => treeSentence switch + { + "published" => " " + AgentRetryContinuity.LostHostPublishedBranchHint(workspaceRef!), + "redo" => " " + AgentRetryContinuity.HonestNoContinuityHint, + _ => "", + }; + + [Fact] + public void A_unit_resumed_from_a_completed_attempt_claims_no_checkpoint() + { + // The other side of the same fork: an attempt that FINISHED left its workspace behind, so its transcript is + // a capture — fail-closed if unreadable — and it owes no lost-host sentence. + // MUTATION: mark every resume a checkpoint → red, and an unreadable captured ref would silently cold-start. + var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli" }; + var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); + + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef: "agent/prior"); + + resumed.RestoredTranscriptIsCheckpoint.ShouldBeFalse(); + resumed.ResumedFromCheckpointAt.ShouldBeNull(); + resumed.ResumedFromAgentRunId.ShouldBeNull(); + resumed.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive); + } + // ── BuildBlockedSpawnOutcome: the wire shape resolve's conflict reader consumes ───── [Fact] diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs index 54c42af2b..3f937db6e 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs @@ -1207,7 +1207,7 @@ public async Task A_repo_less_respawn_is_never_told_its_git_changes_were_lost(st [Fact] public async Task A_respawn_whose_primary_push_failed_repins_the_sibling_but_is_still_told_the_truth() { - // The honesty decision follows the PRIMARY, exactly like the supervisor's workspaceHasPriorWork: a sibling's + // The honesty decision follows the PRIMARY, exactly like the supervisor's workspaceRef: a sibling's // successful push is still conserved (its own branch is repinned), but the agent's primary repo re-clones the // default branch, so the restored conversation MUST be told its changes are not there. An "any repo pushed" // read would suppress the line precisely where the primary lost its work — the worst place to go quiet.