diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs index add042785..292db89d9 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs @@ -5408,7 +5408,7 @@ private sealed record SessionCheckpointTick(Guid TeamId, IAgentHarness Harness, /// The checkpoint coordinates for a run whose envelope OPTED IN, else null — the one place the opt-in is read, so /// the produce side and the consume side cannot disagree about which runs are checkpointed. A run whose failed /// attempt nobody can retry writes nothing, which is what keeps the artifact store free of a per-minute - /// transcript copy for every benchmark cell, review child and supervisor unit on the fleet. + /// transcript copy for every benchmark cell, review child and supervisor unit that no retry could resume. /// private static SessionCheckpointTick? CheckpointTickFor(AgentTask task, Guid teamId, IAgentHarness harness, string? workingDirectory, AgentRunFacts facts) => task.CheckpointSessionTranscript ? new SessionCheckpointTick(teamId, harness, workingDirectory, facts) : null; diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs index 35bffde39..85f729eac 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs @@ -774,13 +774,19 @@ public async Task CancelRunningAsync(Guid runId, string reason, AgentRunAb if (snapshot is null || snapshot.Status != AgentRunStatus.Running) return false; + // 3c: a deliberate cancel is a CLEAN landing, so it releases the mid-run session checkpoint exactly as + // completion does. Nobody owes this run a continuation, and a kept reference would pin the artifact + // Referenced (terminal in the retention ledger) for good — and make a later retry of the same subtask read + // the cancel as a host loss. Only the reconciler's abandon keeps these columns. var cancelled = await _db.AgentRun .Where(r => r.Id == runId && r.Status == AgentRunStatus.Running && r.FenceEpoch == snapshot.FenceEpoch) .ExecuteUpdateAsync(s => s .SetProperty(r => r.Status, AgentRunStatus.Cancelled) .SetProperty(r => r.FenceEpoch, r => r.FenceEpoch + 1) .SetProperty(r => r.Error, reason) - .SetProperty(r => r.CompletedAt, (DateTimeOffset?)DateTimeOffset.UtcNow), cancellationToken) + .SetProperty(r => r.CompletedAt, (DateTimeOffset?)DateTimeOffset.UtcNow) + .SetProperty(r => r.SessionTranscriptCheckpointArtifactId, (Guid?)null) + .SetProperty(r => r.SessionTranscriptCheckpointAt, (DateTimeOffset?)null), cancellationToken) .ConfigureAwait(false); if (cancelled == 0) return false; @@ -956,7 +962,7 @@ await _db.AgentRun.AsNoTracking().SingleOrDefaultAsync(r => r.Id == runId, cance var candidates = await _db.AgentRun.AsNoTracking() .Where(a => a.TeamId == teamId && a.WorkflowRunId == supervisorRunId && a.SessionId != null) .OrderByDescending(a => a.CreatedDate).ThenByDescending(a => a.Id) - .Select(a => new { a.Id, a.SessionId, a.ResultJson, a.TaskJson }) + .Select(a => new { a.Id, a.Status, a.SessionId, a.ResultJson, a.TaskJson, a.SessionTranscriptCheckpointArtifactId, a.SessionTranscriptCheckpointAt }) .ToListAsync(cancellationToken).ConfigureAwait(false); // The most-recent RESUMABLE prior attempt of THIS subtask — skip a captured-but-transcript-less attempt so it @@ -966,11 +972,34 @@ await _db.AgentRun.AsNoTracking().SingleOrDefaultAsync(r => r.Id == runId, cance if (SubtaskIdOf(candidate.TaskJson) != subtaskId) continue; if (TryResumable(candidate.Id, candidate.SessionId, candidate.ResultJson) is { } resumable) return resumable; + + // 3c: a prior attempt whose HOST died has no result at all — the reconciler's abandon writes none — so + // the captured-transcript read above finds nothing for exactly the population a warm retry helps most. + // Its mid-run checkpoint is what survives, and a TERMINAL row still naming one is the signature of an + // abandon: every clean landing — completion or cancel — releases these two columns in its own terminal + // write; only an abandon keeps them. + if (TryResumableFromCheckpoint(candidate.Id, candidate.Status, candidate.SessionId, candidate.SessionTranscriptCheckpointArtifactId, candidate.SessionTranscriptCheckpointAt) is { } continued) return continued; } return null; } + /// + /// A prior attempt's MID-RUN checkpoint as a resumable session — the host-loss counterpart of + /// . Both halves are required for the same reason that one demands both: a session id + /// with no transcript resumes into "No conversation found", and a transcript nothing can address is not + /// resumable at all. + /// + /// And only from a TERMINAL row. The checkpoint columns are written while the attempt is Running, so a + /// row still Running holds the checkpoint of a session that may be live — a kill-wave is best-effort, and a + /// revived parent stops the orphan sweep from selecting it — and resuming it would fork a conversation that is + /// still being written. + /// + private static ResumableSession? TryResumableFromCheckpoint(Guid agentRunId, AgentRunStatus status, string? sessionId, Guid? checkpointArtifactId, DateTimeOffset? checkpointAt) => + AgentRunStateMachine.IsTerminal(status) && sessionId is { Length: > 0 } sid && checkpointArtifactId is { } artifactId && checkpointAt is { } at + ? new ResumableSession(agentRunId, sid, null, artifactId, at) + : null; + /// /// A prior agent run's (id, session id, result json) → its RESUMABLE session, or null when not resumable: no /// session id, no transcript (both-or-neither — a resume without one fails "No conversation found"), or a diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs b/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs index f261c3608..2b87c8f7e 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs @@ -161,6 +161,19 @@ internal static IReadOnlyList DependsOnFor(SupervisorPlannedSubtask? pla internal static DependencyStagingResult PreferPriorAttemptStaging(DependencyStagingResult priorAttemptStaging, DependencyStagingResult dependencyStaging) => priorAttemptStaging.Ref is not null ? priorAttemptStaging : dependencyStaging; + /// + /// 3c: whether a staged unit opts into the mid-run session checkpoint — only when a later retry could actually + /// consume it. A checkpoint costs a whole-file read and an artifact write a minute for as long as the unit runs, + /// and a host loss keeps the survivor for good, so a unit nobody can resume must not pay for one. + /// + /// Two exclusions. A unit with no subtask id can never be FOUND by the retry lookup, which matches on it: + /// that is every resolver unit, whose task is built without one, and a re-resolve stages a fresh resolver rather + /// than resuming the old one. And a unit the run can no longer afford to respawn has nobody to hand a restored + /// conversation to (). + /// + internal static bool CheckpointsSessionTranscript(AgentTask task, SupervisorTurnContext context, int waveSize) => + task.SubtaskId is { Length: > 0 } && SupervisorBounds.CanRespawnAfterWave(context, waveSize); + /// The subtask's target repository, resolved the SAME way will resolve it — a pure pre-computation so dependency staging can look up the right repo's manifest before the task itself is built. private static Guid? ResolveTargetRepositoryId(SupervisorAgentDispatch? spec, SupervisorTurnContext context) { @@ -433,7 +446,11 @@ private async Task ExecuteRetryAsync(SupervisorDecision dec var (escalatedTask, escalation) = await ApplyRetryEscalationAsync(builtTask, priorResult, context, cancellationToken).ConfigureAwait(false); - var task = ApplyRetryDisposition(escalatedTask, prior, priorResult, workspaceHasPriorWork: effectiveStaging.Ref is not null); + // The continuity sentence describes the resumed attempt's OWN git state, so it reads that attempt's own + // pushed branch — never the effective clone ref. For a dependent unit whose attempt pushed nothing, the + // effective ref is the PRODUCER's handoff branch: it says nothing about whether this attempt's work survived, + // and naming it would tell the agent another unit's branch is its own published work. + var task = ApplyRetryDisposition(escalatedTask, prior, priorResult, workspaceRef: priorAttemptStaging.Ref); if (AgentRetryCauses.Classify(priorResult?.Error) == AgentRetryCauses.GatewayFormatFault) _logger.LogWarning("Supervisor retry of subtask {SubtaskId}: the prior attempt died on a gateway FORMAT fault — retrying FRESH (a conversation replay re-triggers the fault) with extended thinking disabled ({EnvVar}=0)", retry.SubtaskId, AgentRetryCauses.MaxThinkingTokensEnvVar); @@ -513,20 +530,44 @@ private async Task ExecuteRetryAsync(SupervisorDecision dec /// World-state continuity (the prior branch tip staging) is decided elsewhere and stays UNCHANGED either way — /// the degrade drops the broken conversation, never the preserved work. /// - internal static AgentTask ApplyRetryDisposition(AgentTask task, ResumableSession? prior, SupervisorAgentResult? priorResult, bool workspaceHasPriorWork) + internal static AgentTask ApplyRetryDisposition(AgentTask task, ResumableSession? prior, SupervisorAgentResult? priorResult, string? workspaceRef) { if (AgentRetryCauses.Classify(priorResult?.Error) == AgentRetryCauses.GatewayFormatFault) return AgentRetryCauses.ApplyFormatFaultMitigation(task); - return prior is null ? task : ApplyResumeRecord(task, prior, workspaceHasPriorWork); + return prior is null ? task : ApplyResumeRecord(task, prior, workspaceRef); } - /// The pure fold of a resumable prior attempt onto the task: always stamps the session/transcript, and — ONLY when is false — appends the honest-redo line so the hint's truth value always matches the actual git state. Internal + static so the honesty branch is unit-pinned directly. - internal static AgentTask ApplyResumeRecord(AgentTask task, ResumableSession prior, bool workspaceHasPriorWork) + /// + /// The pure fold of a resumable prior attempt onto the task: always stamps the session + transcript, then says + /// what is true about the WORLD that conversation refers to. + /// + /// Two shapes, because the prior attempt ended two different ways. An attempt that FINISHED left its + /// workspace behind, so the only open question is whether it pushed a branch — + /// null means it did not, and the honest-redo line says so. An attempt whose + /// HOST died () left nothing but the conversation: its clone is + /// unreachable, so it owes the lost-host block instead, it carries its provenance onto the new run + /// (ResumedFromCheckpointAt for the permanent confinement record, ResumedFromAgentRunId for the + /// column), and its transcript ref is marked a CHECKPOINT so the executor degrades to a cold start rather than + /// failing the attempt when those bytes cannot be read. + /// + /// Internal + static so both honesty branches are unit-pinned directly. + /// + /// The branch the RESUMED attempt itself pushed, or null when it pushed none — never the effective clone ref, which for a dependent unit is its producer's handoff branch and says nothing about this attempt's work. + internal static AgentTask ApplyResumeRecord(AgentTask task, ResumableSession prior, string? workspaceRef) { var resumed = task with { ResumeFromSessionId = prior.SessionId, RestoredTranscript = prior.InlineTranscript, RestoredTranscriptArtifactId = prior.TranscriptArtifactId }; - return workspaceHasPriorWork ? resumed : resumed with { Goal = AgentRetryContinuity.WithHonestNoContinuityHint(resumed.Goal) }; + if (prior.CheckpointAt is { } checkpointAt) + return resumed with + { + RestoredTranscriptIsCheckpoint = true, + ResumedFromCheckpointAt = checkpointAt, + ResumedFromAgentRunId = prior.AgentRunId, + Goal = AgentRetryContinuity.WithLostHostHint(resumed.Goal, workspaceRef, treeOwed: task.RepositoryId is not null), + }; + + return workspaceRef is not null ? resumed : resumed with { Goal = AgentRetryContinuity.WithHonestNoContinuityHint(resumed.Goal) }; } /// @@ -775,9 +816,12 @@ private async Task StageAgentsAndParkAsync(IReadOnlyList<(A var reclaimed = k < orphans.Count; reclaimedAny |= reclaimed; + // 3c: the ONE staging seam every spawn wave, retry and resolve passes through, so the checkpoint opt-in is + // decided once here rather than at each verb's own task build — see CheckpointsSessionTranscript for who + // is excluded and why. var agentRunId = reclaimed ? orphans[k] - : await CreateResolvedAgentRunAsync(tasks[k].Task, tasks[k].Spec, context, cancellationToken).ConfigureAwait(false); + : await CreateResolvedAgentRunAsync(tasks[k].Task with { CheckpointSessionTranscript = CheckpointsSessionTranscript(tasks[k].Task, context, tasks.Count) }, tasks[k].Spec, context, cancellationToken).ConfigureAwait(false); StageAgentWait(context, k, agentRunId); agentRunIds.Add(agentRunId); diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs index c8c4a5c4e..9424be752 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorBounds.cs @@ -91,6 +91,28 @@ private static bool IsReauthorableRejection(SupervisorPriorDecision decision) => return null; } + /// + /// Whether the run could still afford to RESPAWN a unit this wave is about to stage — read from the same + /// total-spawn bound enforces, so the two can never disagree about what the run can + /// still do. + /// + /// A later retry costs exactly one spawn, and refuses it when + /// TotalSpawnedAgents + 1 would exceed the cap. By then this wave's own agents + /// are on the tape, so the room for it has to exist NOW: strictly less than the cap, not at it. + /// + /// The spawn cap and nothing else, deliberately. It is the one bound that says "no further agent can ever + /// be created" — whereas a run near its no-progress cap can still retry, because a wave that makes progress + /// resets that cadence. Gating on no-progress would leave the units most likely to be retried un-checkpointed, + /// which is the opposite of the point. The 3c consumer of this predicate pays a whole-file read and an artifact + /// write per minute, so it is worth asking whether anyone can consume the result. + /// + public static bool CanRespawnAfterWave(SupervisorTurnContext context, int waveSize) + { + ArgumentNullException.ThrowIfNull(context); + + return context.TotalSpawnedAgents + waveSize < (context.MaxTotalSpawns ?? SupervisorLane.DefaultMaxTotalSpawns); + } + /// How many agents the decision would spawn: a spawn fans out its subtaskIds; a retry is exactly one. Best-effort read — a malformed payload reads 0 (it stages nothing, so it can't breach a count bound). internal static int SpawnCount(SupervisorDecision decision) { diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs index c67552d6f..823fa0419 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs @@ -40,9 +40,10 @@ public static class ArtifactRetentionPolicy /// /// A mid-run session-transcript checkpoint. TWO HOURS, not seven days, and the short floor is the whole reason - /// the class exists: a run writes one of these per minute, each supersedes the last, and the run's terminal write - /// clears the column that references the survivor — so on the seven-day floor a single long run would hold every - /// superseded copy of a growing transcript for over a week. Two hours still sits far outside the window in which + /// the class exists: a run writes one of these per minute, each supersedes the last, and a clean landing + /// (completion or a deliberate cancel) clears the column that references the survivor — so on the seven-day floor + /// a single long run would hold every superseded copy of a growing transcript for over a week. An abandon keeps + /// the column on purpose, and that survivor is then Referenced for good; this floor does not collect it. Two hours still sits far outside the window in which /// the reference lands (the stamp is the next statement after the write) and far outside the window in which a /// continuation reads it (an abandon follows the host's death within one liveness window), so the floor costs /// nothing it protects. The quarantine stays the standard 24 h: the second, independent wait is unchanged. diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs index 54225485e..04577032c 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs @@ -506,7 +506,8 @@ private static AgentTask ApplyRespawnResumeHint(AgentTask task, JsonElement? pri /// base ref/pin byte-identical. /// /// The returned flag is whether the honest-redo line is OWED, and it follows the PRIMARY repo alone — - /// the same workspaceHasPriorWork: effectiveStaging.Ref is not null read the supervisor's retry uses. A + /// the same read the supervisor's retry makes of its prior attempt's own pushed branch + /// (workspaceRef: priorAttemptStaging.Ref), which is keyed on the target repository too. A /// multi-repo attempt whose primary push FAILED while a sibling's succeeded still repins that sibling, but its /// primary re-clones the default branch, so the resumed conversation must still be told its changes are not /// there: an OR across repos would suppress the line precisely where the agent's own repo lost its work. Nothing diff --git a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs index a830a0e2b..37d3cac9e 100644 --- a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs +++ b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs @@ -109,9 +109,10 @@ public sealed record AgentTask /// /// An opt-in rather than a default, because a checkpoint nobody will consume is pure waste: it costs a /// whole-file read and an artifact write per minute, per running agent, per worker. Only a producer whose failed - /// attempt can actually be RETRIED sets it — today that is agent.run for a node whose own retry policy - /// allows more than one attempt. The benchmark lanes (one attempt per cell by protocol), review children and - /// supervisor units leave it false. + /// attempt can actually be RETRIED sets it — today agent.run for a node whose own retry policy allows more + /// than one attempt, and a supervisor unit that carries a subtask id while its run can still afford to respawn + /// it. The benchmark lanes (one attempt per cell by protocol), review children and supervisor resolver units + /// leave it false. /// /// [JsonIgnore(WhenWritingDefault)] so an envelope that did not opt in adds nothing to task_json. /// diff --git a/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs b/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs index f5c07e221..a1303f8f0 100644 --- a/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs +++ b/backend/src/CodeSpace.Messages/Agents/ResumableSession.cs @@ -10,4 +10,15 @@ namespace CodeSpace.Messages.Agents; /// off THIS id, never a separately-resolved "latest attempt" id, so a resume hint's honesty claim about git state /// always describes the SAME attempt whose conversation it restores. /// -public sealed record ResumableSession(Guid AgentRunId, string SessionId, string? InlineTranscript, Guid? TranscriptArtifactId); +/// +/// When the transcript this session restores was CHECKPOINTED mid-run, because its attempt's host died before it +/// could finish — null for the ordinary case, where the transcript was captured by an attempt that completed. +/// +/// The two are not interchangeable and the difference is what the consumer owes the agent. A captured +/// transcript comes from an attempt that ran to its end with its workspace intact, so the only open question is +/// whether a branch was pushed. A checkpoint comes from an attempt whose machine is gone: the conversation may +/// describe turns the checkpoint never saw and edits the new sandbox does not contain, so only this one owes the +/// lost-tree sentence — and only this one may degrade to a cold start when its bytes turn out to be unreadable, +/// since a captured ref that cannot be read is a real fault. +/// +public sealed record ResumableSession(Guid AgentRunId, string SessionId, string? InlineTranscript, Guid? TranscriptArtifactId, DateTimeOffset? CheckpointAt = null); diff --git a/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs b/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs index 72593072f..246849d84 100644 --- a/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs +++ b/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs @@ -35,11 +35,13 @@ public enum ArtifactRetentionClass /// A mid-run resumable session transcript, referenced only by /// agent_run.session_transcript_checkpoint_artifact_id. Its own class rather than /// because its lifetime is nothing like an event payload's: exactly one - /// checkpoint per run is ever useful (the newest), each one supersedes the last, and the run's terminal write - /// clears the column — so every intermediate is garbage within minutes and the survivor within hours. Sharing the - /// event class's seven-day floor would hold roughly a gigabyte of superseded transcript per long run for over a - /// week, and the growth gate that makes checkpoints worth taking is exactly what stops the content-addressed - /// store deduplicating them. + /// checkpoint per run is ever useful (the newest), each one supersedes the last, and a clean landing (completion + /// or a deliberate cancel) clears the column — so every intermediate is garbage within minutes and the survivor + /// within hours. An abandon-class ending (the reconciler's abandon, or its spool recovery) deliberately KEEPS the + /// column, because that survivor is what the retry resumes from, and a kept reference is Referenced for good. + /// Sharing the event class's seven-day floor would hold roughly a gigabyte of superseded transcript per long run + /// for over a week, and the growth gate that makes checkpoints worth taking is exactly what stops the + /// content-addressed store deduplicating them. /// SessionTranscriptCheckpoint = 5, } diff --git a/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs b/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs index df69d9a38..da409d5ff 100644 --- a/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs +++ b/backend/tests/CodeSpace.E2ETests/Workflows/RealModelSupervisorWholeLoopE2ETests.cs @@ -1176,6 +1176,13 @@ public async Task The_real_model_reacts_to_a_failed_subtask_by_retrying() // gateway outage is non-gating LOUD infra; a no-secret config skips NOT-EVALUATED. The perpetual-failure scenario // force-STOPs cleanly on a bound (no-progress / total-spawn cap), never a run Failure, so a model that recovered // reads Drove from the ledger and is never mis-gated as a CodeFault. + // + // SCOPE, for the host-loss slice: the retries this arm drives are COLD by construction and must stay so. Each + // agent here reaches a real terminal exit with its tree intact, so the row carries no mid-run checkpoint and + // the respawn resumes nothing — the warm arm needs a machine that never came back, which no live wire can + // stage. That path is pinned deterministically instead (SupervisorRetryWorldStateFlowTests drives the REAL + // reconciler's abandon; SupervisorDependencyStagingTests pins the fold). This arm asserts nothing about the + // opt-in itself — the staging seam's own tests do. var baseUrl = Env(RealModelSupervisorDecisionFlowTests.BaseUrlEnvVar); var apiKey = Env(RealModelSupervisorDecisionFlowTests.ApiKeyEnvVar); var model = Env(RealModelSupervisorDecisionFlowTests.ModelIdEnvVar); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs index e7a81d964..75e58a1e0 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorResolveFlowTests.cs @@ -64,6 +64,29 @@ public async Task Resolve_stages_one_resolver_agent_whose_goal_names_every_branc task.PushProducedBranch.ShouldBe(true, "the resolver MUST push its reconciled branch so a downstream PR-open has a head"); } + [Fact] + public async Task A_resolver_unit_never_opts_into_the_session_checkpoint() + { + // Resolve stages through the same seam as spawn and retry, but its unit carries no subtask id — and the retry + // lookup matches on one, so nothing could ever resume a resolver's checkpoint. It would pay a whole-file read + // and an artifact write a minute for nothing, and keep the survivor for good on a host loss. The run here has + // ample spawn headroom, so only the subtask-id gate can keep the flag off. + // MUTATION: drop the SubtaskId gate from CheckpointsSessionTranscript → the resolver opts in → red. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var runId = await SeedSupervisorRunAsync(teamId, userId); + + var context = ContextWith(runId, teamId, + repositoryId: Guid.NewGuid(), + spawn: SpawnWithBranches("codespace/agent/web", "codespace/agent/api"), + merge: ConflictedMerge("src/Shared.cs")); + + await ExecuteResolveAsync(context); + + var task = JsonSerializer.Deserialize((await StagedAgentRunsAsync(runId)).ShouldHaveSingleItem().TaskJson, AgentJson.Options)!; + task.SubtaskId.ShouldBeNull("precondition: a resolver's task is built without a subtask id"); + task.CheckpointSessionTranscript.ShouldBeFalse("no retry can find a unit without a subtask id, so none may pay for a checkpoint"); + } + [Theory] [InlineData("no-conflict")] // a clean merge on the tape → nothing to resolve [InlineData("no-repo")] // conflict present but no repository bound diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs index 11833fa8b..5c3cdd396 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorRetryWorldStateFlowTests.cs @@ -194,6 +194,422 @@ public async Task With_two_recorded_attempts_the_git_ref_always_matches_the_conv task.Goal.ShouldNotContain(AgentRetryContinuity.HonestNoContinuityHint, customMessage: "the resumed attempt's own branch IS preserved — asserting otherwise would be a lie"); } + // ── 3c: a unit whose HOST died resumes from its mid-run checkpoint ───────────── + + [Fact] + public async Task A_retry_of_a_unit_whose_host_died_resumes_from_its_checkpoint() + { + // The population #1994 excluded. A supervisor unit whose host is lost is abandoned by the reconciler with NO + // result at all, so the captured-transcript lookup every warm retry uses finds nothing. What survives is the + // checkpoint on the row, and a terminal row still naming one is the signature of that abandon: every clean + // landing — completion or cancel — releases those columns in its own terminal write. + // MUTATION: drop the TryResumableFromCheckpoint fall-through → the retry cold-starts with no provenance → red. + // MUTATION: drop the CheckpointAt branch in ApplyResumeRecord → the retry still resumes the checkpoint, but + // unmarked (an unreadable ref would then fail the attempt), with no provenance and the ordinary honest-redo + // line in place of the lost-host block → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var checkpointArtifactId = Guid.NewGuid(); + var lostAttemptRunId = await SeedHostLostAttemptAsync(teamId, repoId, runId, (checkpointArtifactId, "sess-lost-host")); + + var abandoned = await ReconcileUntilTerminalAsync(lostAttemptRunId); + + abandoned.Status.ShouldBe(AgentRunStatus.Failed, "the reconciler terminalizes a run whose host never came back"); + abandoned.SessionTranscriptCheckpointArtifactId.ShouldBe(checkpointArtifactId, "and the abandon KEEPS the checkpoint — releasing it is a clean landing's job, not the abandon's"); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", lostAttemptRunId)); + + var task = await ExecuteRetryAsync(context, "sb"); + + task.ResumeFromSessionId.ShouldBe("sess-lost-host", "the CLI is told WHICH conversation to resume — a transcript with no id names nothing"); + task.RestoredTranscriptArtifactId.ShouldBe(checkpointArtifactId, "the checkpoint rides as a REF the executor resolves just before invocation"); + task.RestoredTranscriptIsCheckpoint.ShouldBeTrue("best-effort bytes: unreadable must cost the conversation, never this retry attempt"); + task.ResumedFromCheckpointAt.ShouldNotBeNull("the launch stamps this onto the run's permanent confinement record"); + task.ResumedFromAgentRunId.ShouldBe(lostAttemptRunId, "which attempt took over from which is a column, not prose"); + task.Goal.ShouldContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive, "a restored conversation describes a machine that is gone, and the agent must be told"); + + using var verify = _fixture.BeginScope(); + var respawn = await verify.Resolve().AgentRun.AsNoTracking() + .Where(r => r.WorkflowRunId == runId && r.Id != lostAttemptRunId).OrderByDescending(r => r.CreatedDate).FirstAsync(); + + respawn.ResumedFromAgentRunId.ShouldBe(lostAttemptRunId, "and the task's provenance is promoted onto the row, like AgentDefinitionId"); + } + + [Fact] + public async Task A_retry_of_a_unit_lost_without_a_checkpoint_is_cold_and_claims_nothing() + { + // The same host loss from a unit that checkpointed nothing — it never opted in, or died before its first + // checkpoint. Production leaves such a row with no session id either: the checkpoint stamp is the only writer + // of a Running row's session id, and it writes both together. Nothing to restore, so the respawn cold-starts + // and claims neither a lost machine nor a restored conversation. + // A NEGATIVE CONTROL, shielded twice: the lookup never loads a row with no session id, and the guard would + // refuse it anyway. No single-line mutation reaches it — dropping the guard alone stays green here. + // MUTATION: drop the lookup's session-id filter AND TryResumableFromCheckpoint's guard → the retry claims a + // conversation that was never written → red. The guard on its own is pinned where production can reach it: + // the deliberate-cancel arm (its row keeps a session id) and the still-running arm. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var lostAttemptRunId = await SeedHostLostAttemptAsync(teamId, repoId, runId, checkpoint: null); + + (await ReconcileUntilTerminalAsync(lostAttemptRunId)).Status.ShouldBe(AgentRunStatus.Failed); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", lostAttemptRunId)); + + ShouldBeColdAndClaimNothing(await ExecuteRetryAsync(context, "sb"), "no checkpoint was taken, so there is no conversation to restore"); + } + + [Fact] + public async Task A_retry_of_a_unit_that_was_deliberately_cancelled_is_cold_and_claims_no_host_loss() + { + // A deliberate cancel (an operator's, or the parent-terminal sweep's) is a CLEAN landing: nobody owes the + // unit a continuation, so it releases its checkpoint exactly as completion does. Left in place, the next + // retry of the same subtask — after an operator's Continue revives the run — would read the cancel as a host + // loss and hand the agent three false claims (a lost machine, a checkpoint provenance stamp, and + // resumed_from_agent_run_id), and the reference would pin the artifact Referenced for good. + // MUTATION: drop the two checkpoint SetProperty calls from CancelRunningAsync → the cancelled row still names + // its checkpoint → red. + // MUTATION: relax TryResumableFromCheckpoint's guard to accept a bare session id (the cancel keeps it) → the + // retry resumes a conversation with no transcript behind it → red, whichever path the resume then takes. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var cancelledRunId = await SeedLiveAttemptAsync(teamId, repoId, runId, (Guid.NewGuid(), "sess-cancelled")); + + (await CancelAsync(cancelledRunId)).ShouldBeTrue("the attempt was Running at the epoch the cancel read"); + + using (var mid = _fixture.BeginScope()) + { + var cancelled = await mid.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == cancelledRunId); + cancelled.Status.ShouldBe(AgentRunStatus.Cancelled); + cancelled.SessionTranscriptCheckpointArtifactId.ShouldBeNull("a clean landing releases the checkpoint"); + cancelled.SessionTranscriptCheckpointAt.ShouldBeNull(); + cancelled.SessionId.ShouldBe("sess-cancelled", "the cancel releases the checkpoint, not the session id — which is what makes the guard reachable here"); + } + + (await FindResumableAsync(teamId, runId)).ShouldBeNull("a cancelled attempt left nothing a retry may resume"); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", cancelledRunId)); + + ShouldBeColdAndClaimNothing(await ExecuteRetryAsync(context, "sb"), "a deliberately cancelled attempt's machine was not lost, and it left no conversation to restore"); + } + + [Fact] + public async Task A_retry_never_resumes_a_prior_attempt_that_is_still_running() + { + // The checkpoint columns are written while an attempt is Running, so a row still Running holds the checkpoint + // of a session that may be live: a kill-wave is best-effort, and once a revived parent is Pending again the + // orphan sweep stops selecting it. Resuming it would fork a conversation that is still being written. + // MUTATION: drop the terminal-status guard from TryResumableFromCheckpoint → the retry resumes the live + // session → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var liveRunId = await SeedLiveAttemptAsync(teamId, repoId, runId, (Guid.NewGuid(), "sess-still-live")); + + (await FindResumableAsync(teamId, runId)).ShouldBeNull("a Running attempt's checkpoint belongs to a session that may still be live"); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: await FailedAttempt(teamId, "sb", liveRunId)); + + ShouldBeColdAndClaimNothing(await ExecuteRetryAsync(context, "sb"), "nothing may resume a conversation another process is still writing"); + } + + // ── 3c: the tree sentence describes the resumed attempt's OWN git state ───────── + + [Fact] + public async Task A_dependent_unit_lost_before_it_pushed_is_told_its_work_is_gone_not_that_its_producers_branch_is_its_own() + { + // A dependent unit is cloned at its producer's handoff branch. When its own attempt lost its host before it + // pushed, that branch holds the PRODUCER's work and none of this attempt's, so the tree sentence must be the + // honest "nothing was preserved" — never "your previous attempt published ``". + // MUTATION: pass the effective clone ref (effectiveStaging.Ref) as the resumed attempt's workspaceRef → the + // goal names the producer's branch as this attempt's published work → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var producer = await RunProducerAsync(teamId, repoId, runId); + var lostAttemptRunId = await SeedHostLostAttemptAsync(teamId, repoId, runId, (Guid.NewGuid(), "sess-dependent")); + + (await ReconcileUntilTerminalAsync(lostAttemptRunId)).Status.ShouldBe(AgentRunStatus.Failed); + + var task = await ExecuteRetryAsync(DependentContext(runId, teamId, repoId, producer.Spawn, await FailedAttempt(teamId, "sb", lostAttemptRunId, sequence: 3)), "sb"); + + task.Workspace!.Repositories.Single().Ref.ShouldBe(producer.Branch, "staging is unchanged: a dependent unit still clones its producer's handoff"); + task.RestoredTranscriptIsCheckpoint.ShouldBeTrue(); + task.Goal.ShouldContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive); + task.Goal.ShouldContain(AgentRetryContinuity.HonestNoContinuityHint, Case.Sensitive, "the lost attempt pushed nothing, so none of its tree survived"); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPublishedBranchHint(producer.Branch), Case.Sensitive, "the producer's branch is not this attempt's published work"); + } + + [Fact] + public async Task A_dependent_unit_resumed_from_a_captured_transcript_is_told_its_own_changes_were_not_preserved() + { + // The same rule on the ordinary path. An attempt that finished without pushing left its changes in a + // workspace nobody will see again; the clone ref is still its producer's handoff, but that is not THIS + // attempt's work, so it must not suppress the honest-redo line — which the effective ref used to do. + // MUTATION: pass effectiveStaging.Ref as workspaceRef → the producer's branch suppresses the line → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var producer = await RunProducerAsync(teamId, repoId, runId); + var (failedAttemptRunId, _) = await RunFailingPriorAttemptAsync(teamId, repoId, runId, "exit 1"); + (await ManifestsAsync(failedAttemptRunId, teamId)).ShouldBeEmpty("precondition: the failed attempt pushed nothing of its own"); + + await StampResumableSessionAsync(failedAttemptRunId, "sess-captured", "the dependent attempt's conversation\n"); + + var task = await ExecuteRetryAsync(DependentContext(runId, teamId, repoId, producer.Spawn, await FailedAttempt(teamId, "sb", failedAttemptRunId, sequence: 3)), "sb"); + + task.ResumeFromSessionId.ShouldBe("sess-captured", "the captured conversation is still resumed"); + task.Workspace!.Repositories.Single().Ref.ShouldBe(producer.Branch, "staging is unchanged: a dependent unit still clones its producer's handoff"); + task.Goal.ShouldContain(AgentRetryContinuity.HonestNoContinuityHint, Case.Sensitive, "the restored conversation describes changes this workspace does not contain"); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive, "an attempt that finished did not lose its machine"); + } + + [Fact] + public async Task A_dependent_unit_that_pushed_its_own_branch_before_its_host_died_is_told_that_branch_is_here() + { + // The other side of the same rule: when the lost attempt DID push, its own branch outranks the producer's + // handoff as the clone ref, the published work is there, and the sentence names THAT branch. + // MUTATION: drop the resumed attempt's ref (pass workspaceRef: null) → the goal says nothing was preserved + // while the clone holds this attempt's own pushed work → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var producer = await RunProducerAsync(teamId, repoId, runId); + var (lostAttemptRunId, _) = await RunPriorAttemptAsync(teamId, repoId, runId, "printf 'own work\\n' > own.txt; echo edited"); + var ownBranch = (await SingleManifestAsync(lostAttemptRunId, teamId)).Branch.ShouldNotBeNull("precondition: the attempt published its own branch before its host died"); + + await LoseTheHostAfterPublishingAsync(lostAttemptRunId, (Guid.NewGuid(), "sess-published")); + + var task = await ExecuteRetryAsync(DependentContext(runId, teamId, repoId, producer.Spawn, await FailedAttempt(teamId, "sb", lostAttemptRunId, sequence: 3)), "sb"); + + task.Workspace!.Repositories.Single().Ref.ShouldBe(ownBranch, "the attempt's own pushed branch outranks its producer's handoff"); + task.RestoredTranscriptIsCheckpoint.ShouldBeTrue(); + task.Goal.ShouldContain(AgentRetryContinuity.LostHostPublishedBranchHint(ownBranch), Case.Sensitive, "the published work IS here, and the agent is told which branch holds it"); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPublishedBranchHint(producer.Branch), Case.Sensitive); + task.Goal.ShouldNotContain(AgentRetryContinuity.HonestNoContinuityHint, Case.Sensitive, "asserting its work is gone would be a lie"); + } + + // ── 3c: the checkpoint opt-in at the staging seam ───────────────────────────── + + [Theory] + [InlineData(2, 8, true)] // room for this retry AND another after it + [InlineData(7, 8, false)] // this retry lands ON the cap — nothing can follow, so nobody would read a checkpoint + public async Task A_staged_unit_checkpoints_only_while_the_run_could_still_respawn_it(int totalSpawned, int cap, bool checkpoints) + { + // The opt-in, pinned at the STAGING SEAM rather than only as a pure predicate: this is the one place every + // spawn wave, retry and resolve passes through, so it is where the envelope either carries the flag or not. + // MUTATION: set CheckpointSessionTranscript unconditionally (or never) at that seam → one arm reds. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var context = ContextWith(runId, teamId, repoId, plan: Plan("sb"), priorAttempt: null) with { TotalSpawnedAgents = totalSpawned, MaxTotalSpawns = cap }; + + (await ExecuteRetryAsync(context, "sb")).CheckpointSessionTranscript.ShouldBe(checkpoints); + } + + [Theory] + [InlineData(5, 8, true)] // 5 + 2 = 7 < 8: a retry still fits after the whole wave + [InlineData(6, 8, false)] // 6 + 2 = 8: the WAVE lands on the cap, though one agent alone (6 + 1 = 7) would not + public async Task A_spawn_wave_checkpoints_only_when_a_retry_still_fits_after_the_whole_wave(int totalSpawned, int cap, bool checkpoints) + { + // The seam reads the WAVE's size, not one agent's: by the time any retry is decided, every agent this wave + // stages is already on the tape and counted. + // MUTATION: pass 1 instead of tasks.Count at the staging seam → the (6, 8) arm checkpoints both units → red. + if (!await GitAvailableAsync()) return; + + var teamId = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedWithOneCommitAsync(); + var repoId = await SeedRepositoryAsync(teamId, remote.Url, await SeedCredentialAsync(teamId), RepositoryPublishMode.Branch); + var runId = await SeedSupervisorRunAsync(teamId); + + var context = ContextWith(runId, teamId, repoId, plan: Plan(("sa", null), ("sb", null)), priorAttempt: null) with { TotalSpawnedAgents = totalSpawned, MaxTotalSpawns = cap }; + + var staged = await ExecuteSpawnAsync(context, "sa", "sb"); + + staged.Count.ShouldBe(2, "one wave, two units"); + staged.ShouldAllBe(t => t.CheckpointSessionTranscript == checkpoints); + } + + private static void ShouldBeColdAndClaimNothing(AgentTask task, string because) + { + task.ResumeFromSessionId.ShouldBeNull(because); + task.RestoredTranscript.ShouldBeNull(because); + task.RestoredTranscriptArtifactId.ShouldBeNull(because); + task.RestoredTranscriptIsCheckpoint.ShouldBeFalse(because); + task.ResumedFromCheckpointAt.ShouldBeNull(because); + task.ResumedFromAgentRunId.ShouldBeNull(because); + task.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPreamble, customMessage: because); + task.Goal.ShouldNotContain(AgentRetryContinuity.HonestNoContinuityHint, customMessage: because); + } + + /// + /// A prior attempt of "sb" in the state a HOST LOSS leaves it: created through the REAL + /// (so its envelope, SubtaskId and columns are the production shape), claimed by a + /// worker that is now gone — its owner id and fence epoch still on the row, its lease lapsed — with a durable + /// handle minted on a host that never came back whose own wall clock has passed: the exact shape + /// AgentRunReconcilerService.DeferToTheMintingHostAsync stops deferring and abandons. A checkpoint, when + /// given, is stamped the way the observer's drain tick leaves it — artifact, time and session id together, the + /// only way production writes a Running row's session id. + /// + /// Left un-executed rather than run and then aged, and the reason bounds what these arms prove: the stale + /// sweep deliberately EXCLUDES a run carrying events inside (a streaming + /// agent whose lease merely lapsed must never be abandoned), and agent_run_event is append-only by + /// trigger — so a run that genuinely executed cannot be backdated into this shape. Such a run also published + /// nothing; the published-branch case is 's. + /// + private Task SeedHostLostAttemptAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId, (Guid ArtifactId, string SessionId)? checkpoint) => + SeedClaimedAttemptAsync(teamId, repositoryId, supervisorRunId, checkpoint, hostLost: true); + + /// A prior attempt of "sb" still RUNNING on a live worker — claimed, its lease fresh, a checkpoint on the row. What a deliberate cancel lands on, and what a best-effort kill-wave or a revived parent can leave behind while a retry is decided. + private Task SeedLiveAttemptAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId, (Guid ArtifactId, string SessionId) checkpoint) => + SeedClaimedAttemptAsync(teamId, repositoryId, supervisorRunId, checkpoint, hostLost: false); + + private async Task SeedClaimedAttemptAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId, (Guid ArtifactId, string SessionId)? checkpoint, bool hostLost) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + // The envelope a supervisor unit really carries: SubtaskId is what FindResumableSubtaskAttemptAsync matches + // on, and CheckpointSessionTranscript is the opt-in that let its executor checkpoint at all. + var created = await scope.Resolve().CreateAsync( + new AgentTask { Goal = "do sb", Harness = "scripted", Model = "test-model", RepositoryId = repositoryId, SubtaskId = "sb", CheckpointSessionTranscript = true }, + teamId, supervisorRunId, NodeId, iterationKey: "", cancellationToken: CancellationToken.None); + + var run = await db.AgentRun.SingleAsync(r => r.Id == created.Id); + var claimedAt = DateTimeOffset.UtcNow - (hostLost ? TimeSpan.FromMinutes(20) : TimeSpan.FromMinutes(2)); + + run.Status = AgentRunStatus.Running; + run.OwnerId = Guid.NewGuid(); + run.FenceEpoch = 1; + run.StartedAt = claimedAt; + run.HeartbeatAt = claimedAt; + run.LeaseExpiresAt = hostLost ? claimedAt + AgentRunLiveness.Window : DateTimeOffset.UtcNow + AgentRunLiveness.Window; + run.RunnerHandleJson = hostLost ? JsonSerializer.Serialize(ForeignHostHandle(), AgentJson.Options) : null; + run.SessionId = checkpoint?.SessionId; + run.SessionTranscriptCheckpointArtifactId = checkpoint?.ArtifactId; + run.SessionTranscriptCheckpointAt = checkpoint is null ? null : claimedAt + TimeSpan.FromMinutes(1); + await db.SaveChangesAsync(); + + return run.Id; + } + + /// A durable handle minted on a host that never came back, whose own wall clock has already passed — the reconciler cannot probe it from here and stops deferring to it. + private static SandboxHandle ForeignHostHandle() + { + var spoolDirectory = Path.Combine(Path.GetTempPath(), "cs-supervisor-host-loss-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(spoolDirectory); + + return new SandboxHandle { Kind = "local", ProcessId = 0x7FFFFFFF, LaunchHost = "a-host-that-never-came-back", SpoolDirectory = spoolDirectory, Deadline = DateTimeOffset.UtcNow.AddMinutes(-1) }; + } + + /// + /// Turn an attempt that really ran and PUBLISHED into the row a host loss leaves when the machine dies between + /// the executor's publish and its terminal write: the manifest and the pushed branch are already real, and the + /// abandon then writes Failed with no result while keeping the checkpoint. Written directly rather than through + /// the reconciler because a run that really executed carries fresh events, which the stale sweep skips, and + /// agent_run_event is append-only — the reconciler's own abandon is what + /// drives. + /// + private async Task LoseTheHostAfterPublishingAsync(Guid agentRunId, (Guid ArtifactId, string SessionId) checkpoint) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var run = await db.AgentRun.SingleAsync(r => r.Id == agentRunId); + + run.Status = AgentRunStatus.Failed; + run.ResultJson = null; + run.SessionId = checkpoint.SessionId; + run.SessionTranscriptCheckpointArtifactId = checkpoint.ArtifactId; + run.SessionTranscriptCheckpointAt = DateTimeOffset.UtcNow.AddMinutes(-2); + await db.SaveChangesAsync(); + } + + /// + /// Sweep with the REAL reconciler until the row this test owns is terminal, and return it. The stale sweep is + /// deployment-wide and takes the oldest lapsed leases first, at + /// a time, so rows other tests left behind can fill a sweep; each sweep terminalizes what it takes, so a bounded + /// number of them reaches ours. + /// + private async Task ReconcileUntilTerminalAsync(Guid agentRunId) + { + const int maxSweeps = 10; + + for (var sweep = 0; sweep < maxSweeps; sweep++) + { + using var scope = _fixture.BeginScope(); + await scope.Resolve().ReconcileAsync(CancellationToken.None); + + var row = await scope.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == agentRunId); + if (AgentRunStateMachine.IsTerminal(row.Status)) return row; + } + + throw new ShouldAssertException($"agent run {agentRunId} was still not terminal after {maxSweeps} reconciler sweeps. Check that its lease has lapsed, that it has no event inside AgentRunLiveness.Window, and that its handle's deadline has passed — the reconciler logs 'leaving agent run ... alone' when it defers."); + } + + private async Task CancelAsync(Guid agentRunId) + { + using var scope = _fixture.BeginScope(); + return await scope.Resolve().CancelRunningAsync(agentRunId, "Cancelled by an operator.", AgentRunAbandonCause.OperatorCancelled, CancellationToken.None); + } + + private async Task FindResumableAsync(Guid teamId, Guid supervisorRunId) + { + using var scope = _fixture.BeginScope(); + return await scope.Resolve().FindResumableSubtaskAttemptAsync(teamId, supervisorRunId, "sb", CancellationToken.None); + } + + /// Run the plan's producer "pa" for real and record it as a Succeeded spawn — the handoff a dependent "sb" is cloned at. Returns the producer's pushed branch and that spawn decision. + private async Task<(string Branch, SupervisorPriorDecision Spawn)> RunProducerAsync(Guid teamId, Guid repositoryId, Guid supervisorRunId) + { + var (producerRunId, _) = await RunPriorAttemptAsync(teamId, repositoryId, supervisorRunId, "printf 'producer work\\n' > producer.txt; echo edited", subtaskId: "pa"); + var branch = (await SingleManifestAsync(producerRunId, teamId)).Branch.ShouldNotBeNull("precondition: the producer really pushed, so a dependent is cloned at its branch"); + + return (branch, await SucceededSpawn(teamId, "pa", producerRunId)); + } + // ─── Drive the real executor ────────────────────────────────────────────────── private async Task ExecuteRetryAsync(SupervisorTurnContext context, string subtaskId) @@ -213,6 +629,20 @@ private async Task ExecuteRetryAsync(SupervisorTurnContext context, s return JsonSerializer.Deserialize(run.TaskJson, AgentJson.Options)!; } + /// One spawn wave through the real executor, returning the envelope of every unit it staged. + private async Task> ExecuteSpawnAsync(SupervisorTurnContext context, params string[] subtaskIds) + { + using var scope = _fixture.BeginScope(); + + var payload = JsonSerializer.Serialize(new SupervisorSpawnPayload { SubtaskIds = subtaskIds }, AgentJson.Options); + + await scope.Resolve().ExecuteAsync(new SupervisorDecision { Kind = SupervisorDecisionKinds.Spawn, PayloadJson = payload }, context, CancellationToken.None); + + var runs = await scope.Resolve().AgentRun.AsNoTracking().Where(r => r.WorkflowRunId == context.SupervisorRunId && r.NodeId == NodeId).ToListAsync(); + + return runs.Select(r => JsonSerializer.Deserialize(r.TaskJson, AgentJson.Options)!).ToList(); + } + // ─── Context / decision-tape builders ───────────────────────────────────────── private static SupervisorTurnContext ContextWith(Guid runId, Guid teamId, Guid repositoryId, SupervisorPriorDecision plan, SupervisorPriorDecision? priorAttempt) => new() @@ -226,29 +656,51 @@ private async Task ExecuteRetryAsync(SupervisorTurnContext context, s AgentProfile = new CodeSpace.Messages.Dtos.Agents.SupervisorAgentProfile { RepositoryId = repositoryId }, }; - private static SupervisorPriorDecision Plan(string subtaskId) + /// The dependent shape: "sb" depends on the producer "pa", whose spawn and "sb"'s own recorded attempt follow the plan on the tape. + private static SupervisorTurnContext DependentContext(Guid runId, Guid teamId, Guid repositoryId, SupervisorPriorDecision producerSpawn, SupervisorPriorDecision dependentAttempt) + { + var plan = Plan(("pa", null), ("sb", new[] { "pa" })); + + return ContextWith(runId, teamId, repositoryId, plan, priorAttempt: null) with { PriorDecisions = new[] { plan, producerSpawn, dependentAttempt } }; + } + + private static SupervisorPriorDecision Plan(string subtaskId) => Plan((subtaskId, null)); + + private static SupervisorPriorDecision Plan(params (string Id, string[]? DependsOn)[] subtasks) { var payload = JsonSerializer.Serialize(new SupervisorPlanPayload { Goal = Goal, - Subtasks = new List { new() { Id = subtaskId, Title = subtaskId, Instruction = $"do {subtaskId}" } }, + Subtasks = subtasks.Select(s => new SupervisorPlannedSubtask { Id = s.Id, Title = s.Id, Instruction = $"do {s.Id}", DependsOn = s.DependsOn }).ToList(), }, AgentJson.Options); return new SupervisorPriorDecision { Id = Guid.NewGuid(), Sequence = 1, DecisionKind = SupervisorDecisionKinds.Plan, Status = SupervisorDecisionStatus.Succeeded, PayloadJson = payload, OutcomeJson = "{}" }; } /// A prior FAILED spawn recording (subtaskId, REAL agentRunId) — the positional subtaskIds[i] ↔ agentResults[i] shape reads to find "this subtask's latest attempt", UNFILTERED on success (the whole point of a retry). - private async Task FailedAttempt(Guid teamId, string subtaskId, Guid agentRunId) + private Task FailedAttempt(Guid teamId, string subtaskId, Guid agentRunId, long sequence = 2) => + RecordedSpawn(teamId, subtaskId, agentRunId, ("Failed", "acceptance failed"), sequence); + + /// A prior SUCCEEDED spawn of a producer — what dependency staging reads to hand its branch to a dependent. + private Task SucceededSpawn(Guid teamId, string subtaskId, Guid agentRunId) => + RecordedSpawn(teamId, subtaskId, agentRunId, ("Succeeded", null), sequence: 2); + + private async Task RecordedSpawn(Guid teamId, string subtaskId, Guid agentRunId, (string Status, string? Error) outcome, long sequence) { - using var scope = _fixture.BeginScope(); - var manifests = await scope.Resolve().ListForAgentRunAsync(agentRunId, teamId, CancellationToken.None); + var manifests = await ManifestsAsync(agentRunId, teamId); - var result = new SupervisorAgentResult { AgentRunId = agentRunId, Status = "Failed", Error = "acceptance failed", ProducedBranch = manifests.FirstOrDefault()?.Branch }; + var result = new SupervisorAgentResult { AgentRunId = agentRunId, Status = outcome.Status, Error = outcome.Error, ProducedBranch = manifests.FirstOrDefault()?.Branch }; var payload = JsonSerializer.Serialize(new SupervisorSpawnPayload { SubtaskIds = new[] { subtaskId } }, AgentJson.Options); - var outcome = JsonSerializer.Serialize(new { agentRunIds = new[] { agentRunId }, agentCount = 1, agentResults = new[] { result } }, AgentJson.Options); + var outcomeJson = JsonSerializer.Serialize(new { agentRunIds = new[] { agentRunId }, agentCount = 1, agentResults = new[] { result } }, AgentJson.Options); + + return new SupervisorPriorDecision { Id = Guid.NewGuid(), Sequence = sequence, DecisionKind = SupervisorDecisionKinds.Spawn, Status = SupervisorDecisionStatus.Succeeded, PayloadJson = payload, OutcomeJson = outcomeJson }; + } - return new SupervisorPriorDecision { Id = Guid.NewGuid(), Sequence = 2, DecisionKind = SupervisorDecisionKinds.Spawn, Status = SupervisorDecisionStatus.Succeeded, PayloadJson = payload, OutcomeJson = outcome }; + private async Task> ManifestsAsync(Guid agentRunId, Guid teamId) + { + using var scope = _fixture.BeginScope(); + return await scope.Resolve().ListForAgentRunAsync(agentRunId, teamId, CancellationToken.None); } /// A prior FAILED retry recording (subtaskId, REAL agentRunId), SEQUENCED AFTER — the decision-tape-literal "latest attempt" reads, independent of whether it is actually resumable. diff --git a/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs index 3d2bf0e02..b8393fefe 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/AgentRetryCausesTests.cs @@ -50,7 +50,7 @@ public void A_format_fault_retry_goes_fresh_with_thinking_disabled() var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); var result = Result("API Error: Content block is not a thinking block"); - var task = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, result, workspaceHasPriorWork: true); + var task = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, result, workspaceRef: "agent/prior"); task.ResumeFromSessionId.ShouldBeNull("a conversation replay re-triggers the format fault deterministically — the retry must NOT resume"); task.RestoredTranscript.ShouldBeNull(); @@ -63,11 +63,11 @@ public void An_ordinary_failure_keeps_todays_resume_semantics_byte_identically() { var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); - var resumed = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, Result("acceptance: ./check.sh exited 2"), workspaceHasPriorWork: true); + var resumed = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior, Result("acceptance: ./check.sh exited 2"), workspaceRef: "agent/prior"); resumed.ResumeFromSessionId.ShouldBe("sess-1"); resumed.Environment.ContainsKey(AgentRetryCauses.MaxThinkingTokensEnvVar).ShouldBeFalse("no degrade on an ordinary failure"); - var cold = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior: null, Result("boom"), workspaceHasPriorWork: false); + var cold = RealSupervisorActionExecutor.ApplyRetryDisposition(Task_(), prior: null, Result("boom"), workspaceRef: null); cold.ResumeFromSessionId.ShouldBeNull(); cold.Environment.ContainsKey(AgentRetryCauses.MaxThinkingTokensEnvVar).ShouldBeFalse(); } @@ -125,7 +125,7 @@ public async Task Every_retry_lane_repairs_a_format_fault_through_the_same_helpe const string liveError = "API Error: Content block is not a thinking block"; var supervisorTask = RealSupervisorActionExecutor.ApplyRetryDisposition( - Task_(), new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null), Result(liveError), workspaceHasPriorWork: true); + Task_(), new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null), Result(liveError), workspaceRef: "agent/prior"); var priorAttempt = JsonDocument.Parse($$""" {"status":"Failed","exitReason":"non-zero-exit","error":"{{liveError}}","sessionId":"sess-1","sessionTranscript":"transcript"} diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs index dd403b722..164cd0e94 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBoundsTests.cs @@ -274,6 +274,50 @@ public void The_terminal_reasons_are_pinned() SupervisorStopReasons.PlanInvalid.ShouldBe("plan structurally invalid"); } + // ─── 3c: may a unit this wave stages ever be respawned? ───────────────────────── + + [Theory] + [InlineData(0, 3, 10, true)] // plenty of room after the wave + [InlineData(6, 3, 10, true)] // 9 of 10 spent — one retry still fits + [InlineData(7, 3, 10, false)] // the wave lands exactly ON the cap: no retry can follow + [InlineData(8, 3, 10, false)] // the wave itself would breach it; PostDecision refuses the wave, and nothing follows either + [InlineData(0, 1, 1, false)] // a one-spawn run: its only agent can never be retried + public void A_unit_is_checkpointed_only_while_the_run_could_still_respawn_it(int alreadySpawned, int waveSize, int cap, bool expected) + { + // The 3c opt-in. A checkpoint costs a whole-file read and an artifact write a minute for as long as the unit + // runs, so it is paid for only where somebody could consume it — and the only bound that says "no further + // agent can EVER be created" is the total-spawn cap. A later retry costs exactly one spawn, and by then this + // wave's own agents are on the tape, so the room has to exist now: strictly less than the cap, not at it. + // MUTATION: use <= instead of < → the exactly-on-the-cap arm reds, and every unit of a spent run would pay + // for a checkpoint nobody can consume. + var context = Context(turn: 1, totalSpawned: alreadySpawned) with { MaxTotalSpawns = cap }; + + SupervisorBounds.CanRespawnAfterWave(context, waveSize).ShouldBe(expected); + } + + [Theory] + [InlineData(4)] // an operator-set cap, which rehydrate copies onto the context from the plan + [InlineData(null)] // a context carrying no cap (legacy): the predicate's fallback must be the plan's own default + public void The_respawn_headroom_reads_the_same_cap_the_spawn_bound_enforces(int? configuredCap) + { + // The two must never disagree about what the run can still do: if PostDecision would REFUSE a further retry, + // the unit must not have been checkpointed for one. Driven through both entry points on one context. + // MUTATION: `<=` for `<` → the explicit-cap arm reds; a fallback other than + // SupervisorLane.DefaultMaxTotalSpawns (or none) → the null arm reds, because the plan still enforces it. + var plan = SupervisorGoalPlan.From(new SupervisorGoalConfig { Goal = "g", MaxTotalSpawns = configuredCap }); + var cap = plan.MaxTotalSpawns; + var atCap = Context(turn: 1, totalSpawned: cap - 1) with { MaxTotalSpawns = configuredCap }; + + SupervisorBounds.CanRespawnAfterWave(atCap, waveSize: 1).ShouldBeFalse("the wave lands on the cap, so nothing can follow it"); + SupervisorBounds.PostDecision(atCap with { TotalSpawnedAgents = cap }, plan, Spawn("a")) + .ShouldBe(SupervisorStopReasons.TotalSpawnCapReached, "and the bound agrees: the retry that would have consumed the checkpoint is refused"); + + var withRoom = Context(turn: 1, totalSpawned: cap - 2) with { MaxTotalSpawns = configuredCap }; + + SupervisorBounds.CanRespawnAfterWave(withRoom, waveSize: 1).ShouldBeTrue(); + SupervisorBounds.PostDecision(withRoom with { TotalSpawnedAgents = cap - 1 }, plan, Spawn("a")).ShouldBeNull("the retry really is affordable — without this the predicate could be false-negative everywhere and still pass"); + } + // ─── Helpers ──────────────────────────────────────────────────────────────────── private static SupervisorTurnContext Context(int turn, int totalSpawned = 0, int noProgress = 0, decimal runSpend = 0m) => diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs index d7b3b5174..6ff3bfa52 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyGateTests.cs @@ -227,6 +227,25 @@ public void A_plan_less_legacy_tape_keeps_its_result_fold() SupervisorDependencyGate.LatestAgentRunId(Context(spawn), "legacy").ShouldBe(legacyRunId); } + // ── LatestResultsBySubtask: the roster SupervisorAmendPrecondition.GradedAttempts grades from ── + + [Fact] + public void A_warm_respawn_of_a_lost_host_replaces_the_abandoned_attempt_rather_than_adding_a_second() + { + // The host-loss slice's billing question. A unit whose host died is abandoned with NO result at all and + // then respawned warm from its checkpoint — so the tape carries two entries for one subtask. The graded + // roster must keep exactly ONE, the respawn's, or a single host loss spends two of the run's graded + // attempts and the amend precondition grades a verdict the machine, not the work, produced. + // MUTATION: fold the tape in reverse (or keep the first write per subtask) → the abandoned attempt wins → red. + var tape = new[] { Spawn(("a", "Failed", null)), Retry(("a", "Succeeded", true)) }; + var respawnRunId = SupervisorOutcome.ReadAgentResults(tape[1].OutcomeJson).Single().AgentRunId; + + var graded = SupervisorDependencyGate.LatestResultsBySubtask(tape); + + graded.Count.ShouldBe(1, "one subtask owes one graded attempt no matter how many machines it burned through"); + graded["a"].AgentRunId.ShouldBe(respawnRunId, "the respawn's verdict is the subtask's verdict — the abandoned attempt was never graded by anything but the reconciler"); + } + // ── Frontier: the decider's guidance ─────────────────────────────────────────────── [Fact] diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs index 8da991a09..518f78b7d 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorDependencyStagingTests.cs @@ -417,7 +417,7 @@ public void A_workspace_pinned_to_prior_work_carries_no_honest_redo_hint() var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli" }; var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); - var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceHasPriorWork: true); + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef: "agent/prior"); resumed.ResumeFromSessionId.ShouldBe("sess-1"); resumed.RestoredTranscript.ShouldBe("transcript"); @@ -430,12 +430,69 @@ public void A_workspace_with_no_preserved_git_state_gets_the_honest_redo_hint() var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli" }; var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); - var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceHasPriorWork: false); + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef: null); resumed.ResumeFromSessionId.ShouldBe("sess-1", "the conversation is still restored"); resumed.Goal.ShouldBe($"do the thing\n\n{AgentRetryContinuity.HonestNoContinuityHint}", "the goal now HONESTLY says the git changes are NOT present, so the agent never trusts a restored conversation implying work it can't see"); } + // ── 3c: a unit whose HOST died resumes from its checkpoint, and owes a different sentence ── + + [Theory] + [InlineData("agent/prior", true, "published")] // the lost attempt pushed its own branch — that work IS in the fresh clone + [InlineData(null, true, "redo")] // it pushed nothing — none of its tree survived + [InlineData(null, false, "none")] // no repository at all — there was never a tree to lose + public void A_unit_resumed_from_a_host_loss_checkpoint_carries_its_provenance_and_the_lost_host_block(string? workspaceRef, bool hasRepository, string treeSentence) + { + // A checkpoint is not a captured transcript. The attempt that wrote it never finished: its machine is gone, + // so the conversation may describe turns the checkpoint never saw and edits the new sandbox does not + // contain — which is true even when a branch WAS pushed, because the unpublished remainder died with the + // host. The ordinary honest-redo line only covers "no branch to continue from", a smaller claim. The whole + // goal is asserted, so each arm pins exactly WHICH tree sentence follows the preamble. + // MUTATION: drop the CheckpointAt branch from ApplyResumeRecord → the task carries no provenance, is not + // marked a checkpoint (so an unreadable ref would FAIL the attempt instead of degrading), and the goal says + // only what an ordinary retry says → red. + // MUTATION: pass no branch to WithLostHostHint → the "published" arm reds; owe a tree regardless of the + // repository → the "none" arm reds. + var priorRunId = Guid.NewGuid(); + var checkpointAt = DateTimeOffset.UtcNow.AddMinutes(-2); + var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli", RepositoryId = hasRepository ? Guid.NewGuid() : null }; + var prior = new ResumableSession(priorRunId, "sess-lost", null, Guid.NewGuid(), checkpointAt); + + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef); + + resumed.ResumeFromSessionId.ShouldBe("sess-lost", "the CLI is told WHICH conversation to resume — a transcript with no id names nothing"); + resumed.RestoredTranscriptArtifactId.ShouldBe(prior.TranscriptArtifactId, "the checkpoint rides as a REF the executor resolves just before invocation"); + resumed.RestoredTranscriptIsCheckpoint.ShouldBeTrue("this ref is best-effort: unreadable must cost the conversation, never the attempt"); + resumed.ResumedFromCheckpointAt.ShouldBe(checkpointAt, "the launch stamps this onto the run's permanent confinement record"); + resumed.ResumedFromAgentRunId.ShouldBe(priorRunId, "which attempt took over from which is a column, not prose"); + resumed.Goal.ShouldBe($"do the thing\n\n{AgentRetryContinuity.LostHostPreamble}{ExpectedTreeSentence(treeSentence, workspaceRef)}", "the preamble always, then exactly the one sentence that is true about the tree"); + } + + private static string ExpectedTreeSentence(string treeSentence, string? workspaceRef) => treeSentence switch + { + "published" => " " + AgentRetryContinuity.LostHostPublishedBranchHint(workspaceRef!), + "redo" => " " + AgentRetryContinuity.HonestNoContinuityHint, + _ => "", + }; + + [Fact] + public void A_unit_resumed_from_a_completed_attempt_claims_no_checkpoint() + { + // The other side of the same fork: an attempt that FINISHED left its workspace behind, so its transcript is + // a capture — fail-closed if unreadable — and it owes no lost-host sentence. + // MUTATION: mark every resume a checkpoint → red, and an unreadable captured ref would silently cold-start. + var task = new AgentTask { Goal = "do the thing", Harness = "codex-cli" }; + var prior = new ResumableSession(Guid.NewGuid(), "sess-1", "transcript", null); + + var resumed = RealSupervisorActionExecutor.ApplyResumeRecord(task, prior, workspaceRef: "agent/prior"); + + resumed.RestoredTranscriptIsCheckpoint.ShouldBeFalse(); + resumed.ResumedFromCheckpointAt.ShouldBeNull(); + resumed.ResumedFromAgentRunId.ShouldBeNull(); + resumed.Goal.ShouldNotContain(AgentRetryContinuity.LostHostPreamble, Case.Sensitive); + } + // ── BuildBlockedSpawnOutcome: the wire shape resolve's conflict reader consumes ───── [Fact] diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs index 54c42af2b..3f937db6e 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs @@ -1207,7 +1207,7 @@ public async Task A_repo_less_respawn_is_never_told_its_git_changes_were_lost(st [Fact] public async Task A_respawn_whose_primary_push_failed_repins_the_sibling_but_is_still_told_the_truth() { - // The honesty decision follows the PRIMARY, exactly like the supervisor's workspaceHasPriorWork: a sibling's + // The honesty decision follows the PRIMARY, exactly like the supervisor's workspaceRef: a sibling's // successful push is still conserved (its own branch is repinned), but the agent's primary repo re-clones the // default branch, so the restored conversation MUST be told its changes are not there. An "any repo pushed" // read would suppress the line precisely where the primary lost its work — the worst place to go quiet.