diff --git a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0234_agent_run_session_transcript_checkpoint.sql b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0234_agent_run_session_transcript_checkpoint.sql new file mode 100644 index 000000000..532d09a7e --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0234_agent_run_session_transcript_checkpoint.sql @@ -0,0 +1,67 @@ +-- 0234_agent_run_session_transcript_checkpoint.sql +-- +-- A run whose launch host dies is NOT continued by another host, because nothing it produced mid-run is durable. +-- +-- The resumable CLI session transcript — the file the harness's own `--resume` reads — is captured exactly once, at +-- run END, from the LAUNCHING host's spool (AgentRunExecutor.CaptureSessionTranscriptAsync reads +-- LocalProcessRunner.ConfigHomePath(handle.SpoolDirectory)). When the host holding that spool is gone, the file is +-- gone with it: AgentRunReconcilerService abandons the run (a foreign handle is left alone until its own deadline, +-- then failed) and every reader is left with a cold restart as the only recovery. +-- +-- The retry itself already happens: the reconciler abandons the run, the agent.run node grades that as a retryable +-- failure, and the node's retry policy stages a fresh agent. It is the CONVERSATION that is lost, so the retry is +-- cold. These columns are what make it warm. The observer's own drain tick uploads the live transcript to the +-- artifact store and stamps the reference here; the reconciler's abandon leaves it on the row (where the executor's +-- own terminal write clears it), the completion notifier projects it across the wait boundary, and the node's +-- respawn stamps it onto the fresh attempt's task. +-- +-- Shape, and why each part of it: +-- +-- * NULLABLE with NO DEFAULT. agent_run is a hot table; a nullable ADD COLUMN with no default is catalog-only in +-- PostgreSQL (no table rewrite, no long ACCESS EXCLUSIVE hold), which a DEFAULT would not be on every supported +-- server. NULL is also the escape for the rollback direction: an older binary that has never heard of these +-- columns simply does not select them, and a row they were never written to reads as "no checkpoint" — which is +-- precisely the pre-3c behaviour, a cold start. +-- +-- * The artifact id is a SOFT link, like every other *_artifact_id in this schema — no FK to workflow_artifact. +-- It is registered in ArtifactReferenceOracle.ReferenceSites (test-pinned, and cross-checked against the EF model +-- by ArtifactReferenceOracleTests) so the reaper can never collect a checkpoint a live run still names. +-- +-- * A PARTIAL INDEX on that soft link, for the reason 0144 gives for the ten it added with the oracle: the oracle +-- asks "does ANY row anywhere still reference this artifact id" as one EXISTS per site, and an unindexed site on +-- a run-scale table makes that EXISTS a sequential scan of agent_run on every artifact the reaper claims. Partial +-- on IS NOT NULL because a checkpoint is the exception, so the index stays small on the hot table. It also builds +-- instantly here: the column is NULL on every existing row, so there is nothing to insert into it. +-- +-- LOCKS, stated as 0144 states them: DbUp runs the whole upgrade in ONE transaction, so the CREATE INDEX below holds +-- its SHARE lock on agent_run (blocking writes, not reads) until that transaction commits. CREATE INDEX CONCURRENTLY +-- cannot run inside a transaction, so shortening that window is a change to DbUpRunner, not to this file. +-- +-- * A THIRD column, resumed_from_agent_run_id, so a continuation NAMES the run it took over from in a column +-- rather than only in the sentence an event carries. Same shape as every other agent_run cross-reference: a soft +-- link with no FK (agent runs are managed independently of one another), nullable, no default, no index — nothing +-- queries by it yet, and adding an index for a reader that does not exist would cost every agent_run write. It is +-- NOT an artifact id, so the reference oracle correctly does not probe it. +-- +-- Rollback: ALTER TABLE agent_run DROP COLUMN resumed_from_agent_run_id, DROP COLUMN +-- session_transcript_checkpoint_at, DROP COLUMN session_transcript_checkpoint_artifact_id (the index goes with the +-- column); the stored transcripts then age out under their own retention declaration +-- (ArtifactRetentionClass.SessionTranscriptCheckpoint) once nothing references them. + +ALTER TABLE agent_run + ADD COLUMN session_transcript_checkpoint_artifact_id uuid NULL, + ADD COLUMN session_transcript_checkpoint_at timestamptz NULL, + ADD COLUMN resumed_from_agent_run_id uuid NULL; + +CREATE INDEX ix_agent_run_session_transcript_checkpoint_artifact + ON agent_run (session_transcript_checkpoint_artifact_id) + WHERE session_transcript_checkpoint_artifact_id IS NOT NULL; + +COMMENT ON COLUMN agent_run.session_transcript_checkpoint_artifact_id IS + 'Soft link to workflow_artifact.id holding this run''s resumable session transcript as of its most recent mid-run checkpoint — what another host continues from when this run''s host dies. NULL until the first checkpoint.'; + +COMMENT ON COLUMN agent_run.session_transcript_checkpoint_at IS + 'When session_transcript_checkpoint_artifact_id was taken: the cadence clock for the next checkpoint, and the age that says how much of the conversation survived the lost host.'; + +COMMENT ON COLUMN agent_run.resumed_from_agent_run_id IS + 'The abandoned agent run this one succeeded after a host loss, and was staged to continue from its session checkpoint. Whether that checkpoint could be read is only known at launch, so this records succession, not recovery. Soft link (no FK, like every other agent_run cross-reference); NULL for every ordinary run.'; diff --git a/backend/src/CodeSpace.Core/Persistence/Entities/AgentRun.cs b/backend/src/CodeSpace.Core/Persistence/Entities/AgentRun.cs index bb310a0bb..328c7ddf3 100644 --- a/backend/src/CodeSpace.Core/Persistence/Entities/AgentRun.cs +++ b/backend/src/CodeSpace.Core/Persistence/Entities/AgentRun.cs @@ -67,11 +67,40 @@ public class AgentRun : IEntity, IAuditable /// /// P3.1a: the harness-native session/thread id captured off the run's CLI conversation (Claude's /// session_id, Codex's thread_id) — promoted from result_jsonb to a first-class column so a - /// rerun's CONTINUE lookup is a column read, not a JSON probe. NULL while in-flight, and for a run whose stream - /// carried no session id (a pre-session CLI). Set on completion from AgentRunResult.SessionId. + /// rerun's CONTINUE lookup is a column read, not a JSON probe. NULL for a run whose stream carried no session id + /// (a pre-session CLI). Set on completion from AgentRunResult.SessionId, and — since 3c — already at the + /// run's first session-transcript checkpoint, because a checkpoint nothing can ADDRESS is not resumable: the + /// continuation needs this id to hand the CLI its --resume. Every reader that treats a non-null id as + /// "resumable" also requires a transcript out of (TryResumable, both-or-neither), + /// so an in-flight row carrying one is skipped exactly as a null one was. /// public string? SessionId { get; set; } + /// + /// 3c: the artifact holding this run's resumable session transcript as of its most recent MID-RUN checkpoint — + /// the only thing a run leaves behind that another host can continue from after the launching host dies. NULL + /// until the first checkpoint, and for every run whose harness has no addressable session transcript. A soft link + /// to workflow_artifact.id, and it is exactly that: the reference the artifact reaper's oracle probes + /// (ArtifactReferenceOracle.ReferenceSites) so a live checkpoint is never collected. + /// + public Guid? SessionTranscriptCheckpointArtifactId { get; set; } + + /// When was taken. The cadence clock for the next checkpoint, and the age an operator (or a resumed attempt) reads to know how much of the conversation survived the lost host. + public DateTimeOffset? SessionTranscriptCheckpointAt { get; set; } + + /// + /// 3c: the abandoned run this one was STAGED to continue from a session checkpoint — the structured link, so + /// "which attempt took over from which" is a column rather than a sentence buried in an event. Soft link (no + /// FK), like every other agent-run cross-reference. NULL for every ordinary run. + /// + /// Staged, not necessarily restored: this is written when the row is created, and whether the checkpoint + /// could actually be READ is only knowable at launch (a reaped blob, an unreachable destination — then the + /// attempt runs cold and says so). The honest reading of a non-null value is therefore "this attempt succeeded + /// that one after it lost its host", which is true either way; what was recovered is the confinement record's + /// ResumedFromCheckpointAt, which the degrade clears. + /// + public Guid? ResumedFromAgentRunId { get; set; } + /// Worker liveness ping; a stuck-Running reconciler reads this to recover crashed runs. public DateTimeOffset? HeartbeatAt { get; set; } diff --git a/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/AgentRunConfiguration.cs b/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/AgentRunConfiguration.cs index 5b6d0fd7b..67e398040 100644 --- a/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/AgentRunConfiguration.cs +++ b/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/AgentRunConfiguration.cs @@ -25,6 +25,13 @@ public void Configure(EntityTypeBuilder builder) // is a column read. Nullable (a pre-session CLI / in-flight run has none). builder.Property(r => r.SessionId).HasColumnName("session_id"); + // 3c (migration 0234) — the mid-run session-transcript checkpoint. Nullable with no default: the column is + // metadata-only on this hot table, and NULL is the escape an older binary and every pre-3c run reads as + // "no checkpoint, cold start". + builder.Property(r => r.SessionTranscriptCheckpointArtifactId).HasColumnName("session_transcript_checkpoint_artifact_id"); + builder.Property(r => r.SessionTranscriptCheckpointAt).HasColumnName("session_transcript_checkpoint_at"); + builder.Property(r => r.ResumedFromAgentRunId).HasColumnName("resumed_from_agent_run_id"); + builder.Property(r => r.RunnerHandleJson).HasColumnName("runner_handle").HasColumnType("jsonb"); builder.Property(r => r.SpoolCleanupAttempts).HasColumnName("spool_cleanup_attempts"); builder.Property(r => r.SpoolCleanupLastAttemptAt).HasColumnName("spool_cleanup_last_attempt_at"); diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRetryContinuity.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRetryContinuity.cs index d7820812c..255c1cc01 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRetryContinuity.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRetryContinuity.cs @@ -15,4 +15,50 @@ public static class AgentRetryContinuity /// Append to a resumed task's goal. One composition, so the two lanes cannot drift on the separator either. public static string WithHonestNoContinuityHint(string goal) => $"{goal}\n\n{HonestNoContinuityHint}"; + + /// + /// 3c: what a CROSS-HOST continuation is told, and the reason it needs its own sentence. The two lanes above + /// retry an attempt that FINISHED on a live host, so their only open question is whether a branch was pushed. + /// This lane continues an attempt whose machine was lost mid-run: the conversation comes from a checkpoint taken + /// some time before the loss, and the working tree is simply gone. Both halves have to be said, because the + /// restored transcript will describe edits — possibly edits made after the checkpoint — that the new workspace + /// does not contain, and an agent that is not told will read its own transcript as evidence about files it cannot + /// see. + /// + public const string LostHostPreamble = "Note: the machine running your previous attempt was lost mid-run. Your conversation is restored from a checkpoint taken before that, so it may describe work you did after the checkpoint, and it may be missing your last few turns."; + + /// Said when the lost attempt HAD published a branch: the workspace is checked out at it, so the published work is present and only the unpublished remainder is gone. Takes the branch name so the agent can verify rather than take the claim on trust. + public static string LostHostPublishedBranchHint(string branch) => + $"Your previous attempt published branch `{branch}`, and this workspace is checked out AT that branch — that work is here. Anything you had NOT published to it died with the machine, so check the files before continuing and redo whatever is missing."; + + /// + /// Append the cross-host continuation's honesty block to a resumed task's goal: the preamble always, then what + /// is true about the TREE. + /// + /// Three cases, and the third is why this takes a flag rather than inferring from the branch alone. A + /// published branch means the pushed work IS here and only the unpublished remainder died. No branch on a + /// repository-backed run means nothing was preserved, which is exactly — + /// the sentence the other two lanes already say, reused verbatim so "your prior work is not here" can never mean + /// two different things. And a run with NO repository ( false) had no working tree to + /// lose, so it gets the preamble alone: telling an analysis-only agent to redo file changes would be a claim + /// about a git fact its run never had. + /// + public static string WithLostHostHint(string goal, string? publishedBranch, bool treeOwed) => + $"{goal}\n\n{LostHostPreamble}{TreeStateSentence(publishedBranch, treeOwed)}"; + + /// + /// 3c: said when the lost host's checkpoint could not be READ — reaped, or its storage unreachable. The attempt + /// still runs (failing it would spend the retry this whole path exists to improve), but it runs COLD, and an + /// agent that was going to be handed a conversation must be told it is not getting one. Appended to the + /// lost-host block rather than replacing it: the machine really was lost, which is still the reason. + /// + public const string LostHostCheckpointUnreadableHint = "Your previous conversation could not be recovered either — the checkpoint it was stored in is no longer readable — so you are starting this task from the beginning."; + + /// Replace a resumed goal's promise of a restored conversation with the truth that there is none. Used when the checkpoint ref resolves to nothing at launch, which is the only moment that fact is knowable. + public static string WithUnreadableCheckpointHint(string goal) => $"{goal}\n\n{LostHostCheckpointUnreadableHint}"; + + private static string TreeStateSentence(string? publishedBranch, bool treeOwed) => + !treeOwed ? "" + : string.IsNullOrWhiteSpace(publishedBranch) ? $" {HonestNoContinuityHint}" + : $" {LostHostPublishedBranchHint(publishedBranch)}"; } diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs index 84aad13eb..4660571e3 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs @@ -186,9 +186,24 @@ public sealed class AgentRunExecutor : IAgentRunExecutor, IScopedDependency // one of those is a live host, so a tear-down arm reading them would fail and kill a healthy agent. Optional like // the rest — a null one simply never takes the shutdown branch, which is the safe direction. private readonly Microsoft.Extensions.Hosting.IHostApplicationLifetime? _lifetime; + // The clock the 3c checkpoint cadence is measured on. Injected (TimeProvider is a registered singleton) so a test + // can advance the window instead of sleeping through it; defaulted so a hand-built executor needs to know nothing. + private readonly TimeProvider _clock; + // 3c: makes the run's resumable session transcript durable MID-run, so a host that dies leaves a conversation + // another host can continue. Optional for the same reason _logCapture and _nativeRecords are — a hand-built test + // double must not have to know about it — and a null one simply leaves the run resumable only from its own host, + // which is exactly the behaviour that existed before this seam. + private readonly Recovery.IAgentSessionTranscriptCheckpointer? _sessionCheckpointer; + // The one checkpoint allowed to be in flight, and the two watermarks the stateless checkpointer cannot hold. All + // three are read and written ONLY from the drain tick, which is single-threaded by construction (the durable tail + // loop awaits each onLine then onCheckpoint sequentially) — the background task itself touches none of them, so a + // tick that finds one incomplete simply skips rather than racing it. + private Task _sessionCheckpoint = Task.FromResult(null); + private DateTimeOffset? _sessionCheckpointAttemptedAt; + private readonly SessionCheckpointWatermark _sessionCheckpointWatermark = new(); private readonly ILogger _logger; - public AgentRunExecutor(IAgentRunService runs, IAgentHarnessRegistry harnesses, IHarnessModelReconciler harnessReconciler, ISandboxRunnerRegistry runners, IAgentWorkspaceResolver workspaceResolver, IModelCredentialResolver modelCredentials, IWorkspaceProviderRegistry workspaces, IAgentRunCompletionNotifier notifier, IServiceScopeFactory scopeFactory, CodeSpaceDbContext db, IStructuredCritic critic, IArtifactOffloader offloader, Workflows.Artifacts.IArtifactStore artifacts, IPublishManifestStore manifests, IArtifactManifestStore artifactManifests, Capture.ICaptureIntentService captureIntents, IEnumerable publishGuards, ILogger logger, IAgentRunLogCaptureBridge? logCapture = null, INativeRecordPlane? nativeRecords = null, AgentDefaultRunnerSetting? defaultRunner = null, Services.RunData.IRunDataCompletenessWriter? completeness = null, Credentials.IModelCredentialBroker? credentialBroker = null, AgentRunLogging.IAgentRunLogService? logs = null, Microsoft.Extensions.Hosting.IHostApplicationLifetime? lifetime = null) + public AgentRunExecutor(IAgentRunService runs, IAgentHarnessRegistry harnesses, IHarnessModelReconciler harnessReconciler, ISandboxRunnerRegistry runners, IAgentWorkspaceResolver workspaceResolver, IModelCredentialResolver modelCredentials, IWorkspaceProviderRegistry workspaces, IAgentRunCompletionNotifier notifier, IServiceScopeFactory scopeFactory, CodeSpaceDbContext db, IStructuredCritic critic, IArtifactOffloader offloader, Workflows.Artifacts.IArtifactStore artifacts, IPublishManifestStore manifests, IArtifactManifestStore artifactManifests, Capture.ICaptureIntentService captureIntents, IEnumerable publishGuards, ILogger logger, IAgentRunLogCaptureBridge? logCapture = null, INativeRecordPlane? nativeRecords = null, AgentDefaultRunnerSetting? defaultRunner = null, Services.RunData.IRunDataCompletenessWriter? completeness = null, Credentials.IModelCredentialBroker? credentialBroker = null, AgentRunLogging.IAgentRunLogService? logs = null, Microsoft.Extensions.Hosting.IHostApplicationLifetime? lifetime = null, Recovery.IAgentSessionTranscriptCheckpointer? sessionCheckpointer = null, TimeProvider? clock = null) { _runs = runs; _harnesses = harnesses; @@ -213,6 +228,8 @@ public AgentRunExecutor(IAgentRunService runs, IAgentHarnessRegistry harnesses, _credentialBroker = credentialBroker; _logs = logs; _lifetime = lifetime; + _sessionCheckpointer = sessionCheckpointer; + _clock = clock ?? TimeProvider.System; // Tolerate a null enumerable (a hand-built test double that never exercises the push path) — zero guards // registered is a legitimate state (every push clears), not a constructor-time crash. _publishGuards = (publishGuards ?? Enumerable.Empty()).OrderBy(g => g.Order).ToList(); @@ -467,12 +484,13 @@ public async Task ExecuteAsync(Guid agentRunId, CancellationToken cancellationTo var runContext = new HarnessRunContext { Owner = owner, TeamId = run.TeamId, ActorId = run.CreatedBy, - Harness = harness, Runner = runner, Spec = spec, McpToken = mcpToken, McpSocketPath = mcpToken is null ? null : socketPath, Redactor = redactor, + Harness = harness, Runner = runner, Spec = spec, Task = effectiveTask, McpToken = mcpToken, McpSocketPath = mcpToken is null ? null : socketPath, Redactor = redactor, ModelBrokerRunToken = brokeredCredential?.RunToken, ModelCredentialBrokered = brokeredPosture, ModelBrokerPort = brokeredCredential?.RebindPort, ModelBrokerRoute = brokeredCredential?.RebindRoute, ModelBrokerCredentialId = brokeredCredential is null ? null : modelCredentialId, ModelBrokerProvider = brokeredCredential is null ? null : modelProvider, SpoolKey = ReviseSpoolKey(agentRunId, round: 0), Transcript = transcript, WorkspaceDirectory = workspaceDirectory, WorkspaceBaseSha = workspaceBaseSha, + ResumedFromCheckpointAt = effectiveTask.ResumedFromCheckpointAt, }; var modelPrices = await ResolveSpendPricesAsync(run, effectiveTask, cancellationToken).ConfigureAwait(false); @@ -574,7 +592,7 @@ public async Task ExecuteAsync(Guid agentRunId, CancellationToken cancellationTo break; } - var roundResult = await RunHarnessAsync(runContext with { Spec = reviseSpec, SpoolKey = ReviseSpoolKey(agentRunId, round) }, cancellationToken).ConfigureAwait(false); + var roundResult = await RunHarnessAsync(runContext with { Spec = reviseSpec, Task = reviseTask, SpoolKey = ReviseSpoolKey(agentRunId, round) }, cancellationToken).ConfigureAwait(false); result = AgentRunBudget.Apply(reviseTask with { BudgetSpentUsd = result.CumulativeCostUsd }, roundResult, modelPrices) with { TokenUsage = SumTokenUsage(priorUsage, roundResult.TokenUsage), ReviseRounds = round }; spendClaim = await SettleInvocationSpendAsync(spendClaim, InvocationObservation(result, roundResult), reviseTask, modelPrices, cancellationToken).ConfigureAwait(false); @@ -976,11 +994,19 @@ async Task PersistFrameAsync(SandboxOutputFrame output) var capture = await OpenLogCaptureAsync(new LogCaptureContext(context.TeamId, context.RunId, context.ActorId, context.WorkerFenceEpoch, redactor), context.Durable, handle, cancellationToken).ConfigureAwait(false); if (!ReferenceEquals(capture.Handle, handle) && capture.Handle != handle) await _runs.SetRunnerHandleAsync(context.Owner, JsonSerializer.Serialize(capture.Handle, AgentJson.Options), cancellationToken).ConfigureAwait(false); - var sandbox = await capture.ObserveAsync((capturedHandle, token) => + SandboxResult sandbox; + try + { + sandbox = await capture.ObserveAsync((capturedHandle, token) => + { + var replayHandle = capturedHandle with { StdoutOffset = Math.Min(capturedHandle.StdoutOffset, native.ReplayStartOffset) }; + return context.Durable.AttachAsync(replayHandle, (frame, _) => PersistFrameAsync(frame), token, CheckpointHandleOffset(context.Owner, capturedHandle, new HarnessSinks(writer, native, CheckpointTickFor(context.Task, context.TeamId, context.Harness, context.Task.WorkspaceDirectory, facts)))); + }, cancellationToken).ConfigureAwait(false); + } + finally { - var replayHandle = capturedHandle with { StdoutOffset = Math.Min(capturedHandle.StdoutOffset, native.ReplayStartOffset) }; - return context.Durable.AttachAsync(replayHandle, (frame, _) => PersistFrameAsync(frame), token, CheckpointHandleOffset(context.Owner, capturedHandle, new HarnessSinks(writer, native))); - }, cancellationToken).ConfigureAwait(false); + await DrainSessionTranscriptCheckpointAsync(context.RunId, cancellationToken).ConfigureAwait(false); // 3c: same reason as the live path, including the finally — a failed observe is a retry's best source of conversation + } // Final flush for the terminal-drain lines (no trailing checkpoint), as in the live path. await writer.FlushAsync(cancellationToken).ConfigureAwait(false); @@ -1538,9 +1564,52 @@ private async Task ResolveRestoredTranscriptAsync(AgentTask task, Gui { if (task.RestoredTranscriptArtifactId is not { } artifactId) return task; - var transcript = await _offloader.ResolveRequiredAsync(teamId, task.RestoredTranscript, artifactId, cancellationToken).ConfigureAwait(false); + if (!task.RestoredTranscriptIsCheckpoint) + { + var transcript = await _offloader.ResolveRequiredAsync(teamId, task.RestoredTranscript, artifactId, cancellationToken).ConfigureAwait(false); + + return task with { RestoredTranscript = transcript, RestoredTranscriptArtifactId = null }; + } - return task with { RestoredTranscript = transcript, RestoredTranscriptArtifactId = null }; + return await ResolveCheckpointTranscriptAsync(task, teamId, artifactId, cancellationToken).ConfigureAwait(false); + } + + /// + /// 3c: resolve a mid-run CHECKPOINT ref under the opposite policy to a captured one — unreadable degrades to a + /// COLD start instead of failing the launch. + /// + /// Fail-closed is right for a captured transcript: the attempt that wrote it finished, so an unreadable ref + /// is a genuine fault and cold-starting a named session silently would hide it. A checkpoint is best-effort by + /// construction — its blob may have been collected, its destination may be unreachable — and this task is already + /// a RETRY of a lost host. Failing it would spend the very attempt the checkpoint exists to improve, on the one + /// fault that says nothing about the work. + /// + /// The degrade is total and honest: the session id goes with the ref (a --resume naming a session + /// whose transcript was never restored cold-starts in the CLI anyway, silently), the confinement stamp is cleared + /// so the run's permanent record does not claim a continuation it never had, and the goal is told. + /// + private async Task ResolveCheckpointTranscriptAsync(AgentTask task, Guid teamId, Guid artifactId, CancellationToken cancellationToken) + { + try + { + var transcript = await _offloader.ResolveRequiredAsync(teamId, task.RestoredTranscript, artifactId, cancellationToken).ConfigureAwait(false); + + return task with { RestoredTranscript = transcript, RestoredTranscriptArtifactId = null, RestoredTranscriptIsCheckpoint = false }; + } + catch (Exception unavailable) when (unavailable is not OperationCanceledException) + { + _logger.LogWarning(unavailable, "Agent run: the lost host's session checkpoint {ArtifactId} could not be read, so this retry runs COLD rather than failing on a best-effort recovery aid", artifactId); + + // Every claim the checkpoint bought goes with it, including the source link — the row is promoted from + // this envelope at creation, and "resumed from run X" would be false for an attempt that restored + // nothing from X. + return task with + { + Goal = AgentRetryContinuity.WithUnreadableCheckpointHint(task.Goal), + RestoredTranscript = null, RestoredTranscriptArtifactId = null, RestoredTranscriptIsCheckpoint = false, + ResumeFromSessionId = null, ResumedFromCheckpointAt = null, ResumedFromAgentRunId = null, + }; + } } /// @@ -1553,10 +1622,19 @@ private async Task ResolveRestoredTranscriptAsync(AgentTask task, Gui /// rollout-<id>.jsonl under the linked target — which a search-based locate like Codex's glob surfaces and /// a leaf-only resolve misses). So the check walks EVERY component from just below the config home to the leaf and /// fail-closes on ANY symlink: the CLIs only ever write real files/dirs here, so a symlink component in this subtree - /// is inherently hostile. Capture runs AFTER the agent process exits, so there is no live check-then-read race. + /// is inherently hostile. /// Returns null when the path escapes (the caller logs + skips); a non-existent in-bounds path is returned as-is /// (the caller's existence check then treats it as a cold-start). /// + /// TWO CALLERS, and they do NOT share a threat model. The end-of-run capture runs after the agent process + /// has exited, so its walk and its read cannot be raced — which is what this comment used to claim for everyone. + /// The 3c mid-run checkpoint () runs while the agent is still + /// executing with write access to its own bind-mounted config home, so the walk here and the open that follows + /// ARE two syscalls with a live writer between them; a component swapped in that window would point the read at + /// a worker-readable host file, and its bytes would land in the team's artifact store and be restored into the + /// next attempt. This function cannot close that on its own — it returns a path, not a handle. The checkpointer + /// does, by verifying through /proc/self/fd that the file it OPENED is the path resolved here. + /// /// RESIDUAL (documented, not closed here): a HARDLINK carries no link target, so a per-component symlink walk /// cannot see it. This is NOT a symlink-style escalation under the default hardening — Linux protected_hardlinks=1 /// (the modern default) only permits hardlinking a file the caller can already READ, so a confined agent gains @@ -1581,8 +1659,8 @@ private async Task ResolveRestoredTranscriptAsync(AgentTask task, Gui return lexical; } - /// The session-transcript capture cap in bytes — the env override () when it parses to a positive long, else . - private static long MaxSessionTranscriptBytes() => + /// The session-transcript capture cap in bytes — the env override () when it parses to a positive long, else . Internal, not private, because the mid-run checkpointer (ArtifactSessionTranscriptCheckpointer) reads the SAME file under the SAME limit — one knob for one decision, never a second one that could drift out from under an operator who tuned this. + internal static long MaxSessionTranscriptBytes() => ParseMaxSessionTranscriptBytes(Environment.GetEnvironmentVariable(MaxSessionTranscriptBytesEnvVar), DefaultMaxSessionTranscriptBytes); /// Parse the cap override — a positive long wins; anything else (null / non-numeric / non-positive) falls back to . Pure, so the parse + fallback is unit-pinned without touching the process env. @@ -3871,7 +3949,21 @@ async Task PersistAsync(string line, SandboxOutputFrame? output) // redactor's fingerprint is stamped onto the durable handle so a re-attach can prove it rebuilt the SAME // key before re-tailing the spool (a rotated/deleted key → marker-only, never an unmaskable leak). The MCP // token rides the handle too so a re-attach re-binds the SAME socket+token the agent's declaration carries. - var sandbox = await RunSandboxAsync(context, PersistLineAsync, PersistFrameAsync, new HarnessSinks(writer, native), cancellationToken).ConfigureAwait(false); + SandboxResult sandbox; + try + { + sandbox = await RunSandboxAsync(context, PersistLineAsync, PersistFrameAsync, new HarnessSinks(writer, native, CheckpointTickFor(context.Task, context.TeamId, context.Harness, context.Spec.WorkingDirectory, facts)), cancellationToken).ConfigureAwait(false); + } + finally + { + // 3c: let the in-flight checkpoint land before anything can unwind this round's scope. It is the LAST + // one — the most conversation any retry would get — and it is running against a DbContext this scope + // owns. In a FINALLY because the paths that skip the straight line — an admitted-launch failure, an + // ownership loss — are exactly the ones a retry follows. A CANCELLED token is not among them: the wait + // below throws on it at once and the checkpoint is abandoned, which is the right answer for a worker + // being torn down. + await DrainSessionTranscriptCheckpointAsync(context.RunId, cancellationToken).ConfigureAwait(false); + } // Final flush: the durable runner's terminal-drain paths (CompleteFromSpool/Timeout/Vanished) deliver the last // lines WITHOUT a trailing checkpoint, so anything buffered after the last checkpoint must be flushed here @@ -4024,6 +4116,16 @@ private async Task RecordConfinementAsync(AgentRunOwnerToken owner, SandboxConfi private static SandboxConfinement? WithCredentialPosture(SandboxConfinement? confinement, bool? brokered) => confinement is null ? null : confinement with { ModelCredentialBrokered = brokered }; + /// + /// 3c: add to the launch's posture that this attempt is the CONTINUATION of a run whose host died. Kept separate + /// from so each merge states one fact, and recorded at all because a resumed + /// attempt is narrower than the one it continues — only the conversation was durable — and a reader who is not + /// told will read its transcript as evidence about a working tree it does not have. Unchanged for an ordinary + /// launch (null ) and for a runner that stamped no posture at all. + /// + private static SandboxConfinement? WithResumeProvenance(SandboxConfinement? confinement, DateTimeOffset? resumedFromCheckpointAt) => + confinement is null || resumedFromCheckpointAt is null ? confinement : confinement with { ResumedFromCheckpointAt = resumedFromCheckpointAt }; + /// /// Stamp the run's posture with what a re-attach has just made true: its BROKERED model credential is gone. The /// lease lived in the launching worker's memory — that is what makes revocation real — so this pass re-opens @@ -4804,7 +4906,7 @@ private async Task AcknowledgeAdmittedLaunchAsync(Ha try { await _runs.SetRunnerHandleAsync(context.Owner, JsonSerializer.Serialize(handle, AgentJson.Options), cancellationToken).ConfigureAwait(false); - await RecordConfinementAsync(context.Owner, WithCredentialPosture(handle.Confinement, context.ModelCredentialBrokered), cancellationToken).ConfigureAwait(false); + await RecordConfinementAsync(context.Owner, WithResumeProvenance(WithCredentialPosture(handle.Confinement, context.ModelCredentialBrokered), context.ResumedFromCheckpointAt), cancellationToken).ConfigureAwait(false); var capture = await OpenLogCaptureAsync(new LogCaptureContext(context.TeamId, context.RunId, context.ActorId, context.WorkerFenceEpoch, context.Redactor), durable, handle, cancellationToken).ConfigureAwait(false); if (capture.Handle != handle) await _runs.SetRunnerHandleAsync(context.Owner, JsonSerializer.Serialize(capture.Handle, AgentJson.Options), cancellationToken).ConfigureAwait(false); @@ -4963,6 +5065,11 @@ private async Task OpenLogCaptureAsync(LogCaptureCon /// poll's lines, THEN persist the advanced spool offset onto the handle. The flush-before-offset ordering is the /// durability invariant — the persisted offset must never run ahead of flushed events, so a re-attach at worst /// re-emits the last batch (never loses a line). A pure jsonb UPDATE for the offset; never blocks completion. + /// + /// 3c rides this same tick to make the run's resumable CONVERSATION durable + /// (). It goes LAST, after the two flushes and the offset, + /// because those are what the run's own completion depends on and the checkpoint is only what another host would + /// need if this one died — it is a recovery aid, and it may never be in front of the work. /// private Func CheckpointHandleOffset(AgentRunOwnerToken owner, SandboxHandle handle, HarnessSinks sinks) => async (offset, ct) => @@ -4970,8 +5077,186 @@ private Func CheckpointHandleOffset(AgentRunOwner await sinks.Events.FlushAsync(ct).ConfigureAwait(false); await sinks.Frames.FlushAsync(ct).ConfigureAwait(false); // the frame plane rides the same checkpoint — best-effort, so a refused frame flush stops capture for the round rather than holding the offset back await _runs.SetRunnerHandleAsync(owner, JsonSerializer.Serialize(handle with { StdoutOffset = Math.Max(handle.StdoutOffset, offset) }, AgentJson.Options), ct).ConfigureAwait(false); + + StartSessionTranscriptCheckpoint(owner, handle, sinks.SessionCheckpoint); }; + /// + /// 3c: START a durable checkpoint of this tick's live session transcript, so an attempt whose host dies leaves a + /// conversation its retry can continue. Returns without awaiting it. + /// + /// Off the tick, and in a DI SCOPE OF ITS OWN. This callback runs inside the durable runner's drain loop, + /// whose next poll carries the run's stdout, its exit marker, its wall-clock check and its progress lease — so a + /// read-and-upload on this thread would stall all four for as long as the artifact store took. And the scope is + /// not optional tidiness: the tick is already using this executor's scoped DbContext (the buffered event + /// flush, the spool-offset write), so a checkpoint sharing it would put two operations on one EF context at once. + /// EF refuses that on whichever statement starts second — as often the TICK's, unhandled inside the runner's + /// attach loop — which would kill a healthy run for a best-effort recovery aid. Same shape as + /// and the heartbeat, for the same reason. + /// + /// At most ONE is in flight: a tick that finds the previous one still running skips, so a slow store costs + /// checkpoints rather than queueing them up. Its result is harvested HERE, on the tick, so the growth watermark + /// is only ever touched by one thread. + /// + /// The order of the guards is the cost order. The two FREE reads come first — a harness with no resumable + /// transcript never has one, and a stream that has not yet named its session id cannot address one — because + /// claiming the cadence window for either would burn a whole interval on a tick that was never going to + /// checkpoint. Then the window, then the locate (Codex finds its rollout by a recursive walk, and this fires + /// several times a second), then the SAME security clamp the end-of-run capture uses. + /// + private void StartSessionTranscriptCheckpoint(AgentRunOwnerToken owner, SandboxHandle handle, SessionCheckpointTick? tick) + { + if (_sessionCheckpointer is null || tick is null || !_sessionCheckpoint.IsCompleted) return; + + HarvestSessionTranscriptCheckpoint(); + + if (tick.Harness is not IAgentSessionTranscript resumable || tick.Facts.SessionId is not { Length: > 0 } sessionId) return; + + var now = _clock.GetUtcNow(); + + if (!SessionCheckpointDue(_sessionCheckpointAttemptedAt, now)) return; + + _sessionCheckpointAttemptedAt = now; + + var configHome = LocalProcessRunner.ConfigHomePath(handle.SpoolDirectory); + + if (resumable.SessionTranscriptRelativePath(configHome, tick.WorkingDirectory, sessionId) is not { } relativePath) return; + + if (ResolveSessionTranscriptPath(configHome, relativePath) is not { } path) + { + _logger.LogWarning("Agent run {RunId}: the live session-transcript path escaped the config home (hostile session id?); skipping the checkpoint", owner.RunId); + return; + } + + var request = new Recovery.SessionTranscriptCheckpointRequest(tick.TeamId, owner, path, sessionId, _sessionCheckpointWatermark.Begin(path)); + + _sessionCheckpoint = Task.Run(() => CheckpointInOwnScopeAsync(request), CancellationToken.None); + } + + /// + /// Resolve a FRESH checkpointer (and with it a fresh DbContext and artifact store) for this one upload — + /// see for why sharing the executor's would be a defect rather + /// than a saving — and bound it with a deadline OF ITS OWN. + /// + /// Deliberately NOT the observer's token. That token is cancelled by exactly the terminations a warm retry + /// follows — a wall-clock timeout, a no-progress stall, an operator cancel — so threading it here would abort the + /// last checkpoint on precisely the runs whose conversation is most worth keeping. And deliberately not + /// unbounded either: a checkpoint that cannot finish inside is one + /// the next cadence would overlap. + /// + private async Task CheckpointInOwnScopeAsync(Recovery.SessionTranscriptCheckpointRequest request) + { + using var budget = new CancellationTokenSource(SessionCheckpointUploadBudget); + using var scope = _scopeFactory.CreateScope(); + + return await scope.ServiceProvider.GetRequiredService().CheckpointAsync(request, budget.Token).ConfigureAwait(false); + } + + /// Advance the growth watermark from a COMPLETED checkpoint, on the tick's own thread. + private void HarvestSessionTranscriptCheckpoint() + { + if (_sessionCheckpoint.IsCompletedSuccessfully) _sessionCheckpointWatermark.Landed(_sessionCheckpoint.Result); + } + + /// + /// How long ONE checkpoint's read and upload may take before it is abandoned. Sized for the work: the capture + /// cap is (32 MiB), so a budget that assumed a fast local store + /// would cancel every checkpoint of a long conversation on a throttled or cross-region destination — silently + /// un-recovering exactly the runs this feature exists for. Thirty seconds is comfortably inside the + /// cadence, so a slow upload still cannot overlap the next attempt. + /// + internal static readonly TimeSpan SessionCheckpointUploadBudget = TimeSpan.FromSeconds(30); + + /// + /// How long the round WAITS for an in-flight checkpoint before landing anyway — deliberately much shorter than + /// , because it is bounding a different thing. That budget is how long + /// a checkpoint may take; this is how long a finished run's terminal write may be deferred by one, and a + /// best-effort recovery aid has no business holding a landing open for half a minute. Past it the run lands and + /// the upload is left to finish or not: its stamp is fenced on the run still being Running, so a late one is + /// refused by the database rather than writing onto a terminal row. + /// + internal static readonly TimeSpan SessionCheckpointDrainBudget = TimeSpan.FromSeconds(5); + + /// + /// How long after one checkpoint ATTEMPT the next may be made. Committed here and changed by a pull request — + /// there is no environment override, for the reason the retention policy states about its own windows: the cost + /// of a mistyped value is paid in bytes uploaded from every running agent on every worker, and a code review is + /// the control that belongs in front of it. Sixty seconds is the trade the whole slice rests on: a lost host + /// costs at most that much conversation, and a run is charged one whole-file read a minute rather than one per + /// poll of a loop that ticks several times a second. + /// + internal static readonly TimeSpan SessionCheckpointInterval = TimeSpan.FromSeconds(60); + + /// + /// The growth watermark one run's checkpoints are measured against: how many bytes of WHICH transcript file the + /// last LANDED checkpoint stored. + /// + /// Its own type, and both halves written in one assignment, because splitting them re-creates the defect + /// the path key exists to prevent. A revise round opens a NEW config home whose transcript legitimately starts + /// smaller than the finished previous round's — so if the path advanced when an attempt was DISPATCHED while the + /// byte count advanced only when one LANDED, the ordinary first-tick decline (the CLI has not written its session + /// file yet) would leave the new round's path paired with the old round's byte count, and round two would never + /// checkpoint until it outgrew round one. + /// + /// Touched only from the drain tick, which is single-threaded by construction: the dispatching tick calls + /// and a later tick calls with the finished task's result. The + /// background upload itself touches nothing here. + /// + internal sealed class SessionCheckpointWatermark + { + private (string Path, long Bytes)? _taken; + private string? _pending; + + /// Record that an attempt for is starting, and hand back the byte count it must grow past — the last landed checkpoint's, and only when that checkpoint was of this same path. + public long? Begin(string path) + { + _pending = path; + + return _taken is { } taken && string.Equals(taken.Path, path, StringComparison.Ordinal) ? taken.Bytes : null; + } + + /// Settle the attempt started. A null — the attempt DECLINED (no growth, no complete line, over the cap, lost fence) — leaves the watermark exactly where it was, so the next attempt is measured against what actually landed. An attempt that FAULTED never arrives here at all: the harvest only reads a task that completed successfully, so the watermark keeps its previous value by not being called. + public void Landed(Messages.Agents.SessionTranscriptCheckpoint? checkpoint) + { + if (checkpoint is not null && _pending is { } path) _taken = (path, checkpoint.Bytes); + + _pending = null; + } + } + + /// Whether the checkpoint cadence window is open. Pure, so the gate that decides how often a fleet uploads transcripts is pinned directly rather than through a timing-dependent drive. + internal static bool SessionCheckpointDue(DateTimeOffset? lastAttemptAt, DateTimeOffset now) => + lastAttemptAt is not { } last || now - last >= SessionCheckpointInterval; + + /// + /// Wait for the in-flight checkpoint before this round's scope can unwind. + /// + /// Without it the fire-and-forget outlives its own dependencies: ExecuteAsync returns, the job scope + /// disposes the context and the store underneath a running upload, and the LAST checkpoint — the one holding the + /// most conversation — is lost to a swallowed ObjectDisposedException. That is exactly the minute a host + /// loss would have needed. + /// + /// Bounded by the transcript cap on the read and by on the rest, so a + /// worker tear-down stops it instead of holding the drain budget open. Its result is harvested for the same + /// reason every other completion is: the watermark must reflect what actually landed. + /// + private async Task DrainSessionTranscriptCheckpointAsync(Guid runId, CancellationToken cancellationToken) + { + if (_sessionCheckpoint.IsCompleted) { HarvestSessionTranscriptCheckpoint(); return; } + + try + { + // The DRAIN bound, not the round's token and not the upload's: this wait sits in front of the terminal + // write, so a slow destination must not defer that write for as long as an upload is allowed to take. + await _sessionCheckpoint.WaitAsync(SessionCheckpointDrainBudget, cancellationToken).ConfigureAwait(false); + HarvestSessionTranscriptCheckpoint(); + } + catch (Exception exception) + { + _logger.LogWarning(exception, "Agent run {RunId}: the in-flight session-transcript checkpoint did not land before this round ended; the run stays recoverable only from its previous one", runId); + } + } + /// /// The two durable sinks one harness round streams into: the normalized event log and the native-frame plane. One /// record rather than two parameters because they are flushed together, at the same checkpoints, for the same @@ -5045,7 +5330,43 @@ internal sealed record AcceptanceInvocation(AgentRun Run, AgentTask Task, IWorks private sealed record WorkspaceCaptureContext(AgentRunOwnerToken Owner, Guid TeamId, AgentTask Task, IWorkspaceHandle? Workspace); private sealed record RepositoryPushContext(AgentRunOwnerToken Owner, AgentTask Task, IWorkspaceHandle Workspace, IWorkspacePushHandle PushHandle); - private sealed record HarnessSinks(BufferedEventWriter Events, AgentNativeRecordPump Frames); + private sealed record HarnessSinks(BufferedEventWriter Events, AgentNativeRecordPump Frames, SessionCheckpointTick? SessionCheckpoint = null); + + /// + /// What a checkpoint tick needs to locate and checkpoint the run's RESUMABLE session transcript — everything + /// except the two things the tick already holds (the owner token and the launched handle whose spool the config + /// home lives under). NULL on when this run is not resumable at all, which is the + /// cheapest possible gate: no checkpointer deployed, a hand-built test double, or an envelope that did not opt in + /// () — and a run whose failure nobody can retry must not pay for, or store, + /// a checkpoint. + /// + /// is the directory the CLI process actually ran in + /// (), and nothing else will do: Claude's session path is + /// projects/<sanitized-cwd>/<id>.jsonl, so the encoding keys on the cwd. The primary repo's + /// directory is NOT that cwd for a multi-repo workspace (which runs at the workspace root) and does not exist at + /// all for a repo-less one — passing it silently addressed a file that was never written, and the whole feature + /// went quiet for exactly those runs. + /// + /// The session id is read LIVE off rather than passed in, because it is not + /// known at launch: the harness names it on its own first lifecycle line (Claude's init, Codex's + /// thread.started), which the fold has already consumed by the first checkpoint. Without it the file + /// cannot be addressed at all — both harness layouts key on the id. + /// + /// That is also why a RE-ATTACH may legitimately checkpoint nothing: its fold resumes from a frame + /// checkpoint, so the lifecycle line carrying the id is usually behind the replay head and its fresh facts never + /// see one. The launch's own last checkpoint then stands, which is the right answer — it is the newest + /// conversation anyone can prove. + /// + private sealed record SessionCheckpointTick(Guid TeamId, IAgentHarness Harness, string? WorkingDirectory, AgentRunFacts Facts); + + /// + /// The checkpoint coordinates for a run whose envelope OPTED IN, else null — the one place the opt-in is read, so + /// the produce side and the consume side cannot disagree about which runs are checkpointed. A run whose failed + /// attempt nobody can retry writes nothing, which is what keeps the artifact store free of a per-minute + /// transcript copy for every benchmark cell, review child and supervisor unit on the fleet. + /// + private static SessionCheckpointTick? CheckpointTickFor(AgentTask task, Guid teamId, IAgentHarness harness, string? workingDirectory, AgentRunFacts facts) => + task.CheckpointSessionTranscript ? new SessionCheckpointTick(teamId, harness, workingDirectory, facts) : null; private sealed record HarnessRunContext { @@ -5057,6 +5378,9 @@ private sealed record HarnessRunContext public required IAgentHarness Harness { get; init; } public required ISandboxRunner Runner { get; init; } public required SandboxSpec Spec { get; init; } + + /// The envelope this round is running — carried for the decisions that read the TASK rather than the spec built from it (3c's resume opt-in). Same object was built from, so the two can never describe different work. + public required AgentTask Task { get; init; } public string? McpToken { get; init; } /// The address this run's endpoint bound, minted with an unguessable segment at launch. Carried here so the durable handle can be stamped with it — the only route a re-attach has back to it. @@ -5080,6 +5404,9 @@ private sealed record HarnessRunContext public required AgentTranscriptSpool Transcript { get; init; } public string? WorkspaceDirectory { get; init; } public string? WorkspaceBaseSha { get; init; } + + /// 3c: when this attempt was minted as the continuation of a checkpointed run whose host died — carried from the task so the launch's permanent confinement record can state it. Null for every ordinary launch. + public DateTimeOffset? ResumedFromCheckpointAt { get; init; } } private sealed record ReattachFoldContext diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunReconcilerService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunReconcilerService.cs index 723e7aabd..fe11f6bc5 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunReconcilerService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunReconcilerService.cs @@ -580,6 +580,20 @@ private async Task ReattachAsync(AgentRunReconciliationCandidate c private async Task AbandonAsync(AgentRunReconciliationCandidate candidate, AgentRunAbandonCause cause, CancellationToken cancellationToken, ISandboxDurableRunner? durable = null, SandboxHandle? handle = null) { var runId = candidate.RunId; + + // NOTE the asymmetry with the executor's own terminal write, which RELEASES the run's session-transcript + // checkpoint: this one must KEEP it. An abandon is exactly the case the checkpoint exists for — the agent + // node's respawn reads it off this row to continue the conversation the dead host was holding — so clearing + // it here would delete the only thing that makes the retry warm. + // + // The COST, stated because it is permanent and nothing else says it: this column keeps the artifact + // Referenced, and Referenced is TERMINAL in the retention ledger (the reaper's claim query takes only + // Declared and Quarantined rows), so one full transcript copy is kept FOR EVER per host loss — including + // when the retry ran and finished. The two-hour class does NOT collect it; that class only shortens the age + // floor for checkpoints nothing references. Clearing it once the successor is staged is not available + // either: the successor's ref lives in its own task_json, which the reference oracle does not probe, so the + // artifact would become collectable in the window before that successor launches. A jsonb reference site is + // the fix, and it belongs with the retention plane rather than here. var transitioned = await TerminalizeCandidateAsync(candidate, AgentRunStatus.Failed, AbandonedError, null, cancellationToken).ConfigureAwait(false); // P2 (capture-intent saga): an abandoned attempt died inside (or before) its capture window — every open @@ -620,6 +634,7 @@ private async Task AbandonAsync(AgentRunReconciliationCandidate ca await RecordLogOwnerLossQuietlyAsync(candidate.TeamId, stamp, cancellationToken).ConfigureAwait(false); await TryAppendEventAsync(runId, AgentEventKind.Error, AbandonedError, cancellationToken).ConfigureAwait(false); + return StaleOutcome.Abandoned; } diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.Ownership.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.Ownership.cs index 4e665918c..86d87e18c 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.Ownership.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.Ownership.cs @@ -78,6 +78,22 @@ public async Task SetSandboxConfinementAsync(AgentRunOwnerToken owner, string co if (changed != 1) throw new AgentRunOwnershipLostException(owner.RunId); } + public async Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(checkpoint); + + // Fenced exactly like the handle and confinement writes beside it — a worker whose ownership was reclaimed + // must not be able to point a live run's recovery at ITS stale conversation. The session id rides the same + // statement because a checkpoint nothing can address is not resumable, and COALESCE keeps an id already on + // the row (a completion's, or an earlier checkpoint's when this tick saw none). + var changed = await _db.Database.ExecuteSqlInterpolatedAsync($"WITH locked AS MATERIALIZED (SELECT id FROM agent_run WHERE id = {owner.RunId} FOR UPDATE) UPDATE agent_run AS target SET session_transcript_checkpoint_artifact_id = {checkpoint.ArtifactId}, session_transcript_checkpoint_at = {checkpoint.At}, session_id = COALESCE({sessionId}, target.session_id) FROM locked WHERE target.id = locked.id AND target.status = 'Running' AND target.owner_id = {owner.OwnerId} AND target.fence_epoch = {owner.Epoch} AND target.lease_expires_at > clock_timestamp()", cancellationToken).ConfigureAwait(false); + + // Deliberately NOT a throw, unlike its two neighbours. This write rides an observer tick whose real job is + // flushing the run's events and advancing its spool offset; a lost race here must cost a checkpoint, never + // the tick. + return changed == 1; + } + private void EnsureIndependentOwnershipTransaction() { if (_db.Database.CurrentTransaction is not null || System.Transactions.Transaction.Current is not null) throw new InvalidOperationException("Execution ownership must commit independently before a worker can perform external actions."); diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs index 12399c6c9..cec369a07 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunService.cs @@ -34,6 +34,15 @@ public interface IAgentRunService Task HeartbeatAsync(AgentRunOwnerToken owner, CancellationToken cancellationToken); Task SetRunnerHandleAsync(AgentRunOwnerToken owner, string handleJson, CancellationToken cancellationToken); Task SetSandboxConfinementAsync(AgentRunOwnerToken owner, string confinementJson, CancellationToken cancellationToken); + + /// + /// 3c: record the run's most recent MID-RUN session-transcript checkpoint (and the session id that addresses it), + /// fenced to like the handle and confinement writes. Returns whether the row was won — + /// false, never a throw, because this rides an observer tick whose real work must not be endangered by a + /// best-effort recovery aid. + /// + Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken); + Task AppendEventAsync(AgentRunOwnerToken owner, AgentEvent @event, CancellationToken cancellationToken); Task AppendEventsAsync(AgentRunOwnerToken owner, IReadOnlyList events, CancellationToken cancellationToken); Task AppendSystemEventAsync(Guid runId, AgentEvent @event, CancellationToken cancellationToken); @@ -226,6 +235,7 @@ private async Task PersistCreatedAsync(AgentRunCreation creation, Canc WorkflowRunId = creation.WorkflowRunId, NodeId = creation.NodeId, IterationKey = creation.IterationKey, + ResumedFromAgentRunId = creation.Task.ResumedFromAgentRunId, // promoted from task_jsonb to a column, like AgentDefinitionId Harness = creation.Task.Harness, AgentDefinitionId = creation.Task.AgentDefinitionId, // promoted from task_jsonb to a column so the runs index can filter by agent Status = AgentRunStatus.Queued, @@ -525,16 +535,22 @@ private async Task CompleteCoreAsync(CompletionWriter writer, AgentRunResult res var sessionId = PersistedText.Sanitize(result.SessionId); var error = PersistedText.Sanitize(result.Error); + // 3c: the terminal write RELEASES the mid-run session checkpoint. It existed to let another host continue a + // run whose own host died, and this run has now landed — so the reference is stale by definition, and holding + // it would pin the artifact Referenced (terminal in the retention ledger) beside the end-of-run transcript the + // result already carries. Clearing it here is what turns the survivor of a run's minute-by-minute checkpoints + // into a reap candidate; the intermediates were already unreferenced the moment the next one replaced them. int flipped; if (owner is not null) { - flipped = await _db.Database.ExecuteSqlInterpolatedAsync($"WITH locked AS MATERIALIZED (SELECT id FROM agent_run WHERE id = {runId} FOR UPDATE) UPDATE agent_run AS target SET status = {result.Status.ToString()}, result_jsonb = CAST({resultJson} AS jsonb), session_id = {sessionId}, error = {error}, completed_at = clock_timestamp() FROM locked WHERE target.id = locked.id AND target.status = {current.ToString()} AND target.owner_id = {owner.OwnerId} AND target.fence_epoch = {owner.Epoch} AND target.lease_expires_at > clock_timestamp()", cancellationToken).ConfigureAwait(false); + flipped = await _db.Database.ExecuteSqlInterpolatedAsync($"WITH locked AS MATERIALIZED (SELECT id FROM agent_run WHERE id = {runId} FOR UPDATE) UPDATE agent_run AS target SET status = {result.Status.ToString()}, result_jsonb = CAST({resultJson} AS jsonb), session_id = {sessionId}, error = {error}, completed_at = clock_timestamp(), session_transcript_checkpoint_artifact_id = NULL, session_transcript_checkpoint_at = NULL FROM locked WHERE target.id = locked.id AND target.status = {current.ToString()} AND target.owner_id = {owner.OwnerId} AND target.fence_epoch = {owner.Epoch} AND target.lease_expires_at > clock_timestamp()", cancellationToken).ConfigureAwait(false); if (flipped == 0) throw new AgentRunOwnershipLostException(runId); } else { flipped = await _db.AgentRun.Where(r => r.Id == runId && r.Status == current && r.OwnerId == null && r.ReattachReservationId == null && (expectedEpoch == null || r.FenceEpoch == expectedEpoch)) - .ExecuteUpdateAsync(s => s.SetProperty(r => r.Status, result.Status).SetProperty(r => r.ResultJson, resultJson).SetProperty(r => r.SessionId, sessionId).SetProperty(r => r.Error, error).SetProperty(r => r.CompletedAt, (DateTimeOffset?)DateTimeOffset.UtcNow), cancellationToken).ConfigureAwait(false); + .ExecuteUpdateAsync(s => s.SetProperty(r => r.Status, result.Status).SetProperty(r => r.ResultJson, resultJson).SetProperty(r => r.SessionId, sessionId).SetProperty(r => r.Error, error).SetProperty(r => r.CompletedAt, (DateTimeOffset?)DateTimeOffset.UtcNow) + .SetProperty(r => r.SessionTranscriptCheckpointArtifactId, (Guid?)null).SetProperty(r => r.SessionTranscriptCheckpointAt, (DateTimeOffset?)null), cancellationToken).ConfigureAwait(false); if (flipped == 0) { await EnsureLegacyWriterAsync(runId, cancellationToken).ConfigureAwait(false); @@ -874,20 +890,27 @@ await _db.AgentRun.AsNoTracking().SingleOrDefaultAsync(r => r.Id == runId, cance var chainIds = chain.Select(id => (Guid?)id).ToList(); // Every agent run at this cell anywhere in the chain that captured a session id (team-scoped; served by idx_agent_run_workflow_run). + // NEWEST first within a hop, exactly as FindResumableSubtaskAttemptAsync orders its own candidates. Without an + // ORDER BY the row a hop yields is whatever the plan happened to emit, which is not a decision anybody made — + // and a hop really can hold several session-bearing runs (a rerun of the cell, and since 3c an in-flight one + // that stamps its id at its first checkpoint), so "which attempt does this resume" needs an answer. var candidates = await _db.AgentRun.AsNoTracking() .Where(a => a.TeamId == teamId && chainIds.Contains(a.WorkflowRunId) && a.NodeId == nodeId && a.IterationKey == iterationKey && a.SessionId != null) + .OrderByDescending(a => a.CreatedDate).ThenByDescending(a => a.Id) .Select(a => new { a.Id, a.WorkflowRunId, a.SessionId, a.ResultJson }) .ToListAsync(cancellationToken).ConfigureAwait(false); // The NEAREST ancestor's RESUMABLE session — chain order (nearest first), skipping a captured-but-transcript-less // prior so it never masks an older resumable one, and treating a corrupt prior result as not-resumable (the // contract is cold-start, never a hard failure). Both-or-neither: a session id is resumable ONLY with a transcript. + // Nearest ancestor first, and EVERY candidate at that hop — not just the first row the list happened to hold. + // A hop can carry more than one session-bearing run (a rerun of the same cell, and since 3c an in-flight run + // that stamps its session id at its first checkpoint rather than only at completion), and a single + // FirstOrDefault let one of those mask an older attempt that IS resumable. Same shape as the subtask lookup + // below, which already scans its candidates. foreach (var runId in chain) - { - var candidate = candidates.FirstOrDefault(c => c.WorkflowRunId == runId); - - if (candidate is not null && TryResumable(candidate.Id, candidate.SessionId, candidate.ResultJson) is { } resumable) return resumable; - } + foreach (var candidate in candidates.Where(c => c.WorkflowRunId == runId)) + if (TryResumable(candidate.Id, candidate.SessionId, candidate.ResultJson) is { } resumable) return resumable; return null; } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Recovery/Checkpoints/ArtifactSessionTranscriptCheckpointer.cs b/backend/src/CodeSpace.Core/Services/Agents/Recovery/Checkpoints/ArtifactSessionTranscriptCheckpointer.cs new file mode 100644 index 000000000..027befa1f --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Agents/Recovery/Checkpoints/ArtifactSessionTranscriptCheckpointer.cs @@ -0,0 +1,167 @@ +using CodeSpace.Core.Services.Workflows.Artifacts.Retention; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Artifacts; +using Microsoft.Extensions.Logging; +using Microsoft.Win32.SafeHandles; + +namespace CodeSpace.Core.Services.Agents.Recovery.Checkpoints; + +/// +/// The artifact-store implementation of : upload the live session +/// transcript's COMPLETE prefix, then stamp the run row with the reference. +/// +/// Complete prefix rather than "whatever bytes were there", and that distinction is the difference between a +/// working continuation and a wasted one. The file is a CLI's own JSONL session, appended to while this reads it, so +/// a read that lands mid-append ends in half a line. The harness restores the bytes verbatim for its +/// --resume, and the executor's ResolveRestoredTranscriptAsync fails CLOSED on an unusable transcript +/// — and the continuation's budget is already spent by then, so a torn checkpoint does not degrade to a cold start, +/// it burns the one attempt. Cutting at the last newline yields a shorter conversation, which is exactly the +/// acceptable outcome; keeping the tail yields a corrupt one, which is not. +/// +/// Bounded by — the SAME cap the end-of-run capture +/// reads, deliberately not a second knob. Over it the checkpoint skips exactly as that capture does. +/// +/// STATELESS, and resolved in a DI scope of the caller's own. The executor dispatches this OFF its drain tick, +/// and the tick is using the executor's scoped DbContext every 250 ms for its event flush and spool-offset +/// write — so a checkpointer sharing that context would put two operations on one EF context concurrently. EF refuses +/// that, and it refuses it on whichever statement starts second: as often the TICK's, which is unhandled inside the +/// runner's attach loop and would kill a healthy run for a best-effort recovery aid. Holding no state is what lets the +/// caller give this its own scope per call; the cadence window and the growth watermark live with the caller. +/// +public sealed class ArtifactSessionTranscriptCheckpointer : IAgentSessionTranscriptCheckpointer +{ + /// The content type the CLIs' session files are: newline-delimited JSON. Same value the end-of-run offload stores them under, so both writes of one run's transcript dedupe against each other in the content-addressed store. + private const string TranscriptContentType = "application/x-ndjson"; + + /// The producer identity carried on the retention declaration for diagnosis (never the reference check — the oracle probes every reference site). The holder is the RUN, because the run row is what names the artifact. + internal const string CheckpointHolderKind = "agent-run-session-transcript-checkpoint"; + + private readonly IArtifactRetentionWriter _artifacts; + private readonly IAgentRunService _runs; + private readonly TimeProvider _clock; + private readonly ILogger _logger; + + public ArtifactSessionTranscriptCheckpointer(IArtifactRetentionWriter artifacts, IAgentRunService runs, TimeProvider clock, ILogger logger) + { + _artifacts = artifacts; + _runs = runs; + _clock = clock; + _logger = logger; + } + + public async Task CheckpointAsync(SessionTranscriptCheckpointRequest request, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(request); + if (string.IsNullOrEmpty(request.TranscriptPath)) return null; + + // The ordinary early-run state — the CLI has not written its session yet, or writes it somewhere this run + // cannot address. Checked before the open so it costs a probe rather than an exception and a warning; the + // open below is still authoritative (a file deleted in between simply throws into the catch). + if (!File.Exists(request.TranscriptPath)) return null; + + try + { + // ONE open, shared with the writing CLI (and with a delete, so a rotating harness cannot wedge this). + // Everything below reads THIS handle — the path is never resolved a second time, which is what closes the + // check-then-read window the clamp alone leaves open on a live config home. + using var stream = new FileStream(request.TranscriptPath, FileMode.Open, FileAccess.Read, FileShare.ReadWrite | FileShare.Delete); + + if (!OpenedTheResolvedFile(stream, request.TranscriptPath)) + { + _logger.LogWarning("Agent run {RunId}: the opened session transcript is not the path the clamp resolved (a component was swapped underneath); skipping this checkpoint", request.Owner.RunId); + return null; + } + + if (!IsDue(request.PreviousBytes, stream.Length)) return null; + + return await TakeAsync(request, stream, cancellationToken).ConfigureAwait(false); + } + catch (Exception exception) when (exception is not OperationCanceledException) + { + _logger.LogWarning(exception, "Agent run {RunId}: a session-transcript checkpoint could not be taken; this run stays recoverable only from its previous checkpoint, or cold-starts if it has none", request.Owner.RunId); + + return null; + } + } + + /// + /// Whether this attempt should pay for a checkpoint. Two gates, each naming a different waste: bytes that have + /// not changed carry no new conversation, and a file over the cap cannot be read at all. The CADENCE is not here + /// — the caller applies it before it even locates the file, so a harness that must SEARCH for its transcript + /// (Codex globs its rollout directory) does not pay for that search on every poll. + /// + private static bool IsDue(long? previousBytes, long length) + { + if (previousBytes is { } previous && length <= previous) return false; + + return length <= AgentRunExecutor.MaxSessionTranscriptBytes(); + } + + /// + /// Read the complete prefix, store it and stamp it. Returns null when the file holds no complete line yet, when + /// the prefix has not grown past the last checkpoint, or when the fenced stamp is lost — and the local state + /// advances only on a STAMPED checkpoint, so a failed attempt is retried at the next cadence rather than + /// suppressed by one it never earned. + /// + private async Task TakeAsync(SessionTranscriptCheckpointRequest request, FileStream stream, CancellationToken cancellationToken) + { + var runId = request.Owner.RunId; + var buffer = new byte[stream.Length]; + await stream.ReadExactlyAsync(buffer, cancellationToken).ConfigureAwait(false); + + var complete = CompletePrefixLength(buffer); + + if (complete == 0) return null; // the CLI has written only a partial first line — there is no whole turn to restore yet + + if (request.PreviousBytes is { } previous && complete <= previous) return null; // the growth was entirely inside the torn tail + + var write = await _artifacts.PutDeclaredAsync(new ArtifactRetentionWriteRequest(request.TeamId, buffer.AsMemory(0, complete), TranscriptContentType, ArtifactRetentionClass.SessionTranscriptCheckpoint, CheckpointHolderKind, runId), cancellationToken).ConfigureAwait(false); + var checkpoint = new SessionTranscriptCheckpoint(write.ArtifactId, _clock.GetUtcNow(), complete); + + if (!await _runs.StampSessionTranscriptCheckpointAsync(request.Owner, checkpoint, request.SessionId, cancellationToken).ConfigureAwait(false)) + { + _logger.LogInformation("Agent run {RunId}: a session-transcript checkpoint was stored but its row was won by another generation, so it was not stamped; the owning worker's own checkpoint stands", runId); + + return null; + } + + _logger.LogDebug("Agent run {RunId}: checkpointed {Bytes} bytes of session transcript as artifact {ArtifactId}; another host can continue this conversation if this one dies", runId, checkpoint.Bytes, checkpoint.ArtifactId); + + return checkpoint; + } + + /// + /// The length of the COMPLETE prefix: everything up to and including the last newline. Zero when the file holds + /// no newline at all. The CLIs append whole JSON lines, so a tail after the last newline is a line still being + /// written — restoring it would hand the next CLI a session it cannot parse. + /// + internal static int CompletePrefixLength(ReadOnlySpan transcript) + { + var last = transcript.LastIndexOf((byte)'\n'); + + return last < 0 ? 0 : last + 1; + } + + /// + /// Whether the handle actually opened the file the clamp resolved. The clamp walks the path for symlinks and then + /// the file is opened — two syscalls, and between them the AGENT is still running with write access to its own + /// bind-mounted config home, so a swapped component would otherwise let it point this read at a worker-readable + /// host file and have the bytes uploaded into its team's store and restored into the next attempt. + /// + /// Verified through the kernel's own answer for the OPEN handle (/proc/self/fd/<n>), which no + /// later rename can change. RESIDUAL, stated rather than hidden: a host with no /proc (macOS development) + /// cannot be asked, and there the check passes. That is the honest bound — the attack needs an agent confined + /// inside a bind-mounted config home, which is a Linux-only posture (BubblewrapSandbox), so the platform + /// that can be exploited is the platform that can be verified. + /// + internal static bool OpenedTheResolvedFile(FileStream stream, string resolvedPath) + { + var descriptor = $"/proc/self/fd/{DescriptorOf(stream.SafeFileHandle)}"; + + if (!OperatingSystem.IsLinux() || !File.Exists(descriptor) && !Directory.Exists(descriptor)) return true; + + return string.Equals(new FileInfo(descriptor).LinkTarget, resolvedPath, StringComparison.Ordinal); + } + + private static int DescriptorOf(SafeFileHandle handle) => (int)handle.DangerousGetHandle(); +} diff --git a/backend/src/CodeSpace.Core/Services/Agents/Recovery/IAgentSessionTranscriptCheckpointer.cs b/backend/src/CodeSpace.Core/Services/Agents/Recovery/IAgentSessionTranscriptCheckpointer.cs new file mode 100644 index 000000000..ad765d9ad --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Agents/Recovery/IAgentSessionTranscriptCheckpointer.cs @@ -0,0 +1,59 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Messages.Agents; + +namespace CodeSpace.Core.Services.Agents.Recovery; + +/// +/// Makes a RUNNING agent's resumable session transcript durable, so the conversation survives the loss of the host +/// running it. The seam the executor's observer ticks call; the abandon path on any other worker then continues from +/// what it stamped. +/// +/// This exists because the ordinary capture is END-of-run only: +/// AgentRunExecutor.CaptureSessionTranscriptAsync reads the file out of +/// LocalProcessRunner.ConfigHomePath(handle.SpoolDirectory) once the process has exited. That spool is on the +/// LAUNCHING host, so a worker that never launched the run cannot read it, and a host that dies takes it with it — +/// which is exactly why "a run whose launch host dies is continued by another host" was false. Everything else a +/// resume needs is already durable (agent_run.task_jsonb, the handle's WorkspaceBaseSha, a +/// PublishManifest row when the attempt pushed); the conversation was the hole. +/// +/// Best-effort by contract, and bounded by the SAME cap the end-of-run capture uses. It is dispatched from an +/// observer tick whose real job is flushing the run's events and advancing its spool offset, so nothing here may ever +/// be able to stop, slow or fail that tick: a checkpoint that cannot be taken is a run that retries cold, which is the +/// behaviour that existed before this seam. +/// +/// The marker is ON THE INTERFACE because DI here is marker SCANNING — CodeSpaceModule.RegisterDependency +/// walks the Core assembly for concrete classes assignable to and registers each +/// AsImplementedInterfaces. An implementation therefore needs no registration of its own, and cannot be +/// deployed unregistered by accident. Same shape as IArtifactRetentionWriter. +/// +public interface IAgentSessionTranscriptCheckpointer : IScopedDependency +{ + /// + /// Take one checkpoint of 's transcript, and return it when one was actually taken — + /// null when this attempt declined, which is an ordinary outcome. + /// + /// It declines when the file is absent, is not the file the caller's clamp resolved, holds no complete JSON + /// line yet, has NOT grown past , is over the + /// transcript byte cap, or loses the fenced row write. + /// + /// STATELESS, and that is load-bearing rather than tidy: the caller runs this OFF its drain tick in a DI + /// scope of its own (an agent's stdout and this upload would otherwise use one scoped DbContext + /// concurrently, which EF refuses — on the tick's statement as often as on this one, killing a healthy run for a + /// best-effort aid). A per-call instance can hold no cadence or growth watermark, so the caller owns both and + /// passes what this needs. + /// + Task CheckpointAsync(SessionTranscriptCheckpointRequest request, CancellationToken cancellationToken); +} + +/// +/// One checkpoint attempt's coordinates, as a record rather than a parameter list (the five-parameter cap), and +/// beside its interface exactly as ArtifactRetentionWriteRequest is: this is a call envelope between two Core +/// services, never a persisted or wire-facing DTO — the persisted fact is , +/// which lives in Messages. +/// +/// The run's team — the scope the artifact is stored under and every read is bound to. +/// The calling worker's observation token. The stamp is fenced to it (the brief said run id + epoch; the token is that plus the owner id, which is the fencing every other agent_run write on this path already uses), so a worker whose ownership was reclaimed cannot point a live run's recovery at its own stale conversation. +/// The absolute path of the live session transcript, already clamped within the run's config home by the caller (AgentRunExecutor.ResolveSessionTranscriptPath — the agent can write there, so the path is untrusted until that walk has run). +/// The harness-native session id naming that transcript. Stamped with the checkpoint because a checkpoint nothing can ADDRESS is not resumable — the retry hands this to the CLI as its --resume id. +/// The complete-prefix length of the last checkpoint taken of THIS path, or null when there is none. The growth watermark: bytes that have not changed carry no new conversation, so re-storing them buys nothing. The caller keys it on the path (a revise round opens a new config home whose transcript legitimately starts smaller) and passes null when the path moved. +public sealed record SessionTranscriptCheckpointRequest(Guid TeamId, AgentRunOwnerToken Owner, string TranscriptPath, string? SessionId, long? PreviousBytes); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs index 13619d4da..316f1dc0f 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs @@ -25,6 +25,7 @@ public sealed class ArtifactReferenceOracle : IArtifactReferenceOracle, IScopedD ("artifact_manifest", "content_artifact_id"), ("publish_manifest", "patch_artifact_id"), ("agent_run_event", "data_artifact_id"), + ("agent_run", "session_transcript_checkpoint_artifact_id"), ("workflow_run_model_call", "request_artifact_id"), ("workflow_run_model_call_attempt", "request_artifact_id"), ("workflow_run_model_call_attempt", "response_artifact_id"), diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs index cde55dbcf..c67552d6f 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactRetentionPolicy.cs @@ -38,6 +38,18 @@ public static class ArtifactRetentionPolicy public static readonly ArtifactRetentionRule AgentRunEventData = new(ArtifactRetentionClass.AgentRunEventData, TimeSpan.FromDays(7), TimeSpan.FromHours(24)); + /// + /// A mid-run session-transcript checkpoint. TWO HOURS, not seven days, and the short floor is the whole reason + /// the class exists: a run writes one of these per minute, each supersedes the last, and the run's terminal write + /// clears the column that references the survivor — so on the seven-day floor a single long run would hold every + /// superseded copy of a growing transcript for over a week. Two hours still sits far outside the window in which + /// the reference lands (the stamp is the next statement after the write) and far outside the window in which a + /// continuation reads it (an abandon follows the host's death within one liveness window), so the floor costs + /// nothing it protects. The quarantine stays the standard 24 h: the second, independent wait is unchanged. + /// + public static readonly ArtifactRetentionRule SessionTranscriptCheckpoint = + new(ArtifactRetentionClass.SessionTranscriptCheckpoint, TimeSpan.FromHours(2), TimeSpan.FromHours(24)); + private static readonly IReadOnlyDictionary Rules = new Dictionary { @@ -45,6 +57,7 @@ public static class ArtifactRetentionPolicy [SensitiveRecordPayload.Class] = SensitiveRecordPayload, [ModelCallBodyCapture.Class] = ModelCallBodyCapture, [AgentRunEventData.Class] = AgentRunEventData, + [SessionTranscriptCheckpoint.Class] = SessionTranscriptCheckpoint, }; /// The rule for , or null when the running policy does not register it — which the reaper reads as "cannot tell" and keeps. diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowEngine.cs b/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowEngine.cs index 982f37d67..4d79d2a03 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowEngine.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowEngine.cs @@ -3499,6 +3499,9 @@ private NodeRunContext BuildNodeRunContext(NodeExecution exec) PriorAttemptPayload = exec.PriorAttemptPayload, NodeId = exec.Node.Id, IncomingNodeIds = DirectPredecessorIds(exec), + // The node's own clamped retry policy, from the same RetryPlan the loop above runs on — so a node that + // decides anything by "will a failure of mine be retried" reads the engine's answer instead of guessing. + RetriesOnFailure = RetryPlan.From(exec.Node.Retry).RetriesOnFailure, }; } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowResumeAgentRunCompletionNotifier.cs b/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowResumeAgentRunCompletionNotifier.cs index a11f8ae42..2fc79ce64 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowResumeAgentRunCompletionNotifier.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Engine/WorkflowResumeAgentRunCompletionNotifier.cs @@ -137,6 +137,16 @@ internal static string BuildResumePayload(AgentRun run) sessionId = run.SessionId, sessionTranscript = result?.SessionTranscript, sessionTranscriptArtifactId = result?.SessionTranscriptArtifactId, + // 3c: the MID-RUN checkpoint, read off the ROW rather than the result — which is the whole point. An + // attempt whose host died has no result at all (the reconciler's abandon writes none), so the two keys + // above are null for exactly the population this one serves. Kept DISTINCT from them rather than merged, + // because they do not mean the same thing to the respawn: a captured transcript comes from an attempt + // that finished and left its workspace behind, while a checkpoint comes from one whose machine is gone + // and whose unpublished work went with it — and only the second owes the agent that sentence. + sessionTranscriptCheckpointArtifactId = run.SessionTranscriptCheckpointArtifactId, + sessionTranscriptCheckpointAt = run.SessionTranscriptCheckpointAt, + // The attempt this payload retires, so its successor can NAME it in a column instead of in prose. + agentRunId = run.Id, }, AgentJson.Options); } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs index 9f85bb9ef..e9ae64bd1 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs @@ -218,6 +218,12 @@ public Task RunAsync(NodeRunContext context, CancellationToken cance // force widens the code / document lanes it was written for and leaves research's base intact. Acceptance = acceptance, AcceptanceAuthority = ReadAcceptanceAuthority(context.Config), + // 3c: pay for durable session checkpoints only where a failure can actually be RETRIED. The engine's own + // answer, from this node's clamped retry policy — a single-attempt node has nobody to hand a restored + // conversation to, so checkpointing it would buy an artifact write a minute for a continuation that can + // never happen. This is the ONLY producer that sets it; benchmark cells, review children and supervisor + // units leave it false. + CheckpointSessionTranscript = context.RetriesOnFailure, }; task = ApplyRespawnEscalation(ApplyRespawnResumeHint(task, context.PriorAttemptPayload), context.PriorAttemptPayload); @@ -418,7 +424,7 @@ private static AgentTask ApplyRespawnResumeHint(AgentTask task, JsonElement? pri { if (priorAttemptPayload is not { } payload) return task; - var (repinned, honestyOwed) = RepinWorkspaceToPriorAttempt(task, payload); + var (repinned, honestyOwed, publishedBranch) = RepinWorkspaceToPriorAttempt(task, payload); // The gateway mangled the wire, not the model: the claude CLI dies in seconds with "Content block is not a // thinking block", before any turn. Warm-resuming that attempt re-sends the very transcript the mangled @@ -435,16 +441,46 @@ private static AgentTask ApplyRespawnResumeHint(AgentTask task, JsonElement? pri if (ReadOptionalString(payload, "sessionId") is not { } sessionId) return repinned; + // 3c: the prior attempt's MID-RUN checkpoint, present exactly when its host died holding the conversation + // (the reconciler's abandon leaves no result, so the captured-transcript keys above are null for precisely + // this population). Its existence is also the capability proof: only a harness the executor found to be an + // IAgentSessionTranscript, and whose file it located, ever produces one — so the node needs no registry to + // know this conversation can be restored. + var captured = ReadOptionalGuid(payload, "sessionTranscriptArtifactId"); + var checkpoint = ReadOptionalGuid(payload, "sessionTranscriptCheckpointArtifactId"); + + // A CAPTURED transcript always wins: the attempt that wrote it finished, so it is both newer and complete, + // where a checkpoint is by definition the conversation as of some moment before the end. Everything the + // checkpoint lane owes — the provenance stamps and the lost-host sentence — is therefore gated on the + // checkpoint actually being the ref that WON, never merely on one existing. + var fromCheckpoint = captured is null && checkpoint is { } resumeCheckpoint ? resumeCheckpoint : (Guid?)null; + var resumed = repinned with { ResumeFromSessionId = sessionId, RestoredTranscript = ReadOptionalString(payload, "sessionTranscript"), - RestoredTranscriptArtifactId = ReadOptionalGuid(payload, "sessionTranscriptArtifactId"), + // The REF, never the bytes: the executor resolves it just before invocation, so a long conversation + // never lands inline in this task's persisted envelope. + RestoredTranscriptArtifactId = captured ?? fromCheckpoint, + RestoredTranscriptIsCheckpoint = fromCheckpoint is not null, + ResumedFromCheckpointAt = fromCheckpoint is null ? null : ReadOptionalDateTime(payload, "sessionTranscriptCheckpointAt"), + ResumedFromAgentRunId = fromCheckpoint is null ? null : ReadOptionalGuid(payload, "agentRunId"), }; + // A checkpoint resume owes a DIFFERENT sentence from an ordinary warm retry, and owes it even when a branch + // was pinned: the prior attempt's machine is gone, so the conversation may describe turns the checkpoint + // never saw and edits the new sandbox does not contain. The ordinary line only covers "no branch to continue + // from", which is a smaller claim. + if (fromCheckpoint is not null) + return resumed with { Goal = AgentRetryContinuity.WithLostHostHint(resumed.Goal, publishedBranch, treeOwed: task.RepositoryId is not null) }; + return honestyOwed ? resumed with { Goal = AgentRetryContinuity.WithHonestNoContinuityHint(resumed.Goal) } : resumed; } + /// A payload timestamp, or null when the key is absent or is not a readable ISO-8601 instant. Same defensive shape as the other optional readers — a projection that changed underneath must degrade, never throw into a respawn. + private static DateTimeOffset? ReadOptionalDateTime(JsonElement bag, string key) => + bag.ValueKind == JsonValueKind.Object && bag.TryGetProperty(key, out var value) && value.ValueKind == JsonValueKind.String && value.TryGetDateTimeOffset(out var parsed) ? parsed : null; + /// /// Point this respawn's clone at the branch(es) the RETIRING attempt pushed — the quick lane's counterpart to the /// supervisor's ResolvePriorAttemptStagingAsync, which pins its retry to the prior attempt's @@ -467,16 +503,16 @@ private static AgentTask ApplyRespawnResumeHint(AgentTask task, JsonElement? pri /// is owed when there is no repository at all (an analysis-only run has no tree to have lost — the line would /// assert a git fact about a run that never had one). /// - private static (AgentTask Task, bool HonestyOwed) RepinWorkspaceToPriorAttempt(AgentTask task, JsonElement payload) + private static (AgentTask Task, bool HonestyOwed, string? PublishedBranch) RepinWorkspaceToPriorAttempt(AgentTask task, JsonElement payload) { - if (task.RepositoryId is not { } primaryId) return (task, false); + if (task.RepositoryId is not { } primaryId) return (task, false, null); var produced = ReadProducedBranches(payload); var primaryBranch = ReadOptionalString(payload, "branch") ?? produced.GetValueOrDefault(primaryId); var authored = task.Workspace?.Primary; var (related, relatedRepinned) = RepinRelatedRepositories(RelatedRepositories(task.Workspace), produced); - if (primaryBranch is null && !relatedRepinned) return (task, true); + if (primaryBranch is null && !relatedRepinned) return (task, true, null); var workspace = AgentWorkspaceAuthoring.ResolveAuthoredWorkspace(primaryId, related, primaryRef: primaryBranch ?? authored?.Ref, @@ -485,7 +521,7 @@ private static (AgentTask Task, bool HonestyOwed) RepinWorkspaceToPriorAttempt(A primaryPinnedSha: primaryBranch is null ? authored?.PinnedSha : null, primaryRefRecoverySha: primaryBranch is null ? authored?.RefRecoverySha : null); - return (task with { Workspace = workspace }, primaryBranch is null); + return (task with { Workspace = workspace }, primaryBranch is null, primaryBranch); } /// Each repository the prior attempt PUSHED a branch for, keyed by repository id — the multi-repo half of the pin (the top-level branch mirrors the primary's). An entry with no id or no produced branch contributes nothing: it pushed nothing to continue from. diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs index 2a22afbb2..f4446bd52 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs @@ -113,4 +113,17 @@ public sealed record NodeRunContext /// byte-identical no-op for every other node type. /// public JsonElement? PriorAttemptPayload { get; init; } + + /// + /// Whether this node's OWN retry policy allows more than one attempt — RetryPlan.RetriesOnFailure, read + /// off the node definition the engine is executing. False (the default) for a node with no policy, and off the + /// engine path entirely. + /// + /// It answers exactly one question and no more: "can a failure of this node buy another attempt at all?" + /// It deliberately does NOT say how many attempts are left, because the budget is a cross-cycle ledger the + /// engine counts from persisted attempt records and the node has no business re-deriving. agent.run reads + /// it to decide whether to pay for durable session checkpoints (3c): a node that can never be retried has nobody + /// to hand a restored conversation to, so checkpointing one is pure waste. + /// + public bool RetriesOnFailure { get; init; } } diff --git a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs index 7057ee22b..bf6f990d6 100644 --- a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs +++ b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs @@ -102,6 +102,55 @@ public sealed record AgentTask [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] public Guid? RestoredTranscriptArtifactId { get; init; } + /// + /// 3c: whether this run's harness session transcript is CHECKPOINTED to durable storage while it runs, so an + /// attempt whose host dies leaves a conversation a later one can continue. False (the default, and every envelope + /// persisted before this field) ⇒ no checkpoints, byte-identical to a pre-3c run. + /// + /// An opt-in rather than a default, because a checkpoint nobody will consume is pure waste: it costs a + /// whole-file read and an artifact write per minute, per running agent, per worker. Only a producer whose failed + /// attempt can actually be RETRIED sets it — today that is agent.run for a node whose own retry policy + /// allows more than one attempt. The benchmark lanes (one attempt per cell by protocol), review children and + /// supervisor units leave it false. + /// + /// [JsonIgnore(WhenWritingDefault)] so an envelope that did not opt in adds nothing to task_json. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)] + public bool CheckpointSessionTranscript { get; init; } + + /// + /// 3c: whether is a mid-run CHECKPOINT rather than a completed + /// attempt's captured transcript. The two refs are resolved under opposite policies and the difference is the + /// point: a captured transcript was written by an attempt that FINISHED, so an unreadable one is a real fault + /// and the launch fails closed rather than silently cold-starting a named session. A checkpoint is best-effort + /// by construction — its blob may have been reaped, or its destination may be unreachable — and failing closed + /// on one would spend the very retry attempt it exists to improve. Unreadable ⇒ the attempt runs COLD and says + /// so. [JsonIgnore(WhenWritingDefault)]. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingDefault)] + public bool RestoredTranscriptIsCheckpoint { get; init; } + + /// + /// 3c: the agent run this attempt was staged to continue from a session checkpoint — the prior attempt whose + /// host died. Promoted to agent_run.resumed_from_agent_run_id at creation (the same way + /// is), so "which attempt took over from which" is a column rather than a + /// sentence buried in an event. Cleared in-memory at launch when the checkpoint turns out to be unreadable, so + /// the envelope the agent actually runs under claims nothing it did not get. Null for every ordinary dispatch. + /// [JsonIgnore(WhenWritingNull)]. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] + public Guid? ResumedFromAgentRunId { get; init; } + + /// + /// 3c: when this attempt was minted as the CONTINUATION of a checkpointed run whose host died — null for every + /// ordinary dispatch. Rides the task because it must reach the launch that stamps + /// SandboxConfinement.ResumedFromCheckpointAt, which is the permanent per-run record (the runner handle it + /// would otherwise live on is reaped 24h after the run goes terminal). + /// [JsonIgnore(WhenWritingNull)]. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] + public DateTimeOffset? ResumedFromCheckpointAt { get; init; } + /// /// P3 (D1): the supervisor SUBTASK id this agent was spawned for — the linking key for retry-resume. When the /// supervisor RETRIES a subtask, the producer finds the prior attempt at the SAME subtask in the same run and diff --git a/backend/src/CodeSpace.Messages/Agents/SandboxConfinement.cs b/backend/src/CodeSpace.Messages/Agents/SandboxConfinement.cs index 255850f91..cd0f6ca1a 100644 --- a/backend/src/CodeSpace.Messages/Agents/SandboxConfinement.cs +++ b/backend/src/CodeSpace.Messages/Agents/SandboxConfinement.cs @@ -69,4 +69,17 @@ public sealed record SandboxConfinement /// unbrokered run — those have nothing to lose here, and say so by saying nothing. /// public bool ModelCredentialLeaseLost { get; init; } + + /// + /// When this attempt was minted as the CONTINUATION of an earlier run whose host died, from that run's durable + /// session-transcript checkpoint — null (the default) for every ordinary launch. + /// + /// Recorded here for the same reason the rest of this record is: it is a permanent fact about the run that + /// must outlive the runner handle the spool reaper nulls. And it must be said out loud, because a resumed attempt + /// is NARROWER than the one it continues: only the CONVERSATION was durable. The lost host's working tree is + /// gone, so every edit the earlier attempt had not published is gone with it, and the resumed agent is told so + /// (AgentRetryContinuity.HonestNoContinuityHint) rather than left to infer it from a transcript that + /// describes files its sandbox does not contain. + /// + public DateTimeOffset? ResumedFromCheckpointAt { get; init; } } diff --git a/backend/src/CodeSpace.Messages/Agents/SessionTranscriptCheckpoint.cs b/backend/src/CodeSpace.Messages/Agents/SessionTranscriptCheckpoint.cs new file mode 100644 index 000000000..0b02faebb --- /dev/null +++ b/backend/src/CodeSpace.Messages/Agents/SessionTranscriptCheckpoint.cs @@ -0,0 +1,16 @@ +namespace CodeSpace.Messages.Agents; + +/// +/// A run's resumable CLI session transcript, made durable WHILE the run is still going — the one fact a host that +/// dies mid-run leaves behind that another host can continue from. +/// +/// Before this existed the transcript was captured only at run END, from the launching host's own spool +/// (AgentRunExecutor.CaptureSessionTranscriptAsync), so a run whose host vanished left nothing resumable +/// anywhere and the only recovery was a cold restart. The artifact this names is uploaded from the live config home +/// on the observer's own checkpoint ticks, and the run row keeps the reference — which is what makes the artifact +/// reachable from a DIFFERENT worker after the launching one is gone. +/// +/// The stored transcript's workflow_artifact id, referenced by agent_run.session_transcript_checkpoint_artifact_id so the artifact reaper's oracle can see it. +/// When this checkpoint was taken — the clock the next checkpoint's cadence is measured from, and the honesty stamp a resumed attempt cites. +/// The transcript's size at capture, in bytes. The growth signal for the next tick, and the number an operator reads to tell a checkpoint that captured a conversation from one that captured an empty file. +public sealed record SessionTranscriptCheckpoint(Guid ArtifactId, DateTimeOffset At, long Bytes); diff --git a/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs b/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs index 0c7e2a190..72593072f 100644 --- a/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs +++ b/backend/src/CodeSpace.Messages/Artifacts/ArtifactRetention.cs @@ -30,6 +30,18 @@ public enum ArtifactRetentionClass /// event id is minted before the artifact write and is the declaration's diagnostic holder identity. /// AgentRunEventData = 4, + + /// + /// A mid-run resumable session transcript, referenced only by + /// agent_run.session_transcript_checkpoint_artifact_id. Its own class rather than + /// because its lifetime is nothing like an event payload's: exactly one + /// checkpoint per run is ever useful (the newest), each one supersedes the last, and the run's terminal write + /// clears the column — so every intermediate is garbage within minutes and the survivor within hours. Sharing the + /// event class's seven-day floor would hold roughly a gigabyte of superseded transcript per long run for over a + /// week, and the growth gate that makes checkpoints worth taking is exactly what stops the content-addressed + /// store deduplicating them. + /// + SessionTranscriptCheckpoint = 5, } /// diff --git a/backend/tests/CodeSpace.E2ETests/Agents/RealModelSessionResumeE2ETests.cs b/backend/tests/CodeSpace.E2ETests/Agents/RealModelSessionResumeE2ETests.cs index 2e61ddf15..12a45ac96 100644 --- a/backend/tests/CodeSpace.E2ETests/Agents/RealModelSessionResumeE2ETests.cs +++ b/backend/tests/CodeSpace.E2ETests/Agents/RealModelSessionResumeE2ETests.cs @@ -131,6 +131,177 @@ public async Task A_real_claude_agent_recalls_a_codeword_from_the_restored_conve $"{Provider} '{model}': the resumed agent {(recalled ? "RECALLED" : "did NOT recall")} the codeword {codeword} from the restored conversation — the P3 continue chain {(recalled ? "held end-to-end against the live model" : "did not surface the prior context")}"); } + /// + /// 3c — the CROSS-HOST arm: the conversation is taken from a MID-RUN checkpoint, the way a run whose host dies + /// leaves one behind, and a second agent continues from exactly those bytes. + /// + /// What makes it a different experiment from the arm above, and why both are needed: that one reads the + /// session file AFTER the process exits, from the host that ran it — which is precisely what a dead host cannot + /// offer. This one takes the transcript while the agent is STILL RUNNING, through the same two production calls + /// the executor's observer tick makes (IAgentSessionTranscript.SessionTranscriptRelativePath to locate it + /// and AgentRunExecutor.ResolveSessionTranscriptPath to clamp it inside the config home), then KILLS the + /// agent before it can finish — the host loss — and asks a fresh agent, in a fresh config home, to recall the + /// codeword from those mid-run bytes alone. + /// + /// A mid-run snapshot that does not yet contain the codeword turn is INFRA, not a capability miss: the + /// experiment could not be staged, so there is nothing to measure. Same reporting contract as the arm above — + /// informational, gating only a . + /// + /// EXACTLY what this arm proves, and what it does not. It proves ONE thing no other tier can: that a + /// transcript located mid-run through the production locate + clamp, cut at a line boundary, is bytes a real + /// claude actually resumes from — that the live model USES the pre-loss context. That is a statement about + /// the CLI and the model, not about this codebase's plumbing. + /// + /// It therefore CANNOT go red for anything else 3c wrote, and must not be read as coverage for it. It drives + /// Process.Start directly, so it never touches the executor's tick, the checkpointer, the artifact store, + /// the fenced row stamp, the reconciler's abandon or the agent node's respawn — all of which are pinned, with + /// their own mutations, in AgentSessionTranscriptCheckpointerTests, + /// AgentRunSessionCheckpointFlowTests, AgentCodeNodeTests and AgentNodeFlowTests. It also + /// cannot FAIL the lane on a recall miss (the verdict is informational by the same ruling as its sibling). What + /// it assumes: that the harness's locate and the executor's clamp are the ones this file calls — which they are, + /// at the call sites in , and which is the only reason a layout change in + /// either would surface here at all. + /// + [SkippableFact] + public async Task A_real_claude_agent_recalls_a_codeword_from_a_mid_run_checkpoint() + { + var baseUrl = Environment.GetEnvironmentVariable(RealModelSupervisorDecisionFlowTests.BaseUrlEnvVar); + var apiKey = Environment.GetEnvironmentVariable(RealModelSupervisorDecisionFlowTests.ApiKeyEnvVar); + var model = Environment.GetEnvironmentVariable(RealModelSupervisorDecisionFlowTests.ModelIdEnvVar); + + var present = new[] { baseUrl, apiKey, model }.Count(v => !string.IsNullOrWhiteSpace(v)); + if (present == 0) throw RealModelGate.ReportSkipped(Provider, "CODESPACE_LLM_* absent (fork/local — no live model)"); + present.ShouldBe(3, "CODESPACE_LLM_* is partially configured — set all three (base url / api key / model id) or none."); + + if (OperatingSystem.IsWindows()) return; + if (!await ClaudeReadyAsync()) throw RealModelGate.ReportSkipped(Provider, "the `claude` coding-agent CLI is not installed — the checkpoint-resume gate needs the harness binary (skip ≠ pass)"); + + try + { + await RealModelGate.AssessLiveAsync(Provider, () => RealModelFormatFaultRepair.WithColdRestageAsync(() => DriveCheckpointSourcedResumeAsync(baseUrl!, apiKey!, model!))); + } + finally + { + foreach (var dir in _tempDirs) + try { Directory.Delete(dir, recursive: true); } catch { /* best-effort cleanup */ } + } + } + + /// + /// One COLD-staged measurement of the cross-host chain: checkpoint a LIVE agent's transcript, kill it, continue + /// from the checkpoint. Every resource is minted per call, so a re-stage measures the same configuration. + /// + private async Task<(RealModelOutcome Outcome, string Note)> DriveCheckpointSourcedResumeAsync(string baseUrl, string apiKey, string model) + { + var codeword = "CODESPACE-" + Guid.NewGuid().ToString("N")[..8].ToUpperInvariant(); + var env = Harness.ProjectToEnv(new ResolvedModelCredential { Provider = Provider, ApiKey = apiKey, BaseUrl = baseUrl }); + var cwd = await ResolveRealPathAsync(NewWorkspace()); + var liveConfig = NewDir(); + + // A LONG second instruction so the agent is still working when the checkpoint is taken and the kill lands — + // the host loss must interrupt a live conversation, not race a process that already exited. + var goal = $"Remember this codeword, I will ask you to recall it: {codeword}. Reply 'ok', then count slowly from 1 to 40, one number per message."; + var live = await RunClaudeUntilCheckpointedAsync(Harness.BuildInvocation(Task(cwd, model, env, goal)), liveConfig, cwd, codeword); + + if (live.Checkpoint is not { Length: > 0 } checkpoint) + throw new AgentExecutionInfraException($"no mid-run checkpoint containing the codeword could be taken before the live claude run ended (sessionId={live.SessionId ?? "null"}, snapshots={live.Snapshots}, error={live.Error ?? "none"}) — gateway/exec infra, not a recall verdict"); + + // ── CONTINUE on a FRESH config home: the dead host's spool is gone, so only the checkpoint's bytes travel. ── + var continueTask = Task(cwd, model, env, "What was the codeword I told you to remember? Reply with ONLY the codeword, nothing else.") + with { ResumeFromSessionId = live.SessionId, RestoredTranscript = checkpoint }; + var resumed = await RunClaudeAsync(Harness.BuildInvocation(continueTask), NewDir()); + var resumedResult = Harness.BuildResult(ParseAll(resumed.Stdout), resumed.ExitCode, ""); + + if (resumedResult.Status != AgentRunStatus.Succeeded) + throw new AgentExecutionInfraException($"the checkpoint-resumed claude run did not complete (status={resumedResult.Status}, error={resumedResult.Error ?? "none"}) — gateway/exec infra, not a recall verdict"); + + var modelReply = string.Join("\n", ParseAll(resumed.Stdout) + .Where(e => e.Kind is AgentEventKind.AssistantMessage or AgentEventKind.Completed or AgentEventKind.FinalSummary) + .Select(e => e.Text)); + var recalled = modelReply.Contains(codeword, StringComparison.OrdinalIgnoreCase); + + return (recalled ? RealModelOutcome.Drove : RealModelOutcome.CapabilityMiss, + $"{Provider} '{model}': after a {checkpoint.Length}-byte MID-RUN checkpoint and a kill, the continued agent {(recalled ? "RECALLED" : "did NOT recall")} the codeword {codeword} — cross-host recovery {(recalled ? "held end-to-end against the live model" : "did not surface the pre-loss context")}"); + } + + /// + /// Run the live agent, checkpoint its session transcript WHILE it runs, and kill it — the host loss, staged. + /// + /// The locate is the production pair and nothing else: the harness's own + /// SessionTranscriptRelativePath (the CLI's cwd-encoded layout, which this test must never restate) and + /// AgentRunExecutor.ResolveSessionTranscriptPath (the symlink/traversal clamp, which applies to a LIVE read + /// exactly as it does to a post-exit one). The session id comes off the live stream through the harness's own + /// parse, as the executor's fold gets it. + /// + /// Only a snapshot ending in a newline is kept: the CLI appends whole JSON lines, so a partial tail means + /// the file was caught mid-write and restoring it would hand the next CLI a corrupt session. + /// + private static async Task<(string? SessionId, string? Checkpoint, int Snapshots, string? Error)> RunClaudeUntilCheckpointedAsync(SandboxSpec spec, string configDir, string cwd, string codeword) + { + LocalProcessRunner.WriteConfigHomeFiles(spec.ConfigHomeFiles, configDir); + + var psi = new ProcessStartInfo { FileName = spec.Command, WorkingDirectory = spec.WorkingDirectory, RedirectStandardOutput = true, RedirectStandardError = true, RedirectStandardInput = true, UseShellExecute = false }; + foreach (var arg in spec.Args) psi.ArgumentList.Add(arg); + psi.Environment[ClaudeCodeHarness.ConfigDirEnvVar] = configDir; + foreach (var (k, v) in spec.Environment) psi.Environment[k] = v; + + using var proc = Process.Start(psi)!; + proc.StandardInput.Close(); + + string? sessionId = null; + string? checkpoint = null; + var snapshots = 0; + var deadline = DateTime.UtcNow.AddSeconds(180); + + while (!proc.HasExited && DateTime.UtcNow < deadline) + { + if (await proc.StandardOutput.ReadLineAsync() is not { } line) break; + + sessionId ??= AgentSessionIdReader.TryRead(Harness.ParseEvents(line).ToList()); + + if (sessionId is null) continue; + + if (TryReadLiveTranscript(configDir, cwd, sessionId) is not { } snapshot) continue; + + snapshots++; + + // The experiment needs the codeword turn to be INSIDE the checkpoint — that is the fact the continued + // agent is asked to recall. Anything earlier is a snapshot of a conversation that never heard it. + if (!snapshot.Contains(codeword, StringComparison.Ordinal)) continue; + + checkpoint = snapshot; + break; + } + + // The host loss: the process is killed where it stands, so nothing it would have written after the checkpoint + // — including its post-exit session file — can reach the continuation. + try { proc.Kill(entireProcessTree: true); } catch { /* already gone */ } + try { await proc.WaitForExitAsync(new CancellationTokenSource(TimeSpan.FromSeconds(15)).Token); } catch { /* best-effort */ } + + return (sessionId, checkpoint, snapshots, checkpoint is null ? await proc.StandardError.ReadToEndAsync() : null); + } + + /// The live transcript through the PRODUCTION locate + clamp, or null when it is not addressable yet or was caught mid-write. Never a hand-built path: the cwd encoding is the harness's to own. + private static string? TryReadLiveTranscript(string configDir, string cwd, string sessionId) + { + if (((IAgentSessionTranscript)Harness).SessionTranscriptRelativePath(configDir, cwd, sessionId) is not { } relative) return null; + + if (AgentRunExecutor.ResolveSessionTranscriptPath(configDir, relative) is not { } path || !File.Exists(path)) return null; + + try + { + using var stream = new FileStream(path, FileMode.Open, FileAccess.Read, FileShare.ReadWrite | FileShare.Delete); + using var reader = new StreamReader(stream); + var text = reader.ReadToEnd(); + + return text.EndsWith('\n') ? text : null; + } + catch (IOException) + { + return null; + } + } + private static AgentTask Task(string cwd, string model, IReadOnlyDictionary env, string goal) => new() { Goal = goal, diff --git a/backend/tests/CodeSpace.E2ETests/Workflows/RerunFromNodeAgentFlowTests.cs b/backend/tests/CodeSpace.E2ETests/Workflows/RerunFromNodeAgentFlowTests.cs index fee92abcf..ae8adc19d 100644 --- a/backend/tests/CodeSpace.E2ETests/Workflows/RerunFromNodeAgentFlowTests.cs +++ b/backend/tests/CodeSpace.E2ETests/Workflows/RerunFromNodeAgentFlowTests.cs @@ -229,6 +229,50 @@ public async Task Rerun_from_an_agent_with_a_corrupt_prior_result_cold_starts_wi task.ResumeFromSessionId.ShouldBeNull("a corrupt prior result is not resumable — cold-start, never a hard failure"); } + [Fact] + public async Task An_in_flight_sibling_at_one_hop_cannot_mask_the_resumable_attempt() + { + if (OperatingSystem.IsWindows()) return; + + // The lookup walks ancestors nearest-first and takes the first RESUMABLE candidate at each hop. It used to + // take the first candidate at each hop, full stop — and "resumable" needs BOTH a session id and a transcript + // (TryResumable, both-or-neither), so a row with an id and no transcript could win the hop and mask the + // attempt that actually had one. 3c makes that reachable rather than theoretical: a run now stamps its + // session id at its first MID-RUN checkpoint instead of only at completion, so an in-flight sibling at the + // same cell carries one with no result_jsonb at all. + // MUTATION: restore `candidates.FirstOrDefault(c => c.WorkflowRunId == runId)` → the masking row wins and + // this reds. + using var cli = new SubtaskAwareFakeCli(); + var (teamId, _, originalRunId) = await RunOriginalChainAsync(); + await SeedCapturedSessionAsync(originalRunId, "b", "sess-resumable", inlineTranscript: "B-transcript\n", transcriptArtifactId: null); + await SeedInFlightSiblingAsync(originalRunId, "b", teamId, "sess-in-flight"); + + using var scope = _fixture.BeginScope(); + + (await scope.Resolve().FindResumableSessionAsync(teamId, originalRunId, "b", "", CancellationToken.None)) + .ShouldNotBeNull("an in-flight sibling carrying only a session id must not mask the attempt that is genuinely resumable") + .SessionId.ShouldBe("sess-resumable"); + } + + /// A SECOND run at the same cell of the same workflow run, Running with a mid-run session id and no result — exactly what a checkpointed run looks like before it lands. + private async Task SeedInFlightSiblingAsync(Guid workflowRunId, string nodeId, Guid teamId, string sessionId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + var sibling = await db.AgentRun.AsNoTracking().SingleAsync(r => r.WorkflowRunId == workflowRunId && r.NodeId == nodeId); + + db.AgentRun.Add(new Core.Persistence.Entities.AgentRun + { + Id = Guid.NewGuid(), TeamId = teamId, WorkflowRunId = workflowRunId, NodeId = nodeId, IterationKey = sibling.IterationKey, + Harness = sibling.Harness, Status = Messages.Enums.AgentRunStatus.Running, SessionId = sessionId, TaskJson = sibling.TaskJson, + // NEWER than the resumable attempt, so the hop's newest-first order puts it FIRST — which is what makes + // this a real masking test rather than one that depends on whatever order the plan emitted. + CreatedDate = sibling.CreatedDate.AddMinutes(1), + }); + await db.SaveChangesAsync(); + } + [Fact] public async Task Find_resumable_session_is_team_scoped_and_never_crosses_teams() { diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentNodeFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentNodeFlowTests.cs index 6ef698bd1..9fe5357b5 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentNodeFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentNodeFlowTests.cs @@ -716,6 +716,183 @@ public async Task A_transient_agent_failure_respawns_a_fresh_agent_and_the_retry } } + // ── 3c: an agent node whose HOST died ─────────────────────────────────────── + + [Fact] + public async Task A_host_loss_abandon_of_an_agent_node_buys_a_second_attempt() + { + // The FACT the whole of 3c rests on, pinned before anything is built on it: a run whose host died is + // abandoned by the REAL reconciler (Failed + the abandoned error, no result at all), and the agent.run node + // reads that as a RETRYABLE failure — so the node's own retry policy stages a second agent run without any + // new machinery. 3c's job is not to buy that attempt; it is to make it WARM. + // MUTATION: make the abandon deterministic in AgentCodeNode's verdict (add the abandoned status to the + // non-retryable set) → no second run → red, and every warm-retry test below becomes unreachable. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var workflowId = await CreateWorkflowAsync(teamId, userId, RetryingAgentNodeDefinition(maxAttempts: 2)); + var runId = await WorkflowsTestSeed.SeedManualRunAsync(_fixture, workflowId, teamId); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + jobClient.AutoExecute = false; + + try + { + await RunEngineAsync(runId); + var lostAgent = await GetAgentRunIdAsync(runId); + + await LoseTheHostAsync(lostAgent, checkpointArtifactId: null, sessionId: null); + await ReconcileAsync(); + + using (var mid = _fixture.BeginScope()) + { + var abandoned = await mid.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == lostAgent); + abandoned.Status.ShouldBe(AgentRunStatus.Failed, "the reconciler terminalizes a run whose host never came back"); + abandoned.Error.ShouldNotBeNull().ShouldContain("abandoned", Case.Insensitive); + abandoned.ResultJson.ShouldBeNull("an abandon has no result — which is why a warm retry cannot come from the captured-transcript keys"); + } + + await RunEngineAsync(runId); + + using var verify = _fixture.BeginScope(); + var agents = await verify.Resolve().AgentRun.AsNoTracking().Where(r => r.WorkflowRunId == runId).ToListAsync(); + + agents.Count.ShouldBe(2, $"a host-loss abandon must buy the node's second attempt; the run has {agents.Count} agent run(s)"); + (await verify.Resolve().WorkflowRun.AsNoTracking().SingleAsync(r => r.Id == runId)).Status + .ShouldBe(WorkflowRunStatus.Suspended, "the fresh attempt parks the run again — the host loss was absorbed"); + } + finally + { + jobClient.AutoExecute = true; + } + } + + [Fact] + public async Task A_retried_agent_node_restores_the_lost_hosts_checkpoint() + { + // The consumer, end to end on the production wiring: the abandoned run's MID-RUN checkpoint columns are read + // by the completion notifier off the ROW (an abandon leaves no result), ride the wait boundary as + // PriorAttemptPayload, and land on the SECOND agent run's own persisted task. + // MUTATION: drop the checkpoint branch from AgentCodeNode.ApplyRespawnResumeHint, or the two checkpoint keys + // from WorkflowResumeAgentRunCompletionNotifier.BuildResumePayload → the fresh task is cold → red. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var workflowId = await CreateWorkflowAsync(teamId, userId, RetryingAgentNodeDefinition(maxAttempts: 2)); + var runId = await WorkflowsTestSeed.SeedManualRunAsync(_fixture, workflowId, teamId); + var checkpointArtifactId = Guid.NewGuid(); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + jobClient.AutoExecute = false; + + try + { + await RunEngineAsync(runId); + var lostAgent = await GetAgentRunIdAsync(runId); + + using (var staged = _fixture.BeginScope()) + { + var task = JsonSerializer.Deserialize((await staged.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == lostAgent)).TaskJson, AgentJson.Options)!; + task.CheckpointSessionTranscript.ShouldBeTrue("a node whose policy allows a second attempt opts in to checkpointing — otherwise nothing below could exist"); + } + + await LoseTheHostAsync(lostAgent, checkpointArtifactId, sessionId: "sess-lost-host"); + await ReconcileAsync(); + await RunEngineAsync(runId); + + using var verify = _fixture.BeginScope(); + var db = verify.Resolve(); + var retry = await db.AgentRun.AsNoTracking().Where(r => r.WorkflowRunId == runId && r.Id != lostAgent).SingleAsync(); + var resumed = JsonSerializer.Deserialize(retry.TaskJson, AgentJson.Options)!; + + resumed.ResumeFromSessionId.ShouldBe("sess-lost-host", "the CLI is told WHICH conversation to resume — a transcript with no id names nothing"); + resumed.RestoredTranscriptArtifactId.ShouldBe(checkpointArtifactId, "the checkpoint rides as a REF the executor resolves just before invocation"); + resumed.ResumedFromCheckpointAt.ShouldNotBeNull("the launch stamps this onto the run's permanent confinement record"); + resumed.ResumedFromAgentRunId.ShouldBe(lostAgent, "which attempt took over from which is a column, not prose"); + retry.ResumedFromAgentRunId.ShouldBe(lostAgent, "and the task's provenance is promoted onto the row, like AgentDefinitionId"); + resumed.Goal.ShouldContain("machine running your previous attempt was lost", Case.Sensitive, + "a restored conversation describes a working tree this sandbox does not have, and the agent must be told rather than left to infer it"); + } + finally + { + jobClient.AutoExecute = true; + } + } + + [Fact] + public async Task A_host_loss_with_no_checkpoint_is_retried_cold_and_claims_nothing() + { + // The same host loss from a run that checkpointed nothing — its harness had no addressable session + // transcript, or it died before its first checkpoint. There is nothing to restore, and the fresh attempt must + // say so by saying NOTHING: a task that claims a restored conversation it does not have is worse than one + // that admits it is starting over. + // MUTATION: stamp the lost-host block (or a transcript ref) unconditionally → red. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var workflowId = await CreateWorkflowAsync(teamId, userId, RetryingAgentNodeDefinition(maxAttempts: 2)); + var runId = await WorkflowsTestSeed.SeedManualRunAsync(_fixture, workflowId, teamId); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + jobClient.AutoExecute = false; + + try + { + await RunEngineAsync(runId); + var lostAgent = await GetAgentRunIdAsync(runId); + + await LoseTheHostAsync(lostAgent, checkpointArtifactId: null, sessionId: "sess-uncheckpointed"); + await ReconcileAsync(); + await RunEngineAsync(runId); + + using var verify = _fixture.BeginScope(); + var retry = await verify.Resolve().AgentRun.AsNoTracking().Where(r => r.WorkflowRunId == runId && r.Id != lostAgent).SingleAsync(); + var resumed = JsonSerializer.Deserialize(retry.TaskJson, AgentJson.Options)!; + + resumed.RestoredTranscriptArtifactId.ShouldBeNull("no checkpoint was taken, so there is no conversation to restore"); + resumed.ResumedFromCheckpointAt.ShouldBeNull(); + resumed.ResumedFromAgentRunId.ShouldBeNull(); + retry.ResumedFromAgentRunId.ShouldBeNull(); + resumed.Goal.ShouldNotContain("machine running your previous attempt was lost", Case.Sensitive, "nothing may assert a restored conversation this attempt does not have"); + } + finally + { + jobClient.AutoExecute = true; + } + } + + /// + /// Make a staged agent run look like one whose HOST died: Running, its lease lapsed, and a durable handle minted + /// on a host that never came back whose own wall clock has passed — the exact shape + /// AgentRunReconcilerService.DeferToTheMintingHostAsync stops deferring and abandons. The checkpoint + /// columns are stamped the way the observer's tick would have. + /// + private async Task LoseTheHostAsync(Guid agentRunId, Guid? checkpointArtifactId, string? sessionId) + { + var spoolDirectory = Path.Combine(Path.GetTempPath(), "cs-host-loss-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(spoolDirectory); + + var handle = new SandboxHandle + { + Kind = "local", ProcessId = 0x7FFFFFFF, LaunchHost = "a-host-that-never-came-back", + SpoolDirectory = spoolDirectory, Deadline = DateTimeOffset.UtcNow.AddMinutes(-1), + }; + var stale = DateTimeOffset.UtcNow - TimeSpan.FromMinutes(20); + + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var run = await db.AgentRun.SingleAsync(r => r.Id == agentRunId); + + run.Status = AgentRunStatus.Running; + run.FenceEpoch = 1; + run.StartedAt = stale; + run.HeartbeatAt = stale; + run.LeaseExpiresAt = stale + AgentRunLiveness.Window; + run.RunnerHandleJson = JsonSerializer.Serialize(handle, AgentJson.Options); + run.SessionId = sessionId; + run.SessionTranscriptCheckpointArtifactId = checkpointArtifactId; + run.SessionTranscriptCheckpointAt = checkpointArtifactId is null ? null : DateTimeOffset.UtcNow.AddMinutes(-2); + + await db.SaveChangesAsync(); + } + [Fact] public async Task P2_3_a_respawn_warm_resumes_the_prior_attempts_captured_session() { diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunSessionCheckpointFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunSessionCheckpointFlowTests.cs new file mode 100644 index 000000000..a960b4d66 --- /dev/null +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunSessionCheckpointFlowTests.cs @@ -0,0 +1,663 @@ +using System.Text; +using System.Text.Json; +using Autofac; +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Authority; +using CodeSpace.Core.Services.Agents.Harnesses.Claude; +using CodeSpace.Core.Services.Agents.Recovery; +using CodeSpace.Core.Services.Agents.Recovery.Checkpoints; +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using CodeSpace.Core.Services.Workflows.Artifacts; +using CodeSpace.Core.Services.Workflows.Artifacts.Retention; +using Autofac.Extensions.DependencyInjection; +using CodeSpace.IntegrationTests.Infrastructure; +using Microsoft.Extensions.DependencyInjection; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Artifacts; +using CodeSpace.Messages.Authorization; +using CodeSpace.Messages.Constants; +using CodeSpace.Messages.Enums; +using Microsoft.EntityFrameworkCore; +using Shouldly; + +namespace CodeSpace.IntegrationTests.Workflows; + +/// +/// 3c PRODUCE side — making a running agent's conversation survive the loss of its host. +/// +/// Until now the resumable session transcript was captured only at run END, out of the LAUNCHING host's own +/// spool (AgentRunExecutor.CaptureSessionTranscriptAsync reads +/// LocalProcessRunner.ConfigHomePath(handle.SpoolDirectory)). A worker that never launched the run cannot read +/// that spool, and a host that dies takes it with it — so an abandoned agent node was retried COLD even though the +/// node's retry policy had already bought it a fresh attempt. +/// +/// Everything here drives the REAL executor, the REAL drain tick, the REAL checkpointer, the REAL artifact +/// store and the REAL fenced row write, and asserts on the rows THIS test owns. The CONSUME side — what the fresh +/// attempt does with a checkpoint — lives in AgentCodeNodeTests and AgentNodeCheckpointRetryFlowTests. +/// +[Collection(PostgresCollection.Name)] +[Trait("Category", "Integration")] +public sealed class AgentRunSessionCheckpointFlowTests : IDisposable +{ + private readonly PostgresFixture _fixture; + private readonly List _spoolDirs = []; + + public AgentRunSessionCheckpointFlowTests(PostgresFixture fixture) { _fixture = fixture; } + + public void Dispose() + { + foreach (var dir in _spoolDirs) + try { if (Directory.Exists(dir)) Directory.Delete(dir, recursive: true); } catch { /* best-effort */ } + } + + [Theory] + [InlineData(false)] // scratch workspace — the executor mounts one, and its Repositories are empty + [InlineData(true)] // an authored cwd — no workspace is resolved at all, and the envelope's own path stands + public async Task A_running_agent_checkpoints_its_live_transcript_through_the_real_tick(bool authoredWorkingDirectory) + { + // The produce half, driven end to end: the REAL executor, the REAL drain tick, the REAL checkpointer, the + // REAL artifact store and the REAL fenced row write. Nothing here seeds a column or computes a path. + // + // MUTATION this pins: pass the PRIMARY REPO's directory to the tick instead of the cwd the CLI actually got + // (which is what AgentRunExecutor.cs did before this rework). Claude keys its session file on the cwd + // (projects//.jsonl), so wherever the two differ the located file never exists and the run + // silently never checkpoints. BOTH arms here are no-primary-repo shapes, which is what makes the primary-repo + // directory null and both arms red. The multi-repo shape the review named — a cwd at the workspace ROOT while + // the primary repo sits in a subdirectory, so both paths are non-null and different — is the same read one + // step further along, and is not reproducible here without real clones; it is named rather than faked. + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory); + var harness = new TranscriptWritingHarness(run.SessionId); + + await ExecuteAsync(run.RunId, harness); + + using var verify = _fixture.BeginScope(); + var db = verify.Resolve(); + var row = await db.AgentRun.AsNoTracking().SingleAsync(r => r.Id == run.RunId); + + harness.WrittenAt.ShouldNotBeNull($"the harness never wrote a transcript, so this arm proves nothing — cwd was {harness.ObservedWorkingDirectory ?? "null"}"); + + // The terminal write RELEASES the checkpoint (it exists for a run whose host dies, and this one landed), so + // the LIVE stamp is read from the event the run recorded while it was still going, plus the artifact itself. + var declared = await db.WorkflowArtifactRetention.AsNoTracking() + .Where(d => d.TeamId == team.TeamId && d.RetentionClass == nameof(ArtifactRetentionClass.SessionTranscriptCheckpoint) && d.HolderId == run.RunId) + .ToListAsync(); + + declared.ShouldNotBeEmpty($"the live tick must have checkpointed the transcript the CLI wrote at {harness.WrittenAt}; nothing was stored for run {run.RunId}"); + declared.ShouldAllBe(d => d.HolderKind == ArtifactSessionTranscriptCheckpointer.CheckpointHolderKind, + "the declaration names the RUN as its holder, which is what a later collection is diagnosed from"); + + var bytes = (await verify.Resolve().GetBytesAsync(team.TeamId, declared[0].ArtifactId, CancellationToken.None)).ShouldNotBeNull(); + Encoding.UTF8.GetString(bytes.Bytes).ShouldContain(run.SessionId, Case.Sensitive, + "the stored bytes are the agent's own live session file, not an empty or fabricated one"); + + row.SessionId.ShouldBe(run.SessionId, "the checkpoint stamps the session id too — a checkpoint no CLI can be pointed at is not resumable"); + row.SessionTranscriptCheckpointArtifactId.ShouldBeNull("the run LANDED, so its terminal write released the checkpoint reference; holding it would pin the artifact Referenced for ever beside the end-of-run transcript"); + row.SessionTranscriptCheckpointAt.ShouldBeNull(); + } + + [Fact] + public async Task A_streaming_agent_survives_a_checkpoint_that_is_still_uploading() + { + // N1. The checkpoint runs OFF the drain tick, and the tick is using this executor's scoped DbContext every + // 250 ms for its buffered-event flush and its spool-offset write. A checkpointer sharing that context puts + // two operations on one EF context at once, which EF refuses on whichever statement starts SECOND — as often + // the tick's, unhandled inside the runner's attach loop, killing a healthy run for a best-effort aid. + // + // A single-line agent never shows this: one tick, no overlap. This one streams across several ticks while a + // deliberately SLOW store upload is in flight, which is what a real agent does all the time. + // MUTATION: resolve the checkpointer from the executor's own scope instead of a fresh one (drop + // CheckpointInOwnScopeAsync) → "A second operation was started on this context instance" → red. + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false); + await ExecuteAsync(run.RunId, new TranscriptWritingHarness(run.SessionId, lines: 6), uploadDelay: TimeSpan.FromMilliseconds(700)); + + using var verify = _fixture.BeginScope(); + var db = verify.Resolve(); + var row = await db.AgentRun.AsNoTracking().SingleAsync(r => r.Id == run.RunId); + + row.Status.ShouldBe(AgentRunStatus.Succeeded, $"the run must land normally while a checkpoint uploads; it ended {row.Status} — {row.Error ?? "no error"}"); + (await db.WorkflowArtifactRetention.AsNoTracking().CountAsync(d => d.TeamId == team.TeamId && d.HolderId == run.RunId)) + .ShouldBeGreaterThan(0, "this arm proves nothing unless a checkpoint was actually taken while the ticks kept firing"); + (await db.AgentRunEvent.AsNoTracking().CountAsync(e => e.AgentRunId == run.RunId)).ShouldBeGreaterThan(1, "the tick's own event flush must have kept working throughout"); + } + + [Fact] + public async Task The_last_checkpoint_lands_even_when_the_agent_exits_immediately() + { + // N2. The upload is started fire-and-forget from the tick, so without a drain the ROUND does not wait for it. + // That is not merely untidy: the checkpoint's stamp is fenced on the run still being Running, so an upload + // still in flight when the terminal write lands has its stamp REFUSED — and the last minute of conversation, + // the most valuable minute a host loss could have taken, is silently dropped. The agent here exits the + // instant it finishes writing, so there is no slack to hide behind. + // MUTATION: remove the DrainSessionTranscriptCheckpointAsync call from the live path → the round returns in + // about a second while a four-second upload is still running → red. + var upload = TimeSpan.FromSeconds(2); // comfortably inside AgentRunExecutor.SessionCheckpointDrainBudget + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false); + var clock = System.Diagnostics.Stopwatch.StartNew(); + await ExecuteAsync(run.RunId, new TranscriptWritingHarness(run.SessionId, lines: 2, trailer: ""), uploadDelay: upload); + clock.Stop(); + + using var verify = _fixture.BeginScope(); + + (await verify.Resolve().WorkflowArtifactRetention.AsNoTracking() + .CountAsync(d => d.TeamId == team.TeamId && d.HolderId == run.RunId)) + .ShouldBeGreaterThan(0, "this arm proves nothing unless a checkpoint was actually in flight when the agent exited"); + + clock.Elapsed.ShouldBeGreaterThanOrEqualTo(upload, + $"the round returned in {clock.ElapsedMilliseconds}ms while a {upload.TotalSeconds}s checkpoint was still uploading — its stamp is fenced on the run being Running, so it would be refused by the terminal write and the last conversation lost"); + } + + [Fact] + public async Task A_run_that_did_not_opt_in_writes_no_checkpoint_at_all() + { + // The other half of the produce gate, and the reason it is worth its own test: without it every workflow and + // supervisor agent on the fleet would upload a copy of its transcript once a minute for a continuation that + // can never be bought. MUTATION: build the tick unconditionally → a declaration appears → red. + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false, checkpointSessionTranscript: false); + + await ExecuteAsync(run.RunId, new TranscriptWritingHarness(run.SessionId)); + + using var verify = _fixture.BeginScope(); + (await verify.Resolve().WorkflowArtifactRetention.AsNoTracking() + .CountAsync(d => d.TeamId == team.TeamId && d.HolderId == run.RunId)) + .ShouldBe(0, "a run nobody will ever continue must not pay for, or store, a checkpoint"); + } + + [Fact] + public async Task A_landed_run_releases_its_checkpoint_so_the_reaper_can_collect_it() + { + // The leak the terminal write closes. A checkpoint is Referenced while the column names it, and Referenced is + // TERMINAL in the retention ledger — so a run that landed normally would pin its last transcript copy for + // ever, beside the end-of-run transcript its own result already carries. + // MUTATION: drop the two NULL assignments from CompleteCoreAsync's terminal UPDATE → the oracle still answers + // Referenced → red. + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false); + + await ExecuteAsync(run.RunId, new TranscriptWritingHarness(run.SessionId)); + + using var verify = _fixture.BeginScope(); + var db = verify.Resolve(); + var declared = await db.WorkflowArtifactRetention.AsNoTracking() + .Where(d => d.TeamId == team.TeamId && d.HolderId == run.RunId).Select(d => d.ArtifactId).ToListAsync(); + + declared.ShouldNotBeEmpty("this arm needs a checkpoint to have been taken, or it proves nothing about releasing one"); + + var oracle = verify.Resolve(); + foreach (var artifactId in declared) + (await oracle.ClassifyAsync(db, artifactId, CancellationToken.None)) + .ShouldBe(ArtifactReferenceVerdict.Unreferenced, $"artifact {artifactId} is still named by a live column after its run landed, so the reaper can never collect it"); + } + + [Fact] + public async Task The_checkpoint_stamp_is_fenced_and_writes_the_session_id_once() + { + // The raw fenced UPDATE behind every checkpoint, exercised against Postgres rather than a fake: it must land + // under the CURRENT owner and fence, land NOTHING under a superseded one, and never overwrite a session id + // the row already carries with a null. + // MUTATION: drop the fence_epoch predicate from the statement → the superseded arm reds; replace the + // COALESCE with a bare assignment → the last assertion reds. + var team = await SeedTeamAsync(); + var owner = await SeedRunningOwnedRunAsync(team); + + using var scope = _fixture.BeginScope(); + var runs = scope.Resolve(); + var db = scope.Resolve(); + + var first = new SessionTranscriptCheckpoint(await StoreCheckpointAsync(team.TeamId, "one\n"), DateTimeOffset.UtcNow, 4); + (await runs.StampSessionTranscriptCheckpointAsync(owner, first, "s-live", CancellationToken.None)).ShouldBeTrue("the live owner's stamp must land"); + + var stamped = await db.AgentRun.AsNoTracking().SingleAsync(r => r.Id == owner.RunId); + stamped.SessionTranscriptCheckpointArtifactId.ShouldBe(first.ArtifactId); + stamped.SessionTranscriptCheckpointAt.ShouldNotBeNull(); + stamped.SessionId.ShouldBe("s-live", "a checkpoint no CLI can be pointed at is not resumable, so the id rides the same statement"); + + var superseded = new SessionTranscriptCheckpoint(await StoreCheckpointAsync(team.TeamId, "two\n"), DateTimeOffset.UtcNow, 4); + (await runs.StampSessionTranscriptCheckpointAsync(owner with { Epoch = owner.Epoch - 1 }, superseded, "s-stale", CancellationToken.None)) + .ShouldBeFalse("a worker whose ownership was reclaimed must not point a live run's recovery at its own stale conversation"); + + var later = new SessionTranscriptCheckpoint(await StoreCheckpointAsync(team.TeamId, "three\n"), DateTimeOffset.UtcNow, 6); + (await runs.StampSessionTranscriptCheckpointAsync(owner, later, null, CancellationToken.None)).ShouldBeTrue(); + + var final = await db.AgentRun.AsNoTracking().SingleAsync(r => r.Id == owner.RunId); + final.SessionTranscriptCheckpointArtifactId.ShouldBe(later.ArtifactId, "the newest checkpoint supersedes the last"); + final.SessionId.ShouldBe("s-live", "a later tick that observed no session id must not erase the one the row already carries"); + } + + [Fact] + public async Task A_slow_destination_still_gets_its_checkpoint_stored() + { + // The reason the upload's deadline is NOT the drain's. One checkpoint reads and uploads up to the 32 MiB + // capture cap, so on a throttled or cross-region destination it can legitimately outlast the few seconds a + // finished run may be held open — and if the two bounds were one number, every checkpoint of exactly the + // long conversations this feature protects would be cancelled by its own deadline and the run would + // silently stop being recoverable. + // + // Eight seconds is past the drain bound (the round lands without it) and well inside the upload's, so the + // checkpoint must still be STORED. + // MUTATION: give the upload the drain's 5s budget → its own CTS cancels it → nothing is ever stored → red. + var slow = TimeSpan.FromSeconds(8); + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false); + + await ExecuteAsync(run.RunId, new TranscriptWritingHarness(run.SessionId, lines: 2, trailer: ""), uploadDelay: slow); + + // The round has already landed by here — the drain gave up at its own bound — so the checkpoint is still on + // its way. Wait for it, bounded and loud: the predicate starts FALSE, because the round returned before the + // upload could possibly have finished. + var deadline = DateTimeOffset.UtcNow + slow + TimeSpan.FromSeconds(15); + while (DateTimeOffset.UtcNow < deadline) + { + using var poll = _fixture.BeginScope(); + if (await poll.Resolve().WorkflowArtifactRetention.AsNoTracking().AnyAsync(d => d.TeamId == team.TeamId && d.HolderId == run.RunId)) return; + + await Task.Delay(250); + } + + throw new Xunit.Sdk.XunitException( + $"no checkpoint was ever stored for run {run.RunId} against a {slow.TotalSeconds}s destination. The upload's own deadline (AgentRunExecutor.SessionCheckpointUploadBudget) must be sized for the 32 MiB capture cap, not for how long a landing may be deferred — check whether it was collapsed into SessionCheckpointDrainBudget."); + } + + [Fact] + public async Task A_wedged_store_cannot_defer_the_terminal_write_past_the_checkpoint_budget() + { + // The other half of waiting for the last checkpoint: the wait sits in FRONT of the run's terminal write, so + // an unbounded one would let a wedged storage backend hold a finished run open for as long as its client is + // willing to hang. The checkpoint carries its own deadline and the drain backstops it, so a store that + // ignores cancellation still cannot defer the landing. + // MUTATION: drop SessionCheckpointDrainBudget from the drain's WaitAsync → the round waits for the upload's + // own thirty-second deadline instead → red. + var wedged = TimeSpan.FromSeconds(20); + var team = await SeedTeamAsync(); + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false); + + var clock = System.Diagnostics.Stopwatch.StartNew(); + await ExecuteAsync(run.RunId, new TranscriptWritingHarness(run.SessionId, lines: 2, trailer: ""), uploadDelay: wedged); + clock.Stop(); + + clock.Elapsed.ShouldBeLessThan(wedged, + $"the round took {clock.Elapsed.TotalSeconds:F1}s against a {wedged.TotalSeconds}s wedged store — a best-effort checkpoint must never hold a finished run's terminal write open"); + + using var verify = _fixture.BeginScope(); + (await verify.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == run.RunId)).Status + .ShouldBe(AgentRunStatus.Succeeded, "and the run must still land"); + } + + [Theory] + [InlineData(true)] // a CHECKPOINT ref — best-effort, so an unreadable one must degrade + [InlineData(false)] // a CAPTURED ref — written by an attempt that finished, so an unreadable one is a fault + public async Task An_unreadable_transcript_ref_degrades_only_when_it_is_a_checkpoint(bool isCheckpoint) + { + // The two refs are resolved under opposite policies, and the difference decides whether a lost host costs + // the tenant an attempt. A captured transcript was written by an attempt that FINISHED, so an unreadable one + // is a genuine fault and cold-starting a named session silently would hide it. A checkpoint is best-effort by + // construction — its blob may have been collected, its destination may be unreachable — and this task is + // already a RETRY of a lost host, so failing it would spend the very attempt the checkpoint exists to improve + // on the one fault that says nothing about the work. + // MUTATION: delete the `if (!task.RestoredTranscriptIsCheckpoint)` branch (fail closed for both) → the + // checkpoint arm reds; make both degrade → the captured arm reds. + var team = await SeedTeamAsync(); + var priorRunId = Guid.NewGuid(); + var absent = Guid.NewGuid(); // never written to the artifact store + var harness = new TranscriptWritingHarness("s-unreadable"); + + var run = await SeedQueuedResumableRunAsync(team, authoredWorkingDirectory: false, task => task with + { + ResumeFromSessionId = "s-lost-host", + RestoredTranscriptArtifactId = absent, + RestoredTranscriptIsCheckpoint = isCheckpoint, + ResumedFromCheckpointAt = DateTimeOffset.UtcNow.AddMinutes(-3), + ResumedFromAgentRunId = priorRunId, + }); + + await ExecuteAsync(run.RunId, harness); + + using var verify = _fixture.BeginScope(); + var row = await verify.Resolve().AgentRun.AsNoTracking().SingleAsync(r => r.Id == run.RunId); + + if (!isCheckpoint) + { + row.Status.ShouldBe(AgentRunStatus.Failed, "an unreadable CAPTURED transcript is a real fault — failing closed is what stops a named session silently cold-starting"); + harness.Invocations.ShouldBeEmpty("and the agent must never have been launched at all"); + return; + } + + row.Status.ShouldBe(AgentRunStatus.Succeeded, $"an unreadable checkpoint must cost the conversation, never the attempt; the run ended {row.Status} — {row.Error ?? "no error"}"); + + var launched = harness.Invocations.ShouldHaveSingleItem(); + launched.RestoredTranscriptArtifactId.ShouldBeNull("the ref that resolved to nothing must not ride into the invocation"); + launched.RestoredTranscript.ShouldBeNull(); + launched.ResumeFromSessionId.ShouldBeNull("a --resume naming a session whose transcript was never restored cold-starts in the CLI anyway, silently — so it goes with the ref"); + launched.ResumedFromCheckpointAt.ShouldBeNull("the permanent confinement record must not claim a continuation this attempt never had"); + launched.ResumedFromAgentRunId.ShouldBeNull(); + launched.Goal.ShouldContain(AgentRetryContinuity.LostHostCheckpointUnreadableHint, Case.Sensitive, + "an agent that was going to be handed a conversation must be told it is not getting one"); + + JsonSerializer.Deserialize(row.SandboxConfinementJson!, AgentJson.Options)!.ResumedFromCheckpointAt + .ShouldBeNull("and the run's own record must agree with what the agent was actually given"); + } + + [Fact] + public void The_production_executor_is_wired_to_the_checkpointer() + { + // The whole slice is INERT if this wire is missing, and nothing else would say so: the checkpointer is an + // optional constructor dependency (so a hand-built test double need not know about it), the tick no-ops on a + // null one, and every other test in this class hands the executor its own collaborators. So the one thing no + // other assertion covers is whether the CONTAINER fills it — a capability with no caller. + // MUTATION: drop the IScopedDependency marker from IAgentSessionTranscriptCheckpointer (DI here is marker + // scanning, so nothing else registers it) → the resolve below fails and the injected field is null. + using var scope = _fixture.BeginScope(); + + scope.Resolve().ShouldBeOfType( + "the marker on the interface is what CodeSpaceModule.RegisterDependency scans for; without it no implementation is registered at all"); + + var executor = scope.Resolve().ShouldBeOfType(); + var injected = typeof(AgentRunExecutor) + .GetField("_sessionCheckpointer", System.Reflection.BindingFlags.Instance | System.Reflection.BindingFlags.NonPublic) + .ShouldNotBeNull("the field was renamed — update this wiring check rather than deleting it") + .GetValue(executor); + + injected.ShouldNotBeNull("the container must fill the executor's optional checkpointer; unfilled, every running agent silently stops being continuable after a host loss and no test elsewhere would notice"); + } + + // ── helpers ──────────────────────────────────────────────────────────────── + + /// The seeded team plus its Owner's live principal identity — the four values a standalone authority receipt is minted from. + private readonly record struct SeededTeam(Guid TeamId, Guid MembershipId, Guid UserId, Guid SecurityStamp); + + + /// Drive the REAL — the launch path whose drain tick owns the checkpoint — with the CONTAINER's own checkpointer, so the produce side runs the shape production runs. + private async Task ExecuteAsync(Guid runId, IAgentHarness harness, TimeSpan? uploadDelay = null) + { + using var outer = _fixture.BeginScope(); + + // The executor resolves its checkpointer from a DI scope of its OWN, per checkpoint — so a test that wants a + // slow store decorates the registration those scopes inherit, never an instance handed to the constructor. + // Anything else would stop measuring the production shape at exactly the seam under test. + using var scope = uploadDelay is { } delay + ? outer.BeginLifetimeScope(b => { b.RegisterInstance(new UploadDelay(delay)); b.RegisterType().As().InstancePerLifetimeScope(); }) + : outer.BeginLifetimeScope(); + + await NewExecutor(scope, harness, scope.Resolve(), scope.Resolve()).ExecuteAsync(runId, CancellationToken.None); + } + + /// + /// An rooted at THIS scope, so a per-test registration override is visible to + /// the scopes the executor creates for its own checkpoints. + /// + /// The container's own factory is a singleton holding the ROOT lifetime scope, so every scope it makes is a + /// child of the container and sees nothing a test registered. The shape under test is unchanged — one fresh + /// scope, and therefore one fresh DbContext, per checkpoint — only its parent moves. + /// + private sealed class ScopedServiceScopeFactory : IServiceScopeFactory + { + private readonly ILifetimeScope _scope; + + public ScopedServiceScopeFactory(ILifetimeScope scope) => _scope = scope; + + public IServiceScope CreateScope() => new Scope(_scope.BeginLifetimeScope()); + + private sealed class Scope : IServiceScope + { + private readonly ILifetimeScope _child; + + public Scope(ILifetimeScope child) { _child = child; ServiceProvider = new AutofacServiceProvider(child); } + + public IServiceProvider ServiceProvider { get; } + + public void Dispose() => _child.Dispose(); + } + } + + /// How long a decorated artifact write is held open — long enough to still be in flight when the next drain tick fires, and when the agent exits. + private sealed record UploadDelay(TimeSpan Value); + + /// + /// The REAL checkpointer with a slow DATABASE operation held open in front of it — so a checkpoint really is + /// mid-upload while the drain tick keeps firing and while the run lands, and the checkpoint it finally takes is + /// the production one. + /// + /// A bare Task.Delay would not do for the concurrency arm: the defect there is two operations on one + /// EF context, so the context has to be BUSY for the window, not merely the thread idle. pg_sleep on the + /// SCOPED context is exactly that — harmless when the checkpoint runs in a scope of its own, and a guaranteed + /// collision with the drain tick's own statement when it does not. + /// + /// Registered as a plain override rather than a decorator: the executor resolves its checkpointer from a + /// scope IT creates, and a decorator registered on this test's scope does not reach that far. + /// + private sealed class SlowCheckpointer : IAgentSessionTranscriptCheckpointer + { + private readonly ArtifactSessionTranscriptCheckpointer _inner; + private readonly CodeSpaceDbContext _db; + private readonly UploadDelay _delay; + + public SlowCheckpointer(ArtifactSessionTranscriptCheckpointer inner, CodeSpaceDbContext db, UploadDelay delay) { _inner = inner; _db = db; _delay = delay; } + + public async Task CheckpointAsync(SessionTranscriptCheckpointRequest request, CancellationToken cancellationToken) + { + await _db.Database.ExecuteSqlRawAsync($"SELECT pg_sleep({_delay.Value.TotalSeconds.ToString(System.Globalization.CultureInfo.InvariantCulture)})", cancellationToken).ConfigureAwait(false); + + return await _inner.CheckpointAsync(request, cancellationToken).ConfigureAwait(false); + } + } + + private static AgentRunExecutor NewExecutor(ILifetimeScope scope, IAgentHarness harness, ISandboxRunnerRegistry runners, IAgentSessionTranscriptCheckpointer? checkpointer) + { + return new AgentRunExecutor( + scope.Resolve(), + new AgentHarnessRegistry([harness]), + new HarnessModelReconciler(new AgentHarnessRegistry([harness]), scope.Resolve(), scope.Resolve()), + runners, + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + new ScopedServiceScopeFactory(scope), + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + scope.Resolve(), + scope.Resolve>(), + Microsoft.Extensions.Logging.Abstractions.NullLogger.Instance, + sessionCheckpointer: checkpointer); + } + + /// Store a transcript in the REAL team-scoped artifact store and hand back its id — the same store the checkpointer writes to, so the resume's resolve reads production bytes. + private async Task StoreCheckpointAsync(Guid teamId, string transcript) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().PutAsync(teamId, Encoding.UTF8.GetBytes(transcript), "application/x-ndjson", CancellationToken.None); + } + + /// + /// Seed a QUEUED standalone run the executor can actually launch, opted in (or not) to continuation. + /// + /// is the divergence under test. Both arms produce a cwd that + /// is NOT the primary repository's directory — which is what the tick used to be handed. With no repository the + /// executor either mounts a scratch workspace (whose Repositories are empty, so the primary-repo directory + /// is null) or, when the envelope authors its own directory, resolves no workspace at all and keeps that path. + /// The multi-repo shape the review named is the same read one step further along — its cwd is the workspace + /// ROOT while the primary repo sits in a subdirectory — and is not reproducible here without real clones, so it + /// is named rather than faked. + /// + private async Task<(Guid RunId, string SessionId)> SeedQueuedResumableRunAsync(SeededTeam team, bool authoredWorkingDirectory, Func? shape = null, bool checkpointSessionTranscript = true) + { + var runId = Guid.NewGuid(); + var sessionId = $"s-{Guid.NewGuid():N}"; + var cwd = authoredWorkingDirectory ? NewDirectory() : null; + + var task = new AgentTask + { + Goal = "Fix the failing billing tests", Harness = TranscriptWritingHarness.HarnessKind, TimeoutSeconds = 120, + CheckpointSessionTranscript = checkpointSessionTranscript, WorkspaceDirectory = cwd, ExecutionAuthority = AuthorityOf(team, runId), + }; + + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.AgentRun.Add(new AgentRun + { + Id = runId, TeamId = team.TeamId, Harness = TranscriptWritingHarness.HarnessKind, Status = AgentRunStatus.Queued, + TaskJson = JsonSerializer.Serialize(shape is null ? task : shape(task), AgentJson.Options), + }); + await db.SaveChangesAsync(); + + return (runId, sessionId); + } + + /// Seed a RUNNING run this test class owns the observation token for — the shape the fenced stamp is written under. + private async Task SeedRunningOwnedRunAsync(SeededTeam team) + { + var runId = Guid.NewGuid(); + var ownerId = Guid.NewGuid(); + + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.AgentRun.Add(new AgentRun + { + Id = runId, TeamId = team.TeamId, Harness = TranscriptWritingHarness.HarnessKind, Status = AgentRunStatus.Running, + FenceEpoch = 3, OwnerId = ownerId, StartedAt = DateTimeOffset.UtcNow, HeartbeatAt = DateTimeOffset.UtcNow, + LeaseExpiresAt = DateTimeOffset.UtcNow + AgentRunLiveness.Window, + TaskJson = JsonSerializer.Serialize(new AgentTask { Goal = "Fix the failing billing tests", Harness = TranscriptWritingHarness.HarnessKind }, AgentJson.Options), + }); + await db.SaveChangesAsync(); + + return new AgentRunOwnerToken(runId, ownerId, 3); + } + + private string NewDirectory() + { + var directory = Path.Combine(Path.GetTempPath(), "cs-checkpoint-cwd-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(directory); + _spoolDirs.Add(directory); + + return directory; + } + + private async Task SeedTeamAsync() + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + var userId = Guid.NewGuid(); + var user = new User { Id = userId, Email = $"checkpoint-{userId:N}@test.local", Name = $"checkpoint-{userId:N}" }; + db.User.Add(user); + + var teamId = Guid.NewGuid(); + var membershipId = Guid.NewGuid(); + db.Team.Add(new Team { Id = teamId, Slug = $"checkpoint-{teamId:N}", Name = "Checkpoint Team", Kind = TeamKind.Workspace }); + db.TeamMembership.Add(new TeamMembership { Id = membershipId, TeamId = teamId, UserId = userId, Role = TeamRole.Owner }); + + await db.SaveChangesAsync(); + + var stamp = (await db.User.AsNoTracking().SingleAsync(u => u.Id == userId)).SecurityStamp; + + return new SeededTeam(teamId, membershipId, userId, stamp); + } + + /// The standalone authority receipt a launch mints, so the executor can claim and run the seeded row exactly as production would. + private static AgentExecutionAuthority AuthorityOf(SeededTeam team, Guid logicalRunId) => new() + { + Version = ExecutionAuthorityService.ReceiptVersion, + PolicyVersion = ExecutionAuthorityService.PolicyVersion, + TeamId = team.TeamId, + LogicalRunId = logicalRunId, + SourceKind = "standalone", + DefinitionHash = "", + GrantedCeiling = AgentAutonomyPolicy.DeploymentCeiling, + IssuedAt = DateTimeOffset.UtcNow.AddMinutes(-30), + Subjects = [new AgentAuthoritySubject { Kind = "launcher", UserId = team.UserId, SecurityStamp = team.SecurityStamp, MembershipId = team.MembershipId, IssuedRole = TeamRole.Owner, Permission = TeamPermissions.RunsLaunch, GlobalAdmin = false }], + }; + + /// + /// A harness whose agent really writes a session transcript where the CLI would, and then stays alive long enough + /// for the drain tick to find it. + /// + /// The LOCATE is delegated to the production , in both directions: the + /// executor asks it where the transcript is, and the script below writes it exactly there. The layout + /// (projects/<sanitized-cwd>/<id>.jsonl) is never restated by this test — a test that computed + /// the path itself would keep passing after production started computing a different one, which is the whole + /// defect this class exists to catch. + /// + /// The script writes the FILE first and prints the session line second, so by the time the fold knows the + /// session id the file it names is already on disk — the tick is then deterministic rather than a race with the + /// first poll. + /// + private sealed class TranscriptWritingHarness : IAgentHarness, IAgentSessionTranscript + { + public const string HarnessKind = "scripted"; + private const string ConfigHomeEnvVar = "CS_TEST_CONFIG_DIR"; + private static readonly ClaudeCodeHarness Layout = new(); + + private readonly string _sessionId; + private readonly int _lines; + private readonly string _trailer; + + /// The session the script names on every line, so the fold can address the transcript. + /// How many lines the agent streams. More than one makes the drain tick fire REPEATEDLY while a checkpoint may be in flight — the overlap a single-line agent never produces. + /// What the agent does after its last line. Empty means it exits IMMEDIATELY, leaving the executor no slack to finish an upload in. + public TranscriptWritingHarness(string sessionId, int lines = 1, string trailer = "sleep 2") + { + _sessionId = sessionId; + _lines = lines; + _trailer = trailer; + } + + /// The cwd the executor actually handed the CLI — reported so a failing arm can say WHICH directory it was given. + public string? ObservedWorkingDirectory { get; private set; } + + /// The config-home-relative path the agent wrote to, or null when the production locate could not address one. + public string? WrittenAt { get; private set; } + + /// The task each launch was built from — the call site where "what the agent was actually given" is either true or it is not. + public List Invocations { get; } = []; + + public string Kind => HarnessKind; + public string Version => "test"; + public IReadOnlyList Models { get; } = ["test-model"]; + + public string? SessionTranscriptRelativePath(string configHome, string? workspaceDirectory, string? sessionId) => + ((IAgentSessionTranscript)Layout).SessionTranscriptRelativePath(configHome, workspaceDirectory, sessionId); + + public SandboxSpec BuildInvocation(AgentTask task) + { + Invocations.Add(task); + ObservedWorkingDirectory = task.WorkspaceDirectory; + WrittenAt = SessionTranscriptRelativePath("", task.WorkspaceDirectory, _sessionId); + + var line = $"{{\"type\":\"assistant\",\"session_id\":\"{_sessionId}\",\"text\":\"working\"}}"; + // The transcript is written BEFORE the first line is printed, so by the time the fold knows the session + // id the file it names is already on disk — the tick is then deterministic instead of racing the first + // poll. Each later line APPENDS to it, so a multi-line agent keeps the file growing across ticks exactly + // as a real one does. + var target = $"\"${ConfigHomeEnvVar}/{WrittenAt}\""; + var stream = string.Join(" && ", Enumerable.Range(0, _lines).Select(_ => $"printf '%s\\n' '{line}' >> {target} && printf '%s\\n' '{line}' && sleep 0.3")); + var script = WrittenAt is null + ? "printf 'no addressable transcript\\n'; sleep 1" + : $"mkdir -p \"${ConfigHomeEnvVar}/{Path.GetDirectoryName(WrittenAt)!.Replace('\\', '/')}\" && : > {target} && {stream}{(_trailer.Length == 0 ? "" : " && " + _trailer)}"; + + return new SandboxSpec { Command = "/bin/sh", Args = ["-c", script], WorkingDirectory = task.WorkspaceDirectory, TimeoutSeconds = 120, ConfigHomeEnvVars = [ConfigHomeEnvVar] }; + } + + public IReadOnlyList ParseEvents(string rawLine) => + string.IsNullOrWhiteSpace(rawLine) ? [] : [new AgentEvent { Kind = AgentEventKind.AssistantMessage, Text = rawLine.Trim(), Data = TryParse(rawLine) }]; + + public IAgentEventFolder CreateFolder() => ScriptedFolders.Result(); + + private static JsonElement? TryParse(string rawLine) + { + try { return JsonSerializer.Deserialize(rawLine); } + catch (JsonException) { return null; } + } + } + +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/InstrumentedAgentRunService.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/InstrumentedAgentRunService.cs index def1f6dad..98b042e2f 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/InstrumentedAgentRunService.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/InstrumentedAgentRunService.cs @@ -68,6 +68,9 @@ public Task AppendEventAsync(Guid runId, AgentEvent @event, Cance // ── pure delegation ─────────────────────────────────────────────────────── public Task CreateReviewAsync(CodeSpace.Core.Services.Agents.Review.AgentReviewCreation request, CancellationToken cancellationToken) => _inner.CreateReviewAsync(request, cancellationToken); + + public Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) => _inner.StampSessionTranscriptCheckpointAsync(owner, checkpoint, sessionId, cancellationToken); + public Task CreateAsync(AgentTask task, Guid teamId, Guid? workflowRunId, string? nodeId, string iterationKey = "", CancellationToken cancellationToken = default) => _inner.CreateAsync(task, teamId, workflowRunId, nodeId, iterationKey, cancellationToken); public Task RejectQueuedAsync(Guid runId, AgentRunResult result, CancellationToken cancellationToken) => _inner.RejectQueuedAsync(runId, result, cancellationToken); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/ThrowingAgentRunService.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/ThrowingAgentRunService.cs index 7f11438ea..f9d189d0b 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/ThrowingAgentRunService.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/ThrowingAgentRunService.cs @@ -25,6 +25,9 @@ public sealed class ThrowingAgentRunService : IAgentRunService public Task CreateReviewAsync(CodeSpace.Core.Services.Agents.Review.AgentReviewCreation request, CancellationToken cancellationToken) => throw new NotSupportedException(); + + public Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) => _inner.StampSessionTranscriptCheckpointAsync(owner, checkpoint, sessionId, cancellationToken); + public Task CreateAsync(AgentTask task, Guid teamId, Guid? workflowRunId, string? nodeId, string iterationKey = "", CancellationToken cancellationToken = default) { if (++_calls == _throwOnCall) throw new InvalidOperationException(FaultMessage); diff --git a/backend/tests/CodeSpace.UnitTests/Agents/AgentReviewRunnerTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/AgentReviewRunnerTests.cs index d8e410af3..5bc9e87a8 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/AgentReviewRunnerTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/AgentReviewRunnerTests.cs @@ -78,6 +78,7 @@ public Task CreateAsync(AgentTask task, Guid teamId, Guid? workflowRun public Task CreateReviewAsync(AgentReviewCreation request, CancellationToken cancellationToken) => _createFault is { } fault ? throw fault : Task.FromResult(_createResult!); + public Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) => throw new NotSupportedException(); public Task GetAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); public Task FindResumableSessionAsync(Guid teamId, Guid? parentRunId, string nodeId, string iterationKey, CancellationToken cancellationToken) => throw new NotSupportedException(); public Task FindResumableSubtaskAttemptAsync(Guid teamId, Guid supervisorRunId, string subtaskId, CancellationToken cancellationToken) => throw new NotSupportedException(); diff --git a/backend/tests/CodeSpace.UnitTests/Agents/AgentSessionTranscriptCheckpointerTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/AgentSessionTranscriptCheckpointerTests.cs new file mode 100644 index 000000000..c3dfc58e8 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Agents/AgentSessionTranscriptCheckpointerTests.cs @@ -0,0 +1,370 @@ +using System.Text; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Recovery; +using CodeSpace.Core.Services.Agents.Recovery.Checkpoints; +using CodeSpace.Core.Services.Workflows.Artifacts.Retention; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Artifacts; +using Microsoft.Extensions.Logging.Abstractions; +using Microsoft.Extensions.Time.Testing; +using Shouldly; + +namespace CodeSpace.UnitTests.Agents; + +/// +/// The mid-run session-transcript checkpoint — what makes a run whose HOST dies continuable somewhere else. +/// +/// It hangs off the observer's own checkpoint tick, which fires on every poll of a running agent, so the whole +/// design rests on two gates: a tick whose file has not GROWN carries no new conversation, and a file that is growing +/// still earns at most one upload per . Delete +/// either and a busy fleet uploads whole session files continuously; delete the byte cap and one pathological +/// session reads itself into a worker's heap once a minute. Each of those is pinned below by counting the uploads a +/// FAKE store actually received — never by reading the gate's own return value, which would pass on a checkpointer +/// that never stores anything at all. +/// +/// The clock is a , so the cadence is decided by the test and not by how long the +/// test happened to take. +/// +[Trait("Category", "Unit")] +public sealed class AgentSessionTranscriptCheckpointerTests : IDisposable +{ + private readonly string _root = Path.Combine(Path.GetTempPath(), $"cs-checkpoint-{Guid.NewGuid():N}"); + private readonly FakeTimeProvider _clock = new(new DateTimeOffset(2026, 9, 19, 12, 0, 0, TimeSpan.Zero)); + private readonly RecordingRetentionWriter _store = new(); + private readonly StampingRuns _runs = new(); + + public AgentSessionTranscriptCheckpointerTests() => Directory.CreateDirectory(_root); + + public void Dispose() + { + if (Directory.Exists(_root)) Directory.Delete(_root, recursive: true); + } + + [Fact] + public async Task Checkpointer_uploads_only_when_the_file_grew() + { + // MUTATION this pins: drop the "length <= previous.Bytes" gate in ArtifactSessionTranscriptCheckpointer.IsDue + // (i.e. upload unconditionally once the cadence has elapsed) and the second call stores a second copy of + // identical bytes — the upload COUNT goes 1 → 2 and this test reds. The cadence is advanced past the interval + // before every call, so the only thing being measured here is growth. + var path = WriteTranscript("{\"type\":\"init\",\"session_id\":\"s-1\"}\n"); + var checkpointer = NewCheckpointer(); + + var first = (await checkpointer.CheckpointAsync(Request(path), CancellationToken.None)).ShouldNotBeNull(); + + var unchanged = await checkpointer.CheckpointAsync(Request(path, previousBytes: first.Bytes), CancellationToken.None); + + AppendTranscript(path, "{\"type\":\"assistant\",\"text\":\"重構結帳流程 🚀\"}\n"); + var grown = await checkpointer.CheckpointAsync(Request(path, previousBytes: first.Bytes), CancellationToken.None); + + unchanged.ShouldBeNull("a transcript that has not grown carries no new conversation, so re-uploading it buys nothing"); + grown.ShouldNotBeNull("a transcript that HAS grown is exactly what a later host needs"); + + _store.Writes.Count.ShouldBe(2, $"only the first and the grown transcript may be stored; the store received {_store.Writes.Count} write(s)"); + grown.Bytes.ShouldBe(new FileInfo(path).Length, "the checkpoint reports the bytes it actually stored, which is what an operator reads to tell a captured conversation from an empty file"); + _runs.Stamps.Count.ShouldBe(2, "every stored checkpoint must reach the run row — an artifact no column names is one the reaper is entitled to collect"); + _runs.Stamps[^1].SessionId.ShouldBe("s-1", "the stamp carries the session id, because a checkpoint no CLI can be pointed at is not resumable"); + } + + [Fact] + public void A_new_rounds_watermark_is_never_paired_with_the_old_rounds_byte_count() + { + // The watermark is (path, bytes) written in ONE assignment, and only when a checkpoint LANDS. Splitting it — + // advancing the path when an attempt is dispatched, the byte count when one lands — re-creates the defect + // the path key exists to prevent, and the sequence below is exactly how a revise round hits it: round one + // lands 400 KiB, round two's FIRST tick on a new path declines (the CLI has not written its session file yet, + // the ordinary first-tick outcome), and every later tick would then be measured against round one. + // MUTATION: assign the path eagerly at Begin and the bytes only in Landed → the third Begin returns 400 KiB + // instead of null → round two never checkpoints until it outgrows round one → red. + var watermark = new AgentRunExecutor.SessionCheckpointWatermark(); + + watermark.Begin("/round-0/session.jsonl").ShouldBeNull("nothing has landed yet"); + watermark.Landed(new SessionTranscriptCheckpoint(Guid.NewGuid(), DateTimeOffset.UtcNow, 400 * 1024)); + + watermark.Begin("/round-1/session.jsonl").ShouldBeNull("a different round's transcript is a different conversation, not a shrunken one"); + watermark.Landed(null); // declined: the new round's session file is not on disk yet + + watermark.Begin("/round-1/session.jsonl").ShouldBeNull("the decline landed nothing, so round two is still measured against nothing — never against round one's 400 KiB"); + + watermark.Begin("/round-0/session.jsonl").ShouldBe(400 * 1024, "and round zero's own watermark survives, so returning to it still refuses to re-store bytes that have not changed"); + } + + [Fact] + public void A_landed_checkpoint_is_what_the_next_attempt_must_grow_past() + { + // The other half of the same rule: a checkpoint that DID land must gate the next attempt, or the growth gate + // is dead and every tick re-stores the same bytes. MUTATION: make Landed a no-op → red. + var watermark = new AgentRunExecutor.SessionCheckpointWatermark(); + + watermark.Begin("/session.jsonl"); + watermark.Landed(new SessionTranscriptCheckpoint(Guid.NewGuid(), DateTimeOffset.UtcNow, 8192)); + + watermark.Begin("/session.jsonl").ShouldBe(8192); + } + + [Theory] + [InlineData(0, false)] // the window was just claimed + [InlineData(59, false)] // still inside it + [InlineData(60, true)] // the interval has elapsed + public void The_checkpoint_cadence_reopens_only_after_its_interval(int secondsSinceLastAttempt, bool due) + { + // The gate that decides how often a whole fleet reads and uploads transcripts. It lives on the CALLER + // (AgentRunExecutor), in front of the locate, because locating is not free for every harness — Codex finds + // its rollout by a recursive walk of the config home, and the drain tick fires several times a second. + // MUTATION: return true unconditionally → the 0s and 59s arms red. + var now = new DateTimeOffset(2026, 9, 19, 12, 0, 0, TimeSpan.Zero); + + AgentRunExecutor.SessionCheckpointDue(now.AddSeconds(-secondsSinceLastAttempt), now).ShouldBe(due); + AgentRunExecutor.SessionCheckpointDue(null, now).ShouldBeTrue("the FIRST attempt must not wait — a run whose host dies in its first minute has to leave something behind"); + AgentRunExecutor.SessionCheckpointInterval.ShouldBe(TimeSpan.FromSeconds(60)); + // TWO budgets, because they bound two different things, and collapsing them costs one of the two. + AgentRunExecutor.SessionCheckpointUploadBudget.ShouldBe(TimeSpan.FromSeconds(30), + "one checkpoint may read and upload up to the 32 MiB capture cap, so a budget sized for a fast local store would cancel every checkpoint of a long conversation on a throttled destination"); + AgentRunExecutor.SessionCheckpointUploadBudget.ShouldBeLessThan(AgentRunExecutor.SessionCheckpointInterval, + "an upload that outlived the cadence would overlap the next attempt"); + AgentRunExecutor.SessionCheckpointDrainBudget.ShouldBe(TimeSpan.FromSeconds(5), + "this is how long a finished run's terminal write may be deferred by a best-effort aid — much shorter than the upload's own deadline, deliberately"); + } + + [Fact] + public async Task A_path_whose_component_was_swapped_is_refused() + { + // The clamp walks the path for symlinks and the file is opened AFTER — two syscalls with the agent, which + // has write access to its own bind-mounted config home, still running between them. A component swapped in + // that window points this read at a file the clamp never approved, and its bytes would be uploaded into the + // team's store and restored into the next attempt. The kernel's own answer for the OPEN handle is the check. + // + // Driven through CheckpointAsync, not the predicate, so the WIRING is what is pinned: a symlink is exactly + // the swap the clamp exists to refuse, and it makes the mismatch real rather than argued. + // MUTATION: replace OpenedTheResolvedFile's body with `return true;` → the decoy's bytes are stored → red. + if (!OperatingSystem.IsLinux()) return; + + var honest = WriteTranscript("{\"type\":\"init\",\"session_id\":\"s-1\"}\n", "honest.jsonl"); + var swapped = Path.Combine(_root, "swapped.jsonl"); + File.CreateSymbolicLink(swapped, honest); + + (await NewCheckpointer().CheckpointAsync(Request(swapped), CancellationToken.None)) + .ShouldBeNull("the handle opened the symlink's TARGET, which is not the path the clamp resolved"); + _store.Writes.ShouldBeEmpty($"nothing may be stored for a path whose component was swapped; the store received {_store.Writes.Count} write(s)"); + + (await NewCheckpointer().CheckpointAsync(Request(honest), CancellationToken.None)) + .ShouldNotBeNull("the same file read by its own real path IS checkpointed — without this the refusal would be indistinguishable from a checkpointer that never works"); + } + + [Fact] + public async Task A_torn_trailing_line_is_never_uploaded() + { + // MUTATION this pins: upload the whole file instead of its complete prefix. The CLI appends while this reads, + // so a read that lands mid-append ends in half a line — and the harness restores those bytes verbatim while + // ResolveRestoredTranscriptAsync fails CLOSED on an unusable transcript, by which point the continuation's + // budget is already spent. A torn checkpoint does not degrade to a cold start; it burns the one attempt. + var whole = "{\"type\":\"init\",\"session_id\":\"s-1\"}\n{\"type\":\"assistant\",\"text\":\"done\"}\n"; + var path = WriteTranscript(whole + "{\"type\":\"assis"); + + var taken = (await NewCheckpointer().CheckpointAsync(Request(path), CancellationToken.None)).ShouldNotBeNull(); + + Encoding.UTF8.GetString(_store.Writes.Single().Bytes.ToArray()).ShouldBe(whole, "only the complete lines may be stored — the half-written tail is a line the CLI has not finished"); + taken.Bytes.ShouldBe(Encoding.UTF8.GetByteCount(whole), "and the checkpoint reports the prefix it actually stored, not the file's length"); + } + + [Fact] + public async Task A_file_with_no_complete_line_yet_is_not_checkpointed() + { + // The boundary of the same rule: a first line still being written is not a shorter conversation, it is none. + var path = WriteTranscript("{\"type\":\"ini"); + + (await NewCheckpointer().CheckpointAsync(Request(path), CancellationToken.None)).ShouldBeNull(); + _store.Writes.ShouldBeEmpty("there is no whole turn to restore, so nothing may be stored or stamped"); + _runs.Stamps.ShouldBeEmpty(); + } + + [Fact] + public async Task Growth_confined_to_the_torn_tail_is_not_a_new_checkpoint() + { + // The file grew, but every new byte is inside the line still being written — so the COMPLETE prefix is + // unchanged and re-uploading it would store a byte-identical copy. The growth gate has to measure the prefix, + // not the file. MUTATION: compare stream.Length instead of the prefix length → a second write appears. + var whole = "{\"type\":\"init\",\"session_id\":\"s-1\"}\n"; + var path = WriteTranscript(whole); + var checkpointer = NewCheckpointer(); + + var taken = (await checkpointer.CheckpointAsync(Request(path), CancellationToken.None)).ShouldNotBeNull(); + + AppendTranscript(path, "{\"type\":\"assistant\",\"tex"); + + (await checkpointer.CheckpointAsync(Request(path, previousBytes: taken.Bytes), CancellationToken.None)).ShouldBeNull("the only new bytes are an unfinished line"); + _store.Writes.Count.ShouldBe(1, $"the store must have seen exactly the first checkpoint; it saw {_store.Writes.Count}"); + } + + [Fact] + public async Task Checkpoint_respects_the_transcript_byte_cap() + { + // MUTATION this pins: remove the "length > MaxSessionTranscriptBytes()" gate and the checkpointer reads a + // pathological session file whole into the worker's heap — once per cadence, per running agent, unbounded + // across concurrency. The cap is the SAME one the end-of-run capture applies (deliberately not a second + // knob), so it is read from the executor here rather than restated. + var over = WriteTranscript(Filler(AgentRunExecutor.MaxSessionTranscriptBytes() + 1)); + var checkpointer = NewCheckpointer(); + + var refused = await checkpointer.CheckpointAsync(Request(over), CancellationToken.None); + + refused.ShouldBeNull("a session past the cap is not read whole into memory; the run keeps going and a later continuation cold-starts"); + _store.Writes.ShouldBeEmpty($"nothing may be stored for an over-cap transcript; the store received {_store.Writes.Count} write(s)"); + _runs.Stamps.ShouldBeEmpty("and nothing may be stamped either — a row pointing at bytes that were never written is worse than no row"); + } + + [Fact] + public async Task An_under_cap_transcript_is_checkpointed_so_the_cap_is_not_a_dead_path() + { + // The other half of the pair: without it, "over the cap stores nothing" would still pass on a checkpointer + // that stores nothing ever. + var under = WriteTranscript(Filler(64 * 1024)); + + var taken = await NewCheckpointer().CheckpointAsync(Request(under), CancellationToken.None); + + taken.ShouldNotBeNull(); + _store.Writes.Single().Bytes.ToArray().ShouldBe(await File.ReadAllBytesAsync(under), "the stored bytes are the transcript verbatim — a CLI session restored from a partial file is not a shorter conversation, it is a corrupt one"); + _store.Writes.Single().RetentionClass.ShouldBe(ArtifactRetentionClass.SessionTranscriptCheckpoint, "a checkpoint gets its OWN short-floor class — on the event class's seven-day floor a long run would hold every superseded copy of its transcript for over a week"); + } + + [Fact] + public async Task A_new_rounds_smaller_transcript_is_still_checkpointed() + { + // The growth watermark is keyed on the PATH by the caller. A revise round opens a NEW config home whose + // transcript legitimately starts smaller than the finished previous round's; a count-only gate would silently + // stop checkpointing round two until it outgrew round one, which is precisely the window a lost host would + // fall into. MUTATION: in AgentRunExecutor, pass the watermark without comparing paths → this reds. + var first = WriteTranscript(Filler(8 * 1024), "round-0.jsonl"); + var checkpointer = NewCheckpointer(); + + var round0 = (await checkpointer.CheckpointAsync(Request(first), CancellationToken.None)).ShouldNotBeNull(); + + var second = WriteTranscript("{\"type\":\"init\",\"session_id\":\"s-2\"}\n", "round-1.jsonl"); + + // The CALLER keys the watermark on the path, so a new round's request carries none — which is the whole + // reason the watermark is the caller's and not this instance's. + round0.Bytes.ShouldBeGreaterThan(new FileInfo(second).Length, "this arm is only meaningful while round one's transcript is the LARGER of the two"); + var taken = await checkpointer.CheckpointAsync(Request(second, previousBytes: null) with { SessionId = "s-2" }, CancellationToken.None); + + taken.ShouldNotBeNull("a new round's transcript is a different conversation, not a shrunken one"); + taken.Bytes.ShouldBe(new FileInfo(second).Length); + } + + [Fact] + public async Task A_transcript_that_is_not_on_disk_yet_is_not_a_failure() + { + var absent = Path.Combine(_root, "never-written.jsonl"); + + (await NewCheckpointer().CheckpointAsync(Request(absent), CancellationToken.None)).ShouldBeNull("the CLI may not have written its session yet; a tick that finds nothing declines rather than throwing into the observer loop"); + _store.Writes.ShouldBeEmpty(); + } + + [Fact] + public async Task A_stamp_that_loses_its_fence_takes_no_checkpoint() + { + // The row write is fenced to the owner token, so a worker whose ownership was already reclaimed cannot point + // a live run's recovery at ITS stale conversation. When that race is lost there is no checkpoint — and the + // local cadence state must NOT advance, or the winning generation's next tick would be suppressed by a + // checkpoint that never landed. + var path = WriteTranscript("{\"type\":\"init\",\"session_id\":\"s-1\"}\n"); + _runs.Wins = false; + var checkpointer = NewCheckpointer(); + + (await checkpointer.CheckpointAsync(Request(path), CancellationToken.None)).ShouldBeNull(); + + _runs.Wins = true; + (await checkpointer.CheckpointAsync(Request(path), CancellationToken.None)).ShouldNotBeNull("the retry is immediate — a refused stamp must not advance the growth watermark it never earned"); + } + + private ArtifactSessionTranscriptCheckpointer NewCheckpointer() => + new(_store, _runs, _clock, NullLogger.Instance); + + private static SessionTranscriptCheckpointRequest Request(string path, long? previousBytes = null) => + new(Guid.NewGuid(), new AgentRunOwnerToken(Guid.NewGuid(), Guid.NewGuid(), 7), path, "s-1", previousBytes); + + private string WriteTranscript(string content, string name = "session.jsonl") => WriteTranscript(Encoding.UTF8.GetBytes(content), name); + + private string WriteTranscript(byte[] content, string name = "session.jsonl") + { + var path = Path.Combine(_root, name); + File.WriteAllBytes(path, content); + + return path; + } + + private static void AppendTranscript(string path, string line) => File.AppendAllText(path, line); + + /// A transcript of at least , repeating a realistic mixed-script line so a byte comparison is not satisfied by pure ASCII. + private static byte[] Filler(long atLeastBytes) + { + var block = Encoding.UTF8.GetBytes("{\"role\":\"user\",\"text\":\"重構結帳流程,先寫測試 🚀\"}\n"); + + using var session = new MemoryStream(); + + while (session.Length < atLeastBytes) session.Write(block); + + return session.ToArray(); + } + + /// Records every declaring write so the tests can COUNT uploads — the only signal that distinguishes a gate that holds from a gate that was deleted. + private sealed class RecordingRetentionWriter : IArtifactRetentionWriter + { + public List Writes { get; } = []; + + public Task PutDeclaredAsync(ArtifactRetentionWriteRequest request, CancellationToken cancellationToken) + { + Writes.Add(request); + + return Task.FromResult(new ArtifactRetentionWrite(Guid.NewGuid(), Declared: true)); + } + } + + /// Minimal IAgentRunService: records the checkpoint stamps and can be told to lose the fenced race. Every other member throws — the checkpointer calls none of them. + private sealed class StampingRuns : IAgentRunService + { + public List<(SessionTranscriptCheckpoint Checkpoint, string? SessionId)> Stamps { get; } = []; + + public bool Wins { get; set; } = true; + + public Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) + { + if (!Wins) return Task.FromResult(false); + + Stamps.Add((checkpoint, sessionId)); + return Task.FromResult(true); + } + + public Task CreateAsync(AgentTask task, Guid teamId, Guid? workflowRunId, string? nodeId, string iterationKey = "", CancellationToken cancellationToken = default) => throw new NotSupportedException(); + public Task CreateReviewAsync(CodeSpace.Core.Services.Agents.Review.AgentReviewCreation request, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task GetAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task FindResumableSessionAsync(Guid teamId, Guid? parentRunId, string nodeId, string iterationKey, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task FindResumableSubtaskAttemptAsync(Guid teamId, Guid supervisorRunId, string subtaskId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AppendEventAsync(Guid runId, AgentEvent @event, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AppendEventsAsync(Guid runId, IReadOnlyList events, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AppendSystemEventAsync(Guid runId, AgentEvent @event, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task RejectQueuedAsync(Guid runId, AgentRunResult result, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ReserveReattachAsync(AgentRunReconciliationCandidate candidate, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ClaimOwnershipAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ReserveReattachAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ActivateReattachAsync(AgentRunReattachReservation reservation, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AssertOwnershipAsync(AgentRunOwnerToken owner, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task HeartbeatAsync(AgentRunOwnerToken owner, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task SetRunnerHandleAsync(AgentRunOwnerToken owner, string handleJson, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task SetSandboxConfinementAsync(AgentRunOwnerToken owner, string confinementJson, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AppendEventAsync(AgentRunOwnerToken owner, AgentEvent @event, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AppendEventsAsync(AgentRunOwnerToken owner, IReadOnlyList events, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task CompleteAsync(AgentRunOwnerToken owner, AgentRunResult result, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task MarkRunningAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task HeartbeatAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ReclaimForReattachAsync(Guid runId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task SetRunnerHandleAsync(Guid runId, string handleJson, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task SetSandboxConfinementAsync(Guid runId, string confinementJson, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task CompleteAsync(Guid runId, AgentRunResult result, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task CompleteAsync(Guid runId, AgentRunResult result, long expectedEpoch, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task CancelQueuedAsync(Guid runId, string reason, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task CancelRunningAsync(Guid runId, string reason, AgentRunAbandonCause cause, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task GetSummaryForTeamAsync(Guid runId, Guid teamId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task> GetEventsAsync(Guid runId, Guid teamId, long afterSequence, CancellationToken cancellationToken) => throw new NotSupportedException(); + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs index 7695bf2a2..d463af0c2 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs @@ -911,6 +911,72 @@ public async Task P2_3_a_respawn_carrying_a_prior_transcript_artifact_ref_thread task.RestoredTranscriptArtifactId.ShouldBe(artifactId); } + // ─── 3c: continuing an attempt whose HOST died ────────────────────────────── + + [Fact] + public async Task A_respawn_after_a_host_loss_restores_the_checkpoint_and_says_the_tree_is_gone() + { + // The population 3c exists for. A run whose host died is abandoned by the reconciler with NO result at all, + // so the captured-transcript keys a warm retry normally reads are empty — the only thing left is the MID-RUN + // checkpoint the notifier projects off the run row. Without this the node respawns cold and the whole + // checkpoint is written for nobody. + // MUTATION: drop the `?? checkpoint` (or the whole checkpoint branch) from ApplyRespawnResumeHint → the task + // carries no transcript ref, no provenance and no honesty line → red. + var checkpointId = Guid.NewGuid(); + var priorRunId = Guid.NewGuid(); + var priorAttempt = JsonDocument.Parse($$""" + {"status":"Failed","error":"The agent run was abandoned","sessionId":"sess-lost", + "sessionTranscriptCheckpointArtifactId":"{{checkpointId}}","sessionTranscriptCheckpointAt":"2026-09-19T10:11:12+00:00", + "agentRunId":"{{priorRunId}}"} + """).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(RequiredConfig(), resume: null, priorAttemptPayload: priorAttempt), CancellationToken.None); + + var task = JsonSerializer.Deserialize(result.SuspendUntil!.Payload, AgentJson.Options)!; + task.ResumeFromSessionId.ShouldBe("sess-lost", "the CLI is told WHICH conversation to resume; a transcript with no id names nothing"); + task.RestoredTranscriptArtifactId.ShouldBe(checkpointId, "the checkpoint rides as a REF — the executor resolves it to bytes just before invocation"); + task.RestoredTranscript.ShouldBeNull("an abandoned attempt left no inline transcript, and inventing one would be a claim about bytes nobody has"); + task.ResumedFromCheckpointAt.ShouldBe(new DateTimeOffset(2026, 9, 19, 10, 11, 12, TimeSpan.Zero), "the launch stamps this onto the run's permanent confinement record"); + task.ResumedFromAgentRunId.ShouldBe(priorRunId, "which attempt took over from which is a column, not prose"); + task.Goal.ShouldContain("machine running your previous attempt was lost", Case.Sensitive, "a restored conversation describes a working tree this sandbox does not have, and the agent must be told"); + } + + [Fact] + public async Task No_checkpoint_means_a_cold_retry_that_says_so() + { + // The same abandon, from a run that never checkpointed — an envelope that did not opt in, or a harness with + // no addressable session transcript. There is nothing to restore, so the respawn cold-starts and says + // NOTHING about a lost machine: a run that claims a restored conversation it does not have is worse than one + // that admits it is starting over. + // MUTATION: stamp the lost-host block unconditionally → red. + var priorAttempt = JsonDocument.Parse(""" + {"status":"Failed","error":"The agent run was abandoned","sessionId":"sess-lost"} + """).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(RequiredConfig(), resume: null, priorAttemptPayload: priorAttempt), CancellationToken.None); + + var task = JsonSerializer.Deserialize(result.SuspendUntil!.Payload, AgentJson.Options)!; + task.RestoredTranscriptArtifactId.ShouldBeNull("no checkpoint was taken, so there is no conversation to restore"); + task.ResumedFromCheckpointAt.ShouldBeNull(); + task.ResumedFromAgentRunId.ShouldBeNull(); + task.Goal.ShouldNotContain("machine running your previous attempt was lost", Case.Sensitive, "nothing may assert a restored conversation this attempt does not have"); + } + + [Theory] + [InlineData(true)] // the node's policy allows another attempt — somebody can consume a checkpoint + [InlineData(false)] // one attempt only — a checkpoint would be written for nobody + public async Task A_node_checkpoints_its_session_only_when_a_failure_can_be_retried(bool retriesOnFailure) + { + // The produce-side opt-in. A checkpoint costs a whole-file read and an artifact write a minute, per running + // agent, per worker — so it is paid for only where a failure can actually buy another attempt, which is the + // engine's own answer about THIS node's retry policy rather than the node's guess. + // MUTATION: set CheckpointSessionTranscript unconditionally (or never) → one arm reds. + var result = await new AgentCodeNode().RunAsync(BuildContext(RequiredConfig(), resume: null, retriesOnFailure: retriesOnFailure), CancellationToken.None); + + JsonSerializer.Deserialize(result.SuspendUntil!.Payload, AgentJson.Options)! + .CheckpointSessionTranscript.ShouldBe(retriesOnFailure); + } + [Fact] public async Task P2_3_a_first_pass_with_no_prior_attempt_cold_starts_byte_identical() { @@ -1486,8 +1552,9 @@ public async Task Unrecognized_mode_degrades_to_unset_and_never_throws() ["model"] = Str("gpt-5.3-codex"), }; - private static NodeRunContext BuildContext(Dictionary config, JsonElement? resume, Dictionary? inputs = null, JsonElement? priorAttemptPayload = null) => new() + private static NodeRunContext BuildContext(Dictionary config, JsonElement? resume, Dictionary? inputs = null, JsonElement? priorAttemptPayload = null, bool retriesOnFailure = true) => new() { + RetriesOnFailure = retriesOnFailure, Inputs = inputs ?? new Dictionary(), Config = config, RawInputs = JsonDocument.Parse("{}").RootElement, diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorOutputReviewTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorOutputReviewTests.cs index b93963dc2..ecea9e249 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorOutputReviewTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorOutputReviewTests.cs @@ -781,6 +781,7 @@ public Task AppendEventAsync(Guid runId, AgentEvent @event, Cance public Task FindResumableSubtaskAttemptAsync(Guid teamId, Guid supervisorRunId, string subtaskId, CancellationToken cancellationToken) => Task.FromResult(null); public Task AppendEventsAsync(Guid runId, IReadOnlyList events, CancellationToken cancellationToken) => throw new NotSupportedException(); public Task CreateReviewAsync(CodeSpace.Core.Services.Agents.Review.AgentReviewCreation request, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) => Task.FromResult(true); public Task CreateAsync(AgentTask task, Guid teamId, Guid? workflowRunId, string? nodeId, string iterationKey = "", CancellationToken cancellationToken = default) => throw new NotSupportedException(); public Task RejectQueuedAsync(Guid runId, AgentRunResult result, CancellationToken cancellationToken) => throw new NotSupportedException(); public Task ReserveReattachAsync(AgentRunReconciliationCandidate candidate, CancellationToken cancellationToken) => throw new NotSupportedException(); diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorPushTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorPushTests.cs index 8facf9a0d..b99383080 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorPushTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorPushTests.cs @@ -638,6 +638,7 @@ public Task AppendEventsAsync(Guid runId, IReadOnlyList events, Canc } public Task CreateReviewAsync(CodeSpace.Core.Services.Agents.Review.AgentReviewCreation request, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task StampSessionTranscriptCheckpointAsync(AgentRunOwnerToken owner, SessionTranscriptCheckpoint checkpoint, string? sessionId, CancellationToken cancellationToken) => Task.FromResult(true); public Task CreateAsync(AgentTask task, Guid teamId, Guid? workflowRunId, string? nodeId, string iterationKey = "", CancellationToken cancellationToken = default) => throw new NotSupportedException(); public Task RejectQueuedAsync(Guid runId, AgentRunResult result, CancellationToken cancellationToken) => throw new NotSupportedException(); public Task ReserveReattachAsync(AgentRunReconciliationCandidate candidate, CancellationToken cancellationToken) => throw new NotSupportedException(); diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs index 3a397dfbd..0ee3c8317 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs @@ -29,6 +29,7 @@ public void Every_soft_link_to_an_artifact_is_a_probed_site() ("artifact_manifest", "content_artifact_id"), ("publish_manifest", "patch_artifact_id"), ("agent_run_event", "data_artifact_id"), + ("agent_run", "session_transcript_checkpoint_artifact_id"), ("workflow_run_model_call", "request_artifact_id"), ("workflow_run_model_call_attempt", "request_artifact_id"), ("workflow_run_model_call_attempt", "response_artifact_id"), @@ -63,6 +64,18 @@ public void A_mapped_column_that_names_an_artifact_and_is_not_probed_fails_this_ "a soft link to workflow_artifact that the oracle does not probe would let the reaper delete a referenced object — add it to ArtifactReferenceOracle.ReferenceSites and give it an index"); } + [Fact] + public void ReferenceSites_includes_the_checkpoint_column() + { + // 3c: agent_run.session_transcript_checkpoint_artifact_id is the ONLY column naming a live mid-run session + // checkpoint, and that checkpoint is DECLARED (AgentRunEventData) — so it is a reap candidate. An oracle that + // did not probe this column would answer "unreferenced" about the one artifact a run whose host died needs to + // be continuable, and the reaper would collect it. The literal list above already pins it; this states the + // reason separately so a future edit that deletes the entry cannot read as tidying. + ArtifactReferenceOracle.ReferenceSites.ShouldContain(("agent_run", "session_transcript_checkpoint_artifact_id"), + "the mid-run session-transcript checkpoint is a declared artifact whose only reference is this column — unprobed, the reaper deletes the conversation a cross-host resume restores"); + } + [Fact] public void Legacy_adoption_copies_source_identity_without_creating_a_retention_reference() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactRetentionPolicyTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactRetentionPolicyTests.cs index 6f921c280..56b8eba91 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactRetentionPolicyTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactRetentionPolicyTests.cs @@ -34,7 +34,12 @@ public void Registered_classes_keep_orphan_candidates_for_a_week_then_quarantine .Select(value => ArtifactRetentionPolicy.For(value.ToString()).ShouldNotBeNull()).ToArray(); rules.ShouldAllBe(rule => rule.MinimumAge == TimeSpan.FromDays(7) && rule.QuarantineWindow == TimeSpan.FromHours(24)); - ArtifactRetentionPolicy.MinimumAgeFloor.ShouldBe(TimeSpan.FromDays(7), "the claim query pre-filters on the smallest floor across all classes"); + + // The floor is the SMALLEST across every registered class, and it is a claim-query pre-filter only — the exact + // per-class floor above is still enforced per row, so the short-floor session-transcript checkpoint class + // widens what the sweep LOOKS at without shortening what any of these four are kept for. + ArtifactRetentionPolicy.MinimumAgeFloor.ShouldBe(TimeSpan.FromHours(2), + "the claim query pre-filters on the smallest floor across all classes, which is now the session-transcript checkpoint's"); } [Theory] @@ -106,14 +111,24 @@ public void Every_production_retention_candidate_has_one_oracle_visible_holder_w .Select(path => Path.GetRelativePath(sourceRoot, path).Replace(Path.DirectorySeparatorChar, '/')) .OrderBy(path => path, StringComparer.Ordinal).ToArray(); - callers.ShouldBe(["Services/Agents/Publish/ArtifactManifestStore.cs", "Services/Workflows/Artifacts/ArtifactOffloader.cs", "Services/Workflows/ModelCalls/WorkflowRunModelCallBodyArtifactWriter.cs", "Services/Workflows/Runtime/WorkflowSensitivePayloadStore.cs"], + ArtifactRetentionPolicy.SessionTranscriptCheckpoint.MinimumAge.ShouldBe(TimeSpan.FromHours(2), + "a run writes one checkpoint a minute and each supersedes the last, so the event class's seven-day floor would hold roughly a gigabyte of superseded transcript per long run for over a week"); + ArtifactRetentionPolicy.SessionTranscriptCheckpoint.QuarantineWindow.ShouldBe(TimeSpan.FromHours(24), + "the short floor buys promptness; the second, independent unreferenced-observation wait is deliberately unchanged"); + ArtifactRetentionPolicy.For(nameof(ArtifactRetentionClass.SessionTranscriptCheckpoint)).ShouldNotBeNull( + "an unregistered class settles Indeterminate and is kept for ever — which is exactly the leak this class exists to close"); + + callers.ShouldBe(["Services/Agents/Publish/ArtifactManifestStore.cs", "Services/Agents/Recovery/Checkpoints/ArtifactSessionTranscriptCheckpointer.cs", "Services/Workflows/Artifacts/ArtifactOffloader.cs", "Services/Workflows/ModelCalls/WorkflowRunModelCallBodyArtifactWriter.cs", "Services/Workflows/Runtime/WorkflowSensitivePayloadStore.cs"], "every retention-candidate id must be freshly obtained through PutDeclaredAsync immediately before its one oracle-visible holder write"); typeof(IArtifactManifestStore).GetMethod(nameof(IArtifactManifestStore.CaptureDeclaredAsync))!.ReturnType.ShouldBe(typeof(Task), "manifest capture must not return the candidate artifact id for a later writer to reuse without passing the store's availability/dedup fence again"); File.ReadAllText(Path.Combine(sourceRoot, callers[0])).ShouldContain("ContentArtifactId = artifactId"); - File.ReadAllText(Path.Combine(sourceRoot, callers[1])).ShouldContain("request.HolderId"); - File.ReadAllText(Path.Combine(sourceRoot, callers[2])).ShouldContain("request.CaptureId"); - File.ReadAllText(Path.Combine(sourceRoot, callers[3])).ShouldContain("CiphertextArtifactId = artifactId"); + // 3c: the checkpoint's one oracle-visible holder is agent_run.session_transcript_checkpoint_artifact_id, and + // the id it stamps is the one this call just obtained — never a re-read of the row or a remembered value. + File.ReadAllText(Path.Combine(sourceRoot, callers[1])).ShouldContain("write.ArtifactId"); + File.ReadAllText(Path.Combine(sourceRoot, callers[2])).ShouldContain("request.HolderId"); + File.ReadAllText(Path.Combine(sourceRoot, callers[3])).ShouldContain("request.CaptureId"); + File.ReadAllText(Path.Combine(sourceRoot, callers[4])).ShouldContain("CiphertextArtifactId = artifactId"); typeof(IArtifactRetentionOffloader).IsAssignableFrom(typeof(ArtifactOffloader)).ShouldBeTrue( "the service registered as IArtifactOffloader must carry the holder-aware sibling without widening the ordinary interface"); var agentRunService = File.ReadAllText(Path.Combine(sourceRoot, "Services/Agents/AgentRunService.cs"));