diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs index f18d8da3c..e21bd5ce4 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs @@ -19,6 +19,7 @@ using CodeSpace.Core.Services.Workflows.Llm; using CodeSpace.Core.Services.Workflows.Planning.Planners; using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Exceptions; using CodeSpace.Core.Services.Agents.Sandbox.Isolation; using CodeSpace.Core.Services.Agents.Sandbox.Runners; using CodeSpace.Core.Services.Agents.Tools; @@ -622,17 +623,32 @@ public async Task ExecuteAsync(Guid agentRunId, CancellationToken cancellationTo } } - var reviseSpec = BuildSpec(reviseTask); + // A revise goal restates the contract, so it can cross a harness's own input cap the original goal fit under + // (Codex's 1,048,576 characters) — warm, or once rebuilt cold below. That refusal is about this round, not + // the run: the last round ran and was graded, and its result stands — the loop stops the way a spend + // refusal stops it, with nothing to settle. + SandboxSpec reviseSpec; + bool roundRanCold; - // The same verdict for the round's own session: warm only when the pipe can carry it. Only the - // conversation and the goal change — a model escalation already applied to this round stands. - var roundRanCold = ContinuationOverflowsTheFrame(reviseTask, reviseSpec); - - if (roundRanCold) + try { - var cold = BuildReviseTask(effectiveTask, result, reason, mayResume: false); - reviseTask = reviseTask with { Goal = cold.Goal, ResumeFromSessionId = null, RestoredTranscript = null }; reviseSpec = BuildSpec(reviseTask); + + // The same verdict for the round's own session: warm only when the pipe can carry it. Only the + // conversation and the goal change — a model escalation already applied to this round stands. + roundRanCold = ContinuationOverflowsTheFrame(reviseTask, reviseSpec); + + if (roundRanCold) + { + var cold = BuildReviseTask(effectiveTask, result, reason, mayResume: false); + reviseTask = reviseTask with { Goal = cold.Goal, ResumeFromSessionId = null, RestoredTranscript = null }; + reviseSpec = BuildSpec(reviseTask); + } + } + catch (SandboxArgumentTooLongException refusal) + { + await AppendReviseStopEventAsync(owner, ReviseSizeStoppedPrefix, refusal.Message, cancellationToken).ConfigureAwait(false); + break; } var priorUsage = result.TokenUsage; @@ -648,7 +664,7 @@ public async Task ExecuteAsync(Guid agentRunId, CancellationToken cancellationTo if (spendClaim is { RefusedDetail: { } roundRefusal }) { spendClaim = null; - await AppendReviseBudgetStopEventAsync(owner, roundRefusal, cancellationToken).ConfigureAwait(false); + await AppendReviseStopEventAsync(owner, ReviseBudgetStoppedPrefix, roundRefusal, cancellationToken).ConfigureAwait(false); break; } @@ -2794,17 +2810,20 @@ private async Task AppendReviseStalledEventAsync(AgentRunOwnerToken owner, strin /// The budget-stopped-revision announcement's pinned prefix — the operator-visible marker that a round was never bought, not that it ran and failed. internal const string ReviseBudgetStoppedPrefix = "Revision stopped — the run's cost cap had no headroom for another attempt"; - /// Announce that the revise loop stopped because the ledger refused the NEXT invocation's claim. The round that already succeeded keeps its result: throwing away finished work because the following attempt cannot be afforded would lose what the operator already paid for. Best-effort, exactly like the stalled announcement above. - private async Task AppendReviseBudgetStopEventAsync(AgentRunOwnerToken owner, string detail, CancellationToken cancellationToken) + /// The timeline's account of a revision the harness refused to build for its size: the last graded round's result stands. + internal const string ReviseSizeStoppedPrefix = "Revision stopped — the revised instruction is past what this agent can be handed, so the last round's result stands"; + + /// Announce that the revise loop stopped before the NEXT round was bought — the ledger refused its claim, or the harness refused to build it for its size. The round that already ran keeps its result: throwing away finished work because the following attempt cannot be run would lose what the operator already paid for. Best-effort, exactly like the stalled announcement above. + private async Task AppendReviseStopEventAsync(AgentRunOwnerToken owner, string reason, string detail, CancellationToken cancellationToken) { var runId = owner.RunId; try { - await _runs.AppendEventAsync(owner, new AgentEvent { Kind = AgentEventKind.Warning, Text = $"{ReviseBudgetStoppedPrefix}. {detail}" }, cancellationToken).ConfigureAwait(false); + await _runs.AppendEventAsync(owner, new AgentEvent { Kind = AgentEventKind.Warning, Text = $"{reason}. {detail}" }, cancellationToken).ConfigureAwait(false); } catch (Exception ex) when (ex is not OperationCanceledException and not AgentRunOwnershipLostException) { - _logger.LogWarning(ex, "Agent run {RunId}: could not record the revise-budget-stop event", runId); + _logger.LogWarning(ex, "Agent run {RunId}: could not record the revise-stop event", runId); } } diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentTerminalOutcomeReader.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentTerminalOutcomeReader.cs index e4ef53398..12f49afa9 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentTerminalOutcomeReader.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentTerminalOutcomeReader.cs @@ -13,6 +13,35 @@ namespace CodeSpace.Core.Services.Agents; /// public static class AgentTerminalOutcomeReader { + /// + /// The a harness's folder stamps when its CLI's OWN terminal event says the + /// request is larger than the model's context window — the provider refused it, or the CLI did before sending it: + /// a field or status the CLI wrote, never a phrase in the agent's prose. Harness-agnostic, like the rest of this reader: each folder knows its CLI's shape, and every + /// consumer (the retry-cause classifier, and through it the agent.run node's retry verdict) keys on this one code. + /// Pinned by a unit test (Rule 8) so the producers and the verdict cannot drift apart. + /// + public const string ContextWindowExceededExitReason = "context-window-exceeded"; + + /// + /// What an OpenAI-compatible gateway (vLLM, LiteLLM) says when it refuses a request as over the model's window — + /// the overflow shape neither CLI stamps as its own, because the gateway speaks in its own words and codes (vLLM + /// sends code: 400, LiteLLM code: "400", with the reason only in the message). Read ONLY off a + /// provider refusal body a CLI relays (Claude's 400 api_error result, Codex's turn.failed) — never + /// off an agent's prose, which is how a crash or a rubric used to pass for a diagnosis. Shared so the two folds + /// cannot disagree about the same gateway. + /// + public static bool NamesAContextOverflow(string providerMessage) => + GatewayOverflowMarkers.Any(marker => providerMessage.Contains(marker, StringComparison.OrdinalIgnoreCase)); + + private static readonly string[] GatewayOverflowMarkers = + { + "maximum context length is", // vLLM, and OpenAI's own wording that gateways pass through + "context_length_exceeded", // OpenAI's error code, embedded in a gateway's message + "ContextWindowExceededError", // LiteLLM's class name, stamped only on a window refusal whatever the upstream + "prompt is too long", // Anthropic's refusal, relayed by a gateway in front of a Claude model + "exceed context limit", // Anthropic's other refusal (input length and max_tokens) + }; + /// /// True when the last Completed-or-Error event in the stream is an Error — i.e. the harness itself reported /// the run failed, even if the OS exit code was 0. False when no such event exists (nothing to reconcile diff --git a/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Claude/ClaudeCodeResultFolder.cs b/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Claude/ClaudeCodeResultFolder.cs index 313d4601c..6bbe563d3 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Claude/ClaudeCodeResultFolder.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Claude/ClaudeCodeResultFolder.cs @@ -1,3 +1,4 @@ +using System.Text.Json; using CodeSpace.Core.Services.Agents.Sandbox; using CodeSpace.Messages.Agents; using CodeSpace.Messages.Enums; @@ -13,8 +14,14 @@ namespace CodeSpace.Core.Services.Agents.Harnesses.Claude; internal sealed class ClaudeCodeResultFolder : IAgentEventFolder { private readonly AgentResultFold _fold = new(); + private JsonElement? _lastErrorLine; - public void Add(AgentEvent normalized) => _fold.Add(normalized); + public void Add(AgentEvent normalized) + { + _fold.Add(normalized); + + if (normalized.Kind == AgentEventKind.Error) _lastErrorLine = normalized.Data; + } public AgentRunResult BuildResult(AgentRunFacts facts, int exitCode, string diagnostics) { @@ -52,8 +59,51 @@ public AgentRunResult BuildResult(AgentRunFacts facts, int exitCode, string diag ?? (string.IsNullOrWhiteSpace(summary) ? null : summary) ?? AgentDiagnosticExcerpt.Explain($"claude exited with code {SandboxExitCode.Describe(exitCode)}", diagnostics); - var exitReason = exitCode != 0 ? "non-zero-exit" : "harness-reported-failure"; + var exitReason = _fold.ReportedFailure && RefusedAsOverContextWindow(_lastErrorLine) ? AgentTerminalOutcomeReader.ContextWindowExceededExitReason + : exitCode != 0 ? "non-zero-exit" : "harness-reported-failure"; return new AgentRunResult { Status = AgentRunStatus.Failed, ExitReason = exitReason, Summary = summary, ChangedFiles = changedFiles, Error = error, TokenUsage = usage, SessionId = sessionId, Model = model }; } + + /// + /// Whether the CLI's own terminal result line says the request is larger than the model's context window. Read off + /// fields the CLI writes and, for the gateway shape, off the refusal body it relays — never off the agent's prose, + /// which is how a crash's last message or a rubric's wording would otherwise pass for a diagnosis. + /// + /// Two verdicts are the CLI's own: prompt_too_long, the provider refused the request; and + /// blocking_limit, the CLI's own token estimate is past the window so it refused locally and sent nothing + /// (observed from 2.1.226 and the pinned 2.1.263 for a goal past roughly 760 KB on a 200K-token model). Either way a + /// respawn sends at least as much. The third shape is a gateway's 400 relayed as api_error. A 5xx is the + /// gateway failing, not the model refusing, and a connection-level failure carries api_error_status: null + /// — both stay ordinary, retryable failures whatever their text says. + /// + private static bool RefusedAsOverContextWindow(JsonElement? line) + { + if (line is not { ValueKind: JsonValueKind.Object } result) return false; + + var terminalReason = ReadString(result, "terminal_reason"); + + if (terminalReason is "prompt_too_long" or "blocking_limit") return true; + + return terminalReason == "api_error" && IsClientRefusal(result) && AgentTerminalOutcomeReader.NamesAContextOverflow(ReadString(result, "result")); + } + + /// The result line's api_error_status is a 400. Null (a connection that never got an answer), absent, or any other kind is not. + private static bool IsClientRefusal(JsonElement result) => + result.TryGetProperty("api_error_status", out var status) && status.ValueKind == JsonValueKind.Number && status.TryGetInt32(out var code) && code == 400; + + /// A string field of the CLI's line, or "" when it is absent or another kind. When it is not valid text (an unpaired surrogate escape a relayed gateway body can carry parses, then throws on read), its escaped text stands in: the words the fold looks for are ASCII, and the fold must never be where a run's work is dropped. + private static string ReadString(JsonElement root, string key) + { + if (!root.TryGetProperty(key, out var value) || value.ValueKind != JsonValueKind.String) return ""; + + try + { + return value.GetString() ?? ""; + } + catch (InvalidOperationException) + { + return value.GetRawText(); + } + } } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexHarness.cs b/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexHarness.cs index e52c55c52..ba877da98 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexHarness.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexHarness.cs @@ -1,6 +1,7 @@ using System.Text.Json; using CodeSpace.Core.DependencyInjection; using CodeSpace.Core.Services.Agents.Mcp; +using CodeSpace.Core.Services.Agents.Sandbox.Exceptions; using CodeSpace.Core.Services.Agents.Skills; using CodeSpace.Messages.Agents; using CodeSpace.Messages.Enums; @@ -136,8 +137,18 @@ public sealed class CodexHarness : IAgentHarness, IAgentHarnessBinary, IAgentHar public IReadOnlyList Models { get; } = new[] { "gpt-5.3-codex", "gpt-5.4", "gpt-5.4-codex" }; + /// + /// The most characters Codex accepts in one input. codex-rs refuses a longer turn/start with + /// input_too_large before any model request — verified against 0.142.2 (the worker's pin) and 0.147.0 — and + /// says so on stderr only. Counted as Unicode scalar values, the way Rust counts a string's characters. Pinned by a + /// test; it moves by PR with the CLI. + /// + public const int MaxInputCharacters = 1_048_576; + public SandboxSpec BuildInvocation(AgentTask task) { + EnsureWithinInputCap(task.Goal); + // P3.2: a CONTINUE re-stage rewrites the `exec --json` seed to `exec resume --json` so Codex picks up the // prior thread. The subcommand must follow `exec` directly; --model, the `-c` overrides (incl. the sandbox on // the resume path — see AppendSandbox), and the stdin `-` positional follow. Null (a fresh run) → the plain seed. @@ -287,6 +298,19 @@ private static bool TryReadModelKey(JsonElement obj, out string model) return model.Length > 0; } + /// + /// Refuse a goal Codex would refuse, before a sandbox is provisioned for it. The limit belongs to this CLI, not to + /// the launch (Rule 7), so it lives on this harness; the refusal is the same terminal, non-retried one an argument + /// the kernel cannot take gets, because every attempt meets the identical cap. + /// + private static void EnsureWithinInputCap(string goal) + { + var characters = goal.EnumerateRunes().Count(); + + if (characters > MaxInputCharacters) + throw new SandboxArgumentTooLongException($"the agent's goal is {characters} characters; codex accepts at most {MaxInputCharacters} in one input and refuses anything longer before any model request. This is an input-size limit of the codex CLI, not a memory limit — shorten the goal, or run this agent on a harness without that cap."); + } + /// /// The config-home files the runner materializes: (1) B1 — AGENTS.md carrying the persona + the always-on /// operating contract (Codex's native instruction channel, since exec has no system-prompt flag; codex loads diff --git a/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexResultFolder.cs b/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexResultFolder.cs index da7781d65..6d2408ecb 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexResultFolder.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Harnesses/Codex/CodexResultFolder.cs @@ -1,6 +1,8 @@ +using System.Text.Json; using CodeSpace.Core.Services.Agents.Sandbox; using CodeSpace.Messages.Agents; using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Failures; namespace CodeSpace.Core.Services.Agents.Harnesses.Codex; @@ -12,9 +14,21 @@ namespace CodeSpace.Core.Services.Agents.Harnesses.Codex; /// internal sealed class CodexResultFolder : IAgentEventFolder { + /// Codex's own sentence for a model refusing a streamed request as over its window (a response.failed, reworded) — observed from Codex 0.147.0. + private const string StreamedOverflowSentence = "ran out of room in the model's context window"; + + /// The JSON-RPC error data Codex writes to stderr when it refuses an input past its own character cap, before any request (observed from 0.142.2 and 0.147.0). + private const string InputCapRefusal = "\"input_error_code\":\"input_too_large\""; + private readonly AgentResultFold _fold = new(); + private JsonElement? _lastErrorLine; + + public void Add(AgentEvent normalized) + { + _fold.Add(normalized); - public void Add(AgentEvent normalized) => _fold.Add(normalized); + if (normalized.Kind == AgentEventKind.Error) _lastErrorLine = normalized.Data; + } public AgentRunResult BuildResult(AgentRunFacts facts, int exitCode, string diagnostics) { @@ -48,8 +62,73 @@ public AgentRunResult BuildResult(AgentRunFacts facts, int exitCode, string diag ?? (string.IsNullOrWhiteSpace(summary) ? null : summary) ?? AgentDiagnosticExcerpt.Explain($"codex exited with code {SandboxExitCode.Describe(exitCode)}", diagnostics); - var exitReason = exitCode != 0 ? "non-zero-exit" : "harness-reported-failure"; + var exitReason = _fold.ReportedFailure && RefusedAsOverContextWindow(_lastErrorLine) ? AgentTerminalOutcomeReader.ContextWindowExceededExitReason + : diagnostics.Contains(InputCapRefusal, StringComparison.Ordinal) ? FailureCodes.SandboxArgumentTooLong + : exitCode != 0 ? "non-zero-exit" : "harness-reported-failure"; return new AgentRunResult { Status = AgentRunStatus.Failed, ExitReason = exitReason, Summary = summary, ChangedFiles = changedFiles, Error = error, TokenUsage = usage, SessionId = sessionId, Model = model }; } + + /// + /// Whether Codex's own terminal turn.failed says the model refused the request as larger than its window: + /// the provider refusal body Codex relays verbatim — OpenAI's typed context_length_exceeded code, or a + /// gateway's overflow message () — or Codex's own + /// sentence for a streamed refusal. Only that event, the turn's verdict, is read, so an agent message about context + /// windows never is; and a body whose code is a 5xx is the gateway failing, which stays retryable. + /// + private static bool RefusedAsOverContextWindow(JsonElement? line) + { + if (line is not { ValueKind: JsonValueKind.Object } failed || !failed.TryGetProperty("type", out var type) || type.ValueKind != JsonValueKind.String || type.GetString() != "turn.failed") return false; + + if (!failed.TryGetProperty("error", out var error) || error.ValueKind != JsonValueKind.Object || !error.TryGetProperty("message", out var message) || message.ValueKind != JsonValueKind.String) return false; + + var text = message.GetString() ?? ""; + + if (text.Contains(StreamedOverflowSentence, StringComparison.Ordinal)) return true; + + return ProviderRefusal(text) is { } refusal && !refusal.ServerError && (refusal.Code == "context_length_exceeded" || AgentTerminalOutcomeReader.NamesAContextOverflow(refusal.Message)); + } + + /// + /// The error of a provider body Codex relays as its message verbatim — its code (a string, or a number as + /// text) and message — or null when the message is not one. + /// + private static (string? Code, string Message, bool ServerError)? ProviderRefusal(string message) + { + try + { + using var body = JsonDocument.Parse(message); + + if (body.RootElement.ValueKind != JsonValueKind.Object || !body.RootElement.TryGetProperty("error", out var error) || error.ValueKind != JsonValueKind.Object) return null; + + var code = error.TryGetProperty("code", out var c) ? c.ValueKind switch { JsonValueKind.String => TextOf(c), JsonValueKind.Number => c.GetRawText(), _ => null } : null; + var text = error.TryGetProperty("message", out var m) && m.ValueKind == JsonValueKind.String ? TextOf(m) : ""; + + return (code, text, int.TryParse(code, out var status) && status >= 500); + } + catch (Exception ex) when (ex is JsonException or InvalidOperationException) + { + // Unparseable, or a gateway-authored key that is not valid text (a lookup unescapes keys): not a refusal + // this fold can read, and never a reason to throw. + return null; + } + } + + /// + /// A string of the relayed body. A body is the gateway's own text, so it can be well-formed JSON that is not valid + /// text: an unpaired surrogate escape (\ud83d) parses, then throws on read. Then the escaped text stands in — + /// every word the fold looks for is ASCII, which an escape cannot split — so one bad character neither throws out of + /// the fold (dropping the run's work) nor hides a refusal the rest of the body states plainly. + /// + private static string TextOf(JsonElement value) + { + try + { + return value.GetString() ?? ""; + } + catch (InvalidOperationException) + { + return value.GetRawText(); + } + } } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Exceptions/SandboxArgumentTooLongException.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Exceptions/SandboxArgumentTooLongException.cs index 100548fd4..1273cceef 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Exceptions/SandboxArgumentTooLongException.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Exceptions/SandboxArgumentTooLongException.cs @@ -4,7 +4,8 @@ namespace CodeSpace.Core.Services.Agents.Sandbox.Exceptions; /// /// The invocation cannot be executed as written: one of its argv or environment strings is past the kernel's -/// per-string ceiling, or its standard input is past what the launch pipe can carry. Its own exit scenario rather than a reason, because the two +/// per-string ceiling, its standard input is past what the launch pipe can carry, or its goal is past the harness +/// CLI's own input cap (Codex's, which refuses before any model request). Its own exit scenario rather than a reason, because the two /// differ in the one way a caller acts on. A launch-slot refusal is about THIS host — a foreign boot, a binding /// conflict, a missing bootstrap — and another worker may well admit it, so it stays retryable. These bytes are /// refused by every kernel on every host, so a retry is N identical refusals, each one burning budget and burying the diff --git a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs index a94c9dbee..9c20085c7 100644 --- a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs +++ b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs @@ -1,4 +1,5 @@ using System.Globalization; +using System.Text.RegularExpressions; using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Tasks.Phases.Sources.Nodes; using CodeSpace.Core.Services.Tasks.Phases.Sources.Supervisor; @@ -853,10 +854,15 @@ private static bool IsAuthError(string? error) if (error.Contains(Agents.Credentials.ModelCredentialLeaseLostException.Explanation, StringComparison.Ordinal)) return false; - return new[] { "401", "unauthorized", "authentication", "api key", "api-key", "credential", "invalid_api_key" } + if (UnauthorizedStatus.IsMatch(error)) return true; + + return new[] { "unauthorized", "authentication", "api key", "api-key", "credential", "invalid_api_key" } .Any(m => error.Contains(m, StringComparison.OrdinalIgnoreCase)); } + /// An HTTP 401 as a status on its own — never the digits inside a longer number, such as the length a size refusal reports (a 1401234-character goal is not a rejected key). + private static readonly Regex UnauthorizedStatus = new(@"(? narrativePhases, string? error) { if (status == WorkflowRunStatus.Cancelled) return "This turn was cancelled."; diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/AgentRetryCauses.cs b/backend/src/CodeSpace.Core/Services/Supervisor/AgentRetryCauses.cs index db676df02..35e6a80f2 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/AgentRetryCauses.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/AgentRetryCauses.cs @@ -31,6 +31,21 @@ public static class AgentRetryCauses /// public const string ModelAccessLost = "model-access-lost"; + /// + /// The request is larger than the model's context window — the provider refused it, or the CLI did before sending + /// it () — deterministic on replay. A + /// retry warm-resumes the conversation, so its request carries the goal AGAIN and is longer still; a fresh one + /// carries the same goal. Either way the same refusal comes back and buries the one fact the author needs: the + /// agent is being given more than this model can read. No mitigation, like — its + /// only consumer that matters is AgentCodeNode, which stops respawning it. + /// + /// Typed by the harness that folded the run, from a field its CLI wrote, and read here off the exit reason + /// only. Never from text: a fail-closed acceptance verdict overwrites the error with the rubric's own words, and a + /// review whose rubric is ABOUT context windows would otherwise classify — which switches the escalation trigger + /// off for exactly the failed grade it exists to act on. + /// + public const string ContextWindowExceeded = "context-window-exceeded"; + /// Seen live 2026-08-30 (run wedge postmortem): the gateway's Anthropic-compat layer broke thinking-block continuation and killed the agent tail with exactly this text. private static readonly string[] FormatFaultMarkers = { "is not a thinking block" }; @@ -39,8 +54,12 @@ public static class AgentRetryCauses /// codebase's own diagnosis and can never be prose about one, so it settles the question outright; only an attempt /// that declared nothing falls through to the marker scan. /// - public static string? Classify(string? exitReason, string? error) => - exitReason == Messages.Failures.FailureCodes.ModelCredentialLeaseLost ? ModelAccessLost : Classify(error); + public static string? Classify(string? exitReason, string? error) => exitReason switch + { + Messages.Failures.FailureCodes.ModelCredentialLeaseLost => ModelAccessLost, + Agents.AgentTerminalOutcomeReader.ContextWindowExceededExitReason => ContextWindowExceeded, + _ => Classify(error), + }; /// The prior attempt's retry-relevant cause read from its error TEXT alone, or null for every ordinary failure (default resume semantics stand unchanged). Prefer the overload above wherever the exit reason is in hand. public static string? Classify(string? error) diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Llm/LlmApiException.cs b/backend/src/CodeSpace.Core/Services/Workflows/Llm/LlmApiException.cs index 0ab344569..664107432 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Llm/LlmApiException.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Llm/LlmApiException.cs @@ -98,7 +98,7 @@ 400 or 422 when MentionsContentFilter(body) => LlmErrorCategory.ContentFiltered, // The category is only a refined LABEL on a 400/422 anyway (the degrade decision is status-based), so a miss merely // leaves the generic BadRequest — it can never disable the progressive fallback. private static bool MentionsContextLength(string? body) => - ContainsAny(body, "context length", "context_length", "context window", "maximum context", "maximum_tokens", "reduce the length", "string too long", "too many tokens", "exceeds the maximum"); + ContainsAny(body, "context length", "context_length", "context window", "maximum context", "maximum_tokens", "reduce the length", "string too long", "too many tokens", "exceeds the maximum", "prompt is too long", "exceed context limit"); private static bool MentionsContentFilter(string? body) => ContainsAny(body, "content filter", "content_filter", "content policy", "content_policy", "content was blocked", "blocked by safety", "safety policy", "flagged"); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs index 6a6e629e9..9bb4629b7 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs @@ -134,7 +134,7 @@ public Task RunAsync(NodeRunContext context, CancellationToken cance if (!TryReadCostCap(context.Config, out var maxCostUsd)) return Fail("Config 'maxCostUsd' must be a positive USD amount when set."); // Resumed: the agent run finished. ResumePayload = { status, summary, changedFiles, branch, error }. - if (context.ResumePayload.HasValue) return Task.FromResult(MapResult(context.ResumePayload.Value, maxCostUsd)); + if (context.ResumePayload.HasValue) return Task.FromResult(MapResult(context.ResumePayload.Value, maxCostUsd, context.RetriesOnFailure)); if (!TryReadPriorSpend(context.PriorAttemptPayload, maxCostUsd, out var budgetSpentUsd, out var budgetError)) return Fail(budgetError!); @@ -296,23 +296,35 @@ private static bool TryReadAcceptance(IReadOnlyDictionary c } /// Map the resumed agent-run outcome onto this node's result. Succeeded → outputs; anything else → a clean node failure, marked retryable only when a fresh respawn could change the outcome. - private static NodeResult MapResult(JsonElement payload, decimal? maxCostUsd) + private static NodeResult MapResult(JsonElement payload, decimal? maxCostUsd, bool retriesOnFailure) { var status = ReadString(payload, "status"); + var succeeded = status == nameof(AgentRunStatus.Succeeded); + var unpriced = false; if (maxCostUsd is { } cap) { - if (ReadFlag(payload, "costIndeterminate")) + var indeterminate = ReadFlag(payload, "costIndeterminate"); + var spendMissing = !TryReadNonNegativeDecimal(payload, "cumulativeCostUsd", out var cumulative) || cumulative is null; + + // A SUCCEEDED run that cannot be priced fails closed on its price: its output would otherwise escape the + // cap unaccounted. A FAILED one has already failed, and its own cause is the sentence the author can act + // on — a launch refused for its size started no process, so it never had a spend to read, and the + // missing price used to stand in for "shorten the goal". It keeps its cause and is only never retried + // (below), which is what the retry's own prior-spend check would decide one attempt later anyway. + if (indeterminate && succeeded) return NodeResult.Fail($"Agent run cannot be priced under the monitored ${cap.ToString(CultureInfo.InvariantCulture)} cost cap.", retryable: false); - if (!TryReadNonNegativeDecimal(payload, "cumulativeCostUsd", out var cumulative) || cumulative is null) + if (spendMissing && succeeded) return NodeResult.Fail($"Agent run cannot be priced under the monitored ${cap.ToString(CultureInfo.InvariantCulture)} cost cap because cumulative spend is missing.", retryable: false); - if (cumulative > cap || (cumulative == cap && status != nameof(AgentRunStatus.Succeeded))) - return NodeResult.Fail($"Agent run stopped: {Supervisor.SupervisorStopReasons.CostCapReached} (${cumulative.Value.ToString(CultureInfo.InvariantCulture)} of ${cap.ToString(CultureInfo.InvariantCulture)} observed).", retryable: false); + unpriced = indeterminate || spendMissing; + + if (!unpriced && (cumulative > cap || (cumulative == cap && !succeeded))) + return NodeResult.Fail($"Agent run stopped: {Supervisor.SupervisorStopReasons.CostCapReached} (${cumulative!.Value.ToString(CultureInfo.InvariantCulture)} of ${cap.ToString(CultureInfo.InvariantCulture)} observed).", retryable: false); } - if (status != nameof(AgentRunStatus.Succeeded)) + if (!succeeded) { var error = ReadString(payload, "error"); @@ -373,16 +385,24 @@ private static NodeResult MapResult(JsonElement payload, decimal? maxCostUsd) // that ALREADY ran mitigated (`thinkingDisabled`, projected from its own dispatched envelope) and died // of the SAME fault has proven the repair does not hold here, so a second identical respawn would only // re-bill a broken gateway and bury the one fact the operator needs. - var formatFault = Supervisor.AgentRetryCauses.Classify(error) == Supervisor.AgentRetryCauses.GatewayFormatFault; + var cause = Supervisor.AgentRetryCauses.Classify(exitReason, error); + var formatFault = cause == Supervisor.AgentRetryCauses.GatewayFormatFault; var mitigationSpent = formatFault && ReadFlag(payload, "thinkingDisabled"); - var deterministic = ((status is nameof(AgentRunStatus.NeedsReview) or nameof(AgentRunStatus.Cancelled) || acceptanceFailed || resourceExhausted || argumentTooLong) + // The request is larger than the context window — the model refused it, or the CLI did before sending it — + // typed by the harness from its CLI's own fields, never read from the error text. A respawn warm-resumes, + // so it re-sends the goal inside a LONGER request, and a fresh one sends the same goal: either way it + // sends at least as much and is refused the same way. The escalation escape below never opens for it: the trigger proposes a stronger model only on a + // failed grade, and stands down for any classified cause, so an overflowing attempt carries no proposal. + var contextWindowExceeded = cause == Supervisor.AgentRetryCauses.ContextWindowExceeded; + + var deterministic = ((status is nameof(AgentRunStatus.NeedsReview) or nameof(AgentRunStatus.Cancelled) || acceptanceFailed || resourceExhausted || argumentTooLong || contextWindowExceeded) && !acceptanceInfraFault && !stalled && !escalationAvailable) || mitigationSpent; - return NodeResult.Fail($"Agent run did not succeed: {(string.IsNullOrEmpty(error) ? status : error)}{FailureCauseSuffix(formatFault, mitigationSpent)}", retryable: !deterministic); + return NodeResult.Fail($"Agent run did not succeed: {(string.IsNullOrEmpty(error) ? status : error)}{FailureCauseSuffix(cause, mitigationSpent)}{UnpricedRetrySuffix(unpriced && !deterministic && retriesOnFailure, maxCostUsd)}", retryable: !deterministic && !unpriced); } var outputs = new Dictionary { ["status"] = JsonSerializer.SerializeToElement(nameof(AgentRunStatus.Succeeded)) }; @@ -400,17 +420,25 @@ private static NodeResult MapResult(JsonElement payload, decimal? maxCostUsd) } /// - /// Name the CAUSE on a gateway format fault's failure message — the text the engine persists as this attempt's - /// attempt.failed / node.failed record, so the run's timeline says the gateway mangled the wire - /// instead of only echoing an opaque CLI line. Deliberately states no more than is true at that instant: it - /// never promises a respawn (whether one is bought is the node's retry budget, which this node cannot see), - /// and on the spent-mitigation arm it says the repair already ran. Empty for every other failure, so every - /// existing message stays byte-identical. + /// Name the CAUSE on a classified failure's message — the text the engine persists as this attempt's + /// attempt.failed / node.failed record, so the run's timeline says what happened instead of only + /// echoing an opaque CLI line. Deliberately states no more than is true at that instant: it never promises a + /// respawn (whether one is bought is the node's retry budget, which this node cannot see), and on the + /// spent-mitigation arm it says the repair already ran. A context overflow gets the author's remedy, because the + /// CLI's own text advises its interactive user (trim your tools, start a new thread), which a workflow author + /// cannot act on. Empty for every other failure, so every existing message stays byte-identical. /// - private static string FailureCauseSuffix(bool formatFault, bool mitigationSpent) => - !formatFault ? "" - : mitigationSpent ? $" ({Supervisor.AgentRetryCauses.GatewayFormatFault}: a fresh conversation with extended thinking disabled hit the same fault)" - : $" ({Supervisor.AgentRetryCauses.GatewayFormatFault})"; + private static string FailureCauseSuffix(string? cause, bool mitigationSpent) => cause switch + { + Supervisor.AgentRetryCauses.GatewayFormatFault when mitigationSpent => $" ({Supervisor.AgentRetryCauses.GatewayFormatFault}: a fresh conversation with extended thinking disabled hit the same fault)", + Supervisor.AgentRetryCauses.GatewayFormatFault => $" ({Supervisor.AgentRetryCauses.GatewayFormatFault})", + Supervisor.AgentRetryCauses.ContextWindowExceeded => $" ({Supervisor.AgentRetryCauses.ContextWindowExceeded}: the request is larger than the context window the CLI or the model allows, and a respawn would send at least as much — give the agent less, or choose a model with a larger window)", + _ => "", + }; + + /// Why a failure a respawn could otherwise change is not respawned: its spend is unknown, so a retry cannot be bought under the cap. Empty whenever that is not the deciding fact — including on a node whose own policy never retries — so the cause stays the whole message. + private static string UnpricedRetrySuffix(bool decides, decimal? maxCostUsd) => + decides ? $"; not retried: its spend cannot be priced under the monitored ${maxCostUsd!.Value.ToString(CultureInfo.InvariantCulture)} cost cap" : ""; /// /// P2.3: stamp the retry-resume hint from the RETIRING prior attempt's own resume payload (the same diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Runtime/MapResultsPrompt.cs b/backend/src/CodeSpace.Core/Services/Workflows/Runtime/MapResultsPrompt.cs index 48ab1f68a..d35c3e9bf 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Runtime/MapResultsPrompt.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Runtime/MapResultsPrompt.cs @@ -33,7 +33,7 @@ namespace CodeSpace.Core.Services.Workflows.Runtime; /// that want more. One pathological branch therefore cannot consume the budget and starve its siblings. /// /// Under budget it is byte-identical to the unbounded binding. The within-budget case returns exactly -/// JsonSerializer.Serialize(resultsArray) as its text — the same call VariableResolver's array arm +/// JsonSerializer.Serialize(resultsArray, WorkflowJson.InterpolatedText) as its text — the same call VariableResolver's array arm /// makes on the same element — so the ordinary small fan-out reaches the model unchanged, character for character, /// and its coverage reads complete. /// @@ -53,13 +53,13 @@ public static class MapResultsPrompt /// public static MapResultsProjection Project(JsonElement resultsArray, int budgetChars) { - var whole = JsonSerializer.Serialize(resultsArray); + var whole = JsonSerializer.Serialize(resultsArray, WorkflowJson.InterpolatedText); var total = resultsArray.ValueKind == JsonValueKind.Array ? resultsArray.GetArrayLength() : 0; if (budgetChars <= 0 || whole.Length <= budgetChars) return Whole(whole, total); if (resultsArray.ValueKind != JsonValueKind.Array) return NothingIncluded(Cut(whole, budgetChars), total); - var branches = resultsArray.EnumerateArray().Select(element => JsonSerializer.Serialize(element)).ToList(); + var branches = resultsArray.EnumerateArray().Select(element => JsonSerializer.Serialize(element, WorkflowJson.InterpolatedText)).ToList(); if (branches.Count == 0) return NothingIncluded(Cut(whole, budgetChars), total); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Runtime/VariableResolver.cs b/backend/src/CodeSpace.Core/Services/Workflows/Runtime/VariableResolver.cs index b8eebba51..20906345e 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Runtime/VariableResolver.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Runtime/VariableResolver.cs @@ -134,7 +134,7 @@ public static IReadOnlyDictionary ResolveBag(JsonElement so JsonValueKind.String => value.GetString() ?? "", JsonValueKind.Number or JsonValueKind.True or JsonValueKind.False => value.ToString(), JsonValueKind.Null or JsonValueKind.Undefined => "", - _ => JsonSerializer.Serialize(value) + _ => JsonSerializer.Serialize(value, WorkflowJson.InterpolatedText) }; }); } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/WorkflowJson.cs b/backend/src/CodeSpace.Core/Services/Workflows/WorkflowJson.cs index 33a40b18e..265e64f19 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/WorkflowJson.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/WorkflowJson.cs @@ -1,3 +1,4 @@ +using System.Text.Encodings.Web; using System.Text.Json; using System.Text.Json.Serialization; @@ -14,6 +15,22 @@ public static class WorkflowJson { public static JsonSerializerOptions Options { get; } = BuildOptions(); + /// + /// How an object or array is written into surrounding TEXT — a prompt, a message body — by a {{...}} + /// template, and by any projection that must stay byte-identical to one (MapResultsPrompt). Characters are + /// left as themselves: the HTML-safe default turned every CJK character and every + < > & ' into a + /// six-character \uXXXX, so a pull-request diff bound into a review prompt reached the model as escape + /// codes at up to twice its size. An object now behaves like the string branch beside it, which has always inserted + /// those characters raw. That is not a new exposure, but it is not "the escaping protected nothing" either: in a + /// URL query, a header or a single-quoted sh -c string the escape codes did keep an object's & or + /// ' from reading as syntax — protection a string value there never had, so no template could rely on it, + /// and a value bound into such a context needs that context's own encoding. What JSON itself requires is still + /// escaped (the quote, the backslash, control characters), so a value embedded in a JSON body stays parseable. + /// One exception no built-in encoder lifts: a character outside the Basic Multilingual Plane (an emoji) is still + /// written as its surrogate-pair escape. + /// + public static JsonSerializerOptions InterpolatedText { get; } = new() { Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping }; + private static JsonSerializerOptions BuildOptions() { var options = new JsonSerializerOptions(JsonSerializerDefaults.Web); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunReviseLoopFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunReviseLoopFlowTests.cs index 0bc4dbc43..ae90c67b2 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunReviseLoopFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunReviseLoopFlowTests.cs @@ -4,6 +4,7 @@ using CodeSpace.Core.Persistence.Entities; using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Agents.Harnesses.Claude; +using CodeSpace.Core.Services.Agents.Harnesses.Codex; using CodeSpace.Core.Services.Agents.ModelCredentials; using CodeSpace.Core.Services.Agents.Sandbox; using CodeSpace.Core.Services.Agents.Sandbox.Runners; @@ -130,6 +131,38 @@ public async Task A_cold_revision_refused_for_spend_leaves_no_note_that_it_conti events.ShouldNotContain(AgentRunExecutor.ReviseRanColdNote, "no round ran, so no round continued as a fresh conversation"); } + [Fact] + public async Task A_revision_whose_restated_goal_crosses_codexs_input_cap_stops_revising_and_keeps_the_verified_round() + { + // A cold revise goal restates the whole contract, so it is longer than the goal round 0 ran under. When the + // goal sits just under Codex's own input cap, the revision crosses it and the harness refuses it at build. + // That refusal is about the revision, not the run: round 0 ran, pushed and was graded, and its verdict stands. + // Thrown out of the loop instead, it replaced round 0's whole result — diff, summary, verdict — with a bare + // size refusal, and marked the run as if nothing had run. + if (OperatingSystem.IsWindows()) return; + + var (teamId, userId) = await SeedTeamAsync(); + using var remote = new BareRemote(); + await remote.SeedBaseAsync(CheckScript); + var repoId = await SeedBoundRepositoryAsync(teamId, remote.Url); + var goal = new string('x', CodexHarness.MaxInputCharacters - 500); + var runId = await CreateRunAsync(teamId, userId, TaskWith(repoId) with { Goal = goal, MaxReviseRounds = 1 }); + var harness = new CodexSpecHarness(); + + await ExecuteAsync(runId, harness); + + var (run, result) = await LoadAsync(runId); + var events = await LoadEventsAsync(runId); + + harness.Goals.Count.ShouldBe(2, "fixture check: round 0 was built, and so was the revision the cap refused"); + harness.Goals[1].Length.ShouldBeGreaterThan(CodexHarness.MaxInputCharacters, "fixture check: the restated goal really crosses the cap"); + result.ExitReason.ShouldBe(AgentAcceptanceContract.FailClosedExitReason, $"round 0's verdict stands — the run ended {result.ExitReason}: {run.Error}"); + result.ChangedFiles.ShouldContain("feature.txt", "round 0's work is still the run's result"); + result.ReviseRounds.ShouldBe(0); + events.ShouldContain(e => e.StartsWith(AgentRunExecutor.ReviseSizeStoppedPrefix, StringComparison.Ordinal), "the timeline says why the revision stopped"); + (await remote.BranchFileContentAsync(AgentRunExecutor.BuildBranchName(runId), "feature.txt")).ShouldNotBeNull("and round 0's branch is still there"); + } + [Fact] public async Task A_revision_receives_the_real_oracle_diagnosis_instead_of_only_its_exit_code() { @@ -800,6 +833,33 @@ public SandboxSpec BuildInvocation(AgentTask task) public string? SessionTranscriptRelativePath(string configHome, string? workspaceDirectory, string? sessionId) => _real.SessionTranscriptRelativePath(configHome, workspaceDirectory, sessionId); } + /// A "scripted" harness whose spec is the REAL Codex adapter's — its own input cap included — with only the executable swapped; round 0 writes draft work, a revision the fixed work. + private sealed class CodexSpecHarness : IAgentHarness + { + private readonly CodexHarness _real = new(); + + public string Kind => "scripted"; + public string Version => "test"; + public IReadOnlyList Models { get; } = new[] { "test-model" }; + public List Goals { get; } = new(); + + public SandboxSpec BuildInvocation(AgentTask task) + { + Goals.Add(task.Goal); + var real = _real.BuildInvocation(task); + var revising = task.Goal.StartsWith(AgentRunExecutor.ReviseInstructionPrefix, StringComparison.Ordinal); + return real with { Command = "/bin/sh", Args = new[] { "-c", "cat >/dev/null; " + (revising ? RevisedScript : DraftScript) } }; + } + + public IReadOnlyList ParseEvents(string rawLine) => + string.IsNullOrWhiteSpace(rawLine) ? Array.Empty() : new[] { new AgentEvent { Kind = AgentEventKind.AssistantMessage, Text = rawLine.Trim() } }; + + public IAgentEventFolder CreateFolder() => new TestEventFolder((fold, exitCode) => + exitCode == 0 + ? new AgentRunResult { Status = AgentRunStatus.Succeeded, ExitReason = "completed", Summary = fold.LastText } + : new AgentRunResult { Status = AgentRunStatus.Failed, ExitReason = "non-zero-exit", Error = $"exit {exitCode}" }); + } + private sealed class ReviseAwareHarness : IAgentHarness { private readonly string _first; diff --git a/backend/tests/CodeSpace.UnitTests/Agents/AgentContextWindowRetryTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/AgentContextWindowRetryTests.cs new file mode 100644 index 000000000..7634c5cb7 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Agents/AgentContextWindowRetryTests.cs @@ -0,0 +1,231 @@ +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Harnesses.Claude; +using CodeSpace.Core.Services.Agents.Harnesses.Codex; +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Failures; +using Shouldly; + +namespace CodeSpace.UnitTests.Agents; + +/// +/// A context-window overflow is recognised from what the CLIs actually print, through the real harness folds. +/// +/// The shapes below are terminal lines real CLIs printed (Claude Code 2.1.226 and the pinned 2.1.263, Codex +/// 0.147.0 and the pinned 0.142.2) when a local endpoint answered with a provider's own error body, trimmed to the +/// fields the fold reads — except where a case says it is representative or synthesized. They are run through ParseEvents and BuildResult — the reader production uses — and the +/// cause is read off the exit reason the fold typed from the CLI's own fields, never off the error text: a text scan +/// also matched a rubric that talks about context windows, and switched model escalation off for its failed grade. +/// +/// Why it matters: an overflow used to be an ordinary retryable failure. A retry warm-resumes, so the next +/// request carries the goal twice and overflows harder; a task-launched agent node spends all three default +/// attempts that way and ends on three identical refusals. +/// +[Trait("Category", "Unit")] +public sealed class AgentContextWindowRetryTests +{ + public static TheoryData ClaudeOverflowLines => new() + { + // Anthropic "prompt is too long: N tokens > M maximum" — the CLI's own rewording, stamped terminal_reason=prompt_too_long. + """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"prompt_too_long","api_error_status":400,"session_id":"8d014f31-0881-46c2-8311-2ad17d852c4f","result":"Prompt is too long · the request is ~215000 tokens (limit 200000) but this conversation is only ~30286 tokens — the rest is system prompt, tool definitions, and attachment content. A single-exchange conversation cannot be compacted; reduce attached files/tools or start with less context."}""", + // Anthropic "input length and max_tokens exceed context limit". + """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"prompt_too_long","api_error_status":400,"session_id":"58110d54-f8b2-43f3-9153-dcd32a2d58dc","result":"Prompt is too long · this conversation is a single exchange and cannot be compacted — the request size comes mostly from system prompt, tool definitions, or attachments."}""", + // An OpenAI-compatible gateway (vLLM/LiteLLM) body the CLI passes through verbatim as terminal_reason=api_error. + """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"api_error","api_error_status":400,"session_id":"6406b570-40de-4274-a586-ab044dbf5d48","result":"API Error: 400 This model's maximum context length is 131072 tokens. However, you requested 161234 tokens. Please reduce the length of the messages."}""", + // The CLI's own local refusal: its token estimate is past the window, so it sent no request at all (pinned 2.1.263, a 1.2 MB goal). + """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"blocking_limit","api_error_status":null,"session_id":"ff8b4b95-8351-4462-ada3-c302b4f526d8","result":"Prompt is too long"}""", + }; + + public static TheoryData ClaudeConnectionFailureLines => new() + { + // Pinned 2.1.263 against a dead port and a peer that resets: the status is null because no answer ever came. + { """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"api_error","api_error_status":null,"session_id":"54818d4e-66e4-4f1b-b3ee-73de94f3b9cc","result":"API Error: Connection refused — a firewall or proxy may be blocking it (ConnectionRefused)"}""", "API Error: Connection refused — a firewall or proxy may be blocking it (ConnectionRefused)" }, + { """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"api_error","api_error_status":null,"session_id":"f01a6528-1c83-4400-af5d-4ac6d322c7f9","result":"API Error: Connection dropped (ECONNRESET)"}""", "API Error: Connection dropped (ECONNRESET)" }, + }; + + public static TheoryData CodexOverflowLines => new() + { + // OpenAI Responses API 400 context_length_exceeded. + """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"Your input exceeds the context window of this model. Please adjust your input and try again.\", \"type\": \"invalid_request_error\", \"param\": \"input\", \"code\": \"context_length_exceeded\"}}"}}""", + // The same refusal delivered as a streaming response.failed, which Codex rewords. + """{"type":"turn.failed","error":{"message":"Codex ran out of room in the model's context window. Start a new thread or clear earlier history before retrying."}}""", + // A vLLM gateway's refusal, relayed verbatim by the pinned 0.142.2: the code is the number 400, the reason only in the message. + """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"This model's maximum context length is 131072 tokens. However, you requested 161234 tokens (161234 in the messages, 0 in the completion). Please reduce the length of the messages or completion.\", \"type\": \"BadRequestError\", \"param\": null, \"code\": 400}}"}}""", + // A LiteLLM proxy in front of a Claude model: its class name and Anthropic's own words, neither OpenAI-shaped. + // Representative of LiteLLM's exception mapping (not byte-captured from a live proxy), relayed the way 0.142.2 relays any 400 body. + """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"litellm.ContextWindowExceededError: litellm.BadRequestError: AnthropicException - {\\\"type\\\":\\\"error\\\",\\\"error\\\":{\\\"type\\\":\\\"invalid_request_error\\\",\\\"message\\\":\\\"prompt is too long: 215000 tokens > 200000 maximum\\\"}}\", \"type\": null, \"param\": null, \"code\": \"400\"}}"}}""", + // A LiteLLM proxy's refusal, relayed verbatim by the pinned 0.142.2: the code is the string "400". + """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"litellm.ContextWindowExceededError: litellm.BadRequestError: ContextWindowExceededError: OpenAIException - Error code: 400 - {'error': {'message': \\\"This model's maximum context length is 128000 tokens. However, your messages resulted in 161234 tokens.\\\", 'type': 'invalid_request_error', 'param': 'messages', 'code': 'context_length_exceeded'}}\", \"type\": null, \"param\": null, \"code\": \"400\"}}"}}""", + }; + + [Theory] + [MemberData(nameof(ClaudeOverflowLines))] + public void A_claude_context_overflow_folds_to_an_error_classified_as_one(string terminalLine) + { + var harness = new ClaudeCodeHarness(); + + var result = harness.BuildResult(harness.ParseEvents(terminalLine), exitCode: 1, diagnostics: ""); + + result.Status.ShouldBe(AgentRunStatus.Failed); + result.ExitReason.ShouldBe(AgentTerminalOutcomeReader.ContextWindowExceededExitReason, customMessage: "the CLI's own terminal_reason / status is the evidence — typed on the exit reason, never re-read from prose"); + AgentRetryCauses.Classify(result.ExitReason, result.Error).ShouldBe(AgentRetryCauses.ContextWindowExceeded, customMessage: $"the folded error was: {result.Error}"); + } + + [Theory] + [MemberData(nameof(CodexOverflowLines))] + public void A_codex_context_overflow_folds_to_an_error_classified_as_one(string terminalLine) + { + var harness = new CodexHarness(); + + var result = harness.BuildResult(harness.ParseEvents(terminalLine), exitCode: 1, diagnostics: ""); + + result.Status.ShouldBe(AgentRunStatus.Failed); + result.ExitReason.ShouldBe(AgentTerminalOutcomeReader.ContextWindowExceededExitReason); + AgentRetryCauses.Classify(result.ExitReason, result.Error).ShouldBe(AgentRetryCauses.ContextWindowExceeded, customMessage: $"the folded error was: {result.Error}"); + } + + [Theory] + [MemberData(nameof(ClaudeConnectionFailureLines))] + public void A_claude_connection_failure_folds_to_its_own_cause(string terminalLine, string cliError) + { + // The status is null when no answer came. Reading it as a number threw out of the fold, so the run landed as + // executor-error with a .NET message, and its diff, transcript and session were never captured. + var harness = new ClaudeCodeHarness(); + + var result = harness.BuildResult(harness.ParseEvents(terminalLine), exitCode: 1, diagnostics: ""); + + result.Status.ShouldBe(AgentRunStatus.Failed); + result.ExitReason.ShouldBe("non-zero-exit"); + result.Error.ShouldBe(cliError); + result.SessionId.ShouldNotBeNullOrEmpty(customMessage: "the fold completed, so the session a retry resumes is still there"); + } + + public static TheoryData CodexNonRefusalLines => new() + { + // What the pinned 0.142.2 really prints for a status it does not relay verbatim: a 503, 413 and 422 whose bodies + // name the window. It rewraps them as prose, so they are not a relayed refusal, and a 5xx is worth a retry. + """{"type":"turn.failed","error":{"message":"unexpected status 503 Service Unavailable: upstream prefill timed out; maximum context length is 131072 tokens, url: http://127.0.0.1:19703/v1/responses"}}""", + """{"type":"turn.failed","error":{"message":"unexpected status 413 Payload Too Large: This model's maximum context length is 131072 tokens. However, you requested 161234 tokens., url: http://127.0.0.1:19719/v1/responses"}}""", + """{"type":"turn.failed","error":{"message":"unexpected status 422 Unprocessable Entity: This model's maximum context length is 131072 tokens. However, you requested 161234 tokens., url: http://127.0.0.1:19891/v1/responses"}}""", + // Belt and braces for the 5xx guard: a JSON body with a 5xx code, which the pin does not print today. + """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"upstream timed out after prefilling; maximum context length is 131072 tokens\", \"type\": \"ServiceUnavailableError\", \"code\": 503}}"}}""", + }; + + [Theory] + [MemberData(nameof(CodexNonRefusalLines))] + public void A_codex_failure_that_is_not_a_relayed_refusal_is_not_an_overflow_whatever_it_mentions(string line) + { + var harness = new CodexHarness(); + + var result = harness.BuildResult(harness.ParseEvents(line), exitCode: 1, diagnostics: ""); + + result.ExitReason.ShouldBe("non-zero-exit"); + } + + public static TheoryData CodexUnreadableBodyLines => new() + { + // The pinned 0.142.2 relaying a gateway 400 whose message holds an unpaired surrogate escape (a preview cut in + // the middle of an emoji). The body parses and then throws on read; the rest of it still says what it says. + { """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"Invalid value for input[0]: preview \\ud83d ... (truncated)\", \"type\": \"BadRequestError\", \"param\": null, \"code\": 400}}"}}""", "non-zero-exit" }, + { """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"This model's maximum context length is 131072 tokens. However, you requested 161234 tokens. Prompt starts: \\ud83d\", \"type\": \"BadRequestError\", \"param\": null, \"code\": 400}}"}}""", AgentTerminalOutcomeReader.ContextWindowExceededExitReason }, + // OpenAI's typed code beside an unreadable message, as the pin relays it: the code alone settles it. + { """{"type":"turn.failed","error":{"message":"{\"error\": {\"message\": \"Your input exceeds the context window of this model. Input preview: \\ud83d\", \"type\": \"invalid_request_error\", \"param\": \"input\", \"code\": \"context_length_exceeded\"}}"}}""", AgentTerminalOutcomeReader.ContextWindowExceededExitReason }, + + }; + + [Theory] + [MemberData(nameof(CodexUnreadableBodyLines))] + public void A_relayed_body_that_is_not_valid_text_never_throws_and_still_says_what_it_says(string failedLine, string expectedExitReason) + { + // A throw here lands the run as executor-error and drops its session, diff and transcript — the fold is never + // the place that happens. And one bad character must not hide a refusal the rest of the body states plainly. + var harness = new CodexHarness(); + var events = harness.ParseEvents("""{"type":"thread.started","thread_id":"01a0d18d-0000-7000-8000-000000000001"}""").Concat(harness.ParseEvents(failedLine)).ToList(); + + var result = harness.BuildResult(events, exitCode: 1, diagnostics: ""); + + result.ExitReason.ShouldBe(expectedExitReason); + result.SessionId.ShouldNotBeNullOrEmpty(customMessage: "the fold completed, so the thread a retry resumes is kept"); + } + + [Fact] + public void Codex_refusing_an_input_past_its_own_character_cap_is_that_refusal_not_a_context_overflow() + { + // Codex refuses before any request, on stderr only. It is the CLI's own character cap — the condition the + // harness preflight refuses up front — not the model's window, so "choose a larger-window model" would be + // advice that cannot help. It folds to the same terminal code as the preflight. + const string stderr = """Error: turn/start: turn/start failed: Input exceeds the maximum length of 1048576 characters. (code -32602), data: {"input_error_code":"input_too_large","max_chars":1048576,"actual_chars":1100000}"""; + + var result = new CodexHarness().BuildResult(Array.Empty(), exitCode: 1, diagnostics: stderr); + + result.ExitReason.ShouldBe(FailureCodes.SandboxArgumentTooLong); + result.Error.ShouldContain("Input exceeds the maximum length of 1048576 characters", customMessage: "the CLI's own sentence is the cause the author reads"); + } + + [Fact] + public void A_rubric_that_talks_about_context_windows_does_not_switch_escalation_off() + { + // The regression a text scan made: a fail-closed acceptance verdict overwrites Error with the rubric's own + // requirement and the judge's evidence. A review whose rubric is ABOUT context windows then read as an + // overflow, and the escalation trigger — which stands down for any classified cause — stopped offering the + // stronger model the failed grade was evidence for. A cause is typed by the harness now; prose never is one. + const string graderProse = "The acceptance check did not pass: requirement 'returns context_length_exceeded when the input exceeds the context window' — evidence: the handler throws instead."; + + AgentRetryCauses.Classify(AgentAcceptanceContract.FailClosedExitReason, graderProse).ShouldBeNull(); + AgentModelEscalationTrigger.Reason(AgentContradiction.OverClaim, acceptanceFailed: true, acceptanceDetail: "requirement not met", workPresent: true, error: graderProse, exitReason: AgentAcceptanceContract.FailClosedExitReason) + .ShouldNotBeNull(customMessage: "a failed grade on a claimed success is escalation evidence, whatever words the rubric uses"); + } + + [Fact] + public void A_crashed_run_whose_last_words_mention_an_overflow_is_not_one() + { + // No result line: the harness reports the agent's own last message as the error. That is the agent's prose. + var harness = new ClaudeCodeHarness(); + var lastWords = """{"type":"assistant","message":{"content":[{"type":"text","text":"The docs say a request fails with 'Prompt is too long' past the context window; let me check."}]}}"""; + + var result = harness.BuildResult(harness.ParseEvents(lastWords), exitCode: 137, diagnostics: ""); + + result.ExitReason.ShouldNotBe(AgentTerminalOutcomeReader.ContextWindowExceededExitReason); + AgentRetryCauses.Classify(result.ExitReason, result.Error).ShouldBeNull(customMessage: "a crash stays a retryable crash"); + } + + [Fact] + public void A_transient_gateway_error_whose_body_mentions_context_length_is_not_an_overflow() + { + // Only the model REFUSING the request (a 4xx) is an overflow; a 5xx is the gateway, and it is worth a retry. + var harness = new ClaudeCodeHarness(); + var line = """{"type":"result","subtype":"success","is_error":true,"num_turns":1,"terminal_reason":"api_error","api_error_status":503,"session_id":"s","result":"API Error: 503 upstream timed out after prefilling; maximum context length is 131072 tokens"}"""; + + var result = harness.BuildResult(harness.ParseEvents(line), exitCode: 1, diagnostics: ""); + + result.ExitReason.ShouldNotBe(AgentTerminalOutcomeReader.ContextWindowExceededExitReason); + } + + [Theory] + [InlineData("claude exited with code 1")] + [InlineData("patch did not apply")] + [InlineData("API Error: 529 Overloaded")] + [InlineData("codex exited with code 1 — stderr: npm WARN deprecated glob@7")] + [InlineData("Prompt is too long")] + [InlineData("the handler must return context_length_exceeded when the input exceeds the context window")] + public void Text_alone_is_never_an_overflow(string error) + { + AgentRetryCauses.Classify(error).ShouldBeNull(customMessage: "only a harness-typed exit reason is an overflow; words are not"); + AgentRetryCauses.Classify("non-zero-exit", error).ShouldBeNull(); + } + + [Fact] + public void The_overflow_exit_reason_is_pinned() + { + // Both harness folders stamp it and the node's retry verdict keys on it; a rename that missed either side + // would silently turn an overflow back into three identical, billed refusals. + AgentTerminalOutcomeReader.ContextWindowExceededExitReason.ShouldBe("context-window-exceeded"); + } + + [Fact] + public void The_format_fault_keeps_its_own_cause() + { + AgentRetryCauses.Classify("API Error: 400 messages.3.content.0: Content block is not a thinking block").ShouldBe(AgentRetryCauses.GatewayFormatFault); + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomNarrativeTests.cs b/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomNarrativeTests.cs index 9e0a44179..bbea39a07 100644 --- a/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomNarrativeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomNarrativeTests.cs @@ -1,4 +1,6 @@ using CodeSpace.Core.Services.Agents.Credentials; +using CodeSpace.Core.Services.Agents.Harnesses.Codex; +using CodeSpace.Core.Services.Agents.Sandbox.Exceptions; using CodeSpace.Core.Services.Sessions.Journal.FactsSources; using CodeSpace.Core.Services.Sessions.Room; using CodeSpace.Core.Services.Tasks.Phases; @@ -236,6 +238,32 @@ public void A_lost_model_credential_lease_is_NOT_dressed_up_as_a_rejected_creden diag.Text.ShouldContain("Retry", Case.Insensitive, "and what to do about it"); } + [Fact] + public void A_size_refusal_whose_count_contains_401_is_not_dressed_up_as_a_rejected_credential() + { + // The refusal names how long the goal is, and a length such as 1401234 carries "401" inside it. Read as a + // substring, that sent the author to rotate a working key for a goal that was simply too long. + var refusal = Should.Throw(() => new CodexHarness().BuildInvocation(new AgentTask { Goal = new string('x', 1_401_234), Harness = CodexHarness.HarnessKind, WorkspaceDirectory = "/tmp/ws" })); + var facts = new RoomTurnFacts { RawError = $"Agent run did not succeed: {refusal.Message}" }; + + var diag = Build(Array.Empty(), WorkflowRunStatus.Failure, facts: facts).Blocks.OfType().ShouldHaveSingleItem(); + + refusal.Message.ShouldContain("1401234", customMessage: "fixture check: the producer's own text carries the digits"); + diag.Title.ShouldNotBe("Authentication failed"); + diag.Actions.ShouldNotContain(a => a.Kind == RoomActionKind.FixCredentials); + } + + [Theory] + [InlineData("OpenAI API error: 401 Unauthorized")] + [InlineData("API Error: 401 {\"type\":\"error\"}")] + [InlineData("request failed with status code 401")] + public void A_401_status_still_reads_as_a_rejected_credential(string error) + { + var diag = Build(Array.Empty(), WorkflowRunStatus.Failure, facts: new RoomTurnFacts { RawError = error }).Blocks.OfType().ShouldHaveSingleItem(); + + diag.Title.ShouldBe("Authentication failed"); + } + [Fact] public void A_non_auth_failure_humanizes_the_engine_error() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs index b0f34a4f3..c9e76edc1 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs @@ -547,6 +547,69 @@ public async Task A_resumed_failure_carries_the_retry_verdict_for_the_engine(str result.Retryable.ShouldBe(expectedRetryable, "the node's verdict tells the retry policy whether a fresh agent could change the outcome"); } + [Theory] + [InlineData("Prompt is too long · the request is ~215000 tokens (limit 200000) but this conversation is only ~30286 tokens")] + [InlineData("API Error: 400 This model's maximum context length is 131072 tokens. However, you requested 161234 tokens.")] + [InlineData("Codex ran out of room in the model's context window. Start a new thread or clear earlier history before retrying.")] + public async Task A_context_window_overflow_is_not_respawned(string error) + { + // A respawn warm-resumes the conversation, so it re-sends the goal inside a LONGER request — it overflows again, + // harder. Every attempt after the first is an identical, billed refusal that buries the one fact the author needs. + var resume = JsonDocument.Parse(JsonSerializer.Serialize(new { status = "Failed", error, exitReason = AgentTerminalOutcomeReader.ContextWindowExceededExitReason })).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(new(), resume), CancellationToken.None); + + result.Status.ShouldBe(NodeStatus.Failure); + result.Retryable.ShouldBeFalse(); + result.Error.ShouldContain("context window", Case.Insensitive, customMessage: "the node's own message must name the cause, whatever advice the CLI's text gives"); + } + + [Fact] + public async Task A_launch_refused_for_its_size_on_a_cost_capped_node_says_why_instead_of_cannot_be_priced() + { + // Refused before any CLI started, so nothing was spent and there is nothing to price. The cost cap's + // "cannot be priced — cumulative spend is missing" used to replace the one sentence that says what to shorten. + var config = new Dictionary { ["maxCostUsd"] = Num(5) }; + var resume = JsonDocument.Parse(JsonSerializer.Serialize(new { status = "Failed", error = "the agent's goal is 1100000 characters; codex accepts at most 1048576", exitReason = FailureCodes.SandboxArgumentTooLong })).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(config, resume), CancellationToken.None); + + result.Status.ShouldBe(NodeStatus.Failure); + result.Retryable.ShouldBeFalse(); + result.Error.ShouldContain("codex accepts at most 1048576"); + result.Error.ShouldNotContain("cannot be priced", Case.Insensitive); + } + + [Fact] + public async Task An_unpriced_failure_on_a_cost_capped_node_keeps_its_cause_and_is_not_retried() + { + // A crash is ordinarily worth a respawn, but under a cap an unpriced attempt cannot buy one. The node says so + // AFTER the cause, rather than instead of it. + var config = new Dictionary { ["maxCostUsd"] = Num(5) }; + var resume = JsonDocument.Parse(JsonSerializer.Serialize(new { status = "Failed", error = "claude exited with code 1", exitReason = "non-zero-exit" })).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(config, resume), CancellationToken.None); + + result.Status.ShouldBe(NodeStatus.Failure); + result.Retryable.ShouldBeFalse(); + result.Error.ShouldStartWith("Agent run did not succeed: claude exited with code 1"); + result.Error.ShouldEndWith("; not retried: its spend cannot be priced under the monitored $5 cost cap"); + } + + [Fact] + public async Task An_unpriced_failure_on_a_single_attempt_node_does_not_blame_the_cost_cap() + { + // With no retry policy there was never a retry to refuse, so naming the cap as the reason would send the + // operator after pricing for a retry that could not have happened either way. Still not retryable. + var config = new Dictionary { ["maxCostUsd"] = Num(5) }; + var resume = JsonDocument.Parse(JsonSerializer.Serialize(new { status = "Failed", error = "claude exited with code 1", exitReason = "non-zero-exit" })).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(config, resume, retriesOnFailure: false), CancellationToken.None); + + result.Retryable.ShouldBeFalse(); + result.Error.ShouldBe("Agent run did not succeed: claude exited with code 1"); + } + [Fact] public void The_two_watchdogs_are_classified_alike_because_neither_can_see_why_the_process_went_quiet() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/CodexHarnessTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/CodexHarnessTests.cs index 832af2c8b..be0c8a2ca 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/CodexHarnessTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/CodexHarnessTests.cs @@ -1,3 +1,6 @@ +using CodeSpace.Messages.Failures; +using CodeSpace.Core.Services.Agents.Sandbox.Exceptions; +using System.Globalization; using CodeSpace.Core.Services.Agents.Sandbox; using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Agents.Harnesses.Codex; @@ -51,6 +54,46 @@ public void A_resumed_thread_takes_its_prompt_from_stdin_behind_the_dash() spec.StandardInput.ShouldBe("Fix the failing billing tests"); } + [Fact] + public void Codexs_own_input_cap_is_pinned() + { + // codex-rs refuses a turn/start input past this many characters (input_too_large) before any model request — + // verified against 0.142.2, the worker's pin, and 0.147.0. If a Codex bump moves it, this moves by PR. + CodexHarness.MaxInputCharacters.ShouldBe(1_048_576); + } + + [Fact] + public void A_goal_at_codexs_input_cap_builds_an_invocation() + { + Should.NotThrow(() => Harness.BuildInvocation(Task(goal: new string('x', CodexHarness.MaxInputCharacters)))); + } + + [Fact] + public void A_goal_past_codexs_input_cap_is_refused_before_launch_as_deterministic() + { + // Without this the launch succeeds, `codex exec -` reads stdin, and exits 1 before any request with the refusal + // on stderr only — a run that clones, provisions and launches only to be told no, and that a retry reproduces. + var goal = new string('x', CodexHarness.MaxInputCharacters + 1); + + var refusal = Should.Throw(() => Harness.BuildInvocation(Task(goal: goal))); + + ((IFailure)refusal).Code.ShouldBe(FailureCodes.SandboxArgumentTooLong, customMessage: "every attempt is refused identically, so it must not be retried"); + refusal.Message.ShouldContain(CodexHarness.MaxInputCharacters.ToString(CultureInfo.InvariantCulture)); + refusal.Message.ShouldContain((CodexHarness.MaxInputCharacters + 1).ToString(CultureInfo.InvariantCulture), customMessage: "the author needs to know how far over the goal is"); + refusal.Message.ShouldNotContain("xxxx", Case.Sensitive, "a refusal is host metadata — never the goal itself"); + } + + [Fact] + public void Codexs_input_cap_counts_characters_not_utf16_units() + { + // Codex is Rust: its "characters" are Unicode scalar values. An emoji is one of them and two UTF-16 units, so a + // UTF-16 count would refuse a goal Codex accepts. + var emoji = string.Concat(Enumerable.Repeat("🚀", CodexHarness.MaxInputCharacters / 2 + 1)); + emoji.Length.ShouldBeGreaterThan(CodexHarness.MaxInputCharacters, "fixture check: over the cap in UTF-16 units"); + + Should.NotThrow(() => Harness.BuildInvocation(Task(goal: emoji))); + } + [Fact] public void A_goal_past_the_argv_ceiling_builds_an_invocation_the_kernel_accepts() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/LlmApiExceptionTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/LlmApiExceptionTests.cs index af0a595b3..571cfc5d9 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/LlmApiExceptionTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/LlmApiExceptionTests.cs @@ -31,6 +31,8 @@ public void Classify_maps_status_to_category(int status, LlmErrorCategory expect [InlineData(400, "This model's maximum context length is 8192 tokens")] [InlineData(413, "input is too long for the context window")] [InlineData(422, "reduce the length of the messages")] + [InlineData(400, """{"type":"error","error":{"type":"invalid_request_error","message":"prompt is too long: 215000 tokens > 200000 maximum"}}""")] + [InlineData(400, """{"type":"error","error":{"type":"invalid_request_error","message":"input length and `max_tokens` exceed context limit: 197000 + 32000 > 200000, decrease input length or `max_tokens` and try again"}}""")] public void Classify_refines_a_4xx_to_context_length_on_a_matching_body(int status, string body) { LlmApiException.Classify(status, body).ShouldBe(LlmErrorCategory.ContextLengthExceeded); diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/MapResultsPromptTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/MapResultsPromptTests.cs index 6c0204cbb..45e99c898 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/MapResultsPromptTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/MapResultsPromptTests.cs @@ -2,6 +2,7 @@ using System.Text.RegularExpressions; using CodeSpace.Core.Services.Workflows.Engine; using CodeSpace.Core.Services.Workflows.Runtime; +using WorkflowJson = CodeSpace.Core.Services.Workflows.WorkflowJson; using CodeSpace.Messages.Constants; using CodeSpace.Messages.Dtos.Workflows; using Shouldly; @@ -42,7 +43,7 @@ public void Within_budget_the_projection_is_the_unbounded_serialization_characte // The ordinary fan-out: the model must see EXACTLY what the raw-array binding produced — the same call // VariableResolver's array arm makes on the same element. Not "equivalent JSON": the same characters. - projected.ShouldBe(JsonSerializer.Serialize(results), + projected.ShouldBe(JsonSerializer.Serialize(results, WorkflowJson.InterpolatedText), customMessage: "a fan-out inside the budget must not be reshaped at all — the bound may only bind when it binds"); } @@ -51,7 +52,7 @@ public void A_zero_budget_disables_the_bound_entirely() { var results = Results(new string('x', 50_000)); - MapResultsPrompt.Project(results, 0).Text.ShouldBe(JsonSerializer.Serialize(results), + MapResultsPrompt.Project(results, 0).Text.ShouldBe(JsonSerializer.Serialize(results, WorkflowJson.InterpolatedText), customMessage: "budget <= 0 means no bound — every map that declares none keeps its pre-existing output"); } @@ -107,7 +108,7 @@ public void A_single_pathological_branch_cannot_spend_the_budget_its_siblings_ne var projected = MapResultsPrompt.Project(results, Budget).Text; foreach (var (element, i) in results.EnumerateArray().Select((e, i) => (e, i)).Where(x => x.i != pathologicalIndex)) - projected.ShouldContain(JsonSerializer.Serialize(element), + projected.ShouldContain(JsonSerializer.Serialize(element, WorkflowJson.InterpolatedText), customMessage: $"small sibling {i} must survive in FULL — one oversized branch may not evict it, wherever the oversized branch sits"); CountBranchMarkers(projected).ShouldBe(1, @@ -151,6 +152,9 @@ public void The_within_budget_prompt_resolves_identically_to_the_raw_array_bindi { Obj("""{"status":"Succeeded","summary":"renamed the module","changedFiles":["a.cs","b.cs"]}"""), Obj("""{"status":"Succeeded","summary":"added the tests","changedFiles":["c.cs"]}"""), + // Characters the resolver and the projection must write the same way. With two different encoders these + // two strings agree on plain ASCII and diverge on the first CJK character, '+', '<', '&' or quote. + Obj("""{"status":"Succeeded","summary":"修复 List & 'x' a+b","changedFiles":["d.cs"]}"""), }; var outputs = WorkflowEngine.BuildMapOutputs("results", results, failed: 0, promptBudgetChars: Budget); @@ -245,8 +249,8 @@ public void Complete_is_recorded_true_exactly_when_nothing_was_dropped(int branc var projection = MapResultsPrompt.Project(results, budget); - projection.Coverage.Complete.ShouldBe(projection.Text == JsonSerializer.Serialize(results), - customMessage: $"{branches} branches of {branchChars} chars at a {budget}-char budget recorded Complete={projection.Coverage.Complete} over a text that is {(projection.Text == JsonSerializer.Serialize(results) ? "" : "NOT ")}the whole serialization"); + projection.Coverage.Complete.ShouldBe(projection.Text == JsonSerializer.Serialize(results, WorkflowJson.InterpolatedText), + customMessage: $"{branches} branches of {branchChars} chars at a {budget}-char budget recorded Complete={projection.Coverage.Complete} over a text that is {(projection.Text == JsonSerializer.Serialize(results, WorkflowJson.InterpolatedText) ? "" : "NOT ")}the whole serialization"); } /// diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/VariableResolverTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/VariableResolverTests.cs index 30baddbf5..e0f73b70d 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/VariableResolverTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/VariableResolverTests.cs @@ -44,6 +44,37 @@ public void Sole_template_preserves_native_type() resolved.GetInt32().ShouldBe(42); } + [Fact] + public void An_object_interpolated_into_text_keeps_its_characters_readable() + { + // A PR-review goal binds {{nodes.fetch_diff.outputs.files}} into prose. The object branch used to serialize with + // the HTML-safe default encoder, so every CJK character and every + < > & ' of the diff reached the model as a + // six-character \uXXXX — "修复" became \u4FEE\u590D, every added line began \u002B — at up to twice the size. The + // string branch beside it already inserts all of those raw, so the escaping protected nothing. (A character + // outside the Basic Multilingual Plane — an emoji — is still escaped: every built-in encoder does that.) + var scope = new NodeRunScope { Trigger = ParseDict("""{ "files": [ { "path": "src/a.cs", "patch": "+ // 修复:List & 'x' a+b" } ] }""") }; + + var resolved = VariableResolver.Resolve(ParseElement("\"请审查:\\n{{trigger.files}}\""), scope).GetString()!; + + resolved.ShouldContain("修复", customMessage: "the model must read the diff's own text, not escape codes"); + resolved.ShouldContain("+ // "); + resolved.ShouldContain("List & 'x' a+b"); + resolved.ShouldNotContain("\\u", Case.Insensitive, "no character of the interpolated value may reach the prompt as a \\uXXXX escape"); + } + + [Fact] + public void An_object_interpolated_into_text_is_still_valid_json_that_round_trips() + { + // The relaxed encoder still escapes what JSON itself requires — the quote, the backslash and control + // characters — so a template that embeds a value in a JSON body keeps a parseable body. + var scope = new NodeRunScope { Trigger = ParseDict("""{ "value": { "text": "a \"quoted\" \\ path\nsecond line 修复" } }""") }; + + var resolved = VariableResolver.Resolve(ParseElement("\"{\\\"wrapped\\\": {{trigger.value}}}\""), scope).GetString()!; + + using var document = JsonDocument.Parse(resolved); + document.RootElement.GetProperty("wrapped").GetProperty("text").GetString().ShouldBe("a \"quoted\" \\ path\nsecond line 修复"); + } + [Fact] public void Multiple_templates_concatenate_into_string() {