From 19467a2dd3b10905d54dd252c6717ede714b00dd Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Wed, 7 Oct 2026 04:22:09 +0800 Subject: [PATCH] Grade with the producing run's network and resource posture An acceptance grade's setup step was hard-coded to AllowNetwork=true, so under bubblewrap it shared the host network whatever the producing run's tier, and neither the setup nor the check had a memory or CPU ceiling. A setup like `npm ci` or `pip install` runs manifests the agent wrote, after the agent's own sandbox is gone. So a network-off Standard run could have its planted code executed with full egress, and a runaway install had no cgroup bound. Benchmark and qualification grading ran the fixture's tests over the agent's workspace on the raw local runner, with no ceiling either. Every lane now carries an AcceptanceGradingPosture, derived once from the run whose bytes it grades. The posture is that run's network grant and egress allowlist plus its tier's resource ceilings, clamped by Sandbox:MaxAutonomy the way RunCommandService.BuildSpec clamps agent.run_command. The executor lanes (branch, patch, multi-repo, local) take it from the run's own task. The supervisor's per-unit, baseline, captured, resolve and branchless-stop lanes read the unit's stored task. The stop over the integrated head uses the run profile's tier, which is the tier every unit is clamped to. BenchmarkRunner uses its agent's task; a TaskLaunch cell uses the posture its attempts agree on, and fails closed when they do not. AcceptanceGradingPosturePolicy.Bind wraps the grading runner once, so the setup and every oracle's command run narrowed, narrow-only. The setup keeps the network only when the producer had it. An allowlist producer is filtered to its operator-configured hosts, and an allowlist with no host severs. Both steps run under the tier's ceilings. A request with no posture grades with network off under the Confined ceilings. What the grade reports follows what the sandbox did, not what the spec asked. The runner now says which egress it enforces for a spec (ISandboxEgressEnforcement). Only when it severed or filtered the setup does one notice head the grade's evidence, so an unconfined host never reads "off" for a setup that kept its network. A setup the sandbox severed fails as "setup-failed-network-severed:": still infra, but the posture comes from the same stored task on every attempt, so agent.code no longer re-buys an agent run to sever it again. Grades can now hit a memory ceiling. A check the runner kills there (ResourceExhausted) is "tests-resource-exhausted", an Environment fact like tests-timed-out, instead of a genuine failure that bought revise rounds and recorded a verdict on the code. This changes what operators see: an operator setup on a network-off run no longer downloads where the sandbox confines. The fix is an egress allowlist or a higher tier. Trusted and Unleashed runs keep full egress. Baselines are now memoized per posture, so two units of different tiers off one base are no longer compared across two different sandboxes. EvaluatorVersion moves to v9. --- .../Agents/AgentAcceptanceContract.cs | 23 + .../Services/Agents/AgentRunExecutor.cs | 8 +- .../Agents/Eval/Benchmark/BenchmarkRunner.cs | 2 +- .../Eval/Benchmark/BenchmarkTaskGrading.cs | 13 +- .../Eval/Benchmark/Graders/TestsPassGrader.cs | 19 +- .../TaskLaunchBenchmarkCellRunner.Grade.cs | 15 + .../TaskLaunchBenchmarkCellRunner.cs | 2 +- .../Sandbox/ISandboxEgressEnforcement.cs | 19 + .../Runners/LocalProcessRunner.Durable.cs | 21 + .../Sandbox/Runners/LocalProcessRunner.cs | 2 +- .../Workspace/LocalAcceptanceVerifier.cs | 2 +- .../AcceptanceGradingPosturePolicy.cs | 122 ++++++ .../RealSupervisorActionExecutor.Spawn.cs | 2 +- .../Supervisor/ISupervisorAcceptanceGrader.cs | 13 + .../Supervisor/PostureBoundSandboxRunner.cs | 24 + .../SupervisorAcceptanceGradeRequest.cs | 34 ++ .../Supervisor/SupervisorAcceptanceGrader.cs | 127 ++++-- .../SupervisorTurnService.Rehydrate.cs | 69 ++- .../Workflows/Nodes/Builtin/AgentCodeNode.cs | 8 +- .../Agents/AcceptanceGradingPosture.cs | 30 ++ .../AcceptanceGradingPostureFlowTests.cs | 206 +++++++++ .../Agents/BenchmarkRunnerFlowTests.cs | 56 +++ .../SupervisorAcceptanceFoldFlowTests.cs | 59 ++- ...SupervisorModelAcceptanceSetupFlowTests.cs | 2 +- .../SupervisorUnitAcceptanceFoldFlowTests.cs | 134 ++++++ .../AcceptanceGradingPostureE2ETests.cs | 125 ++++++ .../BenchmarkTaskGradingPostureTests.cs | 90 ++++ .../TaskLaunchBenchmarkCellRunnerTests.cs | 22 + .../Agents/FakeAcceptanceGrader.cs | 9 + .../Agents/SupervisorAcceptanceGraderTests.cs | 5 +- .../SupervisorBranchlessStopGradeTests.cs | 25 +- .../Agents/SupervisorTurnServiceTests.cs | 23 + .../AcceptanceGradePostureInventoryTests.cs | 153 +++++++ .../AcceptanceGradingPostureTests.cs | 410 ++++++++++++++++++ .../CapturedDeliverableGradeTests.cs | 2 +- .../Workflows/AgentCodeNodeTests.cs | 19 + .../AgentRunExecutorAcceptanceTests.cs | 47 ++ 37 files changed, 1870 insertions(+), 72 deletions(-) create mode 100644 backend/src/CodeSpace.Core/Services/Agents/Sandbox/ISandboxEgressEnforcement.cs create mode 100644 backend/src/CodeSpace.Core/Services/Supervisor/AcceptanceGradingPosturePolicy.cs create mode 100644 backend/src/CodeSpace.Core/Services/Supervisor/PostureBoundSandboxRunner.cs create mode 100644 backend/src/CodeSpace.Messages/Agents/AcceptanceGradingPosture.cs create mode 100644 backend/tests/CodeSpace.IntegrationTests/Agents/AcceptanceGradingPostureFlowTests.cs create mode 100644 backend/tests/CodeSpace.SandboxTests/AcceptanceGradingPostureE2ETests.cs create mode 100644 backend/tests/CodeSpace.UnitTests/Agents/Benchmark/BenchmarkTaskGradingPostureTests.cs create mode 100644 backend/tests/CodeSpace.UnitTests/Architecture/AcceptanceGradePostureInventoryTests.cs create mode 100644 backend/tests/CodeSpace.UnitTests/Supervisor/AcceptanceGradingPostureTests.cs diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentAcceptanceContract.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentAcceptanceContract.cs index 46c2f8bba..a7ac6759d 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentAcceptanceContract.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentAcceptanceContract.cs @@ -148,7 +148,12 @@ public static bool IsInfraFailure(string? detail, bool workPresent) || effective.StartsWith("clone-failed:", StringComparison.Ordinal) || effective.StartsWith("setup-failed:", StringComparison.Ordinal) || effective.StartsWith("oracle-restore-failed:", StringComparison.Ordinal) + || effective.StartsWith(SetupSeveredDetailPrefix, StringComparison.Ordinal) || effective is "no-rubric" or "no-schema" or "tests-timed-out" or "setup-timed-out" + // The check was killed at the grade's OWN memory ceiling (the producing run's posture): an + // environment fact like the grader's own wall clock, and what the agent's own run already reports + // for the same kill. Never a verdict on the code, so no revise round is spent on it. + or Eval.Benchmark.Graders.TestsPassGrader.ResourceExhaustedDetail // POSIX "cannot run the command" codes (B6): 127 = command not found, 126 = found but not // executable — the CHECK ITSELF could not run, exactly like a grader start-throw. The two must // classify identically across runners or the same broken oracle reads infra locally (direct exec @@ -169,6 +174,24 @@ public static bool IsInfraFailure(string? detail, bool workPresent) || (effective == "no-branch-or-repo" && workPresent)); } + /// + /// The detail prefix of a contract setup step that failed with the network the producing run's posture took from + /// it, on a host that enforced that (SupervisorAcceptanceGrader writes it only when the runner reports the + /// step severed). Infra-classed like any setup failure — the check never ran — but, unlike setup-failed:, + /// not transient: the posture comes from the same stored task on every attempt. Pinned by a unit test (Rule 8): + /// the grader writes it and reads it across a durable resume payload. + /// + public const string SetupSeveredDetailPrefix = "setup-failed-network-severed:"; + + /// + /// Whether an acceptance-failure DETAIL was decided by the grade's own posture rather than by the work or a + /// transient fault: a respawn of the same task grades under the same posture and reaches the same failure, so it + /// is infra () and still not worth a retry. Sees through the same display + /// tags the infra classification does. + /// + public static bool IsDecidedByGradePosture(string? detail) => + StripGateLabel(StripRepoTag(detail))?.StartsWith(SetupSeveredDetailPrefix, StringComparison.Ordinal) == true; + /// /// The multi-repo grade paths wrap a classifiable detail in a uniform machine-authored display tag /// (repo 'alias': ) — classification must see through it, or a grader crash on one repo of a diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs index 83394808e..b57618eef 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs @@ -3042,14 +3042,15 @@ private async Task GradeAgainstOracleAsync(AcceptanceInvocation var grader = scope.ServiceProvider.GetRequiredService(); var fullSpec = spec with { Command = command }; var timeoutSeconds = spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds; + var posture = AcceptanceGradingPosturePolicy.For(task); // C3: the run's own recorded base is the oracle anchor — the grader restores the acceptance command's // program file from it, so an agent cannot buy its own pass by rewriting the check script it is graded // with. The patch lane always had this anchor; the BRANCH lane discarded it and graded the candidate's // bytes as the judge. grade = hasBranch - ? await grader.GradeAsync(repositoryId, run.TeamId, result.ProducedBranch!, fullSpec, timeoutSeconds, new OracleAnchor(result.BaseSha, OracleFloorPrograms(fullSpec)), cancellationToken).ConfigureAwait(false) - : await grader.GradePatchAsync(repositoryId, run.TeamId, result.BaseSha!, result.Patch, result.PatchArtifactId, fullSpec, timeoutSeconds, OracleFloorPrograms(fullSpec), cancellationToken).ConfigureAwait(false); + ? await grader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = run.TeamId, Branch = result.ProducedBranch!, Spec = fullSpec, TimeoutSeconds = timeoutSeconds, Anchor = new OracleAnchor(result.BaseSha, OracleFloorPrograms(fullSpec)), Posture = posture }, cancellationToken).ConfigureAwait(false) + : await grader.GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = run.TeamId, BaseSha = result.BaseSha!, InlinePatch = result.Patch, PatchArtifactId = result.PatchArtifactId, Spec = fullSpec, TimeoutSeconds = timeoutSeconds, OracleFloorPrograms = OracleFloorPrograms(fullSpec), Posture = posture }, cancellationToken).ConfigureAwait(false); } catch (Exception ex) when (ex is not OperationCanceledException and not AgentRunOwnershipLostException) { @@ -3118,6 +3119,7 @@ private async Task GradeMultiRepoAcceptanceAsync(AgentRun run, A var command = spec.Command.ToArray(); var fullSpec = spec with { Command = command }; var timeoutSeconds = spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds; + var posture = AcceptanceGradingPosturePolicy.For(task); var targets = result.RepositoryResults.Where(r => !string.IsNullOrEmpty(r.ProducedBranch) && r.RepositoryId is not null).ToList(); @@ -3142,7 +3144,7 @@ private async Task GradeMultiRepoAcceptanceAsync(AgentRun run, A try { // C3: each repo's own recorded base anchors ITS oracle restore — same protection as the single-repo lane. - grade = await grader.GradeAsync(target.RepositoryId!.Value, run.TeamId, target.ProducedBranch!, fullSpec, timeoutSeconds, new OracleAnchor(target.BaseSha, OracleFloorPrograms(fullSpec)), cancellationToken).ConfigureAwait(false); + grade = await grader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = target.RepositoryId!.Value, TeamId = run.TeamId, Branch = target.ProducedBranch!, Spec = fullSpec, TimeoutSeconds = timeoutSeconds, Anchor = new OracleAnchor(target.BaseSha, OracleFloorPrograms(fullSpec)), Posture = posture }, cancellationToken).ConfigureAwait(false); } catch (Exception ex) when (ex is not OperationCanceledException and not AgentRunOwnershipLostException) { diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkRunner.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkRunner.cs index 03c194187..2440f9ce7 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkRunner.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkRunner.cs @@ -73,7 +73,7 @@ public async Task RunAsync(BenchmarkTask task, BenchmarkMode mo var attempts = await RunWithFormatFaultRespawnAsync(task, agentTask, context, cancellationToken).ConfigureAwait(false); - var grade = await BenchmarkTaskGrading.GradeAsync(_graders, _runners, new BenchmarkTaskGradingRequest { Task = task, WorkspaceDirectory = workspaceDirectory, TeamId = teamId, ProducerModel = ProducerModelOf(selection, attempts) }, cancellationToken).ConfigureAwait(false); + var grade = await BenchmarkTaskGrading.GradeAsync(_graders, _runners, new BenchmarkTaskGradingRequest { Task = task, WorkspaceDirectory = workspaceDirectory, TeamId = teamId, ProducerModel = ProducerModelOf(selection, attempts), Posture = Supervisor.AcceptanceGradingPosturePolicy.For(agentTask) }, cancellationToken).ConfigureAwait(false); grade = ApplyMcpFabricRule(grade, mode, attempts[^1]); diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkTaskGrading.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkTaskGrading.cs index 62bfc7d9f..e2e11d01b 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkTaskGrading.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkTaskGrading.cs @@ -1,3 +1,5 @@ +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Messages.Agents; using CodeSpace.Messages.Agents.Benchmark; using CodeSpace.Messages.Review; @@ -8,7 +10,9 @@ namespace CodeSpace.Core.Services.Agents.Eval.Benchmark; /// against a post-run workspace directory (BenchmarkRunner for a direct-harness /// cell, TaskLaunch.TaskLaunchBenchmarkCellRunner for a Launch-mode cell) — grading is independent of HOW /// the workspace got there, so both resolve the SAME grader + sandbox runner through this one seam instead of each -/// duplicating the two resolves. +/// duplicating the two resolves. The runner is bound to the producing agent's posture +/// (), exactly as every acceptance grade's is: the check runs bytes +/// that agent wrote, so it never gets more memory, CPU or network than the agent had. /// internal static class BenchmarkTaskGrading { @@ -16,7 +20,9 @@ public static async Task GradeAsync(IBenchmarkGraderRegistry gra { var grader = graders.Resolve(request.Task.Grading); - var context = new BenchmarkGradingContext { Task = request.Task, WorkspaceDirectory = request.WorkspaceDirectory, Runner = runners.Resolve(Sandbox.SandboxKinds.Local), TeamId = request.TeamId, ProducerModel = request.ProducerModel }; + var runner = AcceptanceGradingPosturePolicy.Bind(runners.Resolve(Sandbox.SandboxKinds.Local), request.Posture ?? AcceptanceGradingPosturePolicy.FailClosed); + + var context = new BenchmarkGradingContext { Task = request.Task, WorkspaceDirectory = request.WorkspaceDirectory, Runner = runner, TeamId = request.TeamId, ProducerModel = request.ProducerModel }; return await grader.GradeAsync(context, cancellationToken).ConfigureAwait(false); } @@ -28,4 +34,7 @@ internal sealed record BenchmarkTaskGradingRequest public required string WorkspaceDirectory { get; init; } public Guid? TeamId { get; init; } public ReviewModelIdentity? ProducerModel { get; init; } + + /// The posture of the agent run whose workspace this grades. The fixture's test command imports that agent's code, so it runs under the same ceilings and network cut. Required so no instrument can forget it; null grades under . + public required AcceptanceGradingPosture? Posture { get; init; } } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/Graders/TestsPassGrader.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/Graders/TestsPassGrader.cs index 6b173398a..ac5439e98 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/Graders/TestsPassGrader.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/Graders/TestsPassGrader.cs @@ -13,6 +13,15 @@ namespace CodeSpace.Core.Services.Agents.Eval.Benchmark.Graders; /// public sealed class TestsPassGrader : IBenchmarkGrader, ISingletonDependency { + /// + /// The detail of a check the runner killed at its own memory ceiling (, + /// read from the cgroup's oom_kill counter, never guessed from the exit code). A grade now runs under the producing + /// run's ceilings, so a ceiling can end a check, and that says nothing about whether the code is right: it is an + /// fact like tests-timed-out. Pinned by a unit test (Rule 8): the + /// infra classifier and the agent.code retry verdict read this literal across a durable resume payload. + /// + public const string ResourceExhaustedDetail = "tests-resource-exhausted"; + public BenchmarkGradingKind Kind => BenchmarkGradingKind.TestsPass; public async Task GradeAsync(BenchmarkGradingContext context, CancellationToken cancellationToken) @@ -31,7 +40,7 @@ public async Task GradeAsync(BenchmarkGradingContext context, Ca return result.Status == SandboxStatus.Success ? new BenchmarkGrade { Passed = true, Detail = "tests-passed", EvidenceText = evidence } - : Fail(DetailFor(result), result.Status == SandboxStatus.TimedOut ? GradeFailureClass.Environment : GradeFailureClass.Genuine) with { EvidenceText = evidence }; + : Fail(DetailFor(result), result.Status is SandboxStatus.TimedOut or SandboxStatus.ResourceExhausted ? GradeFailureClass.Environment : GradeFailureClass.Genuine) with { EvidenceText = evidence }; } /// The grading command runs in the post-run workspace with a fresh, short wall-clock cap — the tests are tiny, and a hung test is a fail, not a hang. The env is the runner's scrubbed default (no agent secret injected — the grader is independent of the agent's credential). @@ -49,8 +58,12 @@ private static string EvidenceTextFor(BenchmarkGradingContext context, SandboxRe private static string Tail(string text) => text.Length <= 8_192 ? text : text[^8_192..]; - private static string DetailFor(SandboxResult result) => - result.Status == SandboxStatus.TimedOut ? "tests-timed-out" : $"tests-failed-exit-{result.ExitCode}"; + private static string DetailFor(SandboxResult result) => result.Status switch + { + SandboxStatus.TimedOut => "tests-timed-out", + SandboxStatus.ResourceExhausted => ResourceExhaustedDetail, + _ => $"tests-failed-exit-{result.ExitCode}", + }; private static BenchmarkGrade Fail(string detail, GradeFailureClass? failureClass = null) => new() { Passed = false, Detail = detail, Class = failureClass }; } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.Grade.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.Grade.cs index 23fc3f95c..17c7aca99 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.Grade.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.Grade.cs @@ -37,6 +37,21 @@ private Task> LoadAgentRunsAsync(Guid runId, CancellationToken ca internal static string? ObservedModelOf(IReadOnlyList attempts) => ParseResult(attempts[^1])?.Model ?? attempts.Select(ParseResult).Select(r => r?.Model).FirstOrDefault(model => model is not null); + /// + /// The posture the reconstructed workspace's check runs under: the one every attempt that wrote it ran with, read off + /// each attempt's stored task. Like , it claims one only when every attempt agrees; a + /// union of work from different postures, or an attempt whose task cannot be read, has none — null, so the grade + /// fails closed (AcceptanceGradingPosturePolicy.FailClosed). Internal so the rule is pinned directly. + /// + internal static AcceptanceGradingPosture? GradedPosture(IReadOnlyList attempts) + { + var postures = attempts.Select(attempt => Supervisor.AcceptanceGradingPosturePolicy.ForStoredTask(attempt.TaskJson)).ToList(); + + if (postures.Count == 0 || postures.Any(posture => posture is null)) return null; + + return postures.Select(posture => JsonSerializer.Serialize(posture, AgentJson.Options)).Distinct(StringComparer.Ordinal).Count() == 1 ? postures[0] : null; + } + internal static ReviewModelIdentity ProducerModelOf(BenchmarkAgentSelection? selection, IReadOnlyList attempts) { var observed = attempts.Select(ParseResult).Select(result => result?.Model).Where(model => !string.IsNullOrWhiteSpace(model)).Distinct(StringComparer.OrdinalIgnoreCase).ToList(); diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.cs index 155bdb312..f3f04ca3b 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/TaskLaunch/TaskLaunchBenchmarkCellRunner.cs @@ -64,7 +64,7 @@ public async Task RunAsync(BenchmarkTask task, BenchmarkMode mo await ReconstructWorkspaceAsync(context.WorkspaceDirectory, attempts, cancellationToken).ConfigureAwait(false); - var grade = await BenchmarkTaskGrading.GradeAsync(_graders, _runners, new BenchmarkTaskGradingRequest { Task = task, WorkspaceDirectory = context.WorkspaceDirectory, TeamId = context.TeamId, ProducerModel = ProducerModelOf(context.Selection, attempts) }, cancellationToken).ConfigureAwait(false); + var grade = await BenchmarkTaskGrading.GradeAsync(_graders, _runners, new BenchmarkTaskGradingRequest { Task = task, WorkspaceDirectory = context.WorkspaceDirectory, TeamId = context.TeamId, ProducerModel = ProducerModelOf(context.Selection, attempts), Posture = GradedPosture(attempts) }, cancellationToken).ConfigureAwait(false); var completionMode = await LoadCompletionEnforcementModeAsync(launched.RunId, cancellationToken).ConfigureAwait(false); diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/ISandboxEgressEnforcement.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/ISandboxEgressEnforcement.cs new file mode 100644 index 000000000..cd5d652f7 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/ISandboxEgressEnforcement.cs @@ -0,0 +1,19 @@ +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Messages.Agents; + +namespace CodeSpace.Core.Services.Agents.Sandbox; + +/// +/// Optional capability a sandbox runner MAY implement alongside (Rule 7 / ISP — a sibling +/// interface, never a widening of the base contract): say, before it runs spec, what egress it would actually +/// ENFORCE for it — which is not always what the spec asks. A spec with false is +/// severed only where the runner confines; one with an allowlist is filtered only where the host can filter, and +/// otherwise severed or, where nothing confines, not narrowed at all. The runner is the one place that knows what it +/// can build on this host, so a caller that must report what a narrowing actually did asks it here rather than +/// reading the spec back as if it were the outcome. A runner without this capability makes no claim. +/// +public interface ISandboxEgressEnforcement +{ + /// The egress would actually get on this host: severed, held to its allowlist, the host network. + SandboxEgressMode EnforcedEgress(SandboxSpec spec); +} diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs index fddfe539c..6cbf28d46 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs @@ -383,6 +383,27 @@ internal static void EnsureEgressAdmissible(SandboxSpec spec, bool confines, boo throw new SealedEgressUnavailableException(cause); } + /// + public SandboxEgressMode EnforcedEgress(SandboxSpec spec) + { + var confines = BubblewrapSandbox.Available is not null; + + return EnforcedEgress(spec, confines, HostFiltersAllowlist(confines)); + } + + /// + /// over whether this host and whether it plans + /// an allowlist into a filtered namespace (), so the table is pinned on any host. + /// The same derivation the launch reads, with the one thing it leaves implicit made + /// explicit: a severed policy is enforced only by bubblewrap's fresh namespace, so where nothing confines the child + /// keeps the worker's network. A filtered namespace is built with or without bubblewrap. + /// + internal static SandboxEgressMode EnforcedEgress(SandboxSpec spec, bool confines, bool filtersAllowlist) => EgressPolicyFor(spec, filtersAllowlist).Mode switch + { + SandboxEgressMode.None when !confines => SandboxEgressMode.Full, + var mode => mode, + }; + /// /// Why a brokered child with a network of its own could not reach its broker from this host, or null when it can /// or needs nothing: no broker port, a network it shares with the worker, or both halves of the relay in place. diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs index 5c5b7e7bc..ad06735f7 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs @@ -24,7 +24,7 @@ namespace CodeSpace.Core.Services.Agents.Sandbox.Runners; /// Caller cancellation is honoured distinctly from the spec timeout: it terminates the process and rethrows /// (the durable path differs — see its remarks: cancellation stops observing without killing). /// -public sealed partial class LocalProcessRunner : ISandboxRunner, ISandboxStreamRunner, ISandboxDurableRunner, ISandboxLaunchIdentityRunner, ISandboxDurableLogSource, ISandboxDurableDiagnosticSource, ISandboxEgressAdmission, ISingletonDependency +public sealed partial class LocalProcessRunner : ISandboxRunner, ISandboxStreamRunner, ISandboxDurableRunner, ISandboxLaunchIdentityRunner, ISandboxDurableLogSource, ISandboxDurableDiagnosticSource, ISandboxEgressAdmission, ISandboxEgressEnforcement, ISingletonDependency { /// This runner's registry key. The runner-local spelling of the shared — same constant, so there is one literal. public const string LocalKind = SandboxKinds.Local; diff --git a/backend/src/CodeSpace.Core/Services/Agents/Workspace/LocalAcceptanceVerifier.cs b/backend/src/CodeSpace.Core/Services/Agents/Workspace/LocalAcceptanceVerifier.cs index 34167361d..f5d263285 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Workspace/LocalAcceptanceVerifier.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Workspace/LocalAcceptanceVerifier.cs @@ -57,7 +57,7 @@ public async Task GradeAsync(LocalAcceptanceRequest request, Can if (observation.Failure != null) return observation.Failure; var context = (Context)request.Context!; BenchmarkGrade grade; - try { grade = await grader.GradeDirectoryAsync(observation.Directory!, context.Spec!, request.TeamId, context.Spec!.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, cancellationToken).ConfigureAwait(false); } + try { grade = await grader.GradeDirectoryAsync(new DirectoryAcceptanceGradeRequest { Directory = observation.Directory!, Spec = context.Spec!, TeamId = request.TeamId, TimeoutSeconds = context.Spec!.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, Posture = AcceptanceGradingPosturePolicy.For(request.Task) }, cancellationToken).ConfigureAwait(false); } catch (Exception ex) when (ex is not OperationCanceledException and not Exceptions.AgentRunOwnershipLostException and not AgentAuthorityDeniedException) { return Failed($"local-oracle-unavailable: {ex.GetType().Name}", GradeFailureClass.GraderFault); diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/AcceptanceGradingPosturePolicy.cs b/backend/src/CodeSpace.Core/Services/Supervisor/AcceptanceGradingPosturePolicy.cs new file mode 100644 index 000000000..08a08972b --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Supervisor/AcceptanceGradingPosturePolicy.cs @@ -0,0 +1,122 @@ +using System.Text.Json; +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Settings; +using CodeSpace.Messages.Agents; + +namespace CodeSpace.Core.Services.Supervisor; + +/// +/// Derives the a grade runs under from the run that produced the work, and +/// applies it to each grade step. The grade runs the candidate's own bytes after the producing run's sandbox has gone: +/// a setup step installs from a manifest the agent wrote, and the check imports scripts beside the ones it names. So a +/// grade step gets no more network, egress or resources than the producing run had. +/// +/// NARROW-ONLY at every layer. A step gets the network only when the producer's own grant +/// () has it AND its tier, clamped by the deployment ceiling +/// (Sandbox:MaxAutonomy, the same ceiling RunCommandService.BuildSpec narrows agent.run_command +/// to), derives it (). A clamped tier never derives network its ceiling denies. An +/// producer is narrowed to its operator-configured hosts only. The model API +/// and git hosts it also reached are not added: a grade carries no model key, and its clone is already on disk. An +/// allowlist that names no host severs the step, the same fail-closed rule AgentRunExecutor.ApplyEgressPolicy +/// applies to the agent. The memory and CPU ceilings are the clamped tier's committed row +/// (), narrowed by the operator's host budget. +/// +public static class AcceptanceGradingPosturePolicy +{ + /// + /// The opening of the notice a grade records when its setup step ran with less network than it asked for. A pinned + /// literal (Rule 8): an operator whose setup stopped downloading searches for it. + /// + public const string SetupNetworkNoticePrefix = "setup network: "; + + /// The posture of a grade whose producing run is unknown: network off and the Confined tier's resource ceilings. A grade that cannot say whose work it runs gets the least anyone could have had. + public static AcceptanceGradingPosture FailClosed => For(AgentAutonomyLevel.Confined, AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Confined)); + + /// The posture of a grade of 's work: the tier and permissions it ran with, as stored after admission. + public static AcceptanceGradingPosture For(AgentTask producer) => For(producer.Autonomy, producer.Permissions); + + /// + /// The posture of a grade of the work a run with this stored TaskJson produced, or null when the task cannot + /// be read — the caller then grades . The one reader every lane that only holds the + /// producer's row goes through. + /// + public static AcceptanceGradingPosture? ForStoredTask(string? taskJson) + { + if (string.IsNullOrWhiteSpace(taskJson)) return null; + + try + { + var task = JsonSerializer.Deserialize(taskJson, AgentJson.Options); + return task is null ? null : For(task); + } + catch (JsonException) { return null; } + } + + /// bound to : every spec handed to the result runs narrowed by . The ONE way a grade hands an oracle a runner. + internal static PostureBoundSandboxRunner Bind(ISandboxRunner runner, AcceptanceGradingPosture posture) => new(runner, posture); + + /// The posture of a grade of work produced under and , on this deployment. + public static AcceptanceGradingPosture For(AgentAutonomyLevel autonomy, AgentPermissions permissions) => + Derive(autonomy, permissions, AgentAutonomyPolicy.DeploymentCeiling, RuntimeSettings.Current.AgentMemoryCeilingMb); + + /// + /// over an explicit deployment ceiling and host memory + /// budget, so the table is pinned without staging configuration. An unknown tier reads as Confined: a value this + /// policy cannot recognise must not widen a grade. + /// + internal static AcceptanceGradingPosture Derive(AgentAutonomyLevel autonomy, AgentPermissions permissions, AgentAutonomyLevel deploymentCeiling, int? hostMemoryBudgetMb) + { + var tier = Enum.IsDefined(autonomy) ? AgentAutonomyPolicy.Clamp(autonomy, deploymentCeiling) : AgentAutonomyLevel.Confined; + var ceilings = AgentAutonomyPolicy.Ceilings(tier, hostMemoryBudgetMb); + var network = permissions.Network == AgentNetworkAccess.On && AgentAutonomyPolicy.Derive(tier).Network == AgentNetworkAccess.On; + var allowlist = network && permissions.Egress == AgentEgressPolicy.Allowlist ? EgressAllowlistBuilder.Build(null, null, Array.Empty(), permissions.EgressAllowHosts) : null; + + return new AcceptanceGradingPosture + { + Autonomy = tier, + AllowNetwork = network && allowlist is not { Count: 0 }, + EgressAllowlist = allowlist is { Count: > 0 } ? allowlist : null, + MaxMemoryMb = ceilings.MemoryMb, + MaxCpuPercent = ceilings.CpuPercent, + }; + } + + /// + /// as a grade step may run it under . A grade step says only + /// WHETHER it needs a remote: the setup step asks, the check does not. Which hosts it reaches is the posture's + /// decision alone, and no grade step carries an allowlist of its own. The step keeps network only when it asked + /// and the posture allows. Each resource ceiling is the narrower of the step's and the posture's, where 0 means + /// unlimited. + /// + internal static SandboxSpec Apply(SandboxSpec spec, AcceptanceGradingPosture posture) + { + var network = spec.AllowNetwork && posture.AllowNetwork; + + return spec with + { + AllowNetwork = network, + EgressAllowlist = network ? posture.EgressAllowlist : null, + MaxMemoryMb = Narrower(spec.MaxMemoryMb, posture.MaxMemoryMb), + MaxCpuPercent = Narrower(spec.MaxCpuPercent, posture.MaxCpuPercent), + }; + } + + /// + /// What a grade records when its setup step, which asks for the network, runs under and + /// the runner beneath enforces for it: null when the setup kept the host network — the + /// posture shares it, or nothing on this host narrowed it — or the runner cannot say. Otherwise one line saying the + /// sandbox severed or filtered it, and what an operator whose setup must download can do about it. Keyed on what was + /// ENFORCED, never on the spec, so a host that does not confine never reads "off" for a setup that kept its network. + /// + public static string? SetupNetworkNotice(AcceptanceGradingPosture posture, SandboxEgressMode? enforced) => (enforced, posture.EgressAllowlist) switch + { + (SandboxEgressMode.None, { Count: > 0 } hosts) => $"{SetupNetworkNoticePrefix}off — this host cannot filter to the producing run's egress allowlist ({string.Join(", ", hosts)}), so the sandbox severed it; a setup that downloads needs a host that filters egress, or a higher tier", + (SandboxEgressMode.None, _) => $"{SetupNetworkNoticePrefix}off, as the producing run had it ({posture.Autonomy}) — the sandbox severed it; a setup that downloads needs an egress allowlist or a higher tier", + (SandboxEgressMode.Filtered, { Count: > 0 } hosts) => $"{SetupNetworkNoticePrefix}narrowed to the producing run's egress allowlist ({string.Join(", ", hosts)}) — the sandbox filtered it", + _ => null, + }; + + private static int Narrower(int step, int posture) => step <= 0 ? posture : posture <= 0 ? step : Math.Min(step, posture); +} diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs b/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs index 9a9cf6f62..e8745c5f3 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/Executors/RealSupervisorActionExecutor.Spawn.cs @@ -1351,7 +1351,7 @@ private static string HarnessOf(SupervisorAgentProfile? profile) => /// clamped to the value returned here (), so tightening this tightens every /// spawn. /// - private static AgentAutonomyLevel AutonomyOf(SupervisorAgentProfile? profile) => + internal static AgentAutonomyLevel AutonomyOf(SupervisorAgentProfile? profile) => AgentAutonomyPolicy.Clamp(AgentAutonomyPolicy.Parse(profile?.AutonomyLevel, AgentAutonomyLevel.Standard), AgentAutonomyPolicy.DeploymentCeiling); /// Clamp a model-authored autonomy REQUEST to the run profile's (L4 arc B): the request wins only when it is MORE restrictive than the ceiling (the enum is ordered Confined < Standard < Trusted < Unleashed); an absent / unparseable / equal-or-higher request keeps the ceiling — so the model can lower its own autonomy but NEVER raise it past the operator's grant. No request → the ceiling (byte-identical). diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/ISupervisorAcceptanceGrader.cs b/backend/src/CodeSpace.Core/Services/Supervisor/ISupervisorAcceptanceGrader.cs index b29b0d9f9..37b135277 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/ISupervisorAcceptanceGrader.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/ISupervisorAcceptanceGrader.cs @@ -15,6 +15,11 @@ namespace CodeSpace.Core.Services.Supervisor; /// cloned, OR a check that cannot be RUN (a binary not on PATH, a judge with no model) — returns a FAILED grade (not /// an exception), because acceptance that cannot be verified is "not accepted", never a silent pass and never a /// crash that strands the caller. Only a genuine cancellation propagates. +/// +/// Every production lane grades through a REQUEST overload, because only a request carries the producing run's +/// : the network, egress allowlist and resource ceilings the grade's setup and +/// check run under. The positional overloads carry none, so the real grader runs them with network off under the +/// Confined tier's ceilings. They remain the members a test double implements. /// public interface ISupervisorAcceptanceGrader { @@ -30,6 +35,14 @@ Task GradeCapturedAsync(CapturedAcceptanceGradeRequest request, Task GradePatchAsync(PatchAcceptanceGradeRequest request, CancellationToken cancellationToken) => GradePatchAsync(request.RepositoryId, request.TeamId, request.BaseSha, request.InlinePatch, request.PatchArtifactId, request.Spec, request.TimeoutSeconds, request.OracleFloorPrograms, cancellationToken); + /// Grade a unit's base tree under the posture of the candidate it is compared against. + Task GradeBaseAsync(BaseAcceptanceGradeRequest request, CancellationToken cancellationToken) => + GradeBaseAsync(request.RepositoryId, request.TeamId, request.BaseSha, request.Spec, request.TimeoutSeconds, cancellationToken); + + /// Grade a repo-less run's live workspace under the posture of the run that wrote it. + Task GradeDirectoryAsync(DirectoryAcceptanceGradeRequest request, CancellationToken cancellationToken) => + GradeDirectoryAsync(request.Directory, request.Spec, request.TeamId, request.TimeoutSeconds, cancellationToken); + /// /// Clone at (team-scoped) and grade it with the oracle /// the spec names (Kind null ⇒ TestsPass) against the spec's command + kind-specific payload, capped diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/PostureBoundSandboxRunner.cs b/backend/src/CodeSpace.Core/Services/Supervisor/PostureBoundSandboxRunner.cs new file mode 100644 index 000000000..1986997b5 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Supervisor/PostureBoundSandboxRunner.cs @@ -0,0 +1,24 @@ +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Messages.Agents; + +namespace CodeSpace.Core.Services.Supervisor; + +/// +/// A grading runner bound to one producing run's posture: every spec a grade step hands it is narrowed by +/// before it runs. A grade's setup step and its oracle's own command +/// (handed to graders through BenchmarkGradingContext.Runner) both go through it, so an oracle added later is +/// bound without knowing the posture exists. Built only by ; it carries +/// no DI marker, so the container never registers it. +/// +internal sealed class PostureBoundSandboxRunner(ISandboxRunner inner, AcceptanceGradingPosture posture) : ISandboxRunner +{ + public AcceptanceGradingPosture Posture { get; } = posture; + + public string Kind => inner.Kind; + + public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) => inner.RunAsync(AcceptanceGradingPosturePolicy.Apply(spec, Posture), cancellationToken); + + /// The egress actually gets once narrowed, as the runner beneath says it enforces it (), or null when that runner cannot say. + public SandboxEgressMode? EnforcedEgress(SandboxSpec spec) => inner is ISandboxEgressEnforcement enforcement ? enforcement.EnforcedEgress(AcceptanceGradingPosturePolicy.Apply(spec, Posture)) : null; +} diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGradeRequest.cs b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGradeRequest.cs index 07e19b5b8..e8215964f 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGradeRequest.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGradeRequest.cs @@ -13,6 +13,9 @@ public sealed record RepositoryAcceptanceGradeRequest public required int TimeoutSeconds { get; init; } public OracleAnchor Anchor { get; init; } = OracleAnchor.None; public ReviewModelIdentity? ProducerModel { get; init; } + + /// The producing run's sandbox posture the grade runs under. Required so no lane can forget it; null grades with network off under the Confined tier's ceilings. + public required AcceptanceGradingPosture? Posture { get; init; } } /// A delayed captured-deliverable acceptance grade with the producer identity that authored the files. @@ -23,6 +26,9 @@ public sealed record CapturedAcceptanceGradeRequest public required SupervisorAcceptanceSpec Spec { get; init; } public required int TimeoutSeconds { get; init; } public ReviewModelIdentity? ProducerModel { get; init; } + + /// The producing run's sandbox posture the grade runs under. Required so no lane can forget it; null grades with network off under the Confined tier's ceilings. + public required AcceptanceGradingPosture? Posture { get; init; } } /// A delayed patch acceptance grade with the producer identity that authored the patch. @@ -37,4 +43,32 @@ public sealed record PatchAcceptanceGradeRequest public required int TimeoutSeconds { get; init; } public IReadOnlyList? OracleFloorPrograms { get; init; } public ReviewModelIdentity? ProducerModel { get; init; } + + /// The producing run's sandbox posture the grade runs under. Required so no lane can forget it; null grades with network off under the Confined tier's ceilings. + public required AcceptanceGradingPosture? Posture { get; init; } +} + +/// A baseline grade of a unit's base tree, under the SAME posture as the candidate it is compared against, so the differential measures the work and not two different sandboxes. +public sealed record BaseAcceptanceGradeRequest +{ + public required Guid RepositoryId { get; init; } + public required Guid TeamId { get; init; } + public required string BaseSha { get; init; } + public required SupervisorAcceptanceSpec Spec { get; init; } + public required int TimeoutSeconds { get; init; } + + /// The producing run's sandbox posture the grade runs under. Required so no lane can forget it; null grades with network off under the Confined tier's ceilings. + public required AcceptanceGradingPosture? Posture { get; init; } +} + +/// A grade of a repo-less run's live workspace directory, under the posture of the run that wrote it. +public sealed record DirectoryAcceptanceGradeRequest +{ + public required string Directory { get; init; } + public required Guid TeamId { get; init; } + public required SupervisorAcceptanceSpec Spec { get; init; } + public required int TimeoutSeconds { get; init; } + + /// The producing run's sandbox posture the grade runs under. Required so no lane can forget it; null grades with network off under the Confined tier's ceilings. + public required AcceptanceGradingPosture? Posture { get; init; } } diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGrader.cs b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGrader.cs index 61c9de3ff..0671c4acc 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGrader.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGrader.cs @@ -3,6 +3,7 @@ using CodeSpace.Core.Services.Agents.Eval.Benchmark.Graders; using CodeSpace.Core.Services.Agents.Publish; using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; using CodeSpace.Core.Services.Agents.Workspace; using CodeSpace.Core.Services.Agents.Workspace.Providers; using CodeSpace.Core.Services.Workflows.Artifacts; @@ -29,7 +30,7 @@ public sealed class SupervisorAcceptanceGrader : ISupervisorAcceptanceGrader, IS /// the SAME PR as any change to grading semantics — oracle dispatch, restore/tamper behavior, evidence /// capture, fail-closed arms. Pinned by test; the literal is the wire value on durable receipts. /// - public const string EvaluatorVersion = "supervisor-acceptance/v8"; // v8: every grade step (setup, check, oracle-restore git) runs under a bounded window — a non-positive authored timeout grades at the default instead of arming no wall clock, a longer one is capped at SupervisorLane.MaxAcceptanceGradeTimeoutSeconds + public const string EvaluatorVersion = "supervisor-acceptance/v9"; // v9: the setup and the check run under the PRODUCING run's posture (network, egress allowlist, memory/cpu ceilings), narrow-only; a request with none grades network-off under the Confined ceilings; a check killed at that ceiling grades tests-resource-exhausted (Environment) and a setup the sandbox severed grades setup-failed-network-severed. v8: every grade step runs under a bounded window /// The grading clone + oracle commands run on the worker host's own local runner. NOT the deployment /// default (AgentDefaultRunnerSetting): this funnel never reads a caller-supplied runner kind, and the @@ -62,7 +63,7 @@ public Task GradeAsync(Guid repositoryId, Guid teamId, string br GradeAsync(repositoryId, teamId, branch, spec, timeoutSeconds, OracleAnchor.None, cancellationToken); public Task GradeAsync(Guid repositoryId, Guid teamId, string branch, SupervisorAcceptanceSpec spec, int timeoutSeconds, OracleAnchor anchor, CancellationToken cancellationToken) => - GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = teamId, Branch = branch, Spec = spec, TimeoutSeconds = timeoutSeconds, Anchor = anchor }, cancellationToken); + GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = teamId, Branch = branch, Spec = spec, TimeoutSeconds = timeoutSeconds, Anchor = anchor, Posture = null }, cancellationToken); public async Task GradeAsync(RepositoryAcceptanceGradeRequest request, CancellationToken cancellationToken) { @@ -88,7 +89,7 @@ public async Task GradeAsync(RepositoryAcceptanceGradeRequest re if (protection.Failure is not null) return protection.Failure; - return await GradeWorkspaceAsync(new WorkspaceGradeRequest(workspace.Directory, spec, teamId, timeoutSeconds, request.ProducerModel, protection), cancellationToken).ConfigureAwait(false); + return await GradeWorkspaceAsync(new WorkspaceGradeRequest(workspace.Directory, spec, teamId, timeoutSeconds, request.Posture, request.ProducerModel, protection), cancellationToken).ConfigureAwait(false); } catch (WorkspaceException ex) { @@ -106,16 +107,19 @@ public async Task GradeAsync(RepositoryAcceptanceGradeRequest re } } - public async Task GradeDirectoryAsync(string directory, SupervisorAcceptanceSpec spec, Guid teamId, int timeoutSeconds, CancellationToken cancellationToken) + public Task GradeDirectoryAsync(string directory, SupervisorAcceptanceSpec spec, Guid teamId, int timeoutSeconds, CancellationToken cancellationToken) => + GradeDirectoryAsync(new DirectoryAcceptanceGradeRequest { Directory = directory, Spec = spec, TeamId = teamId, TimeoutSeconds = timeoutSeconds, Posture = null }, cancellationToken); + + public async Task GradeDirectoryAsync(DirectoryAcceptanceGradeRequest request, CancellationToken cancellationToken) { - if (!Directory.Exists(directory)) + if (!Directory.Exists(request.Directory)) return new BenchmarkGrade { Passed = false, Detail = "grade-error: the workspace directory no longer exists", Class = Messages.Agents.Benchmark.GradeFailureClass.Environment }; - return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds), cancellationToken).ConfigureAwait(false); + return await GradeWorkspaceAsync(new WorkspaceGradeRequest(request.Directory, request.Spec, request.TeamId, request.TimeoutSeconds, request.Posture), cancellationToken).ConfigureAwait(false); } public Task GradeCapturedAsync(Guid agentRunId, Guid teamId, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) => - GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = agentRunId, TeamId = teamId, Spec = spec, TimeoutSeconds = timeoutSeconds }, cancellationToken); + GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = agentRunId, TeamId = teamId, Spec = spec, TimeoutSeconds = timeoutSeconds, Posture = null }, cancellationToken); public async Task GradeCapturedAsync(CapturedAcceptanceGradeRequest request, CancellationToken cancellationToken) { @@ -131,7 +135,7 @@ public async Task GradeCapturedAsync(CapturedAcceptanceGradeRequ return Failed(ISupervisorAcceptanceGrader.NoDeliverablesCaptured, GradeFailureClass.Genuine); } - return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds, request.ProducerModel), cancellationToken).ConfigureAwait(false); + return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds, request.Posture, request.ProducerModel), cancellationToken).ConfigureAwait(false); } catch (Exception ex) when (ex is not OperationCanceledException) { @@ -233,7 +237,7 @@ public Task GradePatchAsync(Guid repositoryId, Guid teamId, stri GradePatchAsync(repositoryId, teamId, baseSha, inlinePatch, patchArtifactId, spec, timeoutSeconds, oracleFloorPrograms: null, cancellationToken); public Task GradePatchAsync(Guid repositoryId, Guid teamId, string baseSha, string inlinePatch, Guid? patchArtifactId, SupervisorAcceptanceSpec spec, int timeoutSeconds, IReadOnlyList? oracleFloorPrograms, CancellationToken cancellationToken) => - GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = teamId, BaseSha = baseSha, InlinePatch = inlinePatch, PatchArtifactId = patchArtifactId, Spec = spec, TimeoutSeconds = timeoutSeconds, OracleFloorPrograms = oracleFloorPrograms }, cancellationToken); + GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = teamId, BaseSha = baseSha, InlinePatch = inlinePatch, PatchArtifactId = patchArtifactId, Spec = spec, TimeoutSeconds = timeoutSeconds, OracleFloorPrograms = oracleFloorPrograms, Posture = null }, cancellationToken); public async Task GradePatchAsync(PatchAcceptanceGradeRequest request, CancellationToken cancellationToken) { @@ -277,7 +281,7 @@ public async Task GradePatchAsync(PatchAcceptanceGradeRequest re if (protection.Failure is not null) return protection.Failure; - return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds, request.ProducerModel, protection), cancellationToken).ConfigureAwait(false); + return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds, request.Posture, request.ProducerModel, protection), cancellationToken).ConfigureAwait(false); } catch (WorkspaceException ex) { @@ -295,8 +299,12 @@ public async Task GradePatchAsync(PatchAcceptanceGradeRequest re } } - public async Task GradeBaseAsync(Guid repositoryId, Guid teamId, string baseSha, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) + public Task GradeBaseAsync(Guid repositoryId, Guid teamId, string baseSha, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) => + GradeBaseAsync(new BaseAcceptanceGradeRequest { RepositoryId = repositoryId, TeamId = teamId, BaseSha = baseSha, Spec = spec, TimeoutSeconds = timeoutSeconds, Posture = null }, cancellationToken); + + public async Task GradeBaseAsync(BaseAcceptanceGradeRequest request, CancellationToken cancellationToken) { + var (repositoryId, teamId, baseSha, spec, timeoutSeconds) = (request.RepositoryId, request.TeamId, request.BaseSha, request.Spec, request.TimeoutSeconds); var directory = Path.Combine(LocalGitWorkspaceProvider.WorkspacesRoot, "grade-base-" + Guid.NewGuid().ToString("N")); try @@ -306,7 +314,7 @@ public async Task GradeBaseAsync(Guid repositoryId, Guid teamId, await CloneAtBaseAsync(clone, baseSha, directory, cancellationToken).ConfigureAwait(false); - return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds), cancellationToken).ConfigureAwait(false); + return await GradeWorkspaceAsync(new WorkspaceGradeRequest(directory, spec, teamId, timeoutSeconds, request.Posture), cancellationToken).ConfigureAwait(false); } catch (WorkspaceException ex) { @@ -654,24 +662,36 @@ private static string Flatten(string paths) private static int BoundedGradeWindow(int timeoutSeconds) => timeoutSeconds <= 0 ? SupervisorLane.AcceptanceGradeTimeoutSeconds : Math.Min(timeoutSeconds, SupervisorLane.MaxAcceptanceGradeTimeoutSeconds); + /// + /// The ONE place every lane's grade runs the candidate's bytes: the optional setup step, then the oracle. Both run + /// through a runner bound to the producing run's (a request without one + /// grades ), so no lane can hand a grade more network or + /// resources than the run whose work it executes had. + /// private async Task GradeWorkspaceAsync(WorkspaceGradeRequest request, CancellationToken cancellationToken) { - var (directory, spec, teamId, authoredTimeoutSeconds, producerModel, protection) = request; + var (directory, spec, teamId, authoredTimeoutSeconds, posture, producerModel, protection) = request; var timeoutSeconds = BoundedGradeWindow(authoredTimeoutSeconds); + var runner = AcceptanceGradingPosturePolicy.Bind(_runners.Resolve(GradingRunnerKind), posture ?? AcceptanceGradingPosturePolicy.FailClosed); + var setupSpec = spec.SetupCommand is { Count: > 0 } setupCommand ? SetupSpec(setupCommand, directory, timeoutSeconds) : null; + var setupEgress = setupSpec is null ? null : runner.EnforcedEgress(setupSpec); + var setupNotice = SetupNotice(setupSpec, runner.Posture, setupEgress, directory); - if (spec.SetupCommand is { Count: > 0 } setupCommand) + if (setupSpec is not null) { - var setupFailure = await RunSetupCommandAsync(setupCommand, directory, timeoutSeconds, cancellationToken).ConfigureAwait(false); - if (setupFailure is not null) return setupFailure; + var setupFailure = await RunSetupCommandAsync(runner, setupSpec, severed: setupEgress == SandboxEgressMode.None, cancellationToken).ConfigureAwait(false); + if (setupFailure is not null) return await CaptureEvidenceAsync(WithSetupNotice(setupFailure, setupNotice), teamId, cancellationToken).ConfigureAwait(false); } - var context = BenchmarkGradingContext.ForAcceptance(spec, teamId, timeoutSeconds, directory, _runners.Resolve(GradingRunnerKind)) with { ProducerModel = producerModel }; + var context = BenchmarkGradingContext.ForAcceptance(spec, teamId, timeoutSeconds, directory, runner) with { ProducerModel = producerModel }; var grade = await _graders.Resolve(spec.Kind ?? BenchmarkGradingKind.TestsPass).GradeAsync(context, cancellationToken).ConfigureAwait(false); if (protection.EvidenceNote is not null) grade = grade with { EvidenceText = $"{protection.EvidenceNote}\n{grade.EvidenceText}" }; + grade = WithSetupNotice(grade, setupNotice); + // A PASS keeps nothing else: both folds drop the evidence tail on green (nothing to repair) and the // decider's pass branch renders no evidence at all, so a grade that ran the candidate's own copy of the // file under test reached the brain reading exactly like a protected pass. The clause rides the DETAIL, @@ -688,7 +708,30 @@ private async Task GradeWorkspaceAsync(WorkspaceGradeRequest req return await CaptureEvidenceAsync(grade, teamId, cancellationToken).ConfigureAwait(false); } - private sealed record WorkspaceGradeRequest(string Directory, SupervisorAcceptanceSpec Spec, Guid TeamId, int TimeoutSeconds, ReviewModelIdentity? ProducerModel = null, OracleProtectionOutcome Protection = default); + private sealed record WorkspaceGradeRequest(string Directory, SupervisorAcceptanceSpec Spec, Guid TeamId, int TimeoutSeconds, AcceptanceGradingPosture? Posture, ReviewModelIdentity? ProducerModel = null, OracleProtectionOutcome Protection = default); + + /// The narrowed-setup notice this grade owes, logged once, or null when the contract has no setup step or the sandbox left it the network it asks for. + private string? SetupNotice(SandboxSpec? setupSpec, AcceptanceGradingPosture posture, SandboxEgressMode? enforced, string directory) + { + if (setupSpec is null || AcceptanceGradingPosturePolicy.SetupNetworkNotice(posture, enforced) is not { } notice) return null; + + _logger.LogInformation("Acceptance setup in {Directory} runs under the producing run's posture ({Autonomy}): {Notice}", directory, posture.Autonomy, notice); + + return notice; + } + + /// + /// Put the narrowed-setup notice at the head of the grade's evidence, which is where an operator reads why a + /// setup failed. A FAILURE may gain evidence for it, because evidence never loosens how a failure is classified. + /// A PASS with no evidence of its own does not: admission caps an unevidenced pass, and a notice must never be + /// what lifts that cap. The log line still records it. + /// + private static BenchmarkGrade WithSetupNotice(BenchmarkGrade grade, string? notice) + { + if (notice is null || grade.Passed && string.IsNullOrEmpty(grade.EvidenceText)) return grade; + + return grade with { EvidenceText = string.IsNullOrEmpty(grade.EvidenceText) ? notice : $"{notice}\n{grade.EvidenceText}" }; + } /// /// P3a-1: the oracle run's output becomes a durable CAS artifact — the id a receipt's EvidenceRef binds @@ -726,35 +769,47 @@ internal static BenchmarkGrade WithClippedEvidenceTail(BenchmarkGrade grade) => ? grade : grade with { EvidenceTail = Agents.AcceptanceEvidenceRenderer.ClipTail(grade.EvidenceText) }; + /// + /// The contract's setup step as a spec. A setup INSTALLS what the check needs (a package restore, a toolchain + /// fetch), so it ASKS for the network. How much of it the step gets is the producing run's posture's call — it runs + /// the manifests the agent wrote, so it never reaches further than that agent could. + /// + private static SandboxSpec SetupSpec(IReadOnlyList setupCommand, string directory, int timeoutSeconds) => new() + { + Command = setupCommand[0], + Args = setupCommand.Skip(1).ToList(), + WorkingDirectory = directory, + TimeoutSeconds = timeoutSeconds, + AllowNetwork = true, + }; + /// /// P3.1 part 2: run the contract's OPTIONAL setup step in the SAME workspace before the check — a failure here /// means the check itself never got a chance to run, so it is classified alongside grade-error:/ /// clone-failed: (infra, not a code verdict) rather than as a genuine failing check. Returns null on /// success (proceed to grading); a non-null grade short-circuits . + /// + /// A setup the sandbox (the producing run's posture took its network, and this + /// host enforced it) fails the same way on every attempt, because the posture comes from the same stored task each + /// time. Its detail says so (): still infra, but + /// decided by the grade's posture rather than a transient fault, so an authored retry does not re-buy an agent run + /// to sever it again. /// - private async Task RunSetupCommandAsync(IReadOnlyList setupCommand, string directory, int timeoutSeconds, CancellationToken cancellationToken) + private async Task RunSetupCommandAsync(PostureBoundSandboxRunner runner, SandboxSpec setupSpec, bool severed, CancellationToken cancellationToken) { - var spec = new SandboxSpec - { - Command = setupCommand[0], - Args = setupCommand.Skip(1).ToList(), - WorkingDirectory = directory, - TimeoutSeconds = timeoutSeconds, - // A contract's setup step is what INSTALLS what the check needs (a package restore, a toolchain fetch), - // so it keeps the egress it has always had — stated here rather than inherited, now that a spec that - // says nothing is severed. - AllowNetwork = true, - }; - - var result = await _runners.Resolve(GradingRunnerKind).RunAsync(spec, cancellationToken).ConfigureAwait(false); + var result = await runner.RunAsync(setupSpec, cancellationToken).ConfigureAwait(false); if (result.Status == SandboxStatus.Success) return null; - _logger.LogWarning("Acceptance grading's setup command failed in {Directory}: {Status} (exit {ExitCode}) {Stderr}", directory, result.Status, result.ExitCode, Summarize(result.Stderr)); + _logger.LogWarning("Acceptance grading's setup command failed in {Directory}: {Status} (exit {ExitCode}) {Stderr}", setupSpec.WorkingDirectory, result.Status, result.ExitCode, Summarize(result.Stderr)); - return result.Status == SandboxStatus.TimedOut - ? Failed("setup-timed-out", GradeFailureClass.Environment) - : Failed($"setup-failed: {Summarize(result.Stderr)}", GradeFailureClass.Environment); + return (severed, result.Status) switch + { + (true, SandboxStatus.TimedOut) => Failed($"{Agents.AgentAcceptanceContract.SetupSeveredDetailPrefix} timed out", GradeFailureClass.Environment), + (true, _) => Failed($"{Agents.AgentAcceptanceContract.SetupSeveredDetailPrefix} {Summarize(result.Stderr)}", GradeFailureClass.Environment), + (false, SandboxStatus.TimedOut) => Failed("setup-timed-out", GradeFailureClass.Environment), + _ => Failed($"setup-failed: {Summarize(result.Stderr)}", GradeFailureClass.Environment), + }; } private static BenchmarkGrade Failed(string detail, GradeFailureClass? failureClass = null) => new() { Passed = false, Detail = detail, Class = failureClass }; diff --git a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorTurnService.Rehydrate.cs b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorTurnService.Rehydrate.cs index a2b02d820..7ed283799 100644 --- a/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorTurnService.Rehydrate.cs +++ b/backend/src/CodeSpace.Core/Services/Supervisor/SupervisorTurnService.Rehydrate.cs @@ -694,6 +694,7 @@ private async Task GradeResolveAcceptanceAsync(SupervisorPriorDe // gate IS the floor — so its own program file(s) are the run's oracle inventory (C3 narrowing). var spec = new SupervisorAcceptanceSpec { Command = command }; var oracleFloorPrograms = AcceptanceOracleProtection.ProgramCandidates(command); + var posture = resolver is null ? null : await ProducerPostureAsync(resolver.AgentRunId, teamId, cancellationToken).ConfigureAwait(false); // The branch arm carries the SAME inventory the patch arm below does. The floor-less overload compiles // and greens, but it silently reduces the grade to authored-only protection — which on this lane means @@ -701,12 +702,12 @@ private async Task GradeResolveAcceptanceAsync(SupervisorPriorDe // saying it went unanchored. There is no base sha to pair it with here (this lane resolves none), and // the anchor is what keeps that from reading as an oversight. if (!string.IsNullOrEmpty(resolver?.ProducedBranch)) - return await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, Branch = resolver.ProducedBranch, Spec = spec, TimeoutSeconds = SupervisorLane.AcceptanceGradeTimeoutSeconds, Anchor = new OracleAnchor(null, oracleFloorPrograms), ProducerModel = ProducerModelOf(resolver!) }, cancellationToken).ConfigureAwait(false); + return await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, Branch = resolver.ProducedBranch, Spec = spec, TimeoutSeconds = SupervisorLane.AcceptanceGradeTimeoutSeconds, Anchor = new OracleAnchor(null, oracleFloorPrograms), ProducerModel = ProducerModelOf(resolver!), Posture = posture }, cancellationToken).ConfigureAwait(false); var manifest = resolver is not null ? await ResolveUnitManifestAsync(resolver.AgentRunId, repositoryId.Value, teamId, cancellationToken).ConfigureAwait(false) : null; if (manifest is { PatchArtifactId: not null, BaseSha: not null }) - return await _acceptanceGrader.GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, BaseSha = manifest.BaseSha!, PatchArtifactId = manifest.PatchArtifactId, Spec = spec, TimeoutSeconds = SupervisorLane.AcceptanceGradeTimeoutSeconds, OracleFloorPrograms = oracleFloorPrograms, ProducerModel = ProducerModelOf(resolver!) }, cancellationToken).ConfigureAwait(false); + return await _acceptanceGrader.GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, BaseSha = manifest.BaseSha!, PatchArtifactId = manifest.PatchArtifactId, Spec = spec, TimeoutSeconds = SupervisorLane.AcceptanceGradeTimeoutSeconds, OracleFloorPrograms = oracleFloorPrograms, ProducerModel = ProducerModelOf(resolver!), Posture = posture }, cancellationToken).ConfigureAwait(false); return new BenchmarkGrade { Passed = false, Detail = "no-branch-or-repo" }; } @@ -962,17 +963,19 @@ private async Task GradeUnitAcceptanceAsync(SupervisorAgentResul if (repositoryId is null) return await GradeCapturedUnitAsync(result, spec, timeoutSeconds, teamId, cancellationToken).ConfigureAwait(false); + var posture = await ProducerPostureAsync(result.AgentRunId, teamId, cancellationToken).ConfigureAwait(false); + if (!string.IsNullOrEmpty(result.ProducedBranch)) { var anchor = await OracleAnchorAsync(result.AgentRunId, repositoryId.Value, spec, oracleFloorPrograms, teamId, cancellationToken).ConfigureAwait(false); - return await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, Branch = result.ProducedBranch, Spec = spec, TimeoutSeconds = timeoutSeconds, Anchor = anchor, ProducerModel = ProducerModelOf(result) }, cancellationToken).ConfigureAwait(false); + return await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, Branch = result.ProducedBranch, Spec = spec, TimeoutSeconds = timeoutSeconds, Anchor = anchor, ProducerModel = ProducerModelOf(result), Posture = posture }, cancellationToken).ConfigureAwait(false); } var manifest = await ResolveUnitManifestAsync(result.AgentRunId, repositoryId.Value, teamId, cancellationToken).ConfigureAwait(false); if (manifest is { PatchArtifactId: not null, BaseSha: not null }) - return await _acceptanceGrader.GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, BaseSha = manifest.BaseSha!, PatchArtifactId = manifest.PatchArtifactId, Spec = spec, TimeoutSeconds = timeoutSeconds, OracleFloorPrograms = oracleFloorPrograms, ProducerModel = ProducerModelOf(result) }, cancellationToken).ConfigureAwait(false); + return await _acceptanceGrader.GradePatchAsync(new PatchAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, BaseSha = manifest.BaseSha!, PatchArtifactId = manifest.PatchArtifactId, Spec = spec, TimeoutSeconds = timeoutSeconds, OracleFloorPrograms = oracleFloorPrograms, ProducerModel = ProducerModelOf(result), Posture = posture }, cancellationToken).ConfigureAwait(false); return NotApplicableOrFailed(expectsChanges); } @@ -1010,7 +1013,9 @@ private async Task GradeCapturedUnitAsync(SupervisorAgentResult if (!Agents.AgentAcceptanceContract.GradesFromDeliverables(spec)) return new BenchmarkGrade { Passed = false, Detail = "no-branch-or-repo" }; - var grade = await _acceptanceGrader.GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = result.AgentRunId, TeamId = teamId, Spec = spec, TimeoutSeconds = timeoutSeconds, ProducerModel = ProducerModelOf(result) }, cancellationToken).ConfigureAwait(false); + var posture = await ProducerPostureAsync(result.AgentRunId, teamId, cancellationToken).ConfigureAwait(false); + + var grade = await _acceptanceGrader.GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = result.AgentRunId, TeamId = teamId, Spec = spec, TimeoutSeconds = timeoutSeconds, ProducerModel = ProducerModelOf(result), Posture = posture }, cancellationToken).ConfigureAwait(false); return grade.Detail == ISupervisorAcceptanceGrader.NoDeliverablesCaptured ? DisambiguateEmptyWorld(result, grade) : grade; } @@ -1070,14 +1075,18 @@ private static BenchmarkGrade DisambiguateEmptyWorld(SupervisorAgentResult resul // no-branch-no-patch verdict leaves nothing to differentiate. if (string.IsNullOrEmpty(result.ProducedBranch) && manifest.PatchArtifactId is null) return null; + // The baseline runs under the CANDIDATE's posture: a differential between a base graded with network and + // a candidate graded without it would measure the two sandboxes, not the work. + var posture = await ProducerPostureAsync(result.AgentRunId, teamId, cancellationToken).ConfigureAwait(false); + // The memo key is the FULL spec identity, not just the argv: Kind routes a different oracle, and // SetupCommand/Rubric/Schema change what the same argv means — two subtasks may share a Command yet - // measure different contracts (adversarial-scan M1). - var key = $"{repositoryId}@{manifest.BaseSha}#{System.Text.Json.JsonSerializer.Serialize(spec, Agents.AgentJson.Options)}"; + // measure different contracts (adversarial-scan M1). The posture is part of it for the reason above. + var key = $"{repositoryId}@{manifest.BaseSha}#{System.Text.Json.JsonSerializer.Serialize(spec, Agents.AgentJson.Options)}#{System.Text.Json.JsonSerializer.Serialize(posture, Agents.AgentJson.Options)}"; if (baselines.TryGetValue(key, out var memoized)) return memoized; - var grade = await _acceptanceGrader.GradeBaseAsync(repositoryId.Value, teamId, manifest.BaseSha!, spec, spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, cancellationToken).ConfigureAwait(false); + var grade = await _acceptanceGrader.GradeBaseAsync(new BaseAcceptanceGradeRequest { RepositoryId = repositoryId.Value, TeamId = teamId, BaseSha = manifest.BaseSha!, Spec = spec, TimeoutSeconds = spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, Posture = posture }, cancellationToken).ConfigureAwait(false); // A MEASURED baseline is shareable; an infra-classed one (a transient clone fault) is not — stamping it // onto every sibling off the same base would spread one blip across the whole fan-out. @@ -1109,6 +1118,8 @@ private async Task GradeUnitAcceptanceMultiRepoAsync(SupervisorA if (targets.Count == 0) return NotApplicableOrFailed(expectsChanges); + var posture = await ProducerPostureAsync(result.AgentRunId, teamId, cancellationToken).ConfigureAwait(false); + foreach (var target in targets) { BenchmarkGrade grade; @@ -1116,7 +1127,7 @@ private async Task GradeUnitAcceptanceMultiRepoAsync(SupervisorA { var anchor = await OracleAnchorAsync(result.AgentRunId, target.RepositoryId!.Value, spec, oracleFloorPrograms, teamId, cancellationToken).ConfigureAwait(false); - grade = await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = target.RepositoryId!.Value, TeamId = teamId, Branch = target.ProducedBranch!, Spec = spec, TimeoutSeconds = spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, Anchor = anchor, ProducerModel = ProducerModelOf(result) }, cancellationToken).ConfigureAwait(false); + grade = await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = target.RepositoryId!.Value, TeamId = teamId, Branch = target.ProducedBranch!, Spec = spec, TimeoutSeconds = spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, Anchor = anchor, ProducerModel = ProducerModelOf(result), Posture = posture }, cancellationToken).ConfigureAwait(false); } catch (Workflows.Llm.LlmBudgetExceededException refused) { @@ -1414,7 +1425,7 @@ internal async Task ApplyStopAcceptanceGradeAsync(Superviso // "grader.acceptance" so its spend is recorded + counts toward the cost cap. BenchmarkGrade grade; using (Workflows.Llm.LlmCallContext.Push(new Workflows.Llm.LlmCallScope(context.SupervisorRunId, teamId, context.NodeId, "", GraderAcceptanceCallKind, _recordLogger, _offloader, _budget, context.MaxCostUsd, context.ModelPrices))) - grade = await GradeStopTargetsWithHeartbeatAsync(context.SupervisorRunId, context.NodeId, teamId, targets, gates, oracleBaseShas, AcceptanceOracleProtection.ProgramCandidates(floorCommand), cancellationToken).ConfigureAwait(false); + grade = await GradeStopTargetsWithHeartbeatAsync(context.SupervisorRunId, context.NodeId, teamId, targets, gates, oracleBaseShas, AcceptanceOracleProtection.ProgramCandidates(floorCommand), RunProfilePosture(context.AgentProfile), cancellationToken).ConfigureAwait(false); return execution with { OutcomeJson = SupervisorOutcome.AppendAcceptanceGrade(execution.OutcomeJson, grade.Passed, grade.Detail) }; } @@ -1472,7 +1483,9 @@ internal async Task ApplyStopAcceptanceGradeAsync(Superviso { foreach (var unit in units) { - var grade = await _acceptanceGrader.GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = unit.AgentRunId, TeamId = teamId, Spec = spec, TimeoutSeconds = timeoutSeconds, ProducerModel = ProducerModelOf(unit) }, cancellationToken).ConfigureAwait(false); + var posture = await ProducerPostureAsync(unit.AgentRunId, teamId, cancellationToken).ConfigureAwait(false); + + var grade = await _acceptanceGrader.GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = unit.AgentRunId, TeamId = teamId, Spec = spec, TimeoutSeconds = timeoutSeconds, ProducerModel = ProducerModelOf(unit), Posture = posture }, cancellationToken).ConfigureAwait(false); if (grade.Passed) return (grade, false); @@ -1561,14 +1574,14 @@ private static IReadOnlyList BranchlessUnits(SupervisorTu /// migration) at ; it stops the instant grading /// finishes (success, failure, or exception) via the linked token — never outlives the grade it protects. /// - private async Task GradeStopTargetsWithHeartbeatAsync(Guid supervisorRunId, string nodeId, Guid teamId, IReadOnlyList<(Guid RepositoryId, string Alias, string Branch)> targets, IReadOnlyList<(string Label, SupervisorAcceptanceSpec? Spec)> gates, IReadOnlyDictionary oracleBaseShas, IReadOnlyList oracleFloorPrograms, CancellationToken cancellationToken) + private async Task GradeStopTargetsWithHeartbeatAsync(Guid supervisorRunId, string nodeId, Guid teamId, IReadOnlyList<(Guid RepositoryId, string Alias, string Branch)> targets, IReadOnlyList<(string Label, SupervisorAcceptanceSpec? Spec)> gates, IReadOnlyDictionary oracleBaseShas, IReadOnlyList oracleFloorPrograms, AcceptanceGradingPosture posture, CancellationToken cancellationToken) { using var heartbeatCts = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); var heartbeat = RunGradingHeartbeatLoopAsync(supervisorRunId, nodeId, SupervisorLane.AcceptanceGradeHeartbeatInterval, heartbeatCts.Token, TimeProvider.System); try { - return await GradeStopTargetsAsync(teamId, targets, gates, oracleBaseShas, oracleFloorPrograms, cancellationToken).ConfigureAwait(false); + return await GradeStopTargetsAsync(teamId, targets, gates, oracleBaseShas, oracleFloorPrograms, posture, cancellationToken).ConfigureAwait(false); } finally { @@ -1641,7 +1654,7 @@ private async Task PulseGradingHeartbeatAsync(Guid supervisorRunId, string nodeI /// operator's own workspace, not the model). The first-failure short-circuit keeps the common rejected case cheap; /// a future perf slice could grade with a bounded degree-of-parallelism if a large workspace makes wall-clock bite. /// - private async Task GradeStopTargetsAsync(Guid teamId, IReadOnlyList<(Guid RepositoryId, string Alias, string Branch)> targets, IReadOnlyList<(string Label, SupervisorAcceptanceSpec? Spec)> gates, IReadOnlyDictionary oracleBaseShas, IReadOnlyList oracleFloorPrograms, CancellationToken cancellationToken) + private async Task GradeStopTargetsAsync(Guid teamId, IReadOnlyList<(Guid RepositoryId, string Alias, string Branch)> targets, IReadOnlyList<(string Label, SupervisorAcceptanceSpec? Spec)> gates, IReadOnlyDictionary oracleBaseShas, IReadOnlyList oracleFloorPrograms, AcceptanceGradingPosture posture, CancellationToken cancellationToken) { var oracleNotes = new List(); @@ -1663,7 +1676,7 @@ private async Task GradeStopTargetsAsync(Guid teamId, IReadOnlyL // by rewriting the check script the operator's floor runs. The floor's inventory gates BOTH // gates: the model's own tightening command can name a file the goal required editing, and // restoring that would void the very work the stop is shipping. - grade = await _acceptanceGrader.GradeAsync(target.RepositoryId, teamId, target.Branch, spec, spec?.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, new OracleAnchor(oracleBaseShas.GetValueOrDefault(target.RepositoryId), oracleFloorPrograms), cancellationToken).ConfigureAwait(false); + grade = await _acceptanceGrader.GradeAsync(new RepositoryAcceptanceGradeRequest { RepositoryId = target.RepositoryId, TeamId = teamId, Branch = target.Branch, Spec = spec, TimeoutSeconds = spec.TimeoutSeconds ?? SupervisorLane.AcceptanceGradeTimeoutSeconds, Anchor = new OracleAnchor(oracleBaseShas.GetValueOrDefault(target.RepositoryId), oracleFloorPrograms), Posture = posture }, cancellationToken).ConfigureAwait(false); } catch (Workflows.Llm.LlmBudgetExceededException refused) { @@ -1756,6 +1769,32 @@ private async Task> ReadResolve return rows.ToDictionary(r => r.Id, r => new ResolveContributorRow(r.Id, r.TeamId, r.Status, r.Error, r.ResultJson, r.TaskJson)); } + /// + /// The posture a grade of 's work runs under, read off the unit's own stored task: the + /// tier and permissions it ran with, as admitted. Null when the row is not this team's or its task cannot be read, + /// and the grader then grades fail-closed. Every per-unit lane reads it for the unit whose bytes it runs. + /// + private async Task ProducerPostureAsync(Guid agentRunId, Guid teamId, CancellationToken cancellationToken) + { + var taskJson = await _db.AgentRun.AsNoTracking().Where(r => r.Id == agentRunId && r.TeamId == teamId).Select(r => r.TaskJson).SingleOrDefaultAsync(cancellationToken).ConfigureAwait(false); + + return AcceptanceGradingPosturePolicy.ForStoredTask(taskJson); + } + + /// + /// The posture of the run-level STOP grade, whose head mixes every unit's work: the run's own autonomy grant, which + /// is the tier each unit it spawns is clamped to (RealSupervisorActionExecutor.AutonomyOf), with the + /// permissions that tier derives. A unit may lower its own tier, but none can exceed this one. So the head is never + /// graded with more than the operator granted the run, and never with less than its most capable unit could have + /// run against the same bytes. + /// + private static AcceptanceGradingPosture RunProfilePosture(SupervisorAgentProfile? profile) + { + var tier = Executors.RealSupervisorActionExecutor.AutonomyOf(profile); + + return AcceptanceGradingPosturePolicy.For(tier, AgentAutonomyPolicy.Derive(tier)); + } + /// The producer's durable routing identity from its task envelope, best-effort. It never invents provider observation. private static ReviewModelIdentity? ReadModelIdentity(string? taskJson) { diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs index 9bb4629b7..5e6dbf98b 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs @@ -340,9 +340,15 @@ private static NodeResult MapResult(JsonElement payload, decimal? maxCostUsd, bo // slow suite) is an environment/workload fact, not a code defect, so it gets the SAME fresh-respawn // chance a crash/timeout does (mirrors AgentAcceptanceContract.IsInfraFailure, the same classification // the executor's revise loop / supervisor decider / recitation already apply elsewhere). + // + // ...EXCEPT an infra fault the grade's own posture decided (a setup step the sandbox severed because the + // producing run had no network): the posture is read off the same stored task on every attempt, so a + // respawn re-buys a whole agent run only to fail the same way. Still infra — never a code verdict — but + // deterministic, so it falls back into the non-retryable arm below. var exitReason = ReadString(payload, "exitReason"); var acceptanceFailed = exitReason == AgentAcceptanceContract.FailClosedExitReason; - var acceptanceInfraFault = acceptanceFailed && AgentAcceptanceContract.IsInfraFailure(ReadOptionalString(payload, "acceptanceDetail"), WorkPresent(payload)); + var acceptanceDetail = ReadOptionalString(payload, "acceptanceDetail"); + var acceptanceInfraFault = acceptanceFailed && AgentAcceptanceContract.IsInfraFailure(acceptanceDetail, WorkPresent(payload)) && !AgentAcceptanceContract.IsDecidedByGradePosture(acceptanceDetail); // The SAME carve-out, one status over: NeedsReview is not one fact. The critic flagging an output is a // verdict a respawn cannot change, but the IDLE watchdog killing a silent process is an environment diff --git a/backend/src/CodeSpace.Messages/Agents/AcceptanceGradingPosture.cs b/backend/src/CodeSpace.Messages/Agents/AcceptanceGradingPosture.cs new file mode 100644 index 000000000..bceb967ae --- /dev/null +++ b/backend/src/CodeSpace.Messages/Agents/AcceptanceGradingPosture.cs @@ -0,0 +1,30 @@ +namespace CodeSpace.Messages.Agents; + +/// +/// The sandbox posture an acceptance grade runs its steps under — the setup command and the check — taken from the +/// run that PRODUCED the work being graded: its autonomy tier, its network grant and its egress allowlist, already +/// clamped by the deployment's autonomy ceiling. A grade executes the candidate's own bytes (a manifest a setup step +/// installs from, a script the check imports), after the producing run's sandbox is gone, so it may never hand those +/// bytes more than the run that wrote them had. +/// +/// Derived ONCE per producing run by AcceptanceGradingPosturePolicy and carried on the grade request; +/// applied NARROW-ONLY at the grader's one choke point. A request with no posture grades with network off under the +/// Confined tier's resource ceilings. +/// +public sealed record AcceptanceGradingPosture +{ + /// The producer's tier after the deployment ceiling: the row and come from, and the tier a narrowed-setup notice names. + public required AgentAutonomyLevel Autonomy { get; init; } + + /// Whether a grade step that asks for the network may have it. False ⇒ every step is severed (where the sandbox confines). + public required bool AllowNetwork { get; init; } + + /// The hosts a networked grade step is narrowed to — the producer's operator-configured egress allowlist. Null ⇒ no narrowing beyond . Never empty: an allowlist with no host is false. + public IReadOnlyList? EgressAllowlist { get; init; } + + /// The memory ceiling every grade step runs under, in MiB — the producer tier's committed row, narrowed by the host budget. + public required int MaxMemoryMb { get; init; } + + /// The CPU ceiling every grade step runs under, as a percent of one core — the producer tier's committed row. + public required int MaxCpuPercent { get; init; } +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Agents/AcceptanceGradingPostureFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Agents/AcceptanceGradingPostureFlowTests.cs new file mode 100644 index 000000000..7fac22c09 --- /dev/null +++ b/backend/tests/CodeSpace.IntegrationTests/Agents/AcceptanceGradingPostureFlowTests.cs @@ -0,0 +1,206 @@ +using System.Net; +using System.Net.Sockets; +using System.Text.Json; +using Autofac; +using Autofac.Extensions.DependencyInjection; +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Agents.Publish; +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using CodeSpace.Core.Services.Agents.Workspace; +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Core.Services.Workflows.Artifacts; +using CodeSpace.IntegrationTests.Infrastructure; +using CodeSpace.IntegrationTests.Workflows.Infrastructure; +using CodeSpace.Messages.Agents; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Logging; +using Shouldly; + +namespace CodeSpace.IntegrationTests.Agents; + +/// +/// 🟢 Integration (real Postgres, the production executor, admission and local acceptance lane, the real +/// on the real ; only the agent harness is +/// scripted): an acceptance setup step reaches exactly the network its PRODUCING run had. The setup is an operator +/// command (npm ci, pip install) that executes manifests the agent wrote, and it runs after the agent's +/// sandbox is gone. So a Standard run's setup must not reach a loopback sink the Standard agent itself could not +/// reach, while an Unleashed run's setup still can. +/// +/// Two proofs per run. The specs the grader hands its runner are recorded and fed through +/// on a stand-in bwrap path: the argv a confining Linux host launches, +/// provable on any host. The REAL reachability of the sink is asserted wherever this host confines (bubblewrap +/// present). A macOS development host does not confine, so there it reaches the sink whatever the posture. The +/// sandbox lane asserts the same posture against the real kernel (AcceptanceGradingPostureE2ETests). +/// +[Collection(PostgresCollection.Name)] +[Trait("Category", "Integration")] +public sealed class AcceptanceGradingPostureFlowTests(PostgresFixture fixture) +{ + private const string FakeBwrap = "/usr/bin/bwrap"; + + [Theory] + [InlineData(AgentAutonomyLevel.Standard, false)] + [InlineData(AgentAutonomyLevel.Unleashed, true)] + public async Task An_acceptance_setup_reaches_only_the_network_its_producing_run_had(AgentAutonomyLevel tier, bool runHasNetwork) + { + using var context = new SinkContext(); + + var script = $"printf ok > report.txt; (echo agent >/dev/tcp/127.0.0.1/{context.Port}) 2>/dev/null; echo produced"; + var acceptance = new SupervisorAcceptanceSpec { Command = ["/bin/sh", "-c", "test -f report.txt"], SetupCommand = ["/bin/bash", "-c", $"echo setup >/dev/tcp/127.0.0.1/{context.Port}"] }; + var task = new AgentTask { Goal = "write the report", Harness = PostureHarness.HarnessKind, Model = "test-model", WorkspaceDirectory = context.Workspace, Autonomy = tier, Permissions = AgentAutonomyPolicy.Derive(tier), MaxReviseRounds = 0, TimeoutSeconds = 30, Acceptance = acceptance }; + + var (stored, result, evidence) = await ExecuteAsync(task, new PostureHarness(script), context.Recorded); + + stored.Permissions.Network.ShouldBe(runHasNetwork ? AgentNetworkAccess.On : AgentNetworkAccess.Off, "fixture check: admission kept the tier's network grant — a lowered Sandbox:MaxAutonomy in this environment would void the Unleashed half"); + + var (setup, check) = RecordedSteps(context.Recorded, result); + var expected = AcceptanceGradingPosturePolicy.For(stored); + + setup.AllowNetwork.ShouldBe(runHasNetwork, "the setup asked for the network and got exactly what the producing run had"); + Argv(setup).Contains("--unshare-net").ShouldBe(!runHasNetwork, "on a confining host the Standard setup is severed and the Unleashed one shares the host network"); + check.AllowNetwork.ShouldBeFalse("the check's network cut is unchanged"); + setup.MaxMemoryMb.ShouldBe(expected.MaxMemoryMb, "the setup runs under the producing tier's memory ceiling"); + check.MaxMemoryMb.ShouldBe(expected.MaxMemoryMb, "and so does the check"); + check.MaxCpuPercent.ShouldBe(expected.MaxCpuPercent); + + var severed = !runHasNetwork && BubblewrapSandbox.Available is not null; + evidence.ShouldNotBeNull("both outcomes bind evidence: a TestsPass check's output, or a severed setup's failure"); + evidence.StartsWith(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix, StringComparison.Ordinal).ShouldBe(severed, "a setup the sandbox severed records one notice saying so, at the head of the evidence — and a host that did not confine claims nothing it did not do"); + + var hits = context.DrainHits(); + if (runHasNetwork) hits.ShouldContain("setup", customMessage: $"the Unleashed setup should have reached the loopback sink on port {context.Port} — check that /bin/bash supports /dev/tcp here"); + + if (BubblewrapSandbox.Available is null) return; // this host does not confine: below is the confined-host answer + + hits.Contains("agent").ShouldBe(runHasNetwork, "fixture check: the agent itself reaches the sink only when its tier grants the network"); + hits.Contains("setup").ShouldBe(runHasNetwork, "the setup never reaches a sink its producing agent could not"); + result.AcceptancePassed.ShouldBe(runHasNetwork, result.AcceptanceDetail); + if (!runHasNetwork) result.AcceptanceDetail.ShouldNotBeNull().ShouldStartWith(AgentAcceptanceContract.SetupSeveredDetailPrefix, customMessage: "the severed setup's failure is decided by the posture, so an authored retry does not re-buy the run"); + } + + private async Task<(AgentTask Stored, AgentRunResult Result, string? Evidence)> ExecuteAsync(AgentTask task, PostureHarness harness, List recorded) + { + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(fixture); + Guid runId; + using (var admission = fixture.BeginScopeAs(userId, teamId)) + runId = (await admission.Resolve().CreateAsync(task, teamId, null, null, cancellationToken: CancellationToken.None)).Id; + + using (var execution = fixture.BeginScope(builder => RegisterRecordingGrader(builder, harness, recorded))) + await execution.Resolve().ExecuteAsync(runId, CancellationToken.None); + + using var scope = fixture.BeginScope(); + var run = await scope.Resolve().GetAsync(runId, CancellationToken.None); + var result = JsonSerializer.Deserialize(run.ResultJson!, AgentJson.Options)!; + var stored = JsonSerializer.Deserialize(run.TaskJson, AgentJson.Options)!; + var bytes = result.AcceptanceEvidenceId is { } id ? await scope.Resolve().GetBytesAsync(teamId, id, CancellationToken.None) : null; + + return (stored, result, bytes is null ? null : System.Text.Encoding.UTF8.GetString(bytes.Bytes)); + } + + /// + /// The scripted harness, and the PRODUCTION grader built over a runner registry that records every spec it is + /// handed before running it for real. The executor reaches the local lane's grader through a child scope it opens + /// itself, and the container's own scope factory opens those under the root. So the scope factory is replaced too, + /// and the executor's child scopes open under this test's scope, where the recording grader is registered. + /// + private static void RegisterRecordingGrader(ContainerBuilder builder, PostureHarness harness, List recorded) + { + builder.RegisterInstance(new AgentHarnessRegistry([harness])).As(); + builder.Register(c => new TestScopeFactory(c.Resolve())).As().InstancePerLifetimeScope(); + builder.Register(c => new SupervisorAcceptanceGrader(c.Resolve(), c.Resolve(), new RecordingRunners(c.Resolve(), recorded), c.Resolve(), c.Resolve(), c.Resolve(), c.Resolve(), c.Resolve>())).As().InstancePerLifetimeScope(); + } + + private static (SandboxSpec Setup, SandboxSpec Check) RecordedSteps(List recorded, AgentRunResult result) + { + recorded.Select(s => s.Command).ShouldBe(new[] { "/bin/bash", "/bin/sh" }, $"fixture check: the grade ran its setup step and then its check through the recording registry (run {result.Status}, acceptance {result.AcceptancePassed}: {result.AcceptanceDetail}; error: {result.Error})"); + + return (recorded[0], recorded[1]); + } + + private static IReadOnlyList Argv(SandboxSpec spec) => + LocalProcessRunner.ChildCommand(new LocalProcessRunner.CommandIsolationContext(spec, null, null, Array.Empty(), Array.Empty()), FakeBwrap, prlimit: null); + + /// A loopback sink on a port the kernel picked, the workspace the agent writes into, and the specs the grader ran; all torn down on dispose. + private sealed class SinkContext : IDisposable + { + private readonly TcpListener _listener = new(IPAddress.Loopback, 0); + + public SinkContext() + { + _listener.Start(); + Port = ((IPEndPoint)_listener.LocalEndpoint).Port; + Directory.CreateDirectory(Workspace); + } + + public int Port { get; } + public string Workspace { get; } = Path.Combine(Path.GetTempPath(), "cs-grade-posture-" + Guid.NewGuid().ToString("N")); + public List Recorded { get; } = new(); + + /// Every marker a client wrote before the run ended. Each client closed its connection before exiting, so the run has finished every handshake by now, and what is pending is all there is. + public IReadOnlyList DrainHits() + { + var hits = new List(); + + while (_listener.Pending()) + { + using var client = _listener.AcceptTcpClient(); + using var reader = new StreamReader(client.GetStream()); + hits.Add(reader.ReadLine() ?? ""); + } + + return hits; + } + + public void Dispose() + { + _listener.Stop(); + try { Directory.Delete(Workspace, recursive: true); } catch (IOException) { } + } + } + + private sealed class TestScopeFactory(ILifetimeScope parent) : IServiceScopeFactory + { + public IServiceScope CreateScope() => new TestScope(parent.BeginLifetimeScope()); + } + + private sealed class TestScope(ILifetimeScope scope) : IServiceScope + { + public IServiceProvider ServiceProvider { get; } = new AutofacServiceProvider(scope); + public void Dispose() => scope.Dispose(); + } + + private sealed class RecordingRunners(ISandboxRunnerRegistry inner, List recorded) : ISandboxRunnerRegistry + { + public IReadOnlyList All => inner.All; + public ISandboxRunner Resolve(string kind) => new RecordingRunner(inner.Resolve(kind), recorded); + } + + private sealed class RecordingRunner(ISandboxRunner inner, List recorded) : ISandboxRunner, ISandboxEgressEnforcement + { + public string Kind => inner.Kind; + + /// The real runner's own answer about what it enforces on this host — recording a spec must not change what the grade can claim about it. + public SandboxEgressMode EnforcedEgress(SandboxSpec spec) => ((ISandboxEgressEnforcement)inner).EnforcedEgress(spec); + + public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) + { + recorded.Add(spec); + return inner.RunAsync(spec, cancellationToken); + } + } + + /// A scripted agent whose sandbox asks for the network exactly as the shipped harnesses do: from the task's admitted permissions. + private sealed class PostureHarness(string script) : IAgentHarness + { + public const string HarnessKind = "grading-posture-test"; + public string Kind => HarnessKind; + public string Version => "test"; + public IReadOnlyList Models => ["test-model"]; + public SandboxSpec BuildInvocation(AgentTask task) => new() { Command = "/bin/bash", Args = ["-c", script], WorkingDirectory = task.WorkspaceDirectory, TimeoutSeconds = 30, AllowNetwork = task.Permissions.Network == AgentNetworkAccess.On }; + public IReadOnlyList ParseEvents(string rawLine) => [new() { Kind = AgentEventKind.AssistantMessage, Text = rawLine }]; + public IAgentEventFolder CreateFolder() => ScriptedFolders.Result(); + } +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Agents/BenchmarkRunnerFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Agents/BenchmarkRunnerFlowTests.cs index d180792fd..16c7560b8 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Agents/BenchmarkRunnerFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Agents/BenchmarkRunnerFlowTests.cs @@ -1,7 +1,10 @@ +using System.Text.Json; using Autofac; using CodeSpace.Core.Persistence.Db; using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Agents.Eval.Benchmark.Graders; using CodeSpace.Core.Services.Agents.Harnesses.Codex; using CodeSpace.Core.Services.Supervisor; using CodeSpace.IntegrationTests.Infrastructure; @@ -307,8 +310,61 @@ public async Task A_cell_whose_fixture_cannot_be_re_staged_fails_closed_instead_ await Should.ThrowAsync(() => RunAsync(task, BenchmarkMode.HarnessCli, workspace.Directory, teamId)); } + [Fact] + public async Task The_benchmark_check_runs_under_the_benchmark_agents_own_posture() + { + // The check runs the fixture's test command in the workspace the agent just wrote, so a module the tests + // import is the agent's code. It runs under the posture the agent itself ran with: the same ceilings, and the + // check's own network cut. Before, the grade handed TestsPassGrader the raw local runner, with no ceiling. + if (OperatingSystem.IsWindows()) return; + + using var cli = new FakeBenchmarkCli(); + using var workspace = BenchmarkFixture.StageSolved(); + + var teamId = await SeedTeamAsync(); + var task = TestsPassTask(); + var graders = new RunnerRecordingGraders(); + + BenchmarkResult result; + using (var scope = _fixture.BeginScope(b => + { + b.RegisterInstance(new TestCurrentUser(_operators[teamId], "test", Array.Empty())).As().SingleInstance(); + b.RegisterInstance(new TestCurrentTeam(teamId)).As().SingleInstance(); + b.RegisterInstance(graders).As(); + })) + { + result = await scope.Resolve().RunAsync(task, BenchmarkMode.HarnessCli, new BenchmarkExecutionContext { WorkspaceDirectory = workspace.Directory, TeamId = teamId }, CancellationToken.None); + } + + var expected = AcceptanceGradingPosturePolicy.For(BenchmarkRunner.BuildAgentTask(task, BenchmarkMode.HarnessCli, workspace.Directory, selection: null)); + var runner = graders.Runners.ShouldHaveSingleItem("fixture check: the real TestsPass oracle graded the cell once").ShouldBeOfType("the oracle is handed a runner bound to a posture, never the raw local runner"); + + result.Grade.Passed.ShouldBeTrue("fixture check: the solved fixture still grades pass under the posture"); + JsonSerializer.Serialize(runner.Posture, AgentJson.Options).ShouldBe(JsonSerializer.Serialize(expected, AgentJson.Options), "the posture the benchmark agent itself ran with — not the fail-closed default"); + expected.MaxMemoryMb.ShouldBeGreaterThan(0, "fixture check: a zero ceiling would mean unlimited and pass vacuously"); + } + // ─── Helpers ─── + /// The real TestsPass oracle, recording the runner each grade hands it before it runs the check for real. + private sealed class RunnerRecordingGraders : IBenchmarkGraderRegistry + { + public List Runners { get; } = new(); + + public IBenchmarkGrader Resolve(BenchmarkGradingKind kind) => new Recording(this); + + private sealed class Recording(RunnerRecordingGraders owner) : IBenchmarkGrader + { + public BenchmarkGradingKind Kind => BenchmarkGradingKind.TestsPass; + + public Task GradeAsync(BenchmarkGradingContext context, CancellationToken cancellationToken) + { + owner.Runners.Add(context.Runner); + return new TestsPassGrader().GradeAsync(context, cancellationToken); + } + } + } + /// The cell's M1a four-state verdict, through the REAL classifier over a one-cell manifest — what the evaluator-health floor actually counts, never a re-derivation of it here. private static CorpusCellState CellStateOf(BenchmarkTask task, BenchmarkResult result) => EvalSuite.Classify(EvalSuite.ManifestFor(new[] { task }), new[] { result }, Array.Empty()).Single().State; diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorAcceptanceFoldFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorAcceptanceFoldFlowTests.cs index 37fed8ff4..5ed15e820 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorAcceptanceFoldFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorAcceptanceFoldFlowTests.cs @@ -73,6 +73,46 @@ public async Task A_resolve_is_graded_once_and_the_objective_verdict_is_folded_a .ShouldBe(true, "the grade is PERSISTED onto the durable ledger row (so replay reads it, not re-grades)"); } + [Theory] + [InlineData(true, AgentAutonomyLevel.Standard, null)] // branch arm, a network-off resolver + [InlineData(false, AgentAutonomyLevel.Standard, null)] // patch arm (a guard-blocked push), the same resolver + [InlineData(true, AgentAutonomyLevel.Trusted, "registry.npmjs.org")] // branch arm, an allowlisted resolver + [InlineData(false, AgentAutonomyLevel.Trusted, "registry.npmjs.org")] // patch arm, the same resolver + public async Task A_resolve_is_graded_under_the_resolvers_own_stored_posture(bool pushed, AgentAutonomyLevel tier, string? allowHost) + { + // The resolve grade runs the resolver's reconciled bytes, so its setup and check get the resolver's own tier + // and network — read off its stored task, never the run profile and never the old hard-coded host network — + // on BOTH arms: the pushed branch, and the recorded patch a policy-blocked push leaves behind. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var runId = await SeedSupervisorRunAsync(teamId, userId); + var repoId = Guid.NewGuid(); + var resolverId = Guid.NewGuid(); + var branch = pushed ? "codespace/resolve/x" : null; + var permissions = AgentAutonomyPolicy.Derive(tier) with { Egress = allowHost is null ? AgentEgressPolicy.Full : AgentEgressPolicy.Allowlist, EgressAllowHosts = allowHost is null ? null : new[] { allowHost } }; + var task = new AgentTask { Goal = "reconcile", Harness = "codex-cli", Autonomy = tier, Permissions = permissions }; + var result = new AgentRunResult { Status = AgentRunStatus.Succeeded, ExitReason = "completed", Summary = $"reconciled {Marker}", ProducedBranch = branch }; + + await SeedAgentRunRawAsync(resolverId, teamId, runId, AgentRunStatus.Succeeded, JsonSerializer.Serialize(result, AgentJson.Options), JsonSerializer.Serialize(task, AgentJson.Options)); + await SeedResolveDecisionAsync(runId, teamId, ResolveOutcomeWithAgentId(resolverId, branch, markerPresent: true)); + if (!pushed) await SeedManifestAsync(teamId, resolverId, repoId, branch: null, baseSha: "deadbeef", patchArtifactId: Guid.NewGuid()); + + var expected = AcceptanceGradingPosturePolicy.For(task); + var unleashed = AcceptanceGradingPosturePolicy.For(AgentAutonomyLevel.Unleashed, AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Unleashed)); + JsonSerializer.Serialize(expected, AgentJson.Options).ShouldNotBe(JsonSerializer.Serialize(AcceptanceGradingPosturePolicy.FailClosed, AgentJson.Options), "fixture check: a lane that dropped the posture (fail-closed) must not pass"); + JsonSerializer.Serialize(expected, AgentJson.Options).ShouldNotBe(JsonSerializer.Serialize(unleashed, AgentJson.Options), "fixture check: a lane that widened the posture must not pass"); + + var grader = new RecordingGrader(new BenchmarkGrade { Passed = true, Detail = "tests-passed" }); + await RehydrateAsync(runId, teamId, GoalConfig(repoId, Command), grader); + + (pushed ? grader.CallCount : grader.PatchCallCount).ShouldBe(1, $"fixture check: the resolve was graded on its {(pushed ? "branch" : "patch")} arm"); + var posture = grader.LastPosture.ShouldNotBeNull("the resolve lane passed the resolver's posture"); + posture.Autonomy.ShouldBe(expected.Autonomy); + posture.AllowNetwork.ShouldBe(expected.AllowNetwork); + posture.EgressAllowlist.ShouldBe(expected.EgressAllowlist); + posture.MaxMemoryMb.ShouldBe(expected.MaxMemoryMb); + posture.MaxCpuPercent.ShouldBe(expected.MaxCpuPercent); + } + [Fact] public async Task A_resolver_that_never_pushed_grades_via_its_own_recorded_patch_not_fail_closed() { @@ -1473,14 +1513,14 @@ private async Task SeedSpawnDecisionAsync(Guid runId, Guid teamId, Guid agentRun await db.SaveChangesAsync(); } - private async Task SeedAgentRunRawAsync(Guid agentRunId, Guid teamId, Guid runId, AgentRunStatus status, string? resultJson) + private async Task SeedAgentRunRawAsync(Guid agentRunId, Guid teamId, Guid runId, AgentRunStatus status, string? resultJson, string taskJson = "{}") { using var scope = _fixture.BeginScope(); var db = scope.Resolve(); db.AgentRun.Add(new AgentRun { Id = agentRunId, TeamId = teamId, WorkflowRunId = runId, NodeId = NodeId, Harness = "codex-cli", - Status = status, TaskJson = "{}", ResultJson = resultJson, + Status = status, TaskJson = taskJson, ResultJson = resultJson, }); await db.SaveChangesAsync(); } @@ -1561,6 +1601,15 @@ private sealed class RecordingGrader : ISupervisorAcceptanceGrader public int PatchCallCount { get; private set; } + /// V-B: the producing-run posture the last request-form grade carried. + public AcceptanceGradingPosture? LastPosture { get; private set; } + + public Task GradeAsync(RepositoryAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + LastPosture = request.Posture; + return GradeAsync(request.RepositoryId, request.TeamId, request.Branch, request.Spec, request.TimeoutSeconds, request.Anchor, cancellationToken); + } + /// The C3 anchor the branch grade was handed. Recorded because it is otherwise unobservable: the floor-less overload compiles and greens while silently reducing the grade to authored-only protection — which on this lane is none — so nothing but this would red if the inventory were dropped. public OracleAnchor? LastAnchor { get; private set; } @@ -1578,6 +1627,12 @@ public Task GradeAsync(Guid repositoryId, Guid teamId, string br return Task.FromResult(_grade); } + public Task GradePatchAsync(PatchAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + LastPosture = request.Posture; + return GradePatchAsync(request.RepositoryId, request.TeamId, request.BaseSha, request.InlinePatch, request.PatchArtifactId, request.Spec, request.TimeoutSeconds, cancellationToken); + } + public Task GradePatchAsync(Guid repositoryId, Guid teamId, string baseSha, string inlinePatch, Guid? patchArtifactId, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) { PatchCallCount++; diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorModelAcceptanceSetupFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorModelAcceptanceSetupFlowTests.cs index 1538b4984..328950d77 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorModelAcceptanceSetupFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorModelAcceptanceSetupFlowTests.cs @@ -101,7 +101,7 @@ public async Task The_same_grader_on_the_same_world_runs_an_operator_setup_under using var scope = _fixture.BeginScope(builder => builder.RegisterInstance(new SandboxRunnerRegistry([runner])).As()); var operatorSpec = new SupervisorAcceptanceSpec { Command = new[] { "report.md" }, Kind = BenchmarkGradingKind.ArtifactPresent, SetupCommand = probe.SetupArgv, TimeoutSeconds = 0 }; - var grade = await scope.Resolve().GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = agentRunId, TeamId = teamId, Spec = operatorSpec, TimeoutSeconds = 0 }, CancellationToken.None); + var grade = await scope.Resolve().GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = agentRunId, TeamId = teamId, Spec = operatorSpec, TimeoutSeconds = 0, Posture = null }, CancellationToken.None); grade.Passed.ShouldBeTrue(grade.Detail); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorUnitAcceptanceFoldFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorUnitAcceptanceFoldFlowTests.cs index 17790c08d..071cc22c0 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorUnitAcceptanceFoldFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/SupervisorUnitAcceptanceFoldFlowTests.cs @@ -833,6 +833,123 @@ public async Task A_unit_with_no_branch_no_patch_and_expects_changes_false_is_a_ unit.AcceptanceDetail.ShouldStartWith("not-applicable"); } + // ─── V-B: every per-unit lane grades under the posture of the unit whose bytes it runs ───────────────── + + [Theory] + [InlineData("branch")] + [InlineData("patch")] + [InlineData("multi-repo")] + [InlineData("captured")] + public async Task Every_per_unit_lane_grades_under_the_units_own_stored_posture(string lane) + { + // The unit's task_json is what its sandbox ran under (tier + permissions as admitted). A lane that graded + // under anything else — the old hard-coded host network, or the run's profile — would let a network-off + // unit's planted manifest install with egress, or cut a Trusted unit's legitimate download. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var runId = await SeedSupervisorRunAsync(teamId, userId); + var repoId = Guid.NewGuid(); + var agentId = Guid.NewGuid(); + var task = UnitTask(AgentAutonomyLevel.Trusted, allowHost: "registry.npmjs.org"); + await SeedUnitAgentRunAsync(teamId, agentId, task); + + if (lane == "captured") + { + await SeedPlanAsync(runId, teamId, sequence: 1, ArtifactPlanPayload("s1", new[] { "report.md" })); + await SeedSpawnAsync(runId, teamId, sequence: 2, """{"subtaskIds":["s1"]}""", SpawnOutcome(Unit(agentId, producedBranch: null))); + } + else + { + await SeedPlanAsync(runId, teamId, sequence: 1, PlanPayload(("s1", Check))); + var unit = lane switch + { + "branch" => Unit(agentId, "codespace/agent/s1"), + "patch" => Unit(agentId, producedBranch: null), + _ => Unit(agentId, "web/x") with { RepositoryResults = new[] { new RepositoryRunResult { Alias = "web", RepositoryId = Guid.NewGuid(), ProducedBranch = "web/x", BaseBranch = "main", Access = WorkspaceAccess.Write }, new RepositoryRunResult { Alias = "api", RepositoryId = Guid.NewGuid(), ProducedBranch = "api/x", BaseBranch = "main", Access = WorkspaceAccess.Write } } }, + }; + await SeedSpawnAsync(runId, teamId, sequence: 2, """{"subtaskIds":["s1"]}""", SpawnOutcome(unit)); + if (lane != "multi-repo") await SeedManifestAsync(teamId, agentId, repoId, lane == "branch" ? "codespace/agent/s1" : null, baseSha: "deadbeef", patchArtifactId: lane == "patch" ? Guid.NewGuid() : null); + } + + var grader = new RecordingGrader(new BenchmarkGrade { Passed = true, Detail = "tests-passed" }); + await RehydrateAsync(runId, teamId, lane == "captured" ? RepoLessGoalConfig() : GoalConfig(repoId), grader); + + var expectedLanes = lane switch { "branch" => new[] { "branch", "base" }, "patch" => new[] { "patch", "base" }, "multi-repo" => new[] { "branch", "branch" }, _ => new[] { "captured" } }; + grader.Postures.Select(p => p.Lane).ShouldBe(expectedLanes, "fixture check: the lane under test (and its baseline) actually graded"); + grader.Postures.ShouldAllBe(p => p.Posture != null, "no per-unit lane may drop the producing unit's posture"); + foreach (var (_, posture) in grader.Postures) ShouldBeThePostureOf(posture!, task); + } + + [Fact] + public async Task A_unit_whose_task_cannot_be_read_hands_the_grader_no_posture_so_it_grades_fail_closed() + { + // No agent_run row for this unit in this team: there is no producer to take a posture from, and the grader + // reads a null posture as network off under the Confined ceilings (pinned in AcceptanceGradingPostureTests). + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var runId = await SeedSupervisorRunAsync(teamId, userId); + var repoId = Guid.NewGuid(); + var agentId = Guid.NewGuid(); + + await SeedPlanAsync(runId, teamId, sequence: 1, PlanPayload(("s1", Check))); + await SeedSpawnAsync(runId, teamId, sequence: 2, """{"subtaskIds":["s1"]}""", SpawnOutcome(Unit(agentId, "codespace/agent/s1"))); + + var grader = new RecordingGrader(new BenchmarkGrade { Passed = true, Detail = "tests-passed" }); + await RehydrateAsync(runId, teamId, GoalConfig(repoId), grader); + + grader.Postures.ShouldHaveSingleItem().Posture.ShouldBeNull(); + } + + [Fact] + public async Task Units_of_different_tiers_off_the_same_base_do_not_share_a_baseline_measurement() + { + // The baseline runs under its candidate's posture, so a base measured with the network cannot stand in for a + // network-off sibling's differential: that would compare two sandboxes, not two trees. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var runId = await SeedSupervisorRunAsync(teamId, userId); + var repoId = Guid.NewGuid(); + var agentA = Guid.NewGuid(); + var agentB = Guid.NewGuid(); + await SeedUnitAgentRunAsync(teamId, agentA, UnitTask(AgentAutonomyLevel.Standard, allowHost: null)); + await SeedUnitAgentRunAsync(teamId, agentB, UnitTask(AgentAutonomyLevel.Trusted, allowHost: null)); + + await SeedPlanAsync(runId, teamId, sequence: 1, PlanPayload(("s1", Check), ("s2", Check))); + await SeedSpawnAsync(runId, teamId, sequence: 2, """{"subtaskIds":["s1","s2"]}""", SpawnOutcome(Unit(agentA, producedBranch: "codespace/agent/a"), Unit(agentB, producedBranch: "codespace/agent/b"))); + await SeedManifestAsync(teamId, agentA, repoId, branch: "codespace/agent/a", baseSha: "deadbeef", patchArtifactId: null); + await SeedManifestAsync(teamId, agentB, repoId, branch: "codespace/agent/b", baseSha: "deadbeef", patchArtifactId: null); + + var grader = new RecordingGrader(new BenchmarkGrade { Passed = true, Detail = "tests-passed" }); + await RehydrateAsync(runId, teamId, GoalConfig(repoId), grader); + + grader.BaseCalls.Count.ShouldBe(2, "same base and same oracle, but two postures ⇒ two measurements"); + grader.Postures.Where(p => p.Lane == "base").Select(p => p.Posture!.AllowNetwork).ShouldBe(new[] { false, true }, "each baseline ran under its own candidate's posture"); + } + + private static AgentTask UnitTask(AgentAutonomyLevel tier, string? allowHost) => new() + { + Goal = "do s1", + Harness = "codex-cli", + Autonomy = tier, + Permissions = AgentAutonomyPolicy.Derive(tier) with { Egress = allowHost is null ? AgentEgressPolicy.Full : AgentEgressPolicy.Allowlist, EgressAllowHosts = allowHost is null ? null : new[] { allowHost } }, + }; + + private async Task SeedUnitAgentRunAsync(Guid teamId, Guid agentRunId, AgentTask task) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.AgentRun.Add(new AgentRun { Id = agentRunId, TeamId = teamId, Harness = task.Harness, Status = CodeSpace.Messages.Enums.AgentRunStatus.Succeeded, TaskJson = JsonSerializer.Serialize(task, AgentJson.Options) }); + await db.SaveChangesAsync(); + } + + private static void ShouldBeThePostureOf(AcceptanceGradingPosture actual, AgentTask producer) + { + var expected = AcceptanceGradingPosturePolicy.For(producer); + + actual.Autonomy.ShouldBe(expected.Autonomy); + actual.AllowNetwork.ShouldBe(expected.AllowNetwork); + (actual.EgressAllowlist ?? Array.Empty()).ShouldBe(expected.EgressAllowlist ?? Array.Empty()); + actual.MaxMemoryMb.ShouldBe(expected.MaxMemoryMb); + actual.MaxCpuPercent.ShouldBe(expected.MaxCpuPercent); + } + // ─── C2: a REPO-LESS unit is graded against what it captured, not failed closed on "no repo" ────────── /// @@ -1633,9 +1750,13 @@ private sealed class RecordingGrader : ISupervisorAcceptanceGrader public List<(Guid RepositoryId, Guid TeamId, string BaseSha, Guid? PatchArtifactId, IReadOnlyList Command)> PatchCalls { get; } = new(); public List PatchProducerModels { get; } = new(); + /// V-B: the producing-run posture every request-form grade carried, tagged by lane, in call order. + public List<(string Lane, AcceptanceGradingPosture? Posture)> Postures { get; } = new(); + public Task GradeAsync(RepositoryAcceptanceGradeRequest request, CancellationToken cancellationToken) { RepositoryProducerModels.Add(request.ProducerModel); + Postures.Add(("branch", request.Posture)); return GradeAsync(request.RepositoryId, request.TeamId, request.Branch, request.Spec, request.TimeoutSeconds, cancellationToken); } @@ -1656,6 +1777,7 @@ public Task GradePatchAsync(Guid repositoryId, Guid teamId, stri public Task GradePatchAsync(PatchAcceptanceGradeRequest request, CancellationToken cancellationToken) { PatchProducerModels.Add(request.ProducerModel); + Postures.Add(("patch", request.Posture)); return GradePatchAsync(request.RepositoryId, request.TeamId, request.BaseSha, request.InlinePatch, request.PatchArtifactId, request.Spec, request.TimeoutSeconds, cancellationToken); } @@ -1667,6 +1789,12 @@ public Task GradePatchAsync(PatchAcceptanceGradeRequest request, /// When set, GradeBaseAsync throws — proves the candidate grade survives its own baseline's crash. public Exception? ThrowOnBase { get; set; } + public Task GradeBaseAsync(BaseAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + Postures.Add(("base", request.Posture)); + return GradeBaseAsync(request.RepositoryId, request.TeamId, request.BaseSha, request.Spec, request.TimeoutSeconds, cancellationToken); + } + public Task GradeBaseAsync(Guid repositoryId, Guid teamId, string baseSha, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) { BaseCalls.Add((repositoryId, baseSha, spec.Command)); @@ -1680,6 +1808,12 @@ public Task GradeBaseAsync(Guid repositoryId, Guid teamId, strin /// The verdict the rebuilt-world grade returns. Its default is the passing ArtifactPresent verdict a captured, correct report earns. public BenchmarkGrade CapturedGrade { get; set; } = new() { Passed = true, Detail = "artifact-present" }; + public Task GradeCapturedAsync(CapturedAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + Postures.Add(("captured", request.Posture)); + return GradeCapturedAsync(request.AgentRunId, request.TeamId, request.Spec, request.TimeoutSeconds, cancellationToken); + } + public Task GradeCapturedAsync(Guid agentRunId, Guid teamId, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) { CapturedCalls.Add((agentRunId, teamId, spec.Command, spec.Kind ?? BenchmarkGradingKind.TestsPass)); diff --git a/backend/tests/CodeSpace.SandboxTests/AcceptanceGradingPostureE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/AcceptanceGradingPostureE2ETests.cs new file mode 100644 index 000000000..4eaa97a03 --- /dev/null +++ b/backend/tests/CodeSpace.SandboxTests/AcceptanceGradingPostureE2ETests.cs @@ -0,0 +1,125 @@ +using System.Net; +using System.Net.Sockets; +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Agents.Eval.Benchmark.Graders; +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Core.Services.Workflows.Artifacts; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Agents.Benchmark; +using Microsoft.Extensions.Logging.Abstractions; +using Shouldly; + +namespace CodeSpace.SandboxTests; + +/// +/// 🟢 High fidelity: the real on the real +/// under real bubblewrap, with only the evidence store in memory. The grader runs an acceptance setup step whose +/// network comes from the producing run's posture, and the setup tries to reach a listener on the host's loopback. +/// A network-off producer's setup is refused by the kernel, while a network-granting producer's setup connects. +/// Before this, every setup step shared the host network whatever the tier: it ran manifests the agent wrote, with +/// egress the agent never had. The argv half of the same proof runs on any host (AcceptanceGradingPostureTests). +/// +[Trait("Category", "Sandbox")] +public sealed class AcceptanceGradingPostureE2ETests +{ + private const int ConnectionRefused = 7; + + [KernelTheory] + [InlineData(AgentAutonomyLevel.Confined, false)] + [InlineData(AgentAutonomyLevel.Standard, false)] + [InlineData(AgentAutonomyLevel.Trusted, true)] + [InlineData(AgentAutonomyLevel.Unleashed, true)] + public async Task A_grades_setup_reaches_a_host_listener_only_when_its_producing_run_could(AgentAutonomyLevel tier, bool reaches) + { + BubblewrapSandbox.Available.ShouldNotBeNull("the kernel suite must execute confinement, never silently degrade"); + + using var context = new GradeContext(); + var connect = $"import socket,sys\ntry:\n s=socket.create_connection(('127.0.0.1',{context.Port}),timeout=2); s.close()\nexcept OSError:\n sys.exit({ConnectionRefused})"; + var spec = new SupervisorAcceptanceSpec { Command = ["/bin/sh", "-c", "exit 0"], SetupCommand = ["/usr/bin/python3", "-c", connect] }; + var posture = AcceptanceGradingPosturePolicy.Derive(tier, AgentAutonomyPolicy.Derive(tier), AgentAutonomyLevel.Unleashed, hostMemoryBudgetMb: null); + + var grade = await context.Grader.GradeDirectoryAsync(new DirectoryAcceptanceGradeRequest { Directory = context.Workspace, Spec = spec, TeamId = Guid.NewGuid(), TimeoutSeconds = 30, Posture = posture }, CancellationToken.None); + + grade.Passed.ShouldBe(reaches, $"{tier}: {grade.Detail} — the setup's connect to 127.0.0.1:{context.Port} should {(reaches ? "succeed on the shared host network" : "be refused inside a fresh network namespace")}; diagnose with `bwrap --unshare-net python3 -c ...` by hand"); + + if (reaches) return; + + grade.Detail.ShouldStartWith(AgentAcceptanceContract.SetupSeveredDetailPrefix, customMessage: "the refused setup is an Environment failure the posture decided, never a verdict on the work"); + grade.Class.ShouldBe(GradeFailureClass.Environment); + AgentAcceptanceContract.IsInfraFailure(grade.Detail, workPresent: true).ShouldBeTrue("no revise round is spent on it"); + AgentAcceptanceContract.IsDecidedByGradePosture(grade.Detail).ShouldBeTrue("and no respawn: the same stored task severs it again"); + context.Artifacts.Texts.ShouldHaveSingleItem().ShouldStartWith(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix + "off", customMessage: "the evidence says the sandbox severed the setup, so an operator knows why it could not download"); + } + + [KernelFact] + public void A_confining_host_reports_the_severance_it_actually_enforces() + { + // The grade claims a setup was severed only when the runner says it enforces it: on this host bwrap does. + BubblewrapSandbox.Available.ShouldNotBeNull("the kernel suite must execute confinement, never silently degrade"); + + new LocalProcessRunner().EnforcedEgress(new SandboxSpec { Command = "npm", AllowNetwork = false }).ShouldBe(SandboxEgressMode.None, "bwrap gives a network-off spec a fresh, empty namespace"); + new LocalProcessRunner().EnforcedEgress(new SandboxSpec { Command = "npm", AllowNetwork = true }).ShouldBe(SandboxEgressMode.Full); + } + + /// A loopback listener on a kernel-picked port, a GUID-named workspace, and a grader over the real runner; torn down on dispose. + private sealed class GradeContext : IDisposable + { + private readonly TcpListener _listener = new(IPAddress.Loopback, 0); + + public GradeContext() + { + _listener.Start(); + Port = ((IPEndPoint)_listener.LocalEndpoint).Port; + Directory.CreateDirectory(Workspace); + Grader = new SupervisorAcceptanceGrader(null!, null!, new SandboxRunnerRegistry([new LocalProcessRunner()]), new BenchmarkGraderRegistry([new TestsPassGrader()]), null!, Artifacts, null!, NullLogger.Instance); + } + + public int Port { get; } + public string Workspace { get; } = Path.Combine(Path.GetTempPath(), "cs-grade-posture-kernel-" + Guid.NewGuid().ToString("N")); + public MemoryArtifacts Artifacts { get; } = new(); + public SupervisorAcceptanceGrader Grader { get; } + + public void Dispose() + { + _listener.Stop(); + try { Directory.Delete(Workspace, recursive: true); } catch (IOException) { } + } + } + + /// The evidence store, in memory: the grader stores each grade's evidence through it, and the test reads it back. + private sealed class MemoryArtifacts : IArtifactStore + { + public List Texts { get; } = new(); + + public Task PutAsync(Guid teamId, ReadOnlyMemory bytes, string contentType, CancellationToken cancellationToken) + { + Texts.Add(System.Text.Encoding.UTF8.GetString(bytes.Span)); + return Task.FromResult(Guid.NewGuid()); + } + + public Task GetBytesAsync(Guid teamId, Guid artifactId, CancellationToken cancellationToken) => Task.FromResult(null); + public Task GetMetadataAsync(Guid teamId, Guid artifactId, CancellationToken cancellationToken) => Task.FromResult(null); + } + + private sealed class KernelTheoryAttribute : TheoryAttribute + { + public KernelTheoryAttribute() + { + if (BubblewrapSandbox.Available is null && !BubblewrapSandbox.IsRequired) + Skip = "Requires real Linux bubblewrap; the privileged GitHub Actions lane is authoritative."; + } + } + + private sealed class KernelFactAttribute : FactAttribute + { + public KernelFactAttribute() + { + if (BubblewrapSandbox.Available is null && !BubblewrapSandbox.IsRequired) + Skip = "Requires real Linux bubblewrap; the privileged GitHub Actions lane is authoritative."; + } + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/BenchmarkTaskGradingPostureTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/BenchmarkTaskGradingPostureTests.cs new file mode 100644 index 000000000..74e4646e3 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/BenchmarkTaskGradingPostureTests.cs @@ -0,0 +1,90 @@ +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Agents.Eval.Benchmark.Graders; +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Agents.Benchmark; +using Shouldly; + +namespace CodeSpace.UnitTests.Agents.Benchmark; + +/// +/// 🟢 Unit: a benchmark or qualification grade runs the fixture's test command in the workspace the benchmark agent +/// just wrote, so a module the tests import is the agent's code. It runs under the producing agent's posture, as every +/// acceptance grade does, instead of the raw local runner with no memory or CPU ceiling. Driven through the real +/// seam both benchmark instruments grade through, with the real +/// , recording the spec the runner receives. +/// +[Trait("Category", "Unit")] +public sealed class BenchmarkTaskGradingPostureTests +{ + [Theory] + [InlineData(AgentAutonomyLevel.Confined)] + [InlineData(AgentAutonomyLevel.Standard)] + [InlineData(AgentAutonomyLevel.Trusted)] + public async Task The_benchmark_check_runs_under_the_producing_agents_ceilings(AgentAutonomyLevel tier) + { + var posture = AcceptanceGradingPosturePolicy.Derive(tier, AgentAutonomyPolicy.Derive(tier), AgentAutonomyLevel.Unleashed, hostMemoryBudgetMb: null); + var runners = new RecordingRunners(SandboxStatus.Success); + + await GradeAsync(runners, posture); + + var check = runners.Specs.ShouldHaveSingleItem("fixture check: the test command ran once"); + check.MaxMemoryMb.ShouldBe(posture.MaxMemoryMb, "the check imports the agent's code, so it is capped like the agent was"); + check.MaxCpuPercent.ShouldBe(posture.MaxCpuPercent); + check.AllowNetwork.ShouldBeFalse("the check never asked for the network — unchanged"); + posture.MaxMemoryMb.ShouldBeGreaterThan(0, "fixture check: a zero ceiling would mean unlimited and pass vacuously"); + } + + [Fact] + public async Task A_benchmark_grade_with_no_producer_posture_runs_under_the_confined_ceilings() + { + var runners = new RecordingRunners(SandboxStatus.Success); + + await GradeAsync(runners, posture: null); + + var check = runners.Specs.ShouldHaveSingleItem(); + check.MaxMemoryMb.ShouldBe(AcceptanceGradingPosturePolicy.FailClosed.MaxMemoryMb, "a grade that cannot say whose work it runs gets the least anyone could have had"); + check.MaxCpuPercent.ShouldBe(AcceptanceGradingPosturePolicy.FailClosed.MaxCpuPercent); + } + + [Fact] + public async Task A_benchmark_check_killed_at_its_ceiling_is_an_environment_fact() + { + var grade = await GradeAsync(new RecordingRunners(SandboxStatus.ResourceExhausted), AcceptanceGradingPosturePolicy.FailClosed); + + grade.Detail.ShouldBe(TestsPassGrader.ResourceExhaustedDetail); + grade.Class.ShouldBe(GradeFailureClass.Environment, "the ceiling killed it, which says nothing about whether the agent solved the task"); + } + + private static Task GradeAsync(RecordingRunners runners, AcceptanceGradingPosture? posture) => + BenchmarkTaskGrading.GradeAsync(new BenchmarkGraderRegistry(new IBenchmarkGrader[] { new TestsPassGrader() }), runners, new BenchmarkTaskGradingRequest { Task = FixtureTask(), WorkspaceDirectory = Path.GetTempPath(), Posture = posture }, CancellationToken.None); + + private static BenchmarkTask FixtureTask() => new() + { + Id = "posture", + Description = "the check runs under the producing agent's posture", + FixtureRef = "inline", + Goal = "make the check pass", + Grading = BenchmarkGradingKind.TestsPass, + TestCommand = new[] { "sh", "check.sh" }, + Harness = "codex-cli", + Modes = new[] { BenchmarkMode.HarnessCli }, + TimeoutSeconds = 30, + }; + + private sealed class RecordingRunners(SandboxStatus status) : ISandboxRunnerRegistry, ISandboxRunner + { + public List Specs { get; } = new(); + public string Kind => SandboxKinds.Local; + public IReadOnlyList All => new ISandboxRunner[] { this }; + public ISandboxRunner Resolve(string kind) => this; + + public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) + { + Specs.Add(spec); + return Task.FromResult(new SandboxResult { Status = status, ExitCode = status == SandboxStatus.Success ? 0 : 137, Stdout = "", Stderr = "" }); + } + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/TaskLaunchBenchmarkCellRunnerTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/TaskLaunchBenchmarkCellRunnerTests.cs index 083ed2ffb..b7ea1261c 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/TaskLaunchBenchmarkCellRunnerTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/Benchmark/TaskLaunchBenchmarkCellRunnerTests.cs @@ -1,6 +1,7 @@ using CodeSpace.Core.Persistence.Entities; using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Agents.Eval.Benchmark.TaskLaunch; +using CodeSpace.Core.Services.Supervisor; using CodeSpace.Messages.Agents; using CodeSpace.Messages.Agents.Benchmark; using CodeSpace.Messages.Constants; @@ -62,6 +63,27 @@ public void A_multi_branch_launch_only_claims_one_producer_wire_identity_when_ev TaskLaunchBenchmarkCellRunner.ProducerModelOf(selection, [Attempt("wire-a"), Attempt("wire-b")]).ObservedModel.ShouldBeNull("a union of work from different backing models has no single producer identity and cannot establish judge independence"); } + [Fact] + public void The_graded_workspace_is_checked_under_its_producers_posture_only_when_every_attempt_agrees_on_it() + { + var standard = Task(AgentAutonomyLevel.Standard); + var trusted = Task(AgentAutonomyLevel.Trusted); + + TaskLaunchBenchmarkCellRunner.GradedPosture([WithTask(standard), WithTask(standard)]).ShouldBe(AcceptanceGradingPosturePolicy.For(standard), "one producer posture — the check runs under it"); + TaskLaunchBenchmarkCellRunner.GradedPosture([WithTask(standard), WithTask(trusted)]).ShouldBeNull("a union of work from different postures has no single one to run it under — the grade fails closed"); + TaskLaunchBenchmarkCellRunner.GradedPosture([WithTask(standard), WithTaskJson("not json")]).ShouldBeNull("an attempt whose task cannot be read says nothing about what it was allowed — fail closed"); + + static AgentTask Task(AgentAutonomyLevel tier) => new() { Goal = "g", Harness = "test", Autonomy = tier, Permissions = AgentAutonomyPolicy.Derive(tier) }; + static AgentRun WithTask(AgentTask task) => WithTaskJson(System.Text.Json.JsonSerializer.Serialize(task, AgentJson.Options)); + + static AgentRun WithTaskJson(string taskJson) + { + var attempt = Attempt(null); + attempt.TaskJson = taskJson; + return attempt; + } + } + private static AgentRun Attempt(string? model) => new() { Id = Guid.NewGuid(), diff --git a/backend/tests/CodeSpace.UnitTests/Agents/FakeAcceptanceGrader.cs b/backend/tests/CodeSpace.UnitTests/Agents/FakeAcceptanceGrader.cs index 24628ab21..20b00f510 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/FakeAcceptanceGrader.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/FakeAcceptanceGrader.cs @@ -23,6 +23,15 @@ internal sealed class FakeAcceptanceGrader : ISupervisorAcceptanceGrader public int PatchCallCount { get; private set; } public (Guid RepositoryId, Guid TeamId, string BaseSha, Guid? PatchArtifactId)? LastPatchCall { get; private set; } + /// The producing-run posture every request-form repository grade carried, in call order. + public List Postures { get; } = new(); + + public Task GradeAsync(RepositoryAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + Postures.Add(request.Posture); + return GradeAsync(request.RepositoryId, request.TeamId, request.Branch, request.Spec, request.TimeoutSeconds, cancellationToken); + } + public Task GradeAsync(Guid repositoryId, Guid teamId, string branch, SupervisorAcceptanceSpec spec, int timeoutSeconds, CancellationToken cancellationToken) { CallCount++; diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorAcceptanceGraderTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorAcceptanceGraderTests.cs index 4273196a6..10c8138cd 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorAcceptanceGraderTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorAcceptanceGraderTests.cs @@ -1029,9 +1029,12 @@ public void Evaluator_version_constant_pinned() { // The literal is the wire value on durable receipts — a rename/bump is a re-qualification decision, not // an invisible refactor. Bump in the SAME PR as any grading-semantics change. + // v9: the setup and the check run under the PRODUCING run's posture (network, egress allowlist, memory/cpu + // ceilings), narrow-only — a request with none grades network-off under the Confined ceilings; a check killed at + // that ceiling is an Environment fact, and a setup the sandbox severed is decided by the posture. // v8: every grade step runs under a bounded window — a non-positive authored timeout grades at the default // instead of arming no wall clock, and a longer one is capped at SupervisorLane.MaxAcceptanceGradeTimeoutSeconds. - SupervisorAcceptanceGrader.EvaluatorVersion.ShouldBe("supervisor-acceptance/v8"); + SupervisorAcceptanceGrader.EvaluatorVersion.ShouldBe("supervisor-acceptance/v9"); } [Fact] diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBranchlessStopGradeTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBranchlessStopGradeTests.cs index daddc68c1..42281316d 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBranchlessStopGradeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorBranchlessStopGradeTests.cs @@ -49,6 +49,24 @@ public async Task A_branchless_delayed_grade_uses_the_folded_units_observed_prod grader.ProducerModels.ShouldBe(new[] { producer }); } + [Theory] + [InlineData(AgentAutonomyLevel.Standard)] + [InlineData(AgentAutonomyLevel.Trusted)] + public async Task A_branchless_gate_grades_each_units_world_under_that_units_stored_posture(AgentAutonomyLevel tier) + { + var db = Infrastructure.EmptyTestDb.New(); + var task = new AgentTask { Goal = "write the report", Harness = "codex-cli", Autonomy = tier, Permissions = AgentAutonomyPolicy.Derive(tier) }; + db.AgentRun.Add(new CodeSpace.Core.Persistence.Entities.AgentRun { Id = UnitId, TeamId = TeamId, Harness = task.Harness, Status = AgentRunStatus.Succeeded, TaskJson = JsonSerializer.Serialize(task, AgentJson.Options) }); + await db.SaveChangesAsync(); + var grader = new CapturingGrader(new BenchmarkGrade { Passed = true, Detail = "artifact-present" }); + + await GradeAsync(grader, StopWith(BenchmarkGradingKind.ArtifactPresent), db: db); + + var posture = grader.Postures.ShouldHaveSingleItem().ShouldNotBeNull("the captured world is the unit's bytes, graded under the unit's own posture"); + posture.Autonomy.ShouldBe(AcceptanceGradingPosturePolicy.For(task).Autonomy); + posture.AllowNetwork.ShouldBe(AcceptanceGradingPosturePolicy.For(task).AllowNetwork); + } + [Fact] public async Task A_failing_deliverable_kind_stop_records_the_failure_named_by_its_gate() { @@ -225,12 +243,12 @@ public async Task A_missing_rubric_judge_is_not_marked_judgedSummary() private static readonly AcceptanceRubric WithRubric = new() { Criteria = new[] { new AcceptanceRubricCriterion { Id = "sources", Requirement = "names at least one source" } } }; /// Drive the stop grade directly with a branchless context (the fake resolver finds no published branch when the tape carries none). - private static async Task GradeAsync(CapturingGrader grader, string stopPayloadJson, StubRubricJudge? rubricJudge = null, IReadOnlyList? acceptanceChecks = null, ProducerFixture? producers = null) + private static async Task GradeAsync(CapturingGrader grader, string stopPayloadJson, StubRubricJudge? rubricJudge = null, IReadOnlyList? acceptanceChecks = null, ProducerFixture? producers = null, CodeSpace.Core.Persistence.Db.CodeSpaceDbContext? db = null) { // Only the stop-grade path's own collaborators are real here: the grader under test, the branch resolver (the // source of the branchless world), the manifest store the oracle anchor reads, and the budget ledger the call // scope carries. Every other seam is untouched by ApplyStopAcceptanceGradeAsync. - var service = new SupervisorTurnService(null!, null!, null!, db: Infrastructure.EmptyTestDb.New(), grader, null!, null!, null!, null!, + var service = new SupervisorTurnService(null!, null!, null!, db: db ?? Infrastructure.EmptyTestDb.New(), grader, null!, null!, null!, null!, null!, null!, new NoManifests(), new FakeSupervisorPublishedBranchResolver(), null!, new AdmitAllBudgetLedger(), null!, null!, NullLogger.Instance, rubricJudge); @@ -303,9 +321,12 @@ private sealed class CapturingGrader : ISupervisorAcceptanceGrader public List CapturedCalls { get; } = new(); public List ProducerModels { get; } = new(); + public List Postures { get; } = new(); + public Task GradeCapturedAsync(CapturedAcceptanceGradeRequest request, CancellationToken cancellationToken) { ProducerModels.Add(request.ProducerModel); + Postures.Add(request.Posture); return GradeCapturedAsync(request.AgentRunId, request.TeamId, request.Spec, request.TimeoutSeconds, cancellationToken); } diff --git a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorTurnServiceTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorTurnServiceTests.cs index 46c00afcc..5da8cb4a4 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/SupervisorTurnServiceTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/SupervisorTurnServiceTests.cs @@ -728,6 +728,29 @@ public async Task A_full_turn_stop_with_a_model_acceptance_that_FAILS_grades_onc SupervisorOutcome.ReadAcceptanceGradePassed(StopRowOutcome(ledger)).ShouldBe(false, "the verdict is folded durably onto the stop row"); } + [Theory] + [InlineData(null, AgentAutonomyLevel.Standard)] // no profile tier → the spawn default every unit got + [InlineData("Trusted", AgentAutonomyLevel.Trusted)] + [InlineData("confined", AgentAutonomyLevel.Confined)] + public async Task A_stop_grades_its_integrated_head_under_the_runs_own_autonomy_grant(string? profileTier, AgentAutonomyLevel expectedTier) + { + // The head mixes every unit's work, so its grade runs the tier every unit was clamped to — never the host + // network a hard-coded setup step used to get, and never less than the run's most capable unit had. + var ledger = SeedRunWithCleanMerge(); + var grader = new FakeAcceptanceGrader(new BenchmarkGrade { Passed = true, Detail = "tests-passed" }); + var service = ServiceWith(ledger, new StopWithAcceptanceDecider("npm", "test"), grader); + var config = GoalConfigWithRepo() with { AgentProfile = GoalConfigWithRepo().AgentProfile! with { AutonomyLevel = profileTier } }; + + await service.RunTurnAsync(_runId, _teamId, "sup", "goal", null, config, CancellationToken.None); + + var tier = AgentAutonomyPolicy.Clamp(expectedTier, AgentAutonomyPolicy.DeploymentCeiling); + var expected = AcceptanceGradingPosturePolicy.For(tier, AgentAutonomyPolicy.Derive(tier)); + var posture = grader.Postures.ShouldHaveSingleItem("the stop's model gate graded once, through the posture-carrying request").ShouldNotBeNull(); + posture.Autonomy.ShouldBe(expected.Autonomy); + posture.AllowNetwork.ShouldBe(expected.AllowNetwork); + posture.MaxMemoryMb.ShouldBe(expected.MaxMemoryMb); + } + [Fact] public async Task A_full_turn_stop_with_a_model_acceptance_that_PASSES_reports_completed_and_surfaces_the_branch() { diff --git a/backend/tests/CodeSpace.UnitTests/Architecture/AcceptanceGradePostureInventoryTests.cs b/backend/tests/CodeSpace.UnitTests/Architecture/AcceptanceGradePostureInventoryTests.cs new file mode 100644 index 000000000..e5bb35c2e --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Architecture/AcceptanceGradePostureInventoryTests.cs @@ -0,0 +1,153 @@ +using System.Text.RegularExpressions; +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Messages.Agents; +using Shouldly; + +namespace CodeSpace.UnitTests.Architecture; + +/// +/// Fail-closed floor for the lane that forgets the producing run's posture. Only a REQUEST overload of +/// carries an , and the request records +/// make it required. The positional overloads carry none, and the real grader runs them with network off. That +/// is safe, but it is not the producing run's posture, so a Trusted run's setup could no longer download. So every +/// production call site grades through a request. +/// +/// The per-file COUNTS are pinned as well. A new grade lane must change this list, and the reviewer then asks +/// whose posture it passes. The lanes' own tests pin that the posture each one passes is the right producer's. +/// +[Trait("Category", "Unit")] +public sealed class AcceptanceGradePostureInventoryTests +{ + private const string ScannedRoot = "backend/src"; + + /// Where the grader's members are declared and implemented. Its positional overloads forward to its request overloads there, with no posture: that is their documented fail-closed meaning, not a lane. + private static readonly string[] Declarations = + [ + "CodeSpace.Core/Services/Supervisor/ISupervisorAcceptanceGrader.cs", + "CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGrader.cs", + ]; + + /// Every production file that grades an acceptance contract, with how many grade calls it makes, all of them through a posture-carrying request. + private static readonly IReadOnlyDictionary GradeLanes = new Dictionary(StringComparer.Ordinal) + { + ["CodeSpace.Core/Services/Agents/AgentRunExecutor.cs"] = 3, // branch, patch, multi-repo + ["CodeSpace.Core/Services/Agents/Workspace/LocalAcceptanceVerifier.cs"] = 1, // the repo-less live workspace + ["CodeSpace.Core/Services/Supervisor/SupervisorTurnService.Rehydrate.cs"] = 9, // resolve branch + patch, unit branch + patch + captured + base + multi-repo, branchless stop, stop targets + }; + + /// A grade-named call whose receiver is NOT the acceptance grader: the executor's hand-off to the local verifier, which derives the posture itself. + private const string LocalVerifierHandOff = "GetRequiredService().GradeAsync("; + + private static readonly Regex GradeCall = new(@"\.(GradeAsync|GradePatchAsync|GradeBaseAsync|GradeCapturedAsync|GradeDirectoryAsync)\((?[^)]{0,80})", RegexOptions.Compiled); + + private static readonly Regex PostureRequest = new(@"^new (Repository|Patch|Base|Captured|Directory)AcceptanceGradeRequest\b", RegexOptions.Compiled); + + [Fact] + public void Every_production_grade_passes_a_posture_carrying_request() + { + var calls = GradeCalls(); + + calls.ShouldNotBeEmpty("the scan found no grade call at all — every check here would pass vacuously"); + + var offenders = calls.Where(call => !PostureRequest.IsMatch(call.Argument)).Select(call => $"{call.File}:{call.Line} — {call.Argument}").ToList(); + + offenders.ShouldBeEmpty( + "these grade an acceptance contract through a POSITIONAL overload, which carries no producing-run posture " + + "and so grades with network off whatever the run had. Pass a request record with Posture = " + + $"{nameof(AcceptanceGradingPosturePolicy)}.{nameof(AcceptanceGradingPosturePolicy.For)}(the producing run's task):\n " + string.Join("\n ", offenders)); + } + + [Fact] + public void The_set_of_grade_lanes_is_pinned() + { + var found = GradeCalls().GroupBy(call => call.File).ToDictionary(group => group.Key, group => group.Count(), StringComparer.Ordinal); + + found.OrderBy(pair => pair.Key, StringComparer.Ordinal).ShouldBe(GradeLanes.OrderBy(pair => pair.Key, StringComparer.Ordinal), + customMessage: "the acceptance grade lanes changed. A new lane belongs in GradeLanes, after its test proves it passes the posture " + + "of the run whose bytes it executes (not the posture of whoever happens to call it)."); + } + + /// + /// Every production file that hands an oracle a runner, with how many grading contexts it builds. The benchmark and + /// qualification instruments never go through , so the scan above cannot + /// see them, yet they run the same agent-written bytes: each one binds its runner to a posture. + /// + private static readonly IReadOnlyDictionary GradingContextLanes = new Dictionary(StringComparer.Ordinal) + { + ["CodeSpace.Core/Services/Agents/Eval/Benchmark/BenchmarkTaskGrading.cs"] = 1, // BenchmarkRunner + TaskLaunchBenchmarkCellRunner + ["CodeSpace.Core/Services/Supervisor/SupervisorAcceptanceGrader.cs"] = 1, // every acceptance lane above + }; + + private const string GradingContextDeclaration = "CodeSpace.Core/Services/Agents/Eval/Benchmark/IBenchmarkGrader.cs"; + + private static readonly Regex GradingContextConstruction = new(@"new BenchmarkGradingContext\b|BenchmarkGradingContext\.For(Acceptance|Command)\(", RegexOptions.Compiled); + + [Fact] + public void Every_oracle_runner_is_bound_to_a_producing_run_posture() + { + var found = new Dictionary(StringComparer.Ordinal); + var unbound = new List(); + + foreach (var (relative, text) in ProductionSources()) + { + if (relative == GradingContextDeclaration) continue; + + var count = text.Split('\n').Count(line => !line.TrimStart().StartsWith("//", StringComparison.Ordinal) && !line.TrimStart().StartsWith('*') && GradingContextConstruction.IsMatch(line)); + + if (count == 0) continue; + + found[relative] = count; + + if (!text.Contains($"{nameof(AcceptanceGradingPosturePolicy)}.{nameof(AcceptanceGradingPosturePolicy.Bind)}(", StringComparison.Ordinal)) unbound.Add(relative); + } + + found.OrderBy(pair => pair.Key, StringComparer.Ordinal).ShouldBe(GradingContextLanes.OrderBy(pair => pair.Key, StringComparer.Ordinal), + customMessage: "the places that hand an oracle a runner changed. A new one belongs in GradingContextLanes, after its test proves the runner it hands over is bound to the producing run's posture."); + unbound.ShouldBeEmpty($"these hand an oracle a raw runner, so agent-written code they execute runs with no ceilings; bind it with {nameof(AcceptanceGradingPosturePolicy)}.{nameof(AcceptanceGradingPosturePolicy.Bind)}"); + } + + private static IEnumerable<(string Relative, string Text)> ProductionSources() + { + var root = Path.Combine(RepositoryRoot(), ScannedRoot); + + return Directory.EnumerateFiles(root, "*.cs", SearchOption.AllDirectories).Select(file => (Path.GetRelativePath(root, file).Replace(Path.DirectorySeparatorChar, '/'), File.ReadAllText(file))); + } + + /// Every grade call in a production file that holds the acceptance grader, minus the declarations and the local-verifier hand-off. Comment lines are text about a call, not a call. + private static List<(string File, int Line, string Argument)> GradeCalls() + { + var calls = new List<(string, int, string)>(); + + foreach (var file in Directory.EnumerateFiles(Path.Combine(RepositoryRoot(), ScannedRoot), "*.cs", SearchOption.AllDirectories)) + { + var relative = Path.GetRelativePath(Path.Combine(RepositoryRoot(), ScannedRoot), file).Replace(Path.DirectorySeparatorChar, '/'); + var text = File.ReadAllText(file); + + if (Declarations.Contains(relative) || !HoldsTheGrader(text)) continue; + + var lines = text.Split('\n'); + + for (var i = 0; i < lines.Length; i++) + { + var line = lines[i].TrimStart(); + + if (line.StartsWith("//", StringComparison.Ordinal) || line.StartsWith('*') || line.Contains(LocalVerifierHandOff, StringComparison.Ordinal)) continue; + + foreach (Match match in GradeCall.Matches(line)) calls.Add((relative, i + 1, match.Groups["argument"].Value.TrimStart())); + } + } + + return calls; + } + + private static bool HoldsTheGrader(string text) => text.Contains(nameof(ISupervisorAcceptanceGrader), StringComparison.Ordinal) || text.Contains("_acceptanceGrader.", StringComparison.Ordinal); + + private static string RepositoryRoot() + { + var dir = new DirectoryInfo(AppContext.BaseDirectory); + + while (dir is not null && !Directory.Exists(Path.Combine(dir.FullName, "backend"))) dir = dir.Parent; + + return dir?.FullName ?? throw new InvalidOperationException($"repository root not found walking up from {AppContext.BaseDirectory}"); + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Supervisor/AcceptanceGradingPostureTests.cs b/backend/tests/CodeSpace.UnitTests/Supervisor/AcceptanceGradingPostureTests.cs new file mode 100644 index 000000000..d34fcd81f --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Supervisor/AcceptanceGradingPostureTests.cs @@ -0,0 +1,410 @@ +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Agents.Eval.Benchmark.Graders; +using CodeSpace.Core.Services.Agents.Sandbox; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using CodeSpace.Core.Services.Supervisor; +using CodeSpace.Core.Settings; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Agents.Benchmark; +using CodeSpace.UnitTests.Agents; +using Microsoft.Extensions.Configuration; +using Microsoft.Extensions.Logging.Abstractions; +using Shouldly; + +namespace CodeSpace.UnitTests.Supervisor; + +/// +/// 🟢 Unit: an acceptance grade runs the PRODUCING run's posture. Its setup step and its check execute bytes the agent +/// wrote, after the agent's own sandbox is gone, so they may never reach further than that agent could. Before this, +/// the setup step was hard-coded to the host network and neither step had a memory or CPU ceiling, whatever the tier. +/// +/// Three layers, each through production code. First, the derivation table (tier × egress allowlist × +/// deployment ceiling). Second, the real with the real +/// , recording the exact specs it hands its runner. Third, those recorded specs fed +/// through on a stand-in bwrap path: the argv a confining Linux host +/// would launch, provable on any host. The real kernel's answer is the sandbox lane's +/// (AcceptanceGradingPostureE2ETests). +/// +[Trait("Category", "Unit")] +public sealed class AcceptanceGradingPostureTests +{ + private const string Bwrap = "/usr/bin/bwrap"; + private static readonly string[] SetupCommand = { "npm", "ci" }; + private static readonly string[] CheckCommand = { "npm", "test" }; + + // ── The derivation table ────────────────────────────────────────────────────────────────────────────── + + [Theory] + // tier network egress extra hosts deployment ceiling → network allowlist memory cpu + [InlineData(AgentAutonomyLevel.Confined, AgentNetworkAccess.Off, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Unleashed, false, null, 1024, 100)] + [InlineData(AgentAutonomyLevel.Standard, AgentNetworkAccess.Off, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Unleashed, false, null, 4096, 400)] + [InlineData(AgentAutonomyLevel.Trusted, AgentNetworkAccess.On, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Unleashed, true, null, 6144, 400)] + [InlineData(AgentAutonomyLevel.Unleashed, AgentNetworkAccess.On, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Unleashed, true, null, 6144, 400)] + [InlineData(AgentAutonomyLevel.Trusted, AgentNetworkAccess.On, AgentEgressPolicy.Allowlist, "Registry.NPMjs.org ", AgentAutonomyLevel.Unleashed, true, "registry.npmjs.org", 6144, 400)] // narrowed to the operator's hosts, normalized + [InlineData(AgentAutonomyLevel.Trusted, AgentNetworkAccess.On, AgentEgressPolicy.Allowlist, null, AgentAutonomyLevel.Unleashed, false, null, 6144, 400)] // an allowlist with no host severs — never full egress + [InlineData(AgentAutonomyLevel.Trusted, AgentNetworkAccess.On, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Standard, false, null, 4096, 400)] // the deployment ceiling clamps tier AND network + [InlineData(AgentAutonomyLevel.Unleashed, AgentNetworkAccess.On, AgentEgressPolicy.Allowlist, "pypi.org", AgentAutonomyLevel.Standard, false, null, 4096, 400)] + [InlineData(AgentAutonomyLevel.Unleashed, AgentNetworkAccess.On, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Trusted, true, null, 6144, 400)] // a ceiling that grants network clamps nothing + [InlineData(AgentAutonomyLevel.Trusted, AgentNetworkAccess.On, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Confined, false, null, 1024, 100)] + [InlineData(AgentAutonomyLevel.Trusted, AgentNetworkAccess.Off, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Unleashed, false, null, 6144, 400)] // the run's own grant, not its tier's maximum + [InlineData(AgentAutonomyLevel.Standard, AgentNetworkAccess.On, AgentEgressPolicy.Full, null, AgentAutonomyLevel.Unleashed, false, null, 4096, 400)] // a grant its tier cannot hold is not honoured + public void The_posture_is_the_producers_own_grant_clamped_by_the_deployment_ceiling(AgentAutonomyLevel tier, AgentNetworkAccess network, AgentEgressPolicy egress, string? extraHosts, AgentAutonomyLevel deploymentCeiling, bool expectedNetwork, string? expectedAllowlist, int expectedMemoryMb, int expectedCpuPercent) + { + var permissions = new AgentPermissions { Network = network, Egress = egress, EgressAllowHosts = extraHosts is null ? null : new[] { extraHosts } }; + + var posture = AcceptanceGradingPosturePolicy.Derive(tier, permissions, deploymentCeiling, hostMemoryBudgetMb: null); + + posture.AllowNetwork.ShouldBe(expectedNetwork); + posture.EgressAllowlist.ShouldBe(expectedAllowlist is null ? null : new[] { expectedAllowlist }); + posture.MaxMemoryMb.ShouldBe(expectedMemoryMb, "the clamped tier's committed memory row"); + posture.MaxCpuPercent.ShouldBe(expectedCpuPercent, "the clamped tier's committed cpu row"); + posture.Autonomy.ShouldBe(AgentAutonomyPolicy.Clamp(tier, deploymentCeiling)); + } + + [Fact] + public void The_host_memory_budget_narrows_the_grade_as_it_narrows_the_agent() => + AcceptanceGradingPosturePolicy.Derive(AgentAutonomyLevel.Trusted, AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Trusted), AgentAutonomyLevel.Unleashed, hostMemoryBudgetMb: 2048).MaxMemoryMb.ShouldBe(2048); + + [Theory] + // Sandbox:MaxAutonomy Sandbox:AgentMemoryCeilingMb → tier network memory + [InlineData("Standard", "2048", AgentAutonomyLevel.Standard, false, 2048)] // both settings narrow the Trusted producer + [InlineData(null, null, AgentAutonomyLevel.Trusted, true, 6144)] // unset: the producer's own row, unnarrowed + public void The_production_posture_reads_the_deployment_ceiling_and_the_host_memory_budget_from_configuration(string? maxAutonomy, string? memoryCeilingMb, AgentAutonomyLevel expectedTier, bool expectedNetwork, int expectedMemoryMb) + { + // For(task) is the only place either setting reaches a grade. Every lane test computes its expected posture + // through For() itself, and the table above calls Derive with explicit arguments, so only this test sees + // whether For() actually reads the operator's configuration. + var producer = new AgentTask { Goal = "g", Harness = "test", Autonomy = AgentAutonomyLevel.Trusted, Permissions = AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Trusted) }; + var configuration = new Microsoft.Extensions.Configuration.ConfigurationBuilder().AddInMemoryCollection(new Dictionary + { + [RuntimeSettings.MaxAutonomyKey] = maxAutonomy, + [RuntimeSettings.AgentMemoryCeilingMbKey] = memoryCeilingMb, + }).Build(); + + AcceptanceGradingPosture posture; + using (RuntimeSettings.Override(RuntimeSettings.Read(configuration))) posture = AcceptanceGradingPosturePolicy.For(producer); + + posture.Autonomy.ShouldBe(expectedTier, "Sandbox:MaxAutonomy clamps the grade as it clamps the agent"); + posture.AllowNetwork.ShouldBe(expectedNetwork, "a clamped tier never derives network its ceiling denies"); + posture.MaxMemoryMb.ShouldBe(expectedMemoryMb, "Sandbox:AgentMemoryCeilingMb narrows the grade's memory ceiling"); + } + + [Fact] + public void A_tier_the_policy_does_not_know_grades_as_confined() + { + var posture = AcceptanceGradingPosturePolicy.Derive((AgentAutonomyLevel)99, new AgentPermissions { Network = AgentNetworkAccess.On }, AgentAutonomyLevel.Unleashed, hostMemoryBudgetMb: null); + + posture.Autonomy.ShouldBe(AgentAutonomyLevel.Confined); + posture.AllowNetwork.ShouldBeFalse(); + posture.MaxMemoryMb.ShouldBe(1024); + } + + [Fact] + public void A_grade_with_no_producer_is_network_off_under_the_confined_ceilings() + { + var posture = AcceptanceGradingPosturePolicy.FailClosed; + var confined = AgentAutonomyPolicy.Ceilings(AgentAutonomyLevel.Confined, RuntimeSettings.Current.AgentMemoryCeilingMb); + + posture.AllowNetwork.ShouldBeFalse(); + posture.EgressAllowlist.ShouldBeNull(); + posture.MaxMemoryMb.ShouldBe(confined.MemoryMb); + posture.MaxCpuPercent.ShouldBe(confined.CpuPercent); + } + + [Fact] + public void The_narrowed_setup_notice_prefix_is_pinned() => + // Rule 8: an operator whose setup stopped downloading on a network-off run searches the evidence for this. + AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix.ShouldBe("setup network: "); + + [Fact] + public void The_grade_details_the_retry_verdicts_read_are_pinned() + { + // Rule 8: the grader writes these and the infra classifier and the agent.code retry verdict read them, across + // a durable resume payload. A rename on one side alone silently flips which failures re-bill an agent run. + TestsPassGrader.ResourceExhaustedDetail.ShouldBe("tests-resource-exhausted"); + AgentAcceptanceContract.SetupSeveredDetailPrefix.ShouldBe("setup-failed-network-severed:"); + } + + [Theory] + // producer tier allowlist host egress the runner actually enforced → notice + [InlineData(AgentAutonomyLevel.Standard, null, SandboxEgressMode.None, "setup network: off, as the producing run had it (Standard) — the sandbox severed it; a setup that downloads needs an egress allowlist or a higher tier")] + [InlineData(AgentAutonomyLevel.Trusted, "registry.npmjs.org", SandboxEgressMode.Filtered, "setup network: narrowed to the producing run's egress allowlist (registry.npmjs.org) — the sandbox filtered it")] + [InlineData(AgentAutonomyLevel.Trusted, "registry.npmjs.org", SandboxEgressMode.None, "setup network: off — this host cannot filter to the producing run's egress allowlist (registry.npmjs.org), so the sandbox severed it; a setup that downloads needs a host that filters egress, or a higher tier")] + [InlineData(AgentAutonomyLevel.Standard, null, SandboxEgressMode.Full, null)] // an unconfined host: the setup kept the network, so the grade claims nothing + [InlineData(AgentAutonomyLevel.Standard, null, null, null)] // a runner that cannot say what it enforces: no claim either + [InlineData(AgentAutonomyLevel.Trusted, null, SandboxEgressMode.Full, null)] + public void The_notice_says_what_the_sandbox_did_to_the_setups_network_and_is_silent_when_it_did_nothing(AgentAutonomyLevel tier, string? allowHost, SandboxEgressMode? enforced, string? expected) + { + var permissions = AgentAutonomyPolicy.Derive(tier) with { Egress = allowHost is null ? AgentEgressPolicy.Full : AgentEgressPolicy.Allowlist, EgressAllowHosts = allowHost is null ? null : new[] { allowHost } }; + + AcceptanceGradingPosturePolicy.SetupNetworkNotice(AcceptanceGradingPosturePolicy.Derive(tier, permissions, AgentAutonomyLevel.Unleashed, null), enforced).ShouldBe(expected); + } + + [Theory] + // asks network allowlist confines filters allowlist → enforced + [InlineData(false, null, true, false, SandboxEgressMode.None)] // bwrap's --unshare-net + [InlineData(false, null, false, false, SandboxEgressMode.Full)] // nothing on this host severs it + [InlineData(true, null, true, false, SandboxEgressMode.Full)] + [InlineData(true, "pypi.org", true, true, SandboxEgressMode.Filtered)] // the per-run filtered namespace + [InlineData(true, "pypi.org", false, true, SandboxEgressMode.Filtered)] // the filtered namespace needs no bwrap + [InlineData(true, "pypi.org", true, false, SandboxEgressMode.None)] // an allowlist bwrap cannot filter fails closed to severed + [InlineData(true, "pypi.org", false, false, SandboxEgressMode.Full)] // and nothing enforces it where nothing confines + public void The_runner_reports_the_egress_it_actually_enforces_not_the_one_the_spec_asks_for(bool allowNetwork, string? allowHost, bool confines, bool filtersAllowlist, SandboxEgressMode expected) + { + var spec = new SandboxSpec { Command = "npm", AllowNetwork = allowNetwork, EgressAllowlist = allowHost is null ? null : new[] { allowHost } }; + + LocalProcessRunner.EnforcedEgress(spec, confines, filtersAllowlist).ShouldBe(expected); + } + + // ── The grader's choke point, proved down to the argv ───────────────────────────────────────────────── + + [Theory] + [InlineData(AgentAutonomyLevel.Confined)] + [InlineData(AgentAutonomyLevel.Standard)] + public async Task A_network_off_producers_setup_and_check_are_both_severed(AgentAutonomyLevel tier) + { + var posture = Posture(tier); + var (setup, check) = await GradeAsync(posture); + + setup.AllowNetwork.ShouldBeFalse("the setup asked for the network, and the producer had none to give it"); + Argv(setup).ShouldContain("--unshare-net", customMessage: "under bubblewrap the setup gets a fresh, empty network namespace — it cannot reach a host the agent could not"); + Argv(check).ShouldContain("--unshare-net", customMessage: "the check's default network cut is unchanged"); + } + + [Theory] + [InlineData(AgentAutonomyLevel.Trusted)] + [InlineData(AgentAutonomyLevel.Unleashed)] + public async Task A_network_granting_producers_setup_keeps_the_host_network_and_its_check_stays_severed(AgentAutonomyLevel tier) + { + var (setup, check) = await GradeAsync(Posture(tier)); + + setup.AllowNetwork.ShouldBeTrue(); + setup.EgressAllowlist.ShouldBeNull(); + Argv(setup).ShouldNotContain("--unshare-net", customMessage: "a Trusted/Unleashed producer's setup still downloads — the clamp only narrows"); + Argv(check).ShouldContain("--unshare-net", customMessage: "the check never asked for the network, so the posture grants it none"); + } + + [Fact] + public async Task An_allowlist_producers_setup_is_filtered_to_its_hosts_and_never_shares_the_host_network() + { + var permissions = AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Trusted) with { Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = new[] { "registry.npmjs.org" } }; + var (setup, check) = await GradeAsync(AcceptanceGradingPosturePolicy.Derive(AgentAutonomyLevel.Trusted, permissions, AgentAutonomyLevel.Unleashed, null)); + + var policy = LocalProcessRunner.EgressPolicyFor(setup, filtersAllowlist: true); + policy.Mode.ShouldBe(SandboxEgressMode.Filtered, "a host that filters puts the setup in a per-run netns that reaches only the allowlist"); + policy.AllowedHosts.ShouldBe(new[] { "registry.npmjs.org" }, "the producer's operator-configured hosts, and nothing the grade does not need (no model API host, no git host)"); + + string[] netns = { "ip", "netns", "exec", "cs-egr-deadbeef" }; + var filtered = Chain(setup, netns); + filtered.Take(netns.Length).ShouldBe(netns, "the whole chain enters the filtered namespace"); + filtered.ShouldNotContain("--unshare-net", "bwrap inherits the filtered namespace rather than re-unsharing it away"); + + Argv(setup).ShouldContain("--unshare-net", customMessage: "a confining host that cannot filter severs the setup — an allowlist never degrades to the host network"); + Argv(check).ShouldContain("--unshare-net"); + } + + [Fact] + public async Task A_grade_with_no_posture_is_severed_under_the_confined_ceilings() + { + var (setup, check) = await GradeAsync(posture: null); + var confined = AcceptanceGradingPosturePolicy.FailClosed; + + Argv(setup).ShouldContain("--unshare-net", customMessage: "missing posture fails closed — never the hard-coded host network this replaced"); + setup.MaxMemoryMb.ShouldBe(confined.MaxMemoryMb); + check.MaxMemoryMb.ShouldBe(confined.MaxMemoryMb); + check.MaxCpuPercent.ShouldBe(confined.MaxCpuPercent); + } + + [Theory] + [InlineData(AgentAutonomyLevel.Confined)] + [InlineData(AgentAutonomyLevel.Standard)] + [InlineData(AgentAutonomyLevel.Trusted)] + public async Task The_setup_and_the_check_both_run_under_the_producers_resource_ceilings(AgentAutonomyLevel tier) + { + var posture = Posture(tier); + var (setup, check) = await GradeAsync(posture); + + setup.MaxMemoryMb.ShouldBe(posture.MaxMemoryMb, "an agent-planted install script is capped like the agent was"); + setup.MaxCpuPercent.ShouldBe(posture.MaxCpuPercent); + check.MaxMemoryMb.ShouldBe(posture.MaxMemoryMb, "so is the check that imports the agent's code"); + check.MaxCpuPercent.ShouldBe(posture.MaxCpuPercent); + posture.MaxMemoryMb.ShouldBeGreaterThan(0, "fixture check: a zero ceiling would mean unlimited and pass vacuously"); + } + + // ── The notice ──────────────────────────────────────────────────────────────────────────────────────── + + [Fact] + public async Task A_network_narrowed_setup_records_one_notice_at_the_head_of_the_grades_evidence() + { + var artifacts = new SupervisorAcceptanceGraderTests.FakeArtifactStore(); + + await GradeAsync(Posture(AgentAutonomyLevel.Standard), artifacts: artifacts); + + var evidence = artifacts.Puts.ShouldHaveSingleItem("the check's own evidence carries the notice").Text; + evidence.ShouldStartWith(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix + "off"); + evidence.Split(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix).Length.ShouldBe(2, "one notice per grade"); + } + + [Fact] + public async Task A_setup_that_kept_the_network_records_no_notice() + { + var artifacts = new SupervisorAcceptanceGraderTests.FakeArtifactStore(); + + await GradeAsync(Posture(AgentAutonomyLevel.Trusted), artifacts: artifacts); + + artifacts.Puts.ShouldHaveSingleItem().Text.ShouldNotContain(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix); + } + + [Theory] + [InlineData(SandboxStatus.Failed, "setup-failed-network-severed: npm ERR! network request failed")] + [InlineData(SandboxStatus.TimedOut, "setup-failed-network-severed: timed out")] + public async Task A_setup_the_sandbox_severed_that_fails_is_decided_by_the_posture_and_carries_the_notice_as_evidence(SandboxStatus setupStatus, string expectedDetail) + { + // The producer's posture is read off the same stored task on every attempt, so a severed setup fails the same + // way on every respawn. The detail says so, which is the bit an authored agent.code retry reads: infra (never a + // code verdict, never a revise round) AND decided by the grade's posture (never a re-billed respawn). + var artifacts = new SupervisorAcceptanceGraderTests.FakeArtifactStore(); + var runners = new RecordingRunners(setupStatus, "npm ERR! network request failed"); + + var grade = await GradeAsync(Posture(AgentAutonomyLevel.Standard), runners, artifacts); + + grade.Detail.ShouldBe(expectedDetail); + grade.Class.ShouldBe(GradeFailureClass.Environment); + AgentAcceptanceContract.IsInfraFailure(grade, workPresent: true).ShouldBeTrue(); + AgentAcceptanceContract.IsInfraFailure(grade.Detail, workPresent: true).ShouldBeTrue("the string path — the only one the agent.code node's resume payload carries — agrees"); + AgentAcceptanceContract.IsDecidedByGradePosture(grade.Detail).ShouldBeTrue(); + grade.EvidenceArtifactId.ShouldNotBeNull("the failure's evidence is where the operator reads why the setup could not download"); + artifacts.Puts.ShouldHaveSingleItem().Text.ShouldStartWith(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix); + grade.EvidenceTail.ShouldNotBeNull().ShouldContain("needs an egress allowlist or a higher tier"); + } + + [Theory] + [InlineData(SandboxStatus.Failed, "setup-failed: npm ERR! network request failed")] + [InlineData(SandboxStatus.TimedOut, "setup-timed-out")] + public async Task A_network_off_setup_on_a_host_that_does_not_confine_keeps_its_transient_detail_and_claims_no_narrowing(SandboxStatus setupStatus, string expectedDetail) + { + // Nothing on this host severed the setup — it ran with the network it asked for — so its failure is the + // ordinary transient one, and its evidence does not say "off" about a setup that kept the network. + var artifacts = new SupervisorAcceptanceGraderTests.FakeArtifactStore(); + + var grade = await GradeAsync(Posture(AgentAutonomyLevel.Standard), new RecordingRunners(setupStatus, "npm ERR! network request failed", confines: false), artifacts); + + grade.Detail.ShouldBe(expectedDetail); + AgentAcceptanceContract.IsDecidedByGradePosture(grade.Detail).ShouldBeFalse(); + artifacts.Puts.ShouldBeEmpty("no notice, so the failure stays evidence-less exactly as before"); + } + + [Fact] + public async Task An_allowlisted_setup_that_fails_stays_transient_and_says_it_was_filtered() + { + // A filtered setup still reaches its allowlisted hosts, so its failure may be a transient blip on one of them. + var artifacts = new SupervisorAcceptanceGraderTests.FakeArtifactStore(); + var permissions = AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Trusted) with { Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = new[] { "registry.npmjs.org" } }; + var posture = AcceptanceGradingPosturePolicy.Derive(AgentAutonomyLevel.Trusted, permissions, AgentAutonomyLevel.Unleashed, null); + + var grade = await GradeAsync(posture, new RecordingRunners(SandboxStatus.Failed, "npm ERR! 503", filtersAllowlist: true), artifacts); + + grade.Detail.ShouldBe("setup-failed: npm ERR! 503"); + AgentAcceptanceContract.IsDecidedByGradePosture(grade.Detail).ShouldBeFalse(); + artifacts.Puts.ShouldHaveSingleItem().Text.ShouldStartWith(AcceptanceGradingPosturePolicy.SetupNetworkNoticePrefix + "narrowed"); + } + + // ── A check killed at the grade's own ceiling ───────────────────────────────────────────────────────── + + [Theory] + [InlineData(AgentAutonomyLevel.Standard)] + [InlineData(AgentAutonomyLevel.Trusted)] + public async Task A_check_killed_at_the_grades_memory_ceiling_is_an_environment_fact_not_a_code_verdict(AgentAutonomyLevel tier) + { + // Before the grade carried ceilings no grade step had a cgroup, so ResourceExhausted could not happen. Now it + // can, and the agent's own run already reads the same kill as "not a fault in the agent's work". + var posture = Posture(tier); + var runners = new RecordingRunners(SandboxStatus.Success, "", checkStatus: SandboxStatus.ResourceExhausted); + + var grade = await GradeAsync(posture, runners, new SupervisorAcceptanceGraderTests.FakeArtifactStore()); + + runners.Specs[1].MaxMemoryMb.ShouldBe(posture.MaxMemoryMb, "fixture check: the check ran under the posture's ceiling that killed it"); + grade.Passed.ShouldBeFalse(); + grade.Detail.ShouldBe(TestsPassGrader.ResourceExhaustedDetail); + grade.Class.ShouldBe(GradeFailureClass.Environment); + AgentAcceptanceContract.IsInfraFailure(grade, workPresent: true).ShouldBeTrue("the revise loop never buys a round for it"); + AgentAcceptanceContract.IsInfraFailure(grade.Detail, workPresent: true).ShouldBeTrue("the string path the agent.code node reads agrees"); + } + + [Fact] + public async Task A_setup_that_fails_with_the_network_it_asked_for_stays_evidence_less_as_before() + { + var artifacts = new SupervisorAcceptanceGraderTests.FakeArtifactStore(); + + var grade = await GradeAsync(Posture(AgentAutonomyLevel.Trusted), new RecordingRunners(SandboxStatus.Failed, "npm ERR!"), artifacts); + + grade.Detail.ShouldStartWith("setup-failed:"); + grade.EvidenceArtifactId.ShouldBeNull(); + artifacts.Puts.ShouldBeEmpty(); + } + + // ── fixture ─────────────────────────────────────────────────────────────────────────────────────────── + + private static AcceptanceGradingPosture Posture(AgentAutonomyLevel tier) => + AcceptanceGradingPosturePolicy.Derive(tier, AgentAutonomyPolicy.Derive(tier), AgentAutonomyLevel.Unleashed, hostMemoryBudgetMb: null); + + /// Grade a scratch workspace with a setup step through the REAL grader and the real TestsPass oracle, and return the two specs it handed its runner. + private static async Task<(SandboxSpec Setup, SandboxSpec Check)> GradeAsync(AcceptanceGradingPosture? posture, SupervisorAcceptanceGraderTests.FakeArtifactStore? artifacts = null) + { + var runners = new RecordingRunners(SandboxStatus.Success, ""); + await GradeAsync(posture, runners, artifacts ?? new SupervisorAcceptanceGraderTests.FakeArtifactStore()); + + runners.Specs.Select(s => s.Command).ShouldBe(new[] { "npm", "npm" }, "fixture check: the setup step and then the check ran"); + runners.Specs[0].Args.ShouldBe(new[] { "ci" }); + + return (runners.Specs[0], runners.Specs[1]); + } + + private static async Task GradeAsync(AcceptanceGradingPosture? posture, RecordingRunners runners, SupervisorAcceptanceGraderTests.FakeArtifactStore artifacts) + { + var directory = Directory.CreateTempSubdirectory("cs-grade-posture-").FullName; + try + { + // Only the seams GradeDirectoryAsync reaches are real or recorded: the runner, the oracle registry and the + // evidence store. The clone/patch/captured collaborators are never touched on this lane. + var grader = new SupervisorAcceptanceGrader(null!, null!, runners, new BenchmarkGraderRegistry(new IBenchmarkGrader[] { new TestsPassGrader() }), null!, artifacts, null!, NullLogger.Instance); + var spec = new SupervisorAcceptanceSpec { Command = CheckCommand, SetupCommand = SetupCommand }; + + return await grader.GradeDirectoryAsync(new DirectoryAcceptanceGradeRequest { Directory = directory, Spec = spec, TeamId = Guid.NewGuid(), TimeoutSeconds = 30, Posture = posture }, CancellationToken.None); + } + finally { Directory.Delete(directory, recursive: true); } + } + + /// The argv a confining host launches for , with no filtered namespace. + private static IReadOnlyList Argv(SandboxSpec spec) => Chain(spec, Array.Empty()); + + private static IReadOnlyList Chain(SandboxSpec spec, IReadOnlyList egressPrefix) => + LocalProcessRunner.ChildCommand(new LocalProcessRunner.CommandIsolationContext(spec, null, null, egressPrefix, Array.Empty()), Bwrap, prlimit: null); + + /// + /// Records every spec the grader runs; the setup step (the first) answers and the + /// check (the second) . It reports the egress it enforces through the production + /// table () for a host that + /// by default — the host whose narrowing the notice describes. + /// + private sealed class RecordingRunners(SandboxStatus setupStatus, string setupStderr, SandboxStatus checkStatus = SandboxStatus.Success, bool confines = true, bool filtersAllowlist = false) : ISandboxRunnerRegistry, ISandboxRunner, ISandboxEgressEnforcement + { + public List Specs { get; } = new(); + public string Kind => SandboxKinds.Local; + public IReadOnlyList All => new ISandboxRunner[] { this }; + public ISandboxRunner Resolve(string kind) => this; + + public SandboxEgressMode EnforcedEgress(SandboxSpec spec) => LocalProcessRunner.EnforcedEgress(spec, confines, filtersAllowlist); + + public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) + { + Specs.Add(spec); + var status = Specs.Count switch { 1 => setupStatus, 2 => checkStatus, _ => SandboxStatus.Success }; + var exitCode = status switch { SandboxStatus.Success => 0, SandboxStatus.ResourceExhausted => 137, _ => 1 }; + + return Task.FromResult(new SandboxResult { Status = status, ExitCode = exitCode, Stdout = "", Stderr = status == SandboxStatus.Success ? "" : setupStderr }); + } + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Supervisor/CapturedDeliverableGradeTests.cs b/backend/tests/CodeSpace.UnitTests/Supervisor/CapturedDeliverableGradeTests.cs index fce3749ea..8f8ded205 100644 --- a/backend/tests/CodeSpace.UnitTests/Supervisor/CapturedDeliverableGradeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Supervisor/CapturedDeliverableGradeTests.cs @@ -55,7 +55,7 @@ public async Task The_delayed_captured_grade_preserves_the_producers_trusted_mod var producer = new ReviewModelIdentity { ModelCredentialModelId = modelRowId, ConfiguredModel = "configured", ObservedModel = "observed" }; var (grader, oracle, _) = New(new BenchmarkGrade { Passed = true, Detail = "artifact-present" }, Row(runId, teamId, "report.md", "# findings\n")); - await grader.GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = runId, TeamId = teamId, Spec = Spec("report.md"), TimeoutSeconds = 60, ProducerModel = producer }, CancellationToken.None); + await grader.GradeCapturedAsync(new CapturedAcceptanceGradeRequest { AgentRunId = runId, TeamId = teamId, Spec = Spec("report.md"), TimeoutSeconds = 60, ProducerModel = producer, Posture = null }, CancellationToken.None); oracle.LastProducerModel.ShouldBe(producer, "the evaluator must see the producer identity even when grading happens after the worker and workspace are gone"); } diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs index 91c0dbbb4..ac92fd346 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs @@ -676,9 +676,28 @@ public async Task P3_1_a_genuine_acceptance_failure_stays_non_retryable_even_wit result.Retryable.ShouldBeFalse("the same code + the same check would fail again — a respawn cannot change a genuine verdict"); } + [Theory] + [InlineData("setup-failed-network-severed: npm ERR! network request failed")] + [InlineData("setup-failed-network-severed: timed out")] + [InlineData("repo 'web': setup-failed-network-severed: npm ERR! network request failed")] // through the multi-repo display tag + public async Task A_setup_the_grades_posture_severed_is_not_retried(string acceptanceDetail) + { + // The posture comes from the same stored task on every attempt, so a respawn re-buys a whole billed agent run + // only to have its setup severed again. Still infra — never a code verdict — but no longer worth a respawn. + var resume = JsonDocument.Parse($$""" + {"status":"Failed","error":"x","exitReason":"acceptance-failed","acceptanceDetail":"{{acceptanceDetail}}","changedFiles":["src/a.ts"]} + """).RootElement; + + var result = await new AgentCodeNode().RunAsync(BuildContext(new(), resume), CancellationToken.None); + + result.Status.ShouldBe(NodeStatus.Failure); + result.Retryable.ShouldBeFalse("a setup the grade's own posture severed fails identically on every respawn"); + } + [Theory] [InlineData("setup-failed: npm ERR! missing script")] [InlineData("setup-timed-out")] + [InlineData("tests-resource-exhausted")] // the check was killed at the grade's own ceiling: an environment fact, like tests-timed-out public async Task P3_1_part_2_a_setup_command_infra_fault_is_retryable_despite_the_fail_closed_acceptance_exit_reason(string acceptanceDetail) { // The contract's OWN setup step (installing deps, a build) failing/timing out means the CHECK never ran at diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorAcceptanceTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorAcceptanceTests.cs index 04464bc9b..c82b16f00 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorAcceptanceTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunExecutorAcceptanceTests.cs @@ -609,6 +609,38 @@ public async Task A_repo_less_file_oracle_also_requires_a_live_context_instead_o grader.DirectoryCalls.ShouldBe(0); } + // ─── V-B: every repository lane grades under the run's OWN posture ────────── + + [Theory] + [InlineData("branch", AgentAutonomyLevel.Standard)] + [InlineData("branch", AgentAutonomyLevel.Trusted)] + [InlineData("patch", AgentAutonomyLevel.Standard)] + [InlineData("patch", AgentAutonomyLevel.Trusted)] + [InlineData("multi-repo", AgentAutonomyLevel.Standard)] + [InlineData("multi-repo", AgentAutonomyLevel.Trusted)] + public async Task Every_repository_lane_grades_under_the_producing_runs_posture(string lane, AgentAutonomyLevel tier) + { + // The grade runs the candidate's bytes after the agent's sandbox is gone; the posture is what keeps its setup + // from reaching a network this run never had. A lane that dropped it would grade fail-closed — safe, but no + // longer this run's posture, so a Trusted run's setup could not download. + var (executor, grader) = NewExecutor(new BenchmarkGrade { Passed = true, Detail = "tests-passed" }); + var task = TaskWith(Spec("sh", "check.sh")) with { Autonomy = tier, Permissions = AgentAutonomyPolicy.Derive(tier) with { Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = new[] { "registry.npmjs.org" } } }; + var result = lane switch + { + "branch" => Succeeded(), + "patch" => SucceededPatchOnly(), + _ => Succeeded() with { RepositoryResults = new[] { new RepositoryRunResult { RepositoryId = Guid.NewGuid(), Alias = "web", ProducedBranch = "agent/web" }, new RepositoryRunResult { RepositoryId = Guid.NewGuid(), Alias = "api", ProducedBranch = "agent/api" } } }, + }; + + await executor.GradeAcceptanceIfPresentAsync(Run(), task, result, workspace: null, CancellationToken.None); + + var expected = AcceptanceGradingPosturePolicy.For(task); + grader.Postures.Count.ShouldBe(lane == "multi-repo" ? 2 : 1, "fixture check: the lane under test actually graded"); + grader.Postures.ShouldAllBe(p => p != null && p.Autonomy == expected.Autonomy && p.AllowNetwork == expected.AllowNetwork && p.MaxMemoryMb == expected.MaxMemoryMb && p.MaxCpuPercent == expected.MaxCpuPercent); + grader.Postures.ShouldAllBe(p => (p!.EgressAllowlist ?? Array.Empty()).SequenceEqual(expected.EgressAllowlist ?? Array.Empty())); + expected.AllowNetwork.ShouldBe(tier == AgentAutonomyLevel.Trusted, "fixture check: the two tiers really differ on the network"); + } + // ─── fixtures ──────────────────────────────────────────────────────────────── private static AgentRun Run() => new() { Id = Guid.NewGuid(), TeamId = Guid.NewGuid() }; @@ -716,6 +748,21 @@ private sealed class FakeGrader : ISupervisorAcceptanceGrader /// C3 narrowing — the run's own ORACLE INVENTORY each branch grade was handed. This lane's contract IS the run's one gate, so its own program file(s) are the judge. public Dictionary?> OracleFloorProgramsByBranch { get; } = new(); + /// The producing-run posture every request-form grade carried, in call order — a lane that dropped it records null. + public List Postures { get; } = new(); + + public Task GradeAsync(RepositoryAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + Postures.Add(request.Posture); + return GradeAsync(request.RepositoryId, request.TeamId, request.Branch, request.Spec, request.TimeoutSeconds, request.Anchor, cancellationToken); + } + + public Task GradePatchAsync(PatchAcceptanceGradeRequest request, CancellationToken cancellationToken) + { + Postures.Add(request.Posture); + return GradePatchAsync(request.RepositoryId, request.TeamId, request.BaseSha, request.InlinePatch, request.PatchArtifactId, request.Spec, request.TimeoutSeconds, cancellationToken); + } + public Task GradeAsync(Guid repositoryId, Guid teamId, string branch, SupervisorAcceptanceSpec spec, int timeoutSeconds, OracleAnchor anchor, CancellationToken cancellationToken) { OracleBaseShaByBranch[branch] = anchor.BaseSha;