diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs index adda7a55b..5c75433f3 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs @@ -334,6 +334,8 @@ public async Task ExecuteAsync(Guid agentRunId, CancellationToken cancellationTo var reconciliation = await _harnessReconciler.ReconcileAsync(task, run.TeamId, cancellationToken).ConfigureAwait(false); var harness = _harnesses.Resolve(reconciliation.HarnessKind); + task = await HoldToModelPoolAsync(owner, task, reconciliation, cancellationToken).ConfigureAwait(false); + if (reconciliation.Repaired) { _logger.LogWarning("AgentRun {RunId}: {Note}", agentRunId, reconciliation.Note); @@ -4027,6 +4029,32 @@ private static IEnumerable UrlEmbeddedSecrets(string? baseUrl) } } + /// + /// Hold a bounded run to its allowed model pool (): run the reconciler's pooled + /// row — its model on its own credential — and persist it, so a re-attach and every reader see the model the agent + /// actually runs; a model that had to move is named on the run's timeline. A pool that resolves nothing any more fails + /// the run rather than let the agent run outside it. An unbounded task passes through untouched (byte-identical). The + /// task is still the ORIGINAL (no injected secret env), so serializing it is safe. + /// + private async Task HoldToModelPoolAsync(AgentRunOwnerToken owner, AgentTask task, HarnessReconciliation reconciliation, CancellationToken cancellationToken) + { + if (task.AllowedModelIds is not { Count: > 0 }) return task; + + if (reconciliation.PooledModel is not { } pooled) + throw new InvalidOperationException(reconciliation.PoolNote); + + if (reconciliation.PoolNote is { } note) + { + _logger.LogWarning("AgentRun {RunId}: {Note}", owner.RunId, note); + await _runs.AppendEventAsync(owner, new AgentEvent { Kind = AgentEventKind.Warning, Text = note }, cancellationToken).ConfigureAwait(false); + } + + var held = task with { Model = pooled.ModelId, ModelCredentialId = pooled.ModelCredentialId, ModelCredentialModelId = null }; + await PersistRuntimeIdentityAsync(owner, null, JsonSerializer.Serialize(held, AgentJson.Options), cancellationToken).ConfigureAwait(false); + + return held; + } + /// Re-persist the run's stored task with its RESOLVED model filled, so the live projection shows what an "auto" run actually dispatches from the moment it starts (mirrors the harness-reconciliation write). The task is the ORIGINAL (no injected secret env) with only set, so serializing it is safe. private Task PersistResolvedModelAsync(AgentRunOwnerToken owner, AgentTask taskWithModel, CancellationToken cancellationToken) => PersistRuntimeIdentityAsync(owner, null, JsonSerializer.Serialize(taskWithModel, AgentJson.Options), cancellationToken); diff --git a/backend/src/CodeSpace.Core/Services/Agents/HarnessModelReconciler.cs b/backend/src/CodeSpace.Core/Services/Agents/HarnessModelReconciler.cs index 6b45b51e4..556815b3e 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/HarnessModelReconciler.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/HarnessModelReconciler.cs @@ -36,6 +36,12 @@ namespace CodeSpace.Core.Services.Agents; /// row, so a pin repair makes the whole (harness, model, credential) triple runnable. Blanking the model here would /// drop a VALID operator choice in the common consistent case, so this layer never does; it only swaps the harness for /// one that can drive the model's provider. +/// +/// The ONE exception is the run's allowed model pool (, the model analogue of +/// the harness allow-list): a bounded task runs on a POOLED ROW — its named model's pooled row, else (no name, or a name +/// no pooled row carries, such as a planner-authored model outside the pool) the pool's default row — reported as +/// for the executor to run and persist, and the harness is reconciled +/// against THAT row's provider. An unbounded task is reconciled exactly as before. /// public interface IHarnessModelReconciler { @@ -43,8 +49,8 @@ public interface IHarnessModelReconciler Task ReconcileAsync(AgentTask task, Guid teamId, CancellationToken cancellationToken); } -/// The harness KIND to run, whether it was REPAIRED away from the authored one, and a human-facing note for the timeline when it was. -public sealed record HarnessReconciliation(string HarnessKind, bool Repaired, string? Note); +/// The harness KIND to run, whether it was REPAIRED away from the authored one, and a human-facing note for the timeline when it was — plus, for a task bounded to an allowed model pool, the pooled row it runs on (null when unbounded, or when the pool resolves nothing any more) and a note when the authored model had to move or cannot be honoured. +public sealed record HarnessReconciliation(string HarnessKind, bool Repaired, string? Note, ModelDispatchRef? PooledModel = null, string? PoolNote = null); public sealed class HarnessModelReconciler : IHarnessModelReconciler, IScopedDependency { @@ -61,11 +67,15 @@ public HarnessModelReconciler(IAgentHarnessRegistry harnesses, IModelPoolSelecto public async Task ReconcileAsync(AgentTask task, Guid teamId, CancellationToken cancellationToken) { - var provider = await ResolveModelProviderAsync(task, teamId, cancellationToken).ConfigureAwait(false); + var pooled = await ResolvePooledModelAsync(task, teamId, cancellationToken).ConfigureAwait(false); + var poolNote = DescribePoolBound(task, pooled); + + // A bounded task runs on its pooled row, so the harness follows THAT row's provider, not the authored model's. + var provider = pooled?.Provider ?? await ResolveModelProviderAsync(task, teamId, cancellationToken).ConfigureAwait(false); // No provider to reconcile against (no pin AND no pooled model name) → return the authored kind verbatim (the // caller's registry resolves it; a genuinely-unregistered kind surfaces there, unchanged). - if (provider is null) return new HarnessReconciliation(task.Harness, false, null); + if (provider is null) return new HarnessReconciliation(task.Harness, false, null, pooled, poolNote); // The repair chooses from the registry CLAMPED to the run's harness allow-list (null/empty = the whole registry, // which is every non-supervisor path and every pre-field task envelope). Without this clamp the run-time repair @@ -75,7 +85,32 @@ public async Task ReconcileAsync(AgentTask task, Guid tea // admitted one, so the floor stays inside the list too. var pool = AgentHarnessPool.Clamp(_harnesses.All, task.AllowedHarnessKinds); - return Reconcile(task.Harness, provider, pool, AgentHarnessDefaults.DefaultHarness); + return Reconcile(task.Harness, provider, pool, AgentHarnessDefaults.DefaultHarness) with { PooledModel = pooled, PoolNote = poolNote }; + } + + /// + /// The allowed-pool row a bounded task runs on: its named model's pooled row, else — no name, or a name no pooled row + /// carries — the pool's default row, ranked by the same agent-plane precedence the supervisor's pool-bound default uses + /// (names repeat across credentials, so both lookups are over the pool's ROWS). Null for an unbounded task, and when + /// nothing in the pool resolves any more. + /// + private async Task ResolvePooledModelAsync(AgentTask task, Guid teamId, CancellationToken cancellationToken) + { + if (task.AllowedModelIds is not { Count: > 0 } pool) return null; + + var named = string.IsNullOrWhiteSpace(task.Model) ? null : await _modelSelector.ResolveDispatchAsync(teamId, task.Model, pool, cancellationToken).ConfigureAwait(false); + + return named ?? await _modelSelector.ResolvePoolDefaultAsync(teamId, pool, cancellationToken).ConfigureAwait(false); + } + + /// Why a bounded task's model did not run as authored — it was outside the pool, or nothing in the pool resolves any more. Null when unbounded, when the named model is pooled, and when no model was named (the pool's default is then simply the model it runs). + private static string? DescribePoolBound(AgentTask task, ModelDispatchRef? pooled) + { + if (task.AllowedModelIds is not { Count: > 0 }) return null; + if (pooled is null) return "None of this run's allowed models resolves to an enabled model under an active credential, so the agent cannot run inside its allowed model pool."; + if (string.IsNullOrWhiteSpace(task.Model) || string.Equals(task.Model.Trim(), pooled.ModelId, StringComparison.OrdinalIgnoreCase)) return null; + + return $"Model '{task.Model}' is not in this run's allowed model pool; running the pool's default, '{pooled.ModelId}', instead."; } /// diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Effort/EffortRouter.cs b/backend/src/CodeSpace.Core/Services/Tasks/Effort/EffortRouter.cs index b9862643a..da8b5bf86 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/Effort/EffortRouter.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/Effort/EffortRouter.cs @@ -2,6 +2,7 @@ using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Tasks.Bounds; using CodeSpace.Core.Services.Tasks.Capabilities; +using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Core.Services.Tasks.Recipes; using CodeSpace.Messages.Agents; using CodeSpace.Messages.Tasks; @@ -15,8 +16,9 @@ namespace CodeSpace.Core.Services.Tasks.Effort; /// type: every branch point is a registry lookup, so a new classification / recipe / bounds / capability /// strategy needs zero edit here (the fake-probe + fake-recipe contract test proves it). The pipeline: resolve /// the decision (operator short-circuit vs the default classifier) → policy-decide the effort mode → resolve the -/// recipe (fail-open) → resolve the projection → DEGRADE if the recipe's required capability is unavailable → -/// resolve the bounds preset + merge any caps override → assemble the RoutePlan + a derived confirm card. +/// recipe (fail-open) → keep an auto route with an operator acceptance floor on a projection that grades it → resolve +/// the projection → DEGRADE if the recipe's required capability is unavailable → resolve the bounds preset + merge any +/// caps override → assemble the RoutePlan + a derived confirm card. /// public sealed class EffortRouter : IEffortRouter, IScopedDependency { @@ -24,13 +26,15 @@ public sealed class EffortRouter : IEffortRouter, IScopedDependency private readonly ITaskRecipeRegistry _recipes; private readonly IBoundsPresetRegistry _bounds; private readonly ICapabilityProbeRegistry _capabilities; + private readonly ITaskProjectionRegistry _projections; - public EffortRouter(IEffortClassifierRegistry classifiers, ITaskRecipeRegistry recipes, IBoundsPresetRegistry bounds, ICapabilityProbeRegistry capabilities) + public EffortRouter(IEffortClassifierRegistry classifiers, ITaskRecipeRegistry recipes, IBoundsPresetRegistry bounds, ICapabilityProbeRegistry capabilities, ITaskProjectionRegistry projections) { _classifiers = classifiers; _recipes = recipes; _bounds = bounds; _capabilities = capabilities; + _projections = projections; } public async Task RouteAsync(EffortRouteRequest request, CancellationToken ct) @@ -41,24 +45,60 @@ public async Task RouteAsync(EffortRouteRequest request, Cancellation var recipe = ResolveRecipe(request, decision); + (effortMode, recipe, var floorReason) = KeepOperatorFloorGradable(request, wasAutoClassified, decision.Signals, effortMode, recipe); + var projectionKind = request.RequestedProjection ?? recipe.DefaultProjectionKind; var (effectiveRecipe, effectiveProjection, degradedReason) = DegradeIfCapabilityUnavailable(request, recipe, projectionKind); var (preset, caps) = ResolveCaps(request, effortMode, effectiveRecipe); - // A risky / irreversible task ALWAYS surfaces the confirm card regardless of the model's self-confidence — the - // classifier emits the risk signal, but the ROUTER (not the model's confidence) decides the human gate, so an - // over-confident model can't suppress the operator's escalation affordance on destructive work. This restores the - // pre-LLM always-confirm floor for risk while keeping the confident-routing win for ordinary tasks (model emits - // data, policy decides — the same tighten-only convention as the autonomy ceiling). - var needsConfirmCard = wasAutoClassified && (decision.Confidence < EffortPolicy.ConfirmConfidenceFloor || decision.Signals.RiskySideEffects); + var needsConfirmCard = NeedsConfirmCard(decision, wasAutoClassified); var confirm = needsConfirmCard ? BuildConfirmCard(decision) : null; - return BuildPlan(decision, wasAutoClassified, effortMode, effectiveRecipe, effectiveProjection, preset, caps, needsConfirmCard, confirm, degradedReason); + return BuildPlan(decision, wasAutoClassified, effortMode, effectiveRecipe, effectiveProjection, preset, caps, needsConfirmCard, confirm, JoinReasons(floorReason, degradedReason)); + } + + /// + /// Whether an auto route must be confirmed by the operator before it runs: a confidence below the floor, a risky / + /// irreversible task, or an AMBIGUOUS one. The classifier emits the signals, but the ROUTER (not the model's + /// confidence) decides the human gate, so an over-confident model can't suppress the operator's escalation affordance + /// on destructive work, or route an under-specified goal as though it were understood — the confirm card says + /// exactly that. Model emits data, policy decides — the same tighten-only convention as the autonomy ceiling. An + /// explicit operator tier is already a decision and never confirms. + /// + private static bool NeedsConfirmCard(EffortDecision decision, bool wasAutoClassified) => + wasAutoClassified && (decision.Confidence < EffortPolicy.ConfirmConfidenceFloor || decision.Signals.RiskySideEffects || decision.Signals.Ambiguous); + + /// + /// An operator acceptance floor is a routing signal. When the AUTO path's classified shape lands on a projection whose + /// builder does not grade an operator command, re-decide the tier with every such tier set aside — the policy's next + /// matching row — and say so on the route. An explicit tier, a pinned recipe or a pinned projection is the operator's + /// own choice and stays where it is: the launch refuses that combination with its reason instead of moving it. + /// + private (string EffortMode, ITaskRecipe Recipe, string? Reason) KeepOperatorFloorGradable(EffortRouteRequest request, bool wasAutoClassified, EffortSignals signals, string effortMode, ITaskRecipe recipe) + { + if (!request.HasOperatorFloor || !wasAutoClassified || request.RequestedRecipe is not null || request.RequestedProjection is not null) return (effortMode, recipe, null); + if (GradesOperatorFloor(recipe.DefaultProjectionKind)) return (effortMode, recipe, null); + + var admitted = EffortPolicy.Decide(signals, requestedEffort: null, mode => GradesOperatorFloor(_recipes.RecipeForEffort(mode).DefaultProjectionKind)); + var rerouted = _recipes.RecipeForEffort(admitted); + + // No tier the policy admits grades the floor either: leave the route as classified — the launch refuses it by name. + if (!GradesOperatorFloor(rerouted.DefaultProjectionKind)) return (effortMode, recipe, null); + + return (admitted, rerouted, $"the operator's acceptance check needs a route that grades it, and '{recipe.DefaultProjectionKind}' does not; moved from {effortMode} to {admitted} ('{rerouted.DefaultProjectionKind}')"); } + /// Whether 's builder advertises that it grades an operator command — the same advertisement the route preview's acceptance verdict and the launch's floor disposition read. + private bool GradesOperatorFloor(string projectionKind) => + _projections.TryResolve(projectionKind, out var builder) && builder.OperatorAcceptance.AcceptsCommand == true; + + /// Every reason the route moved off what was asked for, in the order the moves happened — each is named, none is dropped. + private static string? JoinReasons(string? first, string? second) => + first is null ? second : second is null ? first : $"{first}; {second}"; + /// /// When the resolved recipe DECLARES a required capability (ITaskRecipe.RequiresCapability) that the /// probe registry reports unavailable, DEGRADE to the recipe's fallback (DegradesToRecipe, else the @@ -244,7 +284,7 @@ private static string BuildHint(RouteCaps caps) => NeedsPlanReview = recipe.RequiresPlanReview, WasAutoClassified = wasAutoClassified, ClassifierConfidence = decision.Confidence, - DegradedReason = degradedReason, // set (non-null) when a capability degrade fired, null otherwise — never silent + DegradedReason = degradedReason, // set (non-null) when the route moved — an operator floor it could not grade, a capability degrade — null otherwise; never silent Decision = decision, Confirm = confirm, }; diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Launch/Exceptions/TaskLaunchControlRefusedException.cs b/backend/src/CodeSpace.Core/Services/Tasks/Launch/Exceptions/TaskLaunchControlRefusedException.cs new file mode 100644 index 000000000..8d49e867b --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Tasks/Launch/Exceptions/TaskLaunchControlRefusedException.cs @@ -0,0 +1,20 @@ +using CodeSpace.Messages.Failures; +using CodeSpace.Messages.Tasks; + +namespace CodeSpace.Core.Services.Tasks.Launch.Exceptions; + +/// +/// The resolved route cannot honour a launch control the operator set, so the launch stops before any session or run +/// exists rather than run without it. Carries each refused control with its reason — the same dispositions the route +/// preview reports for this input — so every caller can say what to change. +/// +public sealed class TaskLaunchControlRefusedException : Exception, IFailure +{ + private readonly IReadOnlyList _refused; + + public TaskLaunchControlRefusedException(IReadOnlyList refused) : base(string.Join(" ", refused.Select(d => $"{d.Control}: {d.Reason}"))) { _refused = refused; } + + public FailureKind Kind => FailureKind.Unprocessable; + public string Code => FailureCodes.TaskLaunchControlRefused; + public IReadOnlyDictionary Details => new Dictionary { ["controls"] = _refused }; +} diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Launch/ILaunchControlResolver.cs b/backend/src/CodeSpace.Core/Services/Tasks/Launch/ILaunchControlResolver.cs new file mode 100644 index 000000000..7a044c9f4 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Tasks/Launch/ILaunchControlResolver.cs @@ -0,0 +1,17 @@ +using CodeSpace.Core.Services.Agents.ModelCredentials; +using CodeSpace.Messages.Tasks; + +namespace CodeSpace.Core.Services.Tasks.Launch; + +/// +/// What a resolved route does with each operator control whose effect depends on the route. ONE method for both callers: +/// the launch refuses on a Refused disposition and applies the model clamp, and the route preview reports the same +/// dispositions — so the preview can never state a disposition the launch would not reach. +/// +public interface ILaunchControlResolver +{ + Task ResolveAsync(TaskLaunchRequest request, RoutePlan route, CancellationToken cancellationToken); +} + +/// The dispositions of one launch's route-dependent controls, whether its route grades an operator acceptance floor, and the pooled row the single agent runs on when its pinned model fell outside the allowed pool (null when nothing was clamped). +public sealed record LaunchControlResolution(IReadOnlyList Dispositions, bool GradesOperatorFloor, ModelDispatchRef? ModelClamp); diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Launch/ILaunchGroundingResolver.cs b/backend/src/CodeSpace.Core/Services/Tasks/Launch/ILaunchGroundingResolver.cs new file mode 100644 index 000000000..624d82835 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Tasks/Launch/ILaunchGroundingResolver.cs @@ -0,0 +1,14 @@ +using CodeSpace.Messages.Tasks; + +namespace CodeSpace.Core.Services.Tasks.Launch; + +/// +/// The grounding a launch is primed with — shared by the launch and its route preview, because the route is classified +/// on it too: a follow-up turn has to reach the classifier AS a follow-up, which it cannot when the prior-turn digest is +/// only composed after routing. +/// +public interface ILaunchGroundingResolver +{ + /// On a CONTINUE, the session's prior-turn digest composed over any seed grounding; on a fresh launch, only the seed's own grounding (null for chat). + Task ResolveAsync(TaskLaunchRequest request, TaskLaunchSeed seed, CancellationToken cancellationToken); +} diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Launch/LaunchControlResolver.cs b/backend/src/CodeSpace.Core/Services/Tasks/Launch/LaunchControlResolver.cs new file mode 100644 index 000000000..3b75a3d4b --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Tasks/Launch/LaunchControlResolver.cs @@ -0,0 +1,139 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Core.Services.Agents.ModelCredentials; +using CodeSpace.Core.Services.Tasks.Projection; +using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Tasks; + +namespace CodeSpace.Core.Services.Tasks.Launch; + +/// +/// Default . Each control's disposition follows what the route's lane does with it: +/// the supervisor lane bakes every one; a plan-map lane plans and dispatches inside the model pool, confirms and reviews +/// its plan, and gates the launch's persona; the single-agent lane holds its agent to the model pool, grades the floor +/// and gates the persona. A lane that does not consume a control names it NotApplicable with the reason, never drops it. +/// The operator floor follows the builder's own OperatorAcceptance advertisement — the fact the preview reports. +/// +public sealed class LaunchControlResolver : ILaunchControlResolver, IScopedDependency +{ + /// Why a route whose builder does not grade an operator command refuses one — word for word the route preview's acceptance verdict for that route. + public const string FloorNotGradedReason = "The resolved route does not consume an operator-command acceptance floor. Its plan items use separate acceptance contracts."; + + /// Why a route whose builder never advertised an operator-command adapter refuses a check: nothing on it promises to run the argv. + public const string FloorUnadvertisedReason = "The resolved route has not advertised an operator-command acceptance adapter, so nothing on it would run this check."; + + private readonly ITaskProjectionRegistry _projections; + private readonly IModelPoolSelector _models; + + public LaunchControlResolver(ITaskProjectionRegistry projections, IModelPoolSelector models) + { + _projections = projections; + _models = models; + } + + public async Task ResolveAsync(TaskLaunchRequest request, RoutePlan route, CancellationToken cancellationToken) + { + var (modelPool, clamp) = await DisposeModelPoolAsync(request, route, cancellationToken).ConfigureAwait(false); + + var dispositions = new[] + { + modelPool, + DisposeAgentPool(request, route), + DisposeOperatorFloor(request, route), + SupervisorOnly(LaunchControls.DecisionReviewMode, request.DecisionReviewMode != ReviewMode.None, route, "Only the supervisor lane reviews decisions."), + SupervisorOnly(LaunchControls.DeliverySpec, request.DeliverySpec is not null, route, "Only the supervisor lane carries a delivery spec."), + PlanningOnly(LaunchControls.RequirePlanConfirmation, request.RequirePlanConfirmation == true, route, "This route authors no plan to confirm."), + PlanningOnly(LaunchControls.PlannerReviewMode, request.PlannerReviewMode != ReviewMode.None, route, "This route authors no plan to review."), + }; + + return new LaunchControlResolution(dispositions.OfType().ToList(), AcceptsOperatorCommand(route) == true, clamp); + } + + /// + /// The allowed model pool. The supervisor gates every spawn inside it. Plan-map and single-agent hold every agent to it + /// at dispatch (a model outside it runs the pool's default row); an operator-PINNED model outside the pool is reported + /// Clamped here, and on the single-agent lane — whose one agent IS the pin — the clamp is returned for the launch to + /// bake, so the frozen agent config shows the model that runs. Refused when nothing in the pool resolves any more. + /// + private async Task<(LaunchControlDisposition? Disposition, ModelDispatchRef? Clamp)> DisposeModelPoolAsync(TaskLaunchRequest request, RoutePlan route, CancellationToken cancellationToken) + { + if (request.AllowedModelIds is not { Count: > 0 } pool) return (null, null); + if (IsSupervisor(route)) return (Applied(LaunchControls.AllowedModelIds), null); + if (!IsPlanMap(route) && !IsSingleAgent(route)) return (NotApplicable(LaunchControls.AllowedModelIds, $"The '{route.ProjectionKind}' route does not hold its agents to a model pool."), null); + + if (await DescribeUnpooledPinAsync(request, pool, cancellationToken).ConfigureAwait(false) is not { } pin) return (Applied(LaunchControls.AllowedModelIds, HeldAtDispatchReason(route)), null); + + if (await _models.ResolvePoolDefaultAsync(request.TeamId, pool, cancellationToken).ConfigureAwait(false) is not { } fallback) + return (Refused(LaunchControls.AllowedModelIds, "None of the allowed models resolves to an enabled model under an active credential."), null); + + var runs = IsSingleAgent(route) ? "the agent runs" : "every branch runs"; + + return (Clamped(LaunchControls.AllowedModelIds, $"The pinned model {pin} is not in the allowed model pool; {runs} the pool's default, '{fallback.ModelId}', instead."), IsSingleAgent(route) ? fallback : null); + } + + /// The operator's pinned agent model, described, when it lies outside the pool: a pinned credentialed-model row the pool does not list, or a pinned model name no pooled row carries (names repeat across credentials, so a name is looked up among the pool's rows). Null when nothing is pinned or the pin is pooled. + private async Task DescribeUnpooledPinAsync(TaskLaunchRequest request, IReadOnlyList pool, CancellationToken cancellationToken) + { + if (request.Overrides.ModelCredentialModelId is { } row) return pool.Contains(row) ? null : $"(row {row})"; + if (string.IsNullOrWhiteSpace(request.Overrides.Model)) return null; + + return await _models.ResolveDispatchAsync(request.TeamId, request.Overrides.Model, pool, cancellationToken).ConfigureAwait(false) is null ? $"'{request.Overrides.Model}'" : null; + } + + private static string HeldAtDispatchReason(RoutePlan route) => IsSingleAgent(route) + ? "The agent's model is held to the pool at dispatch; a model outside it runs the pool's default instead." + : "The planner is offered only these models, and every branch is held to them at dispatch; a model outside them runs the pool's default instead."; + + /// + /// The allowed persona pool. The supervisor gates every spawn inside it. Neither plan-map nor single-agent authors a + /// persona, so the launch's own persona is the only one any agent there runs as, and the pool is its gate: pooled ⇒ + /// applied, excluded ⇒ refused (the operator excluded it), none named ⇒ not applicable. + /// + private static LaunchControlDisposition? DisposeAgentPool(TaskLaunchRequest request, RoutePlan route) + { + if (request.AllowedAgentDefinitionIds is not { Count: > 0 } pool) return null; + if (IsSupervisor(route)) return Applied(LaunchControls.AllowedAgentDefinitionIds); + if (!IsPlanMap(route) && !IsSingleAgent(route)) return NotApplicable(LaunchControls.AllowedAgentDefinitionIds, $"The '{route.ProjectionKind}' route does not dispatch personas from a pool."); + + if (request.Overrides.AgentDefinitionId is not { } persona) + return NotApplicable(LaunchControls.AllowedAgentDefinitionIds, IsPlanMap(route) ? "Plan-map branches carry no persona: the launch names none, and the planner never assigns one." : "The agent carries no persona: the launch names none."); + + return pool.Contains(persona) ? Applied(LaunchControls.AllowedAgentDefinitionIds) : Refused(LaunchControls.AllowedAgentDefinitionIds, $"The launch's persona {persona} is not in the allowed agent pool, and every agent on this route runs as it."); + } + + /// The operator's executable acceptance floor: applied where the route's builder grades an operator command, refused where it advertises that it does not — or advertises nothing. + private LaunchControlDisposition? DisposeOperatorFloor(TaskLaunchRequest request, RoutePlan route) + { + if (request.AcceptanceChecks is not { Count: > 0 }) return null; + + return AcceptsOperatorCommand(route) switch + { + true => Applied(LaunchControls.AcceptanceChecks), + false => Refused(LaunchControls.AcceptanceChecks, FloorNotGradedReason), + null => Refused(LaunchControls.AcceptanceChecks, FloorUnadvertisedReason), + }; + } + + private bool? AcceptsOperatorCommand(RoutePlan route) => _projections.TryResolve(route.ProjectionKind, out var builder) ? builder.OperatorAcceptance.AcceptsCommand : null; + + /// A control only the supervisor lane consumes — applied there, not applicable everywhere else. + private static LaunchControlDisposition? SupervisorOnly(string control, bool requested, RoutePlan route, string reason) => + !requested ? null : IsSupervisor(route) ? Applied(control) : NotApplicable(control, reason); + + /// A control over an authored plan — applied on the lanes that author one (supervisor, plan-map), not applicable on the rest. + private static LaunchControlDisposition? PlanningOnly(string control, bool requested, RoutePlan route, string reason) => + !requested ? null : IsSupervisor(route) || IsPlanMap(route) ? Applied(control) : NotApplicable(control, reason); + + private static bool IsSupervisor(RoutePlan route) => route.ProjectionKind == TaskProjectionKinds.Supervisor; + + private static bool IsPlanMap(RoutePlan route) => route.ProjectionKind is TaskProjectionKinds.PlanMapSynth or TaskProjectionKinds.PlanMapDynamic; + + private static bool IsSingleAgent(RoutePlan route) => route.ProjectionKind == TaskProjectionKinds.SingleAgent; + + private static LaunchControlDisposition Applied(string control, string? reason = null) => new() { Control = control, Outcome = LaunchControlOutcome.Applied, Reason = reason }; + + private static LaunchControlDisposition Clamped(string control, string reason) => new() { Control = control, Outcome = LaunchControlOutcome.Clamped, Reason = reason }; + + private static LaunchControlDisposition NotApplicable(string control, string reason) => new() { Control = control, Outcome = LaunchControlOutcome.NotApplicable, Reason = reason }; + + private static LaunchControlDisposition Refused(string control, string reason) => new() { Control = control, Outcome = LaunchControlOutcome.Refused, Reason = reason }; +} diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Launch/LaunchGroundingResolver.cs b/backend/src/CodeSpace.Core/Services/Tasks/Launch/LaunchGroundingResolver.cs new file mode 100644 index 000000000..2a52de616 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Tasks/Launch/LaunchGroundingResolver.cs @@ -0,0 +1,40 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Core.Services.Sessions; +using CodeSpace.Messages.Tasks; + +namespace CodeSpace.Core.Services.Tasks.Launch; + +/// Default — the thread's rolling summary brought up to date, then its prior-turn digest composed over the seed's own grounding. +public sealed class LaunchGroundingResolver : ILaunchGroundingResolver, IScopedDependency +{ + private readonly ISessionContextBuilder _sessionContext; + private readonly ISessionSummarizer _sessionSummarizer; + + public LaunchGroundingResolver(ISessionContextBuilder sessionContext, ISessionSummarizer sessionSummarizer) + { + _sessionContext = sessionContext; + _sessionSummarizer = sessionSummarizer; + } + + public async Task ResolveAsync(TaskLaunchRequest request, TaskLaunchSeed seed, CancellationToken cancellationToken) + { + if (request.ContinueSessionId is not { } sessionId) return seed.GroundingContext; + + // Fold any turns that scrolled out of the recent window into the thread's rolling summary BEFORE building the + // digest, so a long thread's early context is preserved. Best-effort + fail-open (no model / error leaves it). + await _sessionSummarizer.EnsureSummaryUpToDateAsync(sessionId, request.TeamId, cancellationToken).ConfigureAwait(false); + + var priorTurns = await _sessionContext.BuildAsync(sessionId, request.TeamId, cancellationToken).ConfigureAwait(false); + + return ComposeGrounding(priorTurns, seed.GroundingContext); + } + + /// Join the prior-turn digest and the seed's own grounding (either may be absent) into one block, digest first. + private static string? ComposeGrounding(string? priorTurns, string? seedGrounding) + { + if (string.IsNullOrWhiteSpace(priorTurns)) return seedGrounding; + if (string.IsNullOrWhiteSpace(seedGrounding)) return priorTurns; + + return $"{priorTurns}\n\n{seedGrounding}"; + } +} diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/AgentNodeMapping.cs b/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/AgentNodeMapping.cs index e3fabfd00..23fccf190 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/AgentNodeMapping.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/AgentNodeMapping.cs @@ -110,12 +110,17 @@ public static JsonElement BuildAgentConfig(string goal, ResolvedAgentProfile? pr } /// Add the route-owned monitored cost ceiling without widening the already broad profile-mapping signature. Null leaves the serialized config byte-identical. - public static JsonElement WithCostCap(JsonElement config, decimal? maxCostUsd) - { - if (maxCostUsd is null) return config; + public static JsonElement WithCostCap(JsonElement config, decimal? maxCostUsd) => + maxCostUsd is null ? config : WithKey(config, "maxCostUsd", JsonSerializer.SerializeToElement(maxCostUsd.Value)); + + /// Hold the agent's model to the operator's allowed pool (credentialed-model row ids, as string uuids — the shape the supervisor node bakes): the node carries it onto AgentTask.AllowedModelIds, where dispatch runs the model on a pooled row. Null / empty leaves the config byte-identical (the whole team pool). + public static JsonElement WithAllowedModels(JsonElement config, IReadOnlyList? allowedModelIds) => + allowedModelIds is not { Count: > 0 } pool ? config : WithKey(config, "allowedModelIds", JsonSerializer.SerializeToElement(pool.Select(id => id.ToString()).ToList())); + private static JsonElement WithKey(JsonElement config, string key, JsonElement value) + { var mapped = config.EnumerateObject().ToDictionary(p => p.Name, p => p.Value.Clone()); - mapped["maxCostUsd"] = JsonSerializer.SerializeToElement(maxCostUsd.Value); + mapped[key] = value; return JsonSerializer.SerializeToElement(mapped); } diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/PlanMap/PlanMapBuilderBase.cs b/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/PlanMap/PlanMapBuilderBase.cs index b5583cc3f..5d6ee5d25 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/PlanMap/PlanMapBuilderBase.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/PlanMap/PlanMapBuilderBase.cs @@ -88,8 +88,10 @@ private IReadOnlyList BuildNodes(TaskBuildContext context) new() { Id = "ms", TypeKey = "flow.map_start", Label = "Subtask", ParentId = "map", Config = Empty(), Inputs = Empty() }, + // The operator's allowed model pool rides every branch, so a planner-authored {{item.model}} outside it runs + // the pool's default row at dispatch instead of escaping the pool. new() { Id = "agent", TypeKey = "agent.run", Label = "Work the subtask", ParentId = "map", Retry = AgentNodeMapping.DefaultRetry, - Config = AgentNodeMapping.BuildAgentConfig(BranchGoal, context.AgentProfile, BranchMode, grounding: context.GroundingContext, acceptance: "{{item.acceptance}}", fallbackModel: BranchModel), Inputs = AgentNodeMapping.BuildAgentInputs(context) }, + Config = AgentNodeMapping.WithAllowedModels(AgentNodeMapping.BuildAgentConfig(BranchGoal, context.AgentProfile, BranchMode, grounding: context.GroundingContext, acceptance: "{{item.acceptance}}", fallbackModel: BranchModel), context.AllowedModelIds), Inputs = AgentNodeMapping.BuildAgentInputs(context) }, }); // P4 (the plan-map integrated candidate): a repo-bound fan-out integrates its produced work into ONE @@ -136,7 +138,7 @@ private static IReadOnlyList BuildEdges(TaskBuildContext context return edges; } - /// The plan.author Config — always a FLAT plan (the parallel map cannot honor ordering), plus the launch's pinned planner model row + the operator's planner critic (reviewMode / reviewerModelId, omitted when off — byte-identical). + /// The plan.author Config — always a FLAT plan (the parallel map cannot honor ordering), plus the launch's pinned planner model row, the operator's allowed model pool (the catalog the planner allocates subtasks from; omitted when unbounded) + the operator's planner critic (reviewMode / reviewerModelId, omitted when off — byte-identical). private static JsonElement PlannerConfig(TaskBuildContext context) { var config = new Dictionary @@ -145,6 +147,7 @@ private static JsonElement PlannerConfig(TaskBuildContext context) }; AddIfPresent(config, "plannerModelId", context.PlannerModelRowId?.ToString()); + AddIfPresent(config, "allowedModelIds", context.AllowedModelIds is { Count: > 0 } pool ? pool.Select(id => id.ToString()).ToList() : null); AddIfPresent(config, "reviewMode", context.PlannerReviewMode != ReviewMode.None ? (int)context.PlannerReviewMode : null); AddIfPresent(config, "reviewerModelId", context.PlannerReviewMode != ReviewMode.None ? context.ReviewerModelId?.ToString() : null); // D① grounded plan review — a real read-only agent verifies the plan against the bound repository's tree. diff --git a/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/SingleAgent/SingleAgentDefinitionBuilder.cs b/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/SingleAgent/SingleAgentDefinitionBuilder.cs index 7f2db0b48..68f401588 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/SingleAgent/SingleAgentDefinitionBuilder.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/Projection/Builders/SingleAgent/SingleAgentDefinitionBuilder.cs @@ -92,10 +92,12 @@ private static IReadOnlyList RubricCriteria(TaskBuildContext context) // staked requirement rows never credit the operator with a shape-derived contract the server composed. // B2: the SHAPE decides the agent's mode (a question is not a coding run) and, absent an operator floor, // which oracle grades it — a code-shaped launch with no floor stays byte-identical (no mode, no oracle). - Config = AgentNodeMapping.WithCostCap(AgentNodeMapping.BuildAgentConfig(context.Seed.Goal, context.AgentProfile, mode: DeliverableShapes.AgentModeFor(context.Route.DeliverableShape), + // The operator's allowed model pool rides the node too, so the one agent is held to it at dispatch — the + // launch already clamped a pinned model outside it, and this covers the model a persona supplies. + Config = AgentNodeMapping.WithAllowedModels(AgentNodeMapping.WithCostCap(AgentNodeMapping.BuildAgentConfig(context.Seed.Goal, context.AgentProfile, mode: DeliverableShapes.AgentModeFor(context.Route.DeliverableShape), grounding: context.GroundingContext, acceptance: QuickAcceptance(context), criteria: context.AcceptanceCriteria, acceptanceAuthority: OperatorArgv(context) is null ? null : nameof(Messages.Contracts.ContractAuthority.Operator), - deliverablePath: DeliverablePath(context)), context.Route.Caps.MaxCostUsd), Inputs = AgentNodeMapping.BuildAgentInputs(context) }, + deliverablePath: DeliverablePath(context)), context.Route.Caps.MaxCostUsd), context.AllowedModelIds), Inputs = AgentNodeMapping.BuildAgentInputs(context) }, new() { Id = "done", TypeKey = "builtin.terminal", Label = "Done", Config = Empty(), Inputs = TerminalInputs(IsMultiRepo(context)) }, diff --git a/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRoutePreviewService.cs b/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRoutePreviewService.cs index d84e2b3a3..f82454aa1 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRoutePreviewService.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRoutePreviewService.cs @@ -17,14 +17,16 @@ public sealed class TaskRoutePreviewService : ITaskRoutePreviewService, IScopedD private readonly ITaskRouteSnapshotService _snapshots; private readonly ITaskProjectionRegistry _projections; private readonly IModeProfileRegistry _modeProfiles; + private readonly ILaunchControlResolver _controls; - public TaskRoutePreviewService(ITaskLaunchSeedProviderRegistry seedProviders, ILaunchRepositoryScopeGuard repositoryScope, ITaskRouteSnapshotService snapshots, ITaskProjectionRegistry projections, IModeProfileRegistry modeProfiles) + public TaskRoutePreviewService(ITaskLaunchSeedProviderRegistry seedProviders, ILaunchRepositoryScopeGuard repositoryScope, ITaskRouteSnapshotService snapshots, ITaskProjectionRegistry projections, IModeProfileRegistry modeProfiles, ILaunchControlResolver controls) { _seedProviders = seedProviders; _repositoryScope = repositoryScope; _snapshots = snapshots; _projections = projections; _modeProfiles = modeProfiles; + _controls = controls; } public async Task PreviewAsync(TaskLaunchRequest request, CancellationToken cancellationToken) @@ -35,10 +37,14 @@ public async Task PreviewAsync(TaskLaunchRequest request var preview = await _snapshots.CreateAsync(request, seed, cancellationToken).ConfigureAwait(false); + // The SAME resolution the launch refuses and clamps on — a preview cannot report a disposition the launch would not reach. + var controls = await _controls.ResolveAsync(request, preview.Route, cancellationToken).ConfigureAwait(false); + return preview with { AcceptanceCompatibility = DescribeAcceptance(preview.Route, request.RepositoryId ?? seed.RepositoryId), Posture = DescribePosture(request, seed, preview.Route), + ControlDispositions = controls.Dispositions, }; } @@ -79,7 +85,7 @@ private TaskAcceptanceCompatibility DescribeAcceptance(RoutePlan route, Guid? re var unknown = new TaskAcceptanceCompatibility { ProjectionKind = route.ProjectionKind, Detail = "The resolved route has not advertised an operator-command acceptance adapter; compatibility is unknown." }; if (!_projections.TryResolve(route.ProjectionKind, out var builder) || builder.OperatorAcceptance.AcceptsCommand is null) return unknown; var adapter = builder.OperatorAcceptance; - if (adapter.AcceptsCommand == false) return unknown with { State = TaskAcceptanceCompatibilityState.Incompatible, Detail = "The resolved route does not consume an operator-command acceptance floor. Its plan items use separate acceptance contracts." }; + if (adapter.AcceptsCommand == false) return unknown with { State = TaskAcceptanceCompatibilityState.Incompatible, Detail = LaunchControlResolver.FloorNotGradedReason }; if (adapter.GradingKind != BenchmarkGradingKind.TestsPass) return unknown with { Detail = "The resolved acceptance adapter does not advertise the proposed argv input format." }; var requiresRepository = !AgentAcceptanceContract.GradesFromDeliverables(new SupervisorAcceptanceSpec { Kind = adapter.GradingKind, Command = [] }); var compatible = !requiresRepository || repositoryId is not null; diff --git a/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRouteSnapshotService.cs b/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRouteSnapshotService.cs index 22a83a930..899f37644 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRouteSnapshotService.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/RoutePreview/TaskRouteSnapshotService.cs @@ -5,6 +5,7 @@ using CodeSpace.Core.Persistence.Entities; using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Tasks.Effort; +using CodeSpace.Core.Services.Tasks.Launch; using CodeSpace.Core.Services.Tasks.RoutePreview.Exceptions; using CodeSpace.Core.Services.Workflows; using CodeSpace.Messages.Commands.Tasks; @@ -20,14 +21,16 @@ public sealed class TaskRouteSnapshotService : ITaskRouteSnapshotService, IScope { public static readonly TimeSpan Lifetime = TimeSpan.FromMinutes(10); private readonly IEffortRouter _router; + private readonly ILaunchGroundingResolver _grounding; private readonly TaskRoutePolicyFingerprint _policy; private readonly CodeSpaceDbContext _db; private readonly IPostCommitActions _postCommit; private PendingCommitRecovery? _pendingCommitRecovery; - public TaskRouteSnapshotService(IEffortRouter router, TaskRoutePolicyFingerprint policy, CodeSpaceDbContext db, IPostCommitActions postCommit) + public TaskRouteSnapshotService(IEffortRouter router, ILaunchGroundingResolver grounding, TaskRoutePolicyFingerprint policy, CodeSpaceDbContext db, IPostCommitActions postCommit) { _router = router; + _grounding = grounding; _policy = policy; _db = db; _postCommit = postCommit; @@ -36,8 +39,13 @@ public TaskRouteSnapshotService(IEffortRouter router, TaskRoutePolicyFingerprint public async Task CreateAsync(TaskLaunchRequest request, TaskLaunchSeed seed, CancellationToken cancellationToken) { await ResolvePendingCommitAsync().ConfigureAwait(false); + + // The SAME grounding the launch resolves before it routes, so a follow-up turn is classified as one here too. It + // only feeds the router: the seed digest below still binds the provider's seed, as the launch's read validates it. + var grounding = await _grounding.ResolveAsync(request, seed, cancellationToken).ConfigureAwait(false); + var policy = _policy.Capture(); - var route = await _router.RouteAsync(TaskLaunchService.BuildRouteRequest(seed, request), cancellationToken).ConfigureAwait(false); + var route = await _router.RouteAsync(TaskLaunchService.BuildRouteRequest(seed with { GroundingContext = grounding }, request), cancellationToken).ConfigureAwait(false); if (policy != _policy.Capture()) throw new TaskRouteSnapshotMismatchException(); var now = await ReadDatabaseClockAsync(cancellationToken).ConfigureAwait(false); diff --git a/backend/src/CodeSpace.Core/Services/Tasks/TaskLaunchService.cs b/backend/src/CodeSpace.Core/Services/Tasks/TaskLaunchService.cs index 1fa63862d..5099c8b93 100644 --- a/backend/src/CodeSpace.Core/Services/Tasks/TaskLaunchService.cs +++ b/backend/src/CodeSpace.Core/Services/Tasks/TaskLaunchService.cs @@ -7,6 +7,7 @@ using CodeSpace.Core.Services.Workflows.Llm; using CodeSpace.Core.Services.Tasks.Effort; using CodeSpace.Core.Services.Tasks.Launch; +using CodeSpace.Core.Services.Tasks.Launch.Exceptions; using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Core.Services.Tasks.Contracts; using CodeSpace.Core.Services.Completion; @@ -25,7 +26,8 @@ namespace CodeSpace.Core.Services.Tasks; /// /// Default — a flat named-method pipeline (Rule 4/5): resolve the seed provider by -/// the open surface kind → seed → validate the repo TEAM-SCOPED (fail-closed) → route → build the agent profile → +/// the open surface kind → seed → validate the repo TEAM-SCOPED (fail-closed) → ground → route → resolve every +/// route-dependent control (applied / clamped / not applicable, or the launch is refused) → build the agent profile → /// project + start the snapshot run → return the handle + route. Holds no per-surface logic: the ONLY surface /// dispatch is _seedProviders.Resolve(surfaceKind), and the core NEVER reads the surface payload (only the /// resolved provider does), so a new surface plugs in by registering a provider with zero edit here (the generic @@ -39,8 +41,8 @@ public sealed class TaskLaunchService : ITaskLaunchService, IScopedDependency private readonly ITaskRouteSnapshotService _routeSnapshots; private readonly ITaskRunSnapshotFactory _factory; private readonly IWorkSessionService _sessions; - private readonly ISessionContextBuilder _sessionContext; - private readonly ISessionSummarizer _sessionSummarizer; + private readonly ILaunchGroundingResolver _grounding; + private readonly ILaunchControlResolver _controls; private readonly ISessionBranchResolver _sessionBranches; private readonly ILaunchBasePinResolver _basePins; private readonly IModelPoolSelector _modelSelector; @@ -48,7 +50,7 @@ public sealed class TaskLaunchService : ITaskLaunchService, IScopedDependency private readonly CodeSpaceDbContext _db; private readonly ILogger _logger; - public TaskLaunchService(ITaskLaunchSeedProviderRegistry seedProviders, ILaunchRepositoryScopeGuard repositoryScope, IEffortRouter router, ITaskRouteSnapshotService routeSnapshots, ITaskRunSnapshotFactory factory, IWorkSessionService sessions, ISessionContextBuilder sessionContext, ISessionSummarizer sessionSummarizer, ISessionBranchResolver sessionBranches, ILaunchBasePinResolver basePins, IModelPoolSelector modelSelector, ILLMClientRegistry llm, CodeSpaceDbContext db, ILogger logger) + public TaskLaunchService(ITaskLaunchSeedProviderRegistry seedProviders, ILaunchRepositoryScopeGuard repositoryScope, IEffortRouter router, ITaskRouteSnapshotService routeSnapshots, ITaskRunSnapshotFactory factory, IWorkSessionService sessions, ILaunchGroundingResolver grounding, ILaunchControlResolver controls, ISessionBranchResolver sessionBranches, ILaunchBasePinResolver basePins, IModelPoolSelector modelSelector, ILLMClientRegistry llm, CodeSpaceDbContext db, ILogger logger) { _seedProviders = seedProviders; _repositoryScope = repositoryScope; @@ -56,8 +58,8 @@ public TaskLaunchService(ITaskLaunchSeedProviderRegistry seedProviders, ILaunchR _routeSnapshots = routeSnapshots; _factory = factory; _sessions = sessions; - _sessionContext = sessionContext; - _sessionSummarizer = sessionSummarizer; + _grounding = grounding; + _controls = controls; _sessionBranches = sessionBranches; _basePins = basePins; _modelSelector = modelSelector; @@ -78,19 +80,28 @@ public async Task LaunchAsync(TaskLaunchRequest request, Cance var preview = request.RouteSnapshotId is not null ? await _routeSnapshots.ReadAsync(request, seed, cancellationToken).ConfigureAwait(false) : null; if (preview?.PreviousResult is { } previous) return previous; - var route = preview?.Route ?? await _router.RouteAsync(BuildRouteRequest(seed, request), cancellationToken).ConfigureAwait(false); + + await EnsureSessionCanContinueAsync(request, cancellationToken).ConfigureAwait(false); + + // On a CONTINUE, prime the run with the thread's prior-turn digest — the projection folds this grounding into + // the agent's prompt so the follow-up builds on earlier work. Resolved BEFORE routing: the classifier reads the + // same grounding, and a follow-up it cannot see is classified as a fresh task. A fresh launch carries only the + // seed's own grounding (so its route request is unchanged). + var grounding = await _grounding.ResolveAsync(request, seed, cancellationToken).ConfigureAwait(false); + + var route = preview?.Route ?? await _router.RouteAsync(BuildRouteRequest(seed with { GroundingContext = grounding }, request), cancellationToken).ConfigureAwait(false); EnsureRouteConfirmed(route); - EnsureAcceptanceMandate(request, route); + // Every route-dependent control the operator set is applied, clamped, named as not applicable, or refused — the + // same resolution the route preview reports. A refusal stops the launch here, before any session or run exists. + var controls = await _controls.ResolveAsync(request, route, cancellationToken).ConfigureAwait(false); - await EnsureSessionCanContinueAsync(request, cancellationToken).ConfigureAwait(false); + EnsureControlsHonoured(controls); - var profile = BuildAgentProfile(request, seed, route); + EnsureAcceptanceMandate(request, controls.GradesOperatorFloor); - // On a CONTINUE, prime the run with the thread's prior-turn digest — the projection folds this grounding into - // the agent's prompt so the follow-up builds on earlier work. A fresh launch carries only the seed's own grounding. - var grounding = await ResolveGroundingAsync(request, seed, cancellationToken).ConfigureAwait(false); + var profile = ClampAgentModel(BuildAgentProfile(request, seed, route), controls.ModelClamp); // …and clone EACH repo (primary + related) at the prior turn's produced branch for it, so the follow-up builds // on earlier CODE (not just the narrative). Empty on a fresh launch / no repo / no prior branch ⇒ default branches. @@ -115,10 +126,23 @@ public async Task LaunchAsync(TaskLaunchRequest request, Cance // Potential model/Git preparation above runs without holding the snapshot row lock. The callback below // performs only transactional session/run staging and is entered once across competing workers. return request.RouteSnapshotId is not null - ? await _routeSnapshots.ConsumeAsync(new TaskRouteSnapshotConsumption(request, seed, () => StageAsync(request, context, cancellationToken)), cancellationToken).ConfigureAwait(false) - : await StageAsync(request, context, cancellationToken).ConfigureAwait(false); + ? await _routeSnapshots.ConsumeAsync(new TaskRouteSnapshotConsumption(request, seed, () => StageAsync(request, context, controls.Dispositions, cancellationToken)), cancellationToken).ConfigureAwait(false) + : await StageAsync(request, context, controls.Dispositions, cancellationToken).ConfigureAwait(false); + } + + /// A control the resolved route refuses stops the launch before any session or run exists, named with its reason — never silently dropped. + private static void EnsureControlsHonoured(LaunchControlResolution controls) + { + var refused = controls.Dispositions.Where(d => d.Outcome == LaunchControlOutcome.Refused).ToList(); + + if (refused.Count > 0) + throw new TaskLaunchControlRefusedException(refused); } + /// The single agent's model when its pinned one fell outside the allowed pool: the pool's default row — its model on its own credential — so the frozen agent config shows the model that actually runs. No clamp ⇒ the profile verbatim. + private static ResolvedAgentProfile ClampAgentModel(ResolvedAgentProfile profile, ModelDispatchRef? clamp) => + clamp is null ? profile : profile with { Model = clamp.ModelId, ModelCredentialId = clamp.ModelCredentialId, ModelCredentialModelId = null }; + /// /// A low-confidence or risky auto route is a question, not authorization to execute. The router owns the generic /// decision and its registry-derived choices; this single launch chokepoint enforces it for HTTP, jobs and future @@ -138,7 +162,7 @@ private async Task EnsureSessionCanContinueAsync(TaskLaunchRequest request, Canc if (status != WorkSessionStatus.Open) throw new InvalidOperationException($"Session {id} is {status} and cannot take a new turn."); } - private async Task StageAsync(TaskLaunchRequest request, TaskBuildContext context, CancellationToken cancellationToken) + private async Task StageAsync(TaskLaunchRequest request, TaskBuildContext context, IReadOnlyList dispositions, CancellationToken cancellationToken) { var seed = context.Seed; var route = context.Route; @@ -171,6 +195,7 @@ private async Task StageAsync(TaskLaunchRequest request, TaskB Route = route, SurfaceKind = seed.SurfaceKind, LinkedEntity = seed.LinkedEntity, + ControlDispositions = dispositions, }; } @@ -245,29 +270,6 @@ private static string SelectionCapability(TaskLaunchRequest request, TaskLaunchS private sealed record StructuredBrainSelection(Guid RowId, bool PinIneligible, bool Pinned, ModelSelectionReceipt Receipt); - /// The grounding the run is primed with: on a CONTINUE, the session's prior-turn digest composed over any seed grounding; on a fresh launch, only the seed's own grounding (null for chat). The projection folds this into the agent prompt. - private async Task ResolveGroundingAsync(TaskLaunchRequest request, TaskLaunchSeed seed, CancellationToken cancellationToken) - { - if (request.ContinueSessionId is not { } sessionId) return seed.GroundingContext; - - // Fold any turns that scrolled out of the recent window into the thread's rolling summary BEFORE building the - // digest, so a long thread's early context is preserved. Best-effort + fail-open (no model / error leaves it). - await _sessionSummarizer.EnsureSummaryUpToDateAsync(sessionId, request.TeamId, cancellationToken).ConfigureAwait(false); - - var priorTurns = await _sessionContext.BuildAsync(sessionId, request.TeamId, cancellationToken).ConfigureAwait(false); - - return ComposeGrounding(priorTurns, seed.GroundingContext); - } - - /// Join the prior-turn digest and the seed's own grounding (either may be absent) into one block, digest first. - private static string? ComposeGrounding(string? priorTurns, string? seedGrounding) - { - if (string.IsNullOrWhiteSpace(priorTurns)) return seedGrounding; - if (string.IsNullOrWhiteSpace(seedGrounding)) return priorTurns; - - return $"{priorTurns}\n\n{seedGrounding}"; - } - /// On a CONTINUE, the prior turn's produced branch for EACH repo the run touches (primary + related) — the projection clones each repo's workspace at its own ref. Empty on a fresh launch, an analysis-only run (no repos), or when no prior turn produced a branch for any (⇒ default branches — the safe fallback). A repo absent from the map clones at its default. private async Task> ResolveBaseRefsAsync(TaskLaunchRequest request, TaskLaunchSeed seed, ResolvedAgentProfile profile, CancellationToken cancellationToken) { @@ -333,7 +335,7 @@ private async Task EnsureAgentDefinitionsInTeamAsync(TaskLaunchRequest request, throw new KeyNotFoundException($"Agent {string.Join(", ", ids.Except(inTeam))} not found or not accessible."); } - /// Maps the seed + the operator's effort/recipe/autonomy + safety-budget caps onto the router input. The router TIGHTENS the effort preset's caps with CapsOverride (null ⇒ preset-only, byte-identical). Internal (not private) so the read-only route PREVIEW routes through the SAME mapping — a preview that built its own request would be free to drift from the launch it claims to predict. + /// Maps the seed + the operator's effort/recipe/autonomy + safety-budget caps + acceptance floor onto the router input. The router TIGHTENS the effort preset's caps with CapsOverride (null ⇒ preset-only, byte-identical) and keeps an auto route with an operator floor on a projection that grades it. Internal (not private) so the read-only route PREVIEW routes through the SAME mapping — a preview that built its own request would be free to drift from the launch it claims to predict. internal static EffortRouteRequest BuildRouteRequest(TaskLaunchSeed seed, TaskLaunchRequest request) => new() { Seed = seed, @@ -341,6 +343,7 @@ private async Task EnsureAgentDefinitionsInTeamAsync(TaskLaunchRequest request, RequestedRecipe = request.RequestedRecipe, CapsOverride = request.CapsOverride, DeliverableShape = request.DeliverableShape, + HasOperatorFloor = request.AcceptanceChecks is { Count: > 0 }, }; /// Pure mapping: the request overrides + (seed repo ?? request repo) + each related repo + the CLAMPED autonomy → the agent envelope the projection stamps. Every field optional, folding to agent.run's own defaults. Related repos require a primary (fail-loud, mirroring the agent.run node — a workspace has nowhere to anchor without one). Internal (not private) so the clamp + related-repo choke point is unit-pinned directly (InternalsVisibleTo), not only through integration coverage. @@ -410,20 +413,20 @@ private static string ClampAutonomy(TaskLaunchRequest request, RoutePlan route) } /// - /// P3.2: Delivery/Unattended quality on a SUPERVISOR-projected launch (the only projection AcceptanceChecks - /// already has any effect on — S4b) MUST carry an executable acceptance floor, so a caller cannot claim - /// Delivery-grade verification while skipping the one lever that actually gates the terminal stop. Prototype (or - /// an unset tier) is unchanged — self-report stands, byte-identical. Inert on a non-supervisor projection — this - /// doesn't invent new acceptance-floor plumbing for single-agent/plan-map launches (AcceptanceChecks has no - /// effect there today either); such a launch still gets its output-review floor () - /// even though this specific mandate is inert for it. There is no sensible SERVER-SYNTHESIZED check to default to - /// (the argv is domain-specific) — an omission fails LOUD here rather than silently shipping ungated, mirroring + /// P3.2: Delivery/Unattended quality on a launch whose route GRADES an operator floor (its builder advertises an + /// operator-command adapter — the supervisor's terminal stop and the single agent's own oracle both do) MUST carry + /// an executable acceptance floor, so a caller cannot claim Delivery-grade verification while skipping the one lever + /// that actually gates the result. A plan-map route grades no operator floor — its plan items carry their own + /// contracts, and a floor sent to it is refused before this point — so the mandate asks nothing of it; it still gets + /// its output-review floor (). Prototype (or an unset tier) is unchanged — + /// self-report stands, byte-identical. There is no sensible SERVER-SYNTHESIZED check to default to (the argv is + /// domain-specific) — an omission fails LOUD here rather than silently shipping ungated, mirroring /// ILaunchRepositoryScopeGuard's fail-closed launch-time rejection shape (before the session opens). /// - internal static void EnsureAcceptanceMandate(TaskLaunchRequest request, RoutePlan route) + internal static void EnsureAcceptanceMandate(TaskLaunchRequest request, bool gradesOperatorFloor) { if (request.Tier is not (QualityTier.Delivery or QualityTier.Unattended)) return; - if (route.ProjectionKind != TaskProjectionKinds.Supervisor) return; + if (!gradesOperatorFloor) return; if (request.AcceptanceChecks is { Count: > 0 }) return; throw new ArgumentException($"{request.Tier} quality requires an executable acceptance check (acceptanceChecks) — author one, or launch at Prototype quality for a self-reported result."); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs index 9bb4629b7..70cf780c0 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentCodeNode.cs @@ -148,6 +148,8 @@ public Task RunAsync(NodeRunContext context, CancellationToken cance if (!TryReadModelCredentialModelId(context, out var modelCredentialModelId)) return Fail("Config 'modelCredentialModelId' must be a credentialed-model id (uuid)."); + if (!TryReadAllowedModelIds(context.Config, out var allowedModelIds)) return Fail("Config 'allowedModelIds' must be an array of credentialed-model ids (uuid)."); + if (string.IsNullOrWhiteSpace(harness)) return Fail("Config 'harness' is required."); // A persona supplies the prompt floor (its system prompt), so 'goal' is only required without one. @@ -182,6 +184,7 @@ public Task RunAsync(NodeRunContext context, CancellationToken cance AgentDefinitionId = agentDefinitionId, ModelCredentialId = modelCredentialId, ModelCredentialModelId = modelCredentialModelId, + AllowedModelIds = allowedModelIds, Tools = ReadStringArray(context.Config, "tools"), RepositoryId = repositoryId, Workspace = workspace, @@ -686,6 +689,25 @@ private static bool TryReadModelCredentialModelId(NodeRunContext context, out Gu return true; } + /// Read the optional allowedModelIds config — the operator's allowed model pool a task-launch projection bakes, as string uuids. Absent / empty → null (unbounded, byte-identical). Any entry that is not a uuid → false (a clean node failure): a pool is a bound, and one that half-parsed would silently bound the run to a different set of models — or, emptied, to none at all. + private static bool TryReadAllowedModelIds(IReadOnlyDictionary config, out IReadOnlyList? allowedModelIds) + { + allowedModelIds = null; + + if (!config.TryGetValue("allowedModelIds", out var raw) || raw.ValueKind == JsonValueKind.Null) return true; + if (raw.ValueKind != JsonValueKind.Array) return false; + + var ids = new List(); + foreach (var entry in raw.EnumerateArray()) + { + if (entry.ValueKind != JsonValueKind.String || !Guid.TryParse(entry.GetString(), out var id)) return false; + ids.Add(id); + } + + allowedModelIds = ids.Count > 0 ? ids : null; + return true; + } + /// Read the optional baseRef input — the branch/ref to clone the primary repo at (session branch continuity). Absent / blank / non-string → null (the repo default). private static string? ReadBaseRef(NodeRunContext context) => context.Inputs.TryGetValue("baseRef", out var v) && v.ValueKind == JsonValueKind.String && !string.IsNullOrWhiteSpace(v.GetString()) diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/PlanAuthorNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/PlanAuthorNode.cs index 1d3389f35..322598494 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/PlanAuthorNode.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/PlanAuthorNode.cs @@ -240,6 +240,7 @@ internal static WorkflowPlanRequest BuildPlanRequest(IReadOnlyDictionaryA string-uuid array config read (the allowed model pool a projection bakes) — defensive per the node convention: absent / non-array ⇒ null, a non-uuid entry skipped, nothing left ⇒ null. It bounds only the catalog the planner allocates from; each branch's own agent.run is where the pool is enforced. + private static IReadOnlyList? ReadGuids(IReadOnlyDictionary config, string key) + { + var ids = ReadStringArray(config, key).Select(raw => Guid.TryParse(raw, out var id) ? id : (Guid?)null).OfType().ToList(); + + return ids.Count > 0 ? ids : null; + } + /// Defensive bool config read — absent / non-bool ⇒ false, mirroring the node convention. private static bool ReadBool(IReadOnlyDictionary config, string key) => config.TryGetValue(key, out var value) && value.ValueKind == JsonValueKind.True; diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Planning/Planners/LlmWorkflowPlanner.cs b/backend/src/CodeSpace.Core/Services/Workflows/Planning/Planners/LlmWorkflowPlanner.cs index edd5b3ed0..70335c4d5 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Planning/Planners/LlmWorkflowPlanner.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Planning/Planners/LlmWorkflowPlanner.cs @@ -57,10 +57,10 @@ public async Task PlanAsync(WorkflowPlanRequest request, Cancel var (structured, pick) = pickedBrain; - // P2 — render the capability catalog (harnesses + drivable providers, the team's whole credentialed pool) so the - // planner allocates a provider-compatible harness + model PER subtask informed, not blind. The run-time - // reconciler is the backstop. - var pool = await _modelSelector.ListPoolAsync(request.TeamId, allowedRowIds: null, cancellationToken).ConfigureAwait(false); + // P2 — render the capability catalog (harnesses + drivable providers, the team's credentialed pool bounded to the + // operator's allowed models when the launch set any) so the planner allocates a provider-compatible harness + model + // PER subtask informed, not blind. The run-time reconciler is the backstop — for the pool too. + var pool = await _modelSelector.ListPoolAsync(request.TeamId, request.AllowedModelIds, cancellationToken).ConfigureAwait(false); var catalog = CapabilityCatalog.Render(_harnesses.All, pool); // D2 (cross-run learning): the distilled lessons ride the plan prompt — under a deterministic, toggle-free diff --git a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs index 37d3cac9e..c8970c5db 100644 --- a/backend/src/CodeSpace.Messages/Agents/AgentTask.cs +++ b/backend/src/CodeSpace.Messages/Agents/AgentTask.cs @@ -67,6 +67,16 @@ public sealed record AgentTask [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] public IReadOnlyList? AllowedHarnessKinds { get; init; } + /// + /// The credentialed-model ROW ids this run's model must come from — the model analogue of . + /// At dispatch HarnessModelReconciler runs the named model on its pooled row, and a model outside the pool (or + /// none at all) on the pool's default row, naming the move on the run; the run then uses that row's credential. + /// Null / empty (the default, and every task envelope persisted before this field) = UNBOUNDED, byte-identical. + /// The task-launch projections stamp it from the operator's allowed model pool; the supervisor gates its own spawns. + /// + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] + public IReadOnlyList? AllowedModelIds { get; init; } + /// Model id within the chosen harness's catalog, or null/blank to let the harness pick its own default (the Model=empty rule). public string? Model { get; init; } diff --git a/backend/src/CodeSpace.Messages/Commands/Tasks/LaunchTaskCommand.cs b/backend/src/CodeSpace.Messages/Commands/Tasks/LaunchTaskCommand.cs index 1d92c7155..8d9c09da8 100644 --- a/backend/src/CodeSpace.Messages/Commands/Tasks/LaunchTaskCommand.cs +++ b/backend/src/CodeSpace.Messages/Commands/Tasks/LaunchTaskCommand.cs @@ -40,4 +40,7 @@ public sealed record LaunchTaskResult /// The external entity the task was launched from, when the seed carried one. public LinkedEntityRef? LinkedEntity { get; init; } + + /// What the route did with each route-dependent operator control the launch carried — applied, clamped or not applicable, with the reason. A refused control never reaches a result: it stops the launch. Empty when the launch carried none. + public IReadOnlyList ControlDispositions { get; init; } = []; } diff --git a/backend/src/CodeSpace.Messages/Commands/Tasks/TaskLaunchInput.cs b/backend/src/CodeSpace.Messages/Commands/Tasks/TaskLaunchInput.cs index a18553a9b..6922dd4ad 100644 --- a/backend/src/CodeSpace.Messages/Commands/Tasks/TaskLaunchInput.cs +++ b/backend/src/CodeSpace.Messages/Commands/Tasks/TaskLaunchInput.cs @@ -82,9 +82,10 @@ public abstract record TaskLaunchInput /// credentialed-model ROW ids (ModelCredentialModel ids, NOT model names). EVERY id is validated TEAM-SCOPED /// by the service (fail-closed, exactly like the repos): a foreign / disabled / deleted-credential row rejects the /// whole launch. Baked into the projected agent.supervisor node's allowedModelIds, where every - /// dispatched agent's model must resolve to a row in the pool (out of pool ⇒ fails closed at dispatch). Null / - /// empty ⇒ the pool is ALL the team's credentialed models (byte-identical to no pool). Inert on a non-supervisor - /// projection (single-agent / map ignore it). + /// dispatched agent's model must resolve to a row in the pool (out of pool ⇒ fails closed at dispatch). A plan-map + /// run offers its planner only the pool and holds every branch to it; a single-agent run holds its one agent to it + /// (a pinned model outside it is clamped to the pool's default). Null / empty ⇒ the pool is ALL the team's + /// credentialed models (byte-identical to no pool). Its disposition on the route is reported on the launch result. /// public IReadOnlyList? AllowedModelIds { get; init; } @@ -94,8 +95,9 @@ public abstract record TaskLaunchInput /// validated TEAM-SCOPED by the service (fail-closed, exactly like the repos / model pool): a foreign / deleted /// persona rejects the whole launch. Baked into the projected agent.supervisor node's /// allowedAgentDefinitionIds, where every dispatched agent's effective persona (model-authored slug OR the - /// profile default) must be in the pool (out of pool ⇒ fails closed at dispatch). Null / empty ⇒ the pool is ALL - /// the team's personas (byte-identical to no pool). Inert on a non-supervisor projection. + /// profile default) must be in the pool (out of pool ⇒ fails closed at dispatch). On a plan-map / single-agent run + /// the launch's own persona is the only one any agent runs as, so a persona outside the pool refuses the launch. + /// Null / empty ⇒ the pool is ALL the team's personas (byte-identical to no pool). /// public IReadOnlyList? AllowedAgentDefinitionIds { get; init; } @@ -132,7 +134,7 @@ public abstract record TaskLaunchInput /// How an INDEPENDENT critic reviews the AUTHORED PLAN — tier-generic (S4e): the plan-map tiers bake it into plan.author/plan.confirm's reviewMode; Deep bakes it into the supervisor's plan-scoped planReviewMode (plans only — never a critic call on every spawn/merge/stop). None (default, byte-identical) / Gate / Improve. Runs on when set. Inert on quick (no plan). public ReviewMode PlannerReviewMode { get; init; } = ReviewMode.None; - /// The operator's EXECUTABLE acceptance floor (S4b, Deep only) — an argv (e.g. ["sh","check.sh"]) run against the run's reviewable head at the terminal stop; a non-zero exit fails the stop and withholds the branch. DISTINCT from the free-text (prompt-rendered, never executed). Null / empty ⇒ omitted ⇒ byte-identical. Inert on a non-supervisor projection. + /// The operator's EXECUTABLE acceptance floor (S4b) — an argv (e.g. ["sh","check.sh"]) graded on every route that advertises an operator-command adapter: Deep runs it against the run's reviewable head at the terminal stop (a non-zero exit fails the stop and withholds the branch), Quick grades its single agent with it. A plan-map route grades none, so a floor sent there refuses the launch — and on Auto it keeps the route off plan-map. DISTINCT from the free-text (prompt-rendered, never executed). Null / empty ⇒ omitted ⇒ byte-identical. public IReadOnlyList? AcceptanceChecks { get; init; } /// diff --git a/backend/src/CodeSpace.Messages/Dtos/Workflows/Planning/WorkflowPlanRequest.cs b/backend/src/CodeSpace.Messages/Dtos/Workflows/Planning/WorkflowPlanRequest.cs index 128f07921..94ec5cac6 100644 --- a/backend/src/CodeSpace.Messages/Dtos/Workflows/Planning/WorkflowPlanRequest.cs +++ b/backend/src/CodeSpace.Messages/Dtos/Workflows/Planning/WorkflowPlanRequest.cs @@ -67,6 +67,9 @@ public sealed record WorkflowPlanRequest /// public Guid? BrainModelId { get; init; } + /// The operator's allowed model pool (credentialed-model ROW ids) the plan's subtasks are allocated from — the capability catalog the planner reads is bounded to it, so a per-subtask model is picked from the pool. It bounds the catalog only; the planner's own brain is . Null / empty ⇒ the whole team pool (byte-identical). + public IReadOnlyList? AllowedModelIds { get; init; } + /// Whether an INDEPENDENT reviewer model gates / improves the plan (the CriticPlannerDecorator). Default ⇒ no review (byte-identical). public ReviewMode Review { get; init; } = ReviewMode.None; diff --git a/backend/src/CodeSpace.Messages/Failures/FailureCodes.cs b/backend/src/CodeSpace.Messages/Failures/FailureCodes.cs index dd6d551b8..4cfb23b1b 100644 --- a/backend/src/CodeSpace.Messages/Failures/FailureCodes.cs +++ b/backend/src/CodeSpace.Messages/Failures/FailureCodes.cs @@ -56,6 +56,9 @@ public static class FailureCodes public const string WorkflowDefinitionInvalid = "workflow_definition_invalid"; public const string TaskRouteConfirmationRequired = "task_route_confirmation_required"; public const string TaskRouteSnapshotMismatch = "task_route_snapshot_mismatch"; + + /// The resolved route cannot honour a launch control the operator set (an acceptance check a plan-map route cannot grade, a persona the allowed pool excludes, …), so the launch stopped before any session or run. The details name each control and why. Remedy: drop the control or choose a route that consumes it — a retry of the same input is refused again. + public const string TaskLaunchControlRefused = "task_launch_control_refused"; public const string WorkspaceUnresolvable = "workspace_unresolvable"; public const string RerunAlreadyInProgress = "rerun_already_in_progress"; public const string RerunTargetInvalid = "rerun_target_invalid"; diff --git a/backend/src/CodeSpace.Messages/Tasks/Effort/EffortPolicy.cs b/backend/src/CodeSpace.Messages/Tasks/Effort/EffortPolicy.cs index 02510a83e..5121156a7 100644 --- a/backend/src/CodeSpace.Messages/Tasks/Effort/EffortPolicy.cs +++ b/backend/src/CodeSpace.Messages/Tasks/Effort/EffortPolicy.cs @@ -29,6 +29,18 @@ public static string Decide(EffortSignals signals, string? requestedEffort) return Rows.First(row => row.Matches(signals)).Mode; } + /// + /// The tier for among the tiers accepts: the table's first row + /// that both matches and is admitted, so a set-aside tier falls to the next matching row down. When no admitted + /// row matches, the unconstrained pick stands and the caller decides what a tier it cannot use means. + /// + public static string Decide(EffortSignals signals, string? requestedEffort, Func admits) + { + if (IsExplicitOperatorTier(requestedEffort)) return requestedEffort!.Trim(); + + return Rows.FirstOrDefault(row => row.Matches(signals) && admits(row.Mode))?.Mode ?? Decide(signals, requestedEffort); + } + /// An operator tier is explicit when it is non-blank and not the "auto" sentinel — those are honoured verbatim, the rest classify. private static bool IsExplicitOperatorTier(string? requestedEffort) => !string.IsNullOrWhiteSpace(requestedEffort) && !string.Equals(requestedEffort.Trim(), TaskEffortModes.Auto, StringComparison.OrdinalIgnoreCase); diff --git a/backend/src/CodeSpace.Messages/Tasks/Effort/EffortRouteRequest.cs b/backend/src/CodeSpace.Messages/Tasks/Effort/EffortRouteRequest.cs index 1b722b744..da87ea084 100644 --- a/backend/src/CodeSpace.Messages/Tasks/Effort/EffortRouteRequest.cs +++ b/backend/src/CodeSpace.Messages/Tasks/Effort/EffortRouteRequest.cs @@ -38,4 +38,12 @@ public sealed record EffortRouteRequest /// own reading on the auto path, and the code default on the explicit one — byte-identical. /// public string? DeliverableShape { get; init; } + + /// + /// The launch carries an operator acceptance floor (an executable acceptanceChecks argv) — a routing signal: + /// the AUTO path must land on a projection that grades it, so a classified tier whose projection cannot is set aside + /// for the next tier the policy admits. An explicit tier or a pinned recipe / projection is the operator's own + /// choice and is never moved; the launch refuses that combination with its reason instead. False ⇒ unconstrained. + /// + public bool HasOperatorFloor { get; init; } } diff --git a/backend/src/CodeSpace.Messages/Tasks/LaunchControlDisposition.cs b/backend/src/CodeSpace.Messages/Tasks/LaunchControlDisposition.cs new file mode 100644 index 000000000..0258ecc04 --- /dev/null +++ b/backend/src/CodeSpace.Messages/Tasks/LaunchControlDisposition.cs @@ -0,0 +1,35 @@ +using System.Text.Json.Serialization; + +namespace CodeSpace.Messages.Tasks; + +/// What the resolved route did with one operator launch control. Every control a launch carries lands in exactly one of these — never silently dropped. +[JsonConverter(typeof(JsonStringEnumConverter))] +public enum LaunchControlOutcome { Applied, Clamped, NotApplicable, Refused } + +/// +/// One operator launch control's disposition on the resolved route (Rule 18.1, a pure data noun). The launch and its +/// route preview compute it through the same service, so the preview states what the launch will do; a +/// control stops the launch before any session or run exists. +/// +public sealed record LaunchControlDisposition +{ + /// The control's wire name — a value. + public required string Control { get; init; } + + public required LaunchControlOutcome Outcome { get; init; } + + /// Why the route clamped, set aside or refused the control; for an applied one, how it applies when that is not obvious. Null when there is nothing to add. + public string? Reason { get; init; } +} + +/// The launch controls whose effect depends on the route, by their wire names on the launch input. A control every route consumes the same way has no disposition to report. +public static class LaunchControls +{ + public const string AllowedModelIds = "allowedModelIds"; + public const string AllowedAgentDefinitionIds = "allowedAgentDefinitionIds"; + public const string AcceptanceChecks = "acceptanceChecks"; + public const string DecisionReviewMode = "decisionReviewMode"; + public const string DeliverySpec = "deliverySpec"; + public const string RequirePlanConfirmation = "requirePlanConfirmation"; + public const string PlannerReviewMode = "plannerReviewMode"; +} diff --git a/backend/src/CodeSpace.Messages/Tasks/TaskBuildContext.cs b/backend/src/CodeSpace.Messages/Tasks/TaskBuildContext.cs index 66b7245ad..2e68117f9 100644 --- a/backend/src/CodeSpace.Messages/Tasks/TaskBuildContext.cs +++ b/backend/src/CodeSpace.Messages/Tasks/TaskBuildContext.cs @@ -56,7 +56,7 @@ public sealed record TaskBuildContext /// True when the brain row is the operator's own HONORED pin — baked so the decider knows it must resolve verbatim (never fail over). public bool SupervisorBrainModelPinned { get; init; } - /// The operator's allowed model pool (credentialed-model ROW ids) for the agents a Deep run dispatches, validated TEAM-SCOPED at launch — the SupervisorDefinitionBuilder bakes it into the node's allowedModelIds, where a dispatched model out of the pool fails closed. Null / empty ⇒ the pool is all the team's models (the builder omits the key — byte-identical). Inert on a non-supervisor projection (its builder never reads this). + /// The operator's allowed model pool (credentialed-model ROW ids), validated TEAM-SCOPED at launch — the SupervisorDefinitionBuilder bakes it into the node's allowedModelIds, where a dispatched model out of the pool fails closed; the plan-map builders bake it into the planner (its catalog) and every branch's agent.run, and the single-agent builder into its one agent.run, where dispatch holds the model to the pool. Null / empty ⇒ the pool is all the team's models (every builder omits the key — byte-identical). public IReadOnlyList? AllowedModelIds { get; init; } /// The operator's allowed AGENT (persona) pool (AgentDefinition ROW ids), validated TEAM-SCOPED at launch — the SupervisorDefinitionBuilder bakes it into the node's allowedAgentDefinitionIds, where a dispatched persona out of the pool fails closed. Null / empty ⇒ all the team's personas (builder omits the key — byte-identical). Inert on a non-supervisor projection. @@ -80,7 +80,7 @@ public sealed record TaskBuildContext /// How an INDEPENDENT critic reviews the AUTHORED PLAN — tier-generic (S4e): the plan-map builders bake it into plan.author/plan.confirm's reviewMode; the supervisor builder bakes it into the plan-scoped planReviewMode. (the default) ⇒ omitted (byte-identical). Inert on quick. public ReviewMode PlannerReviewMode { get; init; } = ReviewMode.None; - /// The operator's EXECUTABLE acceptance floor (S4b) — an argv (e.g. ["sh","check.sh"]) the SupervisorDefinitionBuilder bakes into the node's acceptanceChecks, enforced at the terminal stop (a non-zero exit fails the stop + withholds the reviewable head). Null / empty ⇒ omitted (byte-identical). Inert on a non-supervisor projection. + /// The operator's EXECUTABLE acceptance floor (S4b) — an argv (e.g. ["sh","check.sh"]) the SupervisorDefinitionBuilder bakes into the node's acceptanceChecks, enforced at the terminal stop (a non-zero exit fails the stop + withholds the reviewable head), and the single-agent builder into its agent's own oracle. Null / empty ⇒ omitted (byte-identical). The plan-map builders grade none — the launch refuses a floor before it reaches them. public IReadOnlyList? AcceptanceChecks { get; init; } /// How an INDEPENDENT critic reviews each supervisor decision — the SupervisorDefinitionBuilder bakes it into the node's decisionReviewMode. (the default) ⇒ the builder omits the key (byte-identical). Inert on a non-supervisor projection. diff --git a/backend/src/CodeSpace.Messages/Tasks/TaskLaunchRequest.cs b/backend/src/CodeSpace.Messages/Tasks/TaskLaunchRequest.cs index 0016a70ac..e6c139495 100644 --- a/backend/src/CodeSpace.Messages/Tasks/TaskLaunchRequest.cs +++ b/backend/src/CodeSpace.Messages/Tasks/TaskLaunchRequest.cs @@ -62,10 +62,10 @@ public sealed record TaskLaunchRequest /// The operator's safety-budget caps projected onto the router's CapsOverride seam (the numeric caps only — autonomy/approval merge tighten-only). Null ⇒ the effort preset's caps stand. The router TIGHTENS the preset with a set cap; a cost cap force-stops the run via the supervisor's bounds. public RouteCaps? CapsOverride { get; init; } - /// The operator's allowed model pool (credentialed-model ROW ids) for the agents a Deep run dispatches — validated TEAM-SCOPED (fail-closed) + baked into the supervisor node's allowedModelIds. Null / empty ⇒ all the team's models (byte-identical). Inert on a non-supervisor projection. + /// The operator's allowed model pool (credentialed-model ROW ids) the run's agents are held to — validated TEAM-SCOPED (fail-closed). The supervisor gates every spawn inside it; plan-map offers its planner only the pool and holds every branch to it; single-agent holds its one agent to it (a pinned model outside it is clamped to the pool's default). Null / empty ⇒ all the team's models (byte-identical). Its disposition on the route is reported on the launch result. public IReadOnlyList? AllowedModelIds { get; init; } - /// The operator's allowed AGENT (persona) pool (AgentDefinition ROW ids) for the agents a Deep run dispatches — validated TEAM-SCOPED (fail-closed) + baked into the supervisor node's allowedAgentDefinitionIds. Null / empty ⇒ all the team's personas (byte-identical). Inert on a non-supervisor projection. + /// The operator's allowed AGENT (persona) pool (AgentDefinition ROW ids) — validated TEAM-SCOPED (fail-closed). The supervisor gates every spawn inside it; on plan-map / single-agent the launch's own persona is the only one any agent runs as, so a persona outside the pool refuses the launch. Null / empty ⇒ all the team's personas (byte-identical). Its disposition on the route is reported on the launch result. public IReadOnlyList? AllowedAgentDefinitionIds { get; init; } /// The operator's free-text ACCEPTANCE CRITERIA — rendered into the supervisor decider prompt as the definition of done (NOT executed; distinct from the acceptanceChecks argv floor). Null / empty ⇒ omitted (byte-identical). Inert on a non-supervisor projection. @@ -77,7 +77,7 @@ public sealed record TaskLaunchRequest /// How an INDEPENDENT critic reviews the AUTHORED PLAN — tier-generic (S4e): plan.author/plan.confirm reviewMode on the plan-map tiers, the supervisor's plan-scoped planReviewMode on Deep. (default) ⇒ omitted (byte-identical). Inert on quick. public ReviewMode PlannerReviewMode { get; init; } = ReviewMode.None; - /// The operator's EXECUTABLE acceptance floor (S4b, Deep only) — an argv baked into the supervisor node's acceptanceChecks, enforced at the terminal stop. Null / empty ⇒ omitted (byte-identical). Inert on a non-supervisor projection. + /// The operator's EXECUTABLE acceptance floor (S4b) — an argv graded on every route whose builder advertises an operator-command adapter: the supervisor's terminal stop, the single agent's own oracle. A plan-map route grades none, so a floor sent to it refuses the launch, and on the auto path a floor keeps the route off plan-map. Null / empty ⇒ omitted (byte-identical). public IReadOnlyList? AcceptanceChecks { get; init; } /// DC-2a: the operator's OWN pre-declared delivery preference — baked into the supervisor node's deliverySpec, PER FIELD authoritative over the model's plan-time proposal. Null ⇒ omitted (byte-identical). Inert on a non-supervisor projection. diff --git a/backend/src/CodeSpace.Messages/Tasks/TaskRoutePreviewResult.cs b/backend/src/CodeSpace.Messages/Tasks/TaskRoutePreviewResult.cs index 9993e28e6..9fa6d653d 100644 --- a/backend/src/CodeSpace.Messages/Tasks/TaskRoutePreviewResult.cs +++ b/backend/src/CodeSpace.Messages/Tasks/TaskRoutePreviewResult.cs @@ -31,6 +31,10 @@ public sealed record TaskRoutePreviewResult /// [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] public TaskRoutePosture? Posture { get; init; } + + /// What the launch will do with each route-dependent operator control this input carries — computed by the same service the launch calls, so a Refused entry here is a launch that will be refused. Null only on the raw result a snapshot store constructs before TaskRoutePreviewService describes it. + [JsonIgnore(Condition = JsonIgnoreCondition.WhenWritingNull)] + public IReadOnlyList? ControlDispositions { get; init; } } /// diff --git a/backend/tests/CodeSpace.IntegrationTests/Agents/HarnessModelReconcilerFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Agents/HarnessModelReconcilerFlowTests.cs index ed7ef5005..6fb85cee7 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Agents/HarnessModelReconcilerFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Agents/HarnessModelReconcilerFlowTests.cs @@ -8,6 +8,7 @@ using CodeSpace.IntegrationTests.Workflows.Infrastructure; using CodeSpace.Messages.Agents; using CodeSpace.Messages.Enums; +using Microsoft.EntityFrameworkCore; using Shouldly; namespace CodeSpace.IntegrationTests.Agents; @@ -197,6 +198,84 @@ public async Task A_foreign_or_missing_credential_is_left_for_the_resolver_to_re result.HarnessKind.ShouldBe("codex-cli", "a missing credential is not a harness mismatch — leave it for the resolver's precise error"); } + // ── The run's allowed model pool: a bounded task runs on a POOLED ROW, and the harness follows that row ── + + [Fact] + public async Task A_bounded_task_naming_a_model_outside_its_pool_runs_the_pools_default_row_and_says_so() + { + // The plan-map shape: the planner authored a model the operator's pool excludes. It runs the pool's default row + // on that row's own credential, and the harness is reconciled to THAT row's provider, not the authored model's. + var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-claude", "Anthropic"); + await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "outside-gpt", "OpenAI"); + var task = new AgentTask { Goal = "g", Harness = "codex-cli", Model = "outside-gpt", AllowedModelIds = [pooled.RowId] }; + + using var scope = _fixture.BeginScope(); + var result = await scope.Resolve().ReconcileAsync(task, teamId, CancellationToken.None); + + result.PooledModel.ShouldNotBeNull("a bounded task always runs on a pooled row"); + result.PooledModel.ModelId.ShouldBe("pooled-claude"); + result.PooledModel.ModelCredentialId.ShouldBe(pooled.CredentialId, "names repeat across credentials — the row decides the key the agent runs on"); + result.PoolNote.ShouldNotBeNull().ShouldContain("outside-gpt", customMessage: "the note names the model that did not fit…"); + result.PoolNote.ShouldContain("pooled-claude", customMessage: "…and the one that runs instead"); + result.HarnessKind.ShouldBe("claude-code", "the harness follows the pooled row's provider (Anthropic), not the excluded OpenAI model's"); + } + + [Fact] + public async Task A_bounded_task_naming_a_pooled_model_runs_that_row_with_no_note() + { + var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-gpt", "OpenAI"); + var task = new AgentTask { Goal = "g", Harness = "codex-cli", Model = "POOLED-GPT", AllowedModelIds = [pooled.RowId] }; + + using var scope = _fixture.BeginScope(); + var result = await scope.Resolve().ReconcileAsync(task, teamId, CancellationToken.None); + + result.PooledModel.ShouldNotBeNull().ModelCredentialId.ShouldBe(pooled.CredentialId, "the pooled row's credential, so a loose name cannot resolve to a key outside the pool"); + result.PoolNote.ShouldBeNull("a pooled model (matched case-insensitively, like every pool lookup) did not move"); + result.HarnessKind.ShouldBe("codex-cli"); + } + + [Fact] + public async Task A_bounded_task_naming_no_model_runs_the_pools_default_row_with_no_note() + { + var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-claude", "Anthropic"); + var task = new AgentTask { Goal = "g", Harness = "codex-cli", AllowedModelIds = [pooled.RowId] }; + + using var scope = _fixture.BeginScope(); + var result = await scope.Resolve().ReconcileAsync(task, teamId, CancellationToken.None); + + result.PooledModel.ShouldNotBeNull().ModelId.ShouldBe("pooled-claude", "no name must not escape to the unbounded team default"); + result.PoolNote.ShouldBeNull("nothing authored moved — the pool's default is simply the model that runs"); + } + + [Fact] + public async Task A_bounded_task_whose_pool_resolves_nothing_any_more_is_named_and_an_unbounded_one_is_untouched() + { + var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-claude", "Anthropic"); + await DisableModelRowAsync(pooled.RowId); + + using var scope = _fixture.BeginScope(); + var reconciler = scope.Resolve(); + var bounded = await reconciler.ReconcileAsync(new AgentTask { Goal = "g", Harness = "codex-cli", Model = "pooled-claude", AllowedModelIds = [pooled.RowId] }, teamId, CancellationToken.None); + var unbounded = await reconciler.ReconcileAsync(new AgentTask { Goal = "g", Harness = "codex-cli", Model = "pooled-claude" }, teamId, CancellationToken.None); + + bounded.PooledModel.ShouldBeNull(); + bounded.PoolNote.ShouldNotBeNull("the executor fails the run on this note rather than let the agent run outside its pool"); + unbounded.PooledModel.ShouldBeNull(); + unbounded.PoolNote.ShouldBeNull("an unbounded task is reconciled exactly as before"); + } + + private async Task DisableModelRowAsync(Guid rowId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + (await db.ModelCredentialModel.SingleAsync(m => m.Id == rowId)).Enabled = false; + await db.SaveChangesAsync(); + } + private async Task SeedCredentialAsync(Guid teamId, string provider) { var id = Guid.NewGuid(); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorTests.cs index 2d12fc9c2..8a474c21f 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorTests.cs @@ -685,6 +685,62 @@ public async Task An_auto_runs_resolved_credential_default_model_is_persisted_on persisted!.Model.ShouldBe("cred-default-model", "the resolved credential-default model is written to the run's task at dispatch, so the live projection shows it before completion"); } + [Fact] + public async Task A_bounded_run_naming_a_model_outside_its_pool_runs_the_pooled_row_persists_it_and_names_the_move() + { + // The plan-map branch shape: the planner authored a model the operator's pool excludes, pinned to that model's + // own credential. At launch the executor must run the POOLED row — its model on its own key — persist it so a + // re-attach and every reader agree, and say on the run's timeline why the model moved. + if (OperatingSystem.IsWindows()) return; + + var teamId = await SeedTeamAsync(); + var pooledCredential = await SeedModelCredentialAsync(teamId, "scripted-provider", "sk-pooled-key"); + var pooledRow = await SeedModelRowAsync(pooledCredential, "pooled-model"); + var outsideCredential = await SeedModelCredentialAsync(teamId, "scripted-provider", "sk-outside-key"); + await SeedModelRowAsync(outsideCredential, "outside-model"); + + var runId = await CreateTaskRunAsync(teamId, new AgentTask { Goal = "scripted", Harness = "scripted-projector", Model = "outside-model", ModelCredentialId = outsideCredential, AllowedModelIds = [pooledRow] }); + + await ExecuteAsync(runId, new ProjectingScriptedHarness("scripted-provider", "SCRIPTED_MODEL_KEY", "if [ \"$SCRIPTED_MODEL_KEY\" = 'sk-pooled-key' ]; then echo POOLED_KEY; else echo OTHER_KEY; fi")); + + using var scope = _fixture.BeginScope(); + var svc = scope.Resolve(); + var run = await svc.GetAsync(runId, CancellationToken.None); + + run.Status.ShouldBe(AgentRunStatus.Succeeded); + var persisted = JsonSerializer.Deserialize(run.TaskJson, AgentJson.Options)!; + persisted.Model.ShouldBe("pooled-model", "the persisted task is the one that ran — a re-attach redacts and folds against it"); + persisted.ModelCredentialId.ShouldBe(pooledCredential); + + var events = (await svc.GetEventsAsync(runId, teamId, 0, CancellationToken.None)).ToList(); + events.Select(e => e.Text).ShouldContain("POOLED_KEY", "the REAL child process ran on the pooled row's credential, not the excluded model's"); + events.ShouldContain(e => e.Kind == AgentEventKind.Warning && e.Text.Contains("outside-model") && e.Text.Contains("pooled-model"), + customMessage: "the move is named on the run's timeline — never a silent swap"); + } + + [Fact] + public async Task A_bounded_run_whose_pool_resolves_nothing_fails_rather_than_run_outside_it() + { + if (OperatingSystem.IsWindows()) return; + + var teamId = await SeedTeamAsync(); + var credential = await SeedModelCredentialAsync(teamId, "scripted-provider", "sk-only-key"); + var pooledRow = await SeedModelRowAsync(credential, "pooled-model"); + await DisableModelRowAsync(pooledRow); // the operator disabled every pooled model after the launch was staged + + var runId = await CreateTaskRunAsync(teamId, new AgentTask { Goal = "scripted", Harness = "scripted-projector", Model = "pooled-model", ModelCredentialId = credential, AllowedModelIds = [pooledRow] }); + + await ExecuteAsync(runId, new ProjectingScriptedHarness("scripted-provider", "SCRIPTED_MODEL_KEY", "echo should-not-run")); + + using var scope = _fixture.BeginScope(); + var svc = scope.Resolve(); + var run = await svc.GetAsync(runId, CancellationToken.None); + + run.Status.ShouldBe(AgentRunStatus.Failed, "an agent bound to a pool that resolves nothing must not run on a model outside it"); + run.Error.ShouldNotBeNull().ShouldContain("allowed model pool"); + (await svc.GetEventsAsync(runId, teamId, 0, CancellationToken.None)).Select(e => e.Text).ShouldNotContain("should-not-run"); + } + [Fact] public async Task A_pinned_credential_from_another_team_lands_the_run_failed_clean() { @@ -1488,6 +1544,26 @@ private async Task SeedDefaultCredentialModelAsync(Guid modelCredentialId, strin await db.SaveChangesAsync(); } + /// Seed one enabled credentialed-model row under and return its id — an allowed-pool entry. + private async Task SeedModelRowAsync(Guid modelCredentialId, string modelId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + var rowId = Guid.NewGuid(); + db.ModelCredentialModel.Add(new ModelCredentialModel { Id = rowId, ModelCredentialId = modelCredentialId, ModelId = modelId, Enabled = true }); + await db.SaveChangesAsync(); + return rowId; + } + + private async Task DisableModelRowAsync(Guid rowId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + (await db.ModelCredentialModel.SingleAsync(m => m.Id == rowId)).Enabled = false; + await db.SaveChangesAsync(); + } + private async Task SeedModelCredentialAsync(Guid teamId, string provider, string plaintextKey) { using var scope = _fixture.BeginScope(); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/PlanMapSynthPlannerRequest.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/PlanMapSynthPlannerRequest.cs index b01825b7e..e6fd5d751 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/PlanMapSynthPlannerRequest.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/Infrastructure/PlanMapSynthPlannerRequest.cs @@ -53,7 +53,8 @@ public static async Task BuildAsync(ILifetimeSco var planRequest = PlanAuthorNode.BuildPlanRequest(config, teamId, new PlanAuthorNode.PlanPromptParts(goal, [], grounding, "")); - var pool = await scope.Resolve().ListPoolAsync(teamId, allowedRowIds: null, cancellationToken).ConfigureAwait(false); + // The production planner lists the pool bounded to the request's allowed models — null for this unbounded launch, so the key is unchanged. + var pool = await scope.Resolve().ListPoolAsync(teamId, planRequest.AllowedModelIds, cancellationToken).ConfigureAwait(false); var catalog = CapabilityCatalog.Render(scope.Resolve().All, pool); return new StructuredLLMCompletionRequest diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/PlannerModelPoolFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/PlannerModelPoolFlowTests.cs new file mode 100644 index 000000000..c7a535c92 --- /dev/null +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/PlannerModelPoolFlowTests.cs @@ -0,0 +1,87 @@ +using System.Text.Json; +using Autofac; +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.ModelCredentials; +using CodeSpace.Core.Services.Learning; +using CodeSpace.Core.Services.Workflows.Llm; +using CodeSpace.Core.Services.Workflows.Planning.Planners; +using CodeSpace.IntegrationTests.Infrastructure; +using CodeSpace.IntegrationTests.Workflows.Infrastructure; +using CodeSpace.Messages.Dtos.Workflows.Planning; +using Shouldly; + +namespace CodeSpace.IntegrationTests.Workflows; + +/// +/// 🟢 Integration (real Postgres + real pool selector + real harness registry + the real planner; a capturing fake at +/// the structured-LLM seam): the capability catalog a plan-map planner allocates subtask models from is bounded to the +/// operator's allowed model pool. It used to list the WHOLE team pool, so a planner could only author a model the +/// dispatch then had to clamp; now it is shown the pool it must choose from, and an unbounded request is unchanged. +/// +[Collection(PostgresCollection.Name)] +[Trait("Category", "Integration")] +public sealed class PlannerModelPoolFlowTests +{ + private readonly PostgresFixture _fixture; + + public PlannerModelPoolFlowTests(PostgresFixture fixture) { _fixture = fixture; } + + [Fact] + public async Task The_planners_catalog_lists_only_the_allowed_pool_and_the_whole_pool_when_unbounded() + { + var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-sonnet"); + await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "outside-opus"); + + var bounded = await PlannerPromptAsync(new WorkflowPlanRequest { TaskText = "split the migration", TeamId = teamId, AllowedModelIds = [pooled.RowId] }); + var unbounded = await PlannerPromptAsync(new WorkflowPlanRequest { TaskText = "split the migration", TeamId = teamId }); + + bounded.ShouldContain("pooled-sonnet", customMessage: "the pooled model is on the planner's menu"); + bounded.ShouldNotContain("outside-opus", customMessage: "a model outside the operator's pool must not be offered to the planner at all"); + unbounded.ShouldContain("pooled-sonnet"); + unbounded.ShouldContain("outside-opus", customMessage: "no pool ⇒ the whole team pool, the catalog the planner has always seen"); + } + + private async Task PlannerPromptAsync(WorkflowPlanRequest request) + { + using var scope = _fixture.BeginScope(); + var client = new CapturingClient(); + var db = scope.Resolve(); + var memory = new PlannerLessonMemory(db, new LessonReader(db, NoLessons.Instance), scope.Resolve>()); + var planner = new LlmWorkflowPlanner(new SingleClient(client), scope.Resolve(), scope.Resolve(), memory); + + await planner.PlanAsync(request, CancellationToken.None); + return client.LastUserPrompt.ShouldNotBeNull("the planner never reached the model"); + } + + /// No lesson is seeded, so relevance is never asked; a call here is a planner reaching for lessons it cannot have. + private sealed class NoLessons : ILessonRelevanceEvaluator + { + public static readonly NoLessons Instance = new(); + public Task EvaluateAsync(LessonRelevanceRequest request, CancellationToken cancellationToken) => throw new NotSupportedException(); + } + + private sealed class SingleClient : ILLMClientRegistry + { + public SingleClient(IStructuredLLMClient structured) => All = new ILLMClient[] { (ILLMClient)structured }; + public IReadOnlyList All { get; } + public ILLMClient Resolve(string provider) => All.First(); + } + + private sealed class CapturingClient : ILLMClient, IStructuredLLMClient + { + public string Provider => "Anthropic"; + public string? LastUserPrompt { get; private set; } + public Task CompleteAsync(LLMCompletionRequest request, CancellationToken ct) => throw new NotSupportedException(); + public Task CompleteStructuredAsync(StructuredLLMCompletionRequest request, CancellationToken ct) + { + LastUserPrompt = request.UserPrompt; + return Task.FromResult(new StructuredLLMCompletion + { + Json = JsonSerializer.SerializeToElement(new { goal = "split it", subtasks = new[] { new { id = "s1", title = "T", instruction = "do it" } } }), + Model = request.Model, + }); + } + } +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchContractFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchContractFlowTests.cs index 5391b43e0..4cf3b7b4c 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchContractFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchContractFlowTests.cs @@ -4,6 +4,8 @@ using CodeSpace.Core.Persistence.Entities; using CodeSpace.Core.Services.Tasks; using CodeSpace.Core.Services.Tasks.Contracts; +using CodeSpace.Core.Services.Tasks.Launch; +using CodeSpace.Core.Services.Tasks.Launch.Exceptions; using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Core.Services.Completion; using CodeSpace.Core.Services.Workflows; @@ -42,10 +44,10 @@ public TaskLaunchContractFlowTests(PostgresFixture fixture) public void Dispose() => _manualExecution.Dispose(); [Theory] - [InlineData(TaskEffortModes.Quick, TaskProjectionKinds.SingleAgent)] - [InlineData(TaskEffortModes.Standard, TaskProjectionKinds.PlanMapSynth)] - [InlineData(TaskEffortModes.Deep, TaskProjectionKinds.Supervisor)] - public async Task Every_launch_lane_records_original_controls_in_the_frozen_run_detail(string effort, string projectionKind) + [InlineData(TaskEffortModes.Quick, TaskProjectionKinds.SingleAgent, true)] + [InlineData(TaskEffortModes.Standard, TaskProjectionKinds.PlanMapSynth, false)] // plan-map grades no operator floor: sending one is refused (pinned below) + [InlineData(TaskEffortModes.Deep, TaskProjectionKinds.Supervisor, true)] + public async Task Every_launch_lane_records_original_controls_in_the_frozen_run_detail(string effort, string projectionKind, bool gradesOperatorFloor) { var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); using var scope = _fixture.BeginScope(); @@ -54,7 +56,7 @@ public async Task Every_launch_lane_records_original_controls_in_the_frozen_run_ TeamId = teamId, ActorUserId = userId, SurfaceKind = TaskLaunchSurfaceKinds.Chat, TaskText = "Preserve the original delivery obligations", RequestedEffort = effort, Autonomy = "Unleashed", CapsOverride = new() { MaxCostUsd = 3.25m, AutonomyCeiling = "Standard" }, - AcceptanceCriteria = ["Use original inputs", "Explain limitations"], AcceptanceChecks = ["sh", "verify.sh"], + AcceptanceCriteria = ["Use original inputs", "Explain limitations"], AcceptanceChecks = gradesOperatorFloor ? ["sh", "verify.sh"] : null, DeliverySpec = new() { OpenPullRequest = false, TargetBranch = "review" }, AllowedModelIds = [], AllowedAgentDefinitionIds = [], RequirePlanConfirmation = false, Overrides = new() { Harness = "codex-cli", AllowedTools = ["Read", "Grep"], PushBranch = false, EnableMcp = false }, @@ -107,6 +109,26 @@ public async Task Every_launch_lane_records_original_controls_in_the_frozen_run_ (await scope.Resolve().AgentRun.CountAsync(r => r.WorkflowRunId == launched.RunId)).ShouldBe(0, "this verifies persistence, not model execution or control enforcement"); } + [Fact] + public async Task A_standard_launch_carrying_an_operator_floor_is_refused_and_records_nothing() + { + // The plan-map lane grades no operator floor, so the floor is refused by name before a session or run exists — + // it used to be recorded in the contract and silently never graded. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + using var scope = _fixture.BeginScope(); + var request = new TaskLaunchRequest + { + TeamId = teamId, ActorUserId = userId, SurfaceKind = TaskLaunchSurfaceKinds.Chat, + TaskText = "Preserve the original delivery obligations", RequestedEffort = TaskEffortModes.Standard, + AcceptanceChecks = ["sh", "verify.sh"], + }; + + var ex = await Should.ThrowAsync(() => scope.Resolve().LaunchAsync(request, CancellationToken.None)); + + ex.Message.ShouldContain(LaunchControlResolver.FloorNotGradedReason); + (await scope.Resolve().WorkflowRun.CountAsync(r => r.TeamId == teamId)).ShouldBe(0, "no run, no contract — nothing records a floor nothing would grade"); + } + [Theory] [InlineData(true)] [InlineData(false)] diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchFlowTests.cs index cb8e4c51e..c049918bc 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskLaunchFlowTests.cs @@ -5,7 +5,11 @@ using CodeSpace.Core.Services.Agents; using CodeSpace.Core.Services.Supervisor; using CodeSpace.Core.Services.Tasks; +using CodeSpace.Core.Services.Tasks.Effort; +using CodeSpace.Core.Services.Tasks.Effort.Classifiers.Heuristic; +using CodeSpace.Core.Services.Tasks.Effort.Classifiers.Llm; using CodeSpace.Core.Services.Tasks.Launch; +using CodeSpace.Core.Services.Tasks.Launch.Exceptions; using CodeSpace.Core.Services.Tasks.RoutePreview; using CodeSpace.IntegrationTests.Infrastructure; using CodeSpace.IntegrationTests.Infrastructure.Jobs; @@ -15,6 +19,7 @@ using CodeSpace.Messages.Commands.Tasks; using CodeSpace.Messages.Constants; using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Failures; using CodeSpace.Messages.Tasks; using CodeSpace.Messages.Tasks.Effort; using MediatR; @@ -984,28 +989,24 @@ public async Task A_deep_launch_at_delivery_quality_with_an_acceptance_check_suc } [Fact] - public async Task A_quick_launch_at_delivery_quality_is_never_rejected_for_a_missing_acceptance_check() + public async Task A_quick_launch_at_delivery_quality_without_an_acceptance_check_is_rejected_before_any_run_is_created() { - // AcceptanceChecks is inert on a non-supervisor projection today — the mandate doesn't invent new - // acceptance-floor plumbing for single-agent launches, so it stays inert there too. - if (OperatingSystem.IsWindows()) return; - - using var cli = new SubtaskAwareFakeCli(); - + // Quick grades its single agent with the operator's argv (the builder advertises the adapter), so a Delivery + // claim there without one is the same unverified claim the Deep mandate refuses — it used to launch, because the + // mandate asked "is this the supervisor?" instead of "does this route grade a floor?". var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); - var jobClient = ResolveJobClient(); - jobClient.Clear(); - jobClient.AutoExecute = true; - - var result = await LaunchAsync(new TaskLaunchRequest + var request = new TaskLaunchRequest { TeamId = teamId, ActorUserId = userId, SurfaceKind = TaskLaunchSurfaceKinds.Chat, TaskText = "Touch nothing important", RequestedEffort = TaskEffortModes.Quick, Tier = QualityTier.Delivery, - }); + }; - result.RunId.ShouldNotBe(Guid.Empty, "a Quick/single-agent Delivery launch is never rejected for the missing acceptance floor — the mandate is inert there"); + var ex = await Should.ThrowAsync(() => LaunchAsync(request)); + + ex.Message.ShouldContain("acceptanceChecks", Case.Insensitive, "the operator needs an actionable name for the missing lever"); + (await CountRunsForTeamAsync(teamId)).ShouldBe(0, "the mandate rejects BEFORE any run/session is created — no orphan"); } [Fact] @@ -1508,6 +1509,204 @@ public async Task A_repo_bound_answer_launch_is_graded_but_publishes_no_branch() } } + // ── 3.3: Auto's controls are applied, clamped, named not-applicable, or refused on every lane — never dropped ── + // + // The auto path below is answered by a CONFIDENT classification (the structured-LLM classifier's slot) — the one case + // the composer's confirm card never sees, where a route to a non-supervisor lane used to drop these controls silently. + + [Fact] + public async Task An_auto_route_to_quick_clamps_a_pinned_model_outside_the_allowed_pool_and_says_so() + { + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-model"); + var outside = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "outside-model"); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + using var manual = jobClient.ManualExecution(); // the frozen snapshot is the evidence; dispatch is pinned in AgentRunExecutorTests + + var request = AutoRequest(teamId, userId, "Fix the typo in the README") with + { + AllowedModelIds = [pooled.RowId], + Overrides = new TaskExecutionOverrides { Harness = "codex-cli", RunnerKind = "local", ModelCredentialModelId = outside.RowId }, + }; + + var result = await LaunchOnConfidentAutoAsync(request, new ConfidentClassifier(QuickSignals, TaskRecipeKinds.SingleAgent)); + + result.ProjectionKind.ShouldBe(TaskProjectionKinds.SingleAgent); + var models = result.ControlDispositions.Single(d => d.Control == LaunchControls.AllowedModelIds); + models.Outcome.ShouldBe(LaunchControlOutcome.Clamped, "the operator pinned a model their own pool excludes — the pool wins, and the result says so"); + models.Reason.ShouldContain("pooled-model"); + + var config = FrozenAgentConfig((await LoadRunAsync(result.RunId)).DefinitionSnapshotJson!); + config.GetProperty("model").GetString().ShouldBe("pooled-model", "the frozen agent runs the pool's default row, not the pin outside it"); + config.GetProperty("modelCredentialId").GetString().ShouldBe(pooled.CredentialId.ToString(), "…on that row's own credential — names repeat across credentials, so the row decides"); + config.TryGetProperty("modelCredentialModelId", out _).ShouldBeFalse("the excluded row pin is gone from the frozen config"); + config.GetProperty("allowedModelIds").EnumerateArray().Select(e => e.GetString()).ShouldBe(new[] { pooled.RowId.ToString() }, "the pool rides the node too, holding a persona-supplied model to it at dispatch"); + } + + [Fact] + public async Task An_auto_route_to_quick_refuses_a_persona_the_allowed_pool_excludes_before_any_run() + { + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var admitted = await SeedPersonaAsync(teamId); + var excluded = await SeedPersonaAsync(teamId); + + var request = AutoRequest(teamId, userId, "Fix the typo in the README") with + { + AllowedAgentDefinitionIds = [admitted], + Overrides = new TaskExecutionOverrides { Harness = "codex-cli", RunnerKind = "local", AgentDefinitionId = excluded }, + }; + + var ex = await Should.ThrowAsync(() => LaunchOnConfidentAutoAsync(request, new ConfidentClassifier(QuickSignals, TaskRecipeKinds.SingleAgent))); + + ex.Code.ShouldBe(FailureCodes.TaskLaunchControlRefused); + var refused = (IReadOnlyList)ex.Details["controls"]!; + refused.ShouldHaveSingleItem().Control.ShouldBe(LaunchControls.AllowedAgentDefinitionIds); + ex.Message.ShouldContain(excluded.ToString(), customMessage: "the refusal names the persona the operator excluded"); + (await CountRunsForTeamAsync(teamId)).ShouldBe(0, "a refused control stops the launch before any session or run exists"); + } + + [Fact] + public async Task An_explicit_standard_launch_with_an_operator_floor_is_refused_with_the_reason_its_preview_gives() + { + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var request = new TaskLaunchRequest + { + TeamId = teamId, ActorUserId = userId, SurfaceKind = TaskLaunchSurfaceKinds.Chat, + TaskText = "Refactor the parser across modules", RequestedEffort = TaskEffortModes.Standard, + AcceptanceChecks = ["sh", "verify.sh"], + }; + + TaskRoutePreviewResult preview; + using (var scope = _fixture.BeginScope()) preview = await scope.Resolve().PreviewAsync(request, CancellationToken.None); + + preview.AcceptanceCompatibility!.State.ShouldBe(TaskAcceptanceCompatibilityState.Incompatible); + var previewed = preview.ControlDispositions!.Single(d => d.Control == LaunchControls.AcceptanceChecks); + previewed.Outcome.ShouldBe(LaunchControlOutcome.Refused); + + var ex = await Should.ThrowAsync(() => LaunchAsync(request with { RouteSnapshotId = preview.RouteSnapshotId })); + + var refused = ((IReadOnlyList)ex.Details["controls"]!).ShouldHaveSingleItem(); + refused.Reason.ShouldBe(preview.AcceptanceCompatibility.Detail, "the launch refuses with the very reason its preview gave — preview and launch can no longer disagree"); + refused.ShouldBe(previewed); + (await CountRunsForTeamAsync(teamId)).ShouldBe(0); + } + + [Fact] + public async Task An_auto_route_with_an_operator_floor_never_lands_on_plan_map_and_grades_the_floor_instead() + { + // The UI dead-end this closes: Auto + Delivery required a check, but an Auto route to plan-map was Incompatible + // with one. A confidently Standard-classified task with a floor now routes to a lane that grades it. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var repoId = await SeedRepositoryAsync(teamId); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + using var manual = jobClient.ManualExecution(); + + var request = AutoRequest(teamId, userId, "Refactor the parser across modules and keep the tests green") with { RepositoryId = repoId, AcceptanceChecks = ["sh", "verify.sh"], Tier = QualityTier.Delivery }; + var classifier = new ConfidentClassifier(StandardSignals, TaskRecipeKinds.MapFanout); + + var preview = await PreviewOnConfidentAutoAsync(request, classifier); + var result = await LaunchOnConfidentAutoAsync(request, classifier); + + preview.Route.ProjectionKind.ShouldBe(TaskProjectionKinds.SingleAgent, "the preview predicts the floor-grading lane"); + preview.AcceptanceCompatibility!.State.ShouldBe(TaskAcceptanceCompatibilityState.Compatible, "Auto + Delivery + a check is launchable again — no dead-end"); + result.ProjectionKind.ShouldBe(TaskProjectionKinds.SingleAgent, "an operator floor keeps an auto route off a lane that cannot grade it"); + result.Route.EffortMode.ShouldBe(TaskEffortModes.Quick, "the Standard row is set aside for the policy's next matching row"); + result.Route.DegradedReason.ShouldNotBeNull().ShouldContain(TaskProjectionKinds.PlanMapSynth, customMessage: "the move is named on the route, never silent"); + result.ControlDispositions.Single(d => d.Control == LaunchControls.AcceptanceChecks).Outcome.ShouldBe(LaunchControlOutcome.Applied); + + var acceptance = FrozenAgentConfig((await LoadRunAsync(result.RunId)).DefinitionSnapshotJson!).GetProperty("acceptance"); + acceptance.GetProperty("kind").GetString().ShouldBe("TestsPass"); + acceptance.GetProperty("command").EnumerateArray().Select(e => e.GetString()).ShouldBe(new[] { "sh", "verify.sh" }, "the operator's floor grades the agent that runs"); + } + + [Fact] + public async Task A_continue_is_routed_on_the_threads_grounding_by_both_the_preview_and_the_launch() + { + // Grounding used to be resolved AFTER routing, so the classifier read "a fresh task" for every chat follow-up. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + using var manual = jobClient.ManualExecution(); + + var first = await LaunchAsync(new TaskLaunchRequest + { + TeamId = teamId, ActorUserId = userId, SurfaceKind = TaskLaunchSurfaceKinds.Chat, + TaskText = "Map every caller of the legacy billing client", RequestedEffort = TaskEffortModes.Quick, + }); + + var followUp = AutoRequest(teamId, userId, "and now migrate them") with { ContinueSessionId = first.SessionId }; + var previewClassifier = new ConfidentClassifier(QuickSignals, TaskRecipeKinds.SingleAgent); + var launchClassifier = new ConfidentClassifier(QuickSignals, TaskRecipeKinds.SingleAgent); + + await PreviewOnConfidentAutoAsync(followUp, previewClassifier); + await LaunchOnConfidentAutoAsync(followUp, launchClassifier); + + foreach (var (surface, classifier) in new[] { ("preview", previewClassifier), ("launch", launchClassifier) }) + { + var seen = classifier.LastRequest.ShouldNotBeNull($"the {surface} never classified the follow-up"); + seen.Seed.GroundingContext.ShouldNotBeNull($"the {surface} classified the follow-up with no thread grounding — it read as a fresh task") + .ShouldContain("Map every caller of the legacy billing client", customMessage: $"the {surface}'s classifier must see the prior turn it is continuing"); + LlmEffortClassifier.BuildUserPrompt(seen).ShouldContain("This is a follow-up turn continuing earlier work: yes"); + } + } + + /// Signals the policy routes to the quick tier (a localized code change). + private static readonly EffortSignals QuickSignals = new() { NeedsCodeChange = true }; + + /// Signals the policy routes to the standard tier (a code change across files needing tests). + private static readonly EffortSignals StandardSignals = new() { NeedsCodeChange = true, CrossFile = true, NeedsTestsOrCi = true }; + + private static TaskLaunchRequest AutoRequest(Guid teamId, Guid userId, string taskText) => new() + { + TeamId = teamId, ActorUserId = userId, SurfaceKind = TaskLaunchSurfaceKinds.Chat, TaskText = taskText, + RequestedEffort = TaskEffortModes.Auto, Autonomy = "Confined", + Overrides = new TaskExecutionOverrides { Harness = "codex-cli", RunnerKind = "local" }, + }; + + /// The production launch graph with only the AUTO classification swapped for — the router, snapshot store, preview and launch services are rebuilt in the scope so they pick it up; everything else is real. + private ILifetimeScope ConfidentAutoScope(IEffortClassifier classifier) => _fixture.BeginScope(b => + { + b.RegisterInstance(new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), classifier })).As(); + b.RegisterType().As().InstancePerLifetimeScope(); + b.RegisterType().As().InstancePerLifetimeScope(); + b.RegisterType().As().InstancePerLifetimeScope(); + b.RegisterType().As().InstancePerLifetimeScope(); + }); + + private async Task LaunchOnConfidentAutoAsync(TaskLaunchRequest request, IEffortClassifier classifier) + { + using var scope = ConfidentAutoScope(classifier); + return await scope.Resolve().LaunchAsync(request, CancellationToken.None); + } + + private async Task PreviewOnConfidentAutoAsync(TaskLaunchRequest request, IEffortClassifier classifier) + { + using var scope = ConfidentAutoScope(classifier); + return await scope.Resolve().PreviewAsync(request, CancellationToken.None); + } + + private static JsonElement FrozenAgentConfig(string definitionSnapshotJson) => + JsonDocument.Parse(definitionSnapshotJson).RootElement.GetProperty("nodes").EnumerateArray().Single(n => n.GetProperty("id").GetString() == "agent").GetProperty("config").Clone(); + + /// A confident auto classification in the structured-LLM classifier's slot (its kind, so the registry's Auto resolves it): fixed signals at 0.9, the policy's tier, a caller-named recipe — and the route request it was asked about, recorded. + private sealed class ConfidentClassifier(EffortSignals signals, string suggestedRecipe) : IEffortClassifier + { + public EffortRouteRequest? LastRequest { get; private set; } + + public string Kind => LlmEffortClassifier.ClassifierKind; + + public Task ClassifyAsync(EffortRouteRequest request, CancellationToken ct) + { + LastRequest = request; + return Task.FromResult(new EffortDecision { Signals = signals, SuggestedEffort = EffortPolicy.Decide(signals, requestedEffort: null), SuggestedRecipe = suggestedRecipe, Confidence = 0.9, Rationale = "Scripted confident classification.", ClassifierKind = Kind }); + } + } + private async Task LaunchAsync(TaskLaunchRequest request) { using var scope = _fixture.BeginScope(); diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskRoutePreviewFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskRoutePreviewFlowTests.cs index 21920f8e2..a2b2c477b 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskRoutePreviewFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/TaskRoutePreviewFlowTests.cs @@ -9,6 +9,7 @@ using CodeSpace.IntegrationTests.Infrastructure; using CodeSpace.IntegrationTests.Infrastructure.Jobs; using CodeSpace.IntegrationTests.Workflows.Infrastructure; +using CodeSpace.Messages.Agents; using CodeSpace.Messages.Commands.Tasks; using CodeSpace.Messages.Constants; using CodeSpace.Messages.Enums; @@ -144,6 +145,44 @@ public async Task A_launch_persists_the_route_it_was_projected_from_on_the_run() persisted.Decision!.ClassifierKind.ShouldBe(result.Route.Decision!.ClassifierKind, "who decided the tier is the whole point of recording it"); } + [Fact] + public async Task The_preview_reports_the_same_control_dispositions_the_launch_then_applies() + { + // One resolver, two callers: a Quick launch that pins a model outside its own pool (Clamped), sends a delivery + // spec and a plan critic Quick has no lane for (NotApplicable) — previewed, then launched on that very preview. + var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(_fixture); + var pooled = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "pooled-model"); + var outside = await WorkflowsTestSeed.SeedCredentialedModelAsync(_fixture, teamId, "outside-model"); + + var jobClient = ResolveJobClient(); + jobClient.Clear(); + using var manual = jobClient.ManualExecution(); + + var request = Request(teamId, userId, "Fix the typo in the README", TaskEffortModes.Quick) with + { + Autonomy = "Confined", + AllowedModelIds = [pooled.RowId], + DeliverySpec = new DeliverySpec { OpenPullRequest = true }, + PlannerReviewMode = ReviewMode.Gate, + Overrides = new TaskExecutionOverrides { Harness = "codex-cli", RunnerKind = "local", ModelCredentialModelId = outside.RowId }, + }; + + TaskRoutePreviewResult preview; + using (var scope = _fixture.BeginScope()) preview = await scope.Resolve().PreviewAsync(request, CancellationToken.None); + + var launched = await LaunchAsync(request with { RouteSnapshotId = preview.RouteSnapshotId }); + + preview.ControlDispositions.ShouldNotBeNull(); + preview.ControlDispositions.Select(d => (d.Control, d.Outcome)).ShouldBe(new[] + { + (LaunchControls.AllowedModelIds, LaunchControlOutcome.Clamped), + (LaunchControls.DeliverySpec, LaunchControlOutcome.NotApplicable), + (LaunchControls.PlannerReviewMode, LaunchControlOutcome.NotApplicable), + }); + JsonSerializer.Serialize(launched.ControlDispositions, Json).ShouldBe(JsonSerializer.Serialize(preview.ControlDispositions, Json), + customMessage: "the launch applied a different disposition than its own preview promised — they must share one resolver"); + } + // ─── Helpers ───────────────────────────────────────────────────────────── private static readonly JsonSerializerOptions Json = new(JsonSerializerDefaults.Web); diff --git a/backend/tests/CodeSpace.UnitTests/Agents/DeploymentAutonomyCeilingTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/DeploymentAutonomyCeilingTests.cs index fa8d6e6a0..c83d39838 100644 --- a/backend/tests/CodeSpace.UnitTests/Agents/DeploymentAutonomyCeilingTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Agents/DeploymentAutonomyCeilingTests.cs @@ -15,6 +15,7 @@ using CodeSpace.Core.Services.Tasks.Recipes.MapFanout; using CodeSpace.Core.Services.Tasks.Recipes.SingleAgent; using CodeSpace.Core.Services.Tasks.Recipes.Supervisor; +using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Core.Services.Workflows.Nodes; using CodeSpace.Core.Services.Workflows.Nodes.Builtin; using CodeSpace.Core.Services.Workflows.Runtime; @@ -268,7 +269,8 @@ public void A_sandbox_spec_that_says_nothing_about_the_network_is_severed() new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier() }), new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe(), new MapFanoutRecipe(), new SupervisorRecipe() }), new BoundsPresetRegistry(withPresets ? new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() } : Array.Empty()), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + new TaskProjectionRegistry(Array.Empty())); private static EffortRouteRequest RouteRequest(string effort) => new() { diff --git a/backend/tests/CodeSpace.UnitTests/Architecture/FailureTaxonomyTests.cs b/backend/tests/CodeSpace.UnitTests/Architecture/FailureTaxonomyTests.cs index 95d85e745..e339aa1f9 100644 --- a/backend/tests/CodeSpace.UnitTests/Architecture/FailureTaxonomyTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Architecture/FailureTaxonomyTests.cs @@ -73,6 +73,7 @@ public void The_wire_codes_are_pinned() FailureCodes.RateLimited.ShouldBe("rate_limited"); FailureCodes.WorkflowDefinitionInvalid.ShouldBe("workflow_definition_invalid"); FailureCodes.TaskRouteConfirmationRequired.ShouldBe("task_route_confirmation_required"); + FailureCodes.TaskLaunchControlRefused.ShouldBe("task_launch_control_refused"); FailureCodes.WorkspaceUnresolvable.ShouldBe("workspace_unresolvable"); FailureCodes.RerunAlreadyInProgress.ShouldBe("rerun_already_in_progress"); FailureCodes.RerunTargetInvalid.ShouldBe("rerun_target_invalid"); diff --git a/backend/tests/CodeSpace.UnitTests/Tasks/LaunchControlResolverTests.cs b/backend/tests/CodeSpace.UnitTests/Tasks/LaunchControlResolverTests.cs new file mode 100644 index 000000000..fdfcaa12e --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Tasks/LaunchControlResolverTests.cs @@ -0,0 +1,188 @@ +using CodeSpace.Core.Services.Agents.ModelCredentials; +using CodeSpace.Core.Services.Tasks.Launch; +using CodeSpace.Core.Services.Tasks.Projection; +using CodeSpace.Core.Services.Tasks.Projection.Builders.PlanMapDynamic; +using CodeSpace.Core.Services.Tasks.Projection.Builders.PlanMapSynth; +using CodeSpace.Core.Services.Tasks.Projection.Builders.SingleAgent; +using CodeSpace.Core.Services.Tasks.Projection.Builders.Supervisor; +using CodeSpace.Messages.Agents; +using CodeSpace.Messages.Dtos.Workflows; +using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Tasks; +using Shouldly; + +namespace CodeSpace.UnitTests.Tasks; + +/// +/// Pins — the one place a route-dependent launch control learns what the resolved +/// route does with it — over the REAL projection builders (their OperatorAcceptance advertisements decide the +/// floor) and a scripted model pool (the only I/O). Every lane × control lands in exactly one disposition, with a reason +/// wherever it is not a plain apply; nothing the operator set is left out of the list. +/// +[Trait("Category", "Unit")] +public class LaunchControlResolverTests +{ + private static readonly Guid PooledRow = Guid.NewGuid(); + private static readonly Guid ForeignRow = Guid.NewGuid(); + private static readonly Guid Persona = Guid.NewGuid(); + private static readonly ModelDispatchRef PoolDefault = new() { ModelId = "pool-default", ModelCredentialId = Guid.NewGuid(), Provider = "Anthropic" }; + + private static LaunchControlResolver Resolver(ScriptedPool pool) => new(new TaskProjectionRegistry(new IWorkflowDefinitionBuilder[] { new SingleAgentDefinitionBuilder(), new PlanMapSynthDefinitionBuilder(), new PlanMapDynamicDefinitionBuilder(), new SupervisorDefinitionBuilder(), new UnadvertisedBuilder() }), pool); + + private static RoutePlan Route(string projectionKind) => new() { ProjectionKind = projectionKind }; + + /// A launch carrying EVERY route-dependent control — a pooled model row, a pooled persona, a floor, both critics, a delivery spec and the plan gate. + private static TaskLaunchRequest EveryControl() => new() + { + TeamId = Guid.NewGuid(), ActorUserId = Guid.NewGuid(), SurfaceKind = "chat", + AllowedModelIds = [PooledRow], AllowedAgentDefinitionIds = [Persona], AcceptanceChecks = ["sh", "check.sh"], + DecisionReviewMode = ReviewMode.Gate, DeliverySpec = new DeliverySpec { OpenPullRequest = true }, RequirePlanConfirmation = true, PlannerReviewMode = ReviewMode.Improve, + Overrides = new TaskExecutionOverrides { ModelCredentialModelId = PooledRow, AgentDefinitionId = Persona }, + }; + + private static LaunchControlOutcome OutcomeOf(LaunchControlResolution resolution, string control) => resolution.Dispositions.Single(d => d.Control == control).Outcome; + + [Theory] + // lane, model pool, persona pool, floor, decision critic, delivery spec, plan gate, plan critic + [InlineData(TaskProjectionKinds.Supervisor, "Applied", "Applied", "Applied", "Applied", "Applied", "Applied", "Applied")] + [InlineData(TaskProjectionKinds.SingleAgent, "Applied", "Applied", "Applied", "NotApplicable", "NotApplicable", "NotApplicable", "NotApplicable")] + [InlineData(TaskProjectionKinds.PlanMapSynth, "Applied", "Applied", "Refused", "NotApplicable", "NotApplicable", "Applied", "Applied")] + [InlineData(TaskProjectionKinds.PlanMapDynamic, "Applied", "Applied", "Refused", "NotApplicable", "NotApplicable", "Applied", "Applied")] + public async Task Every_control_the_launch_carries_gets_exactly_one_disposition_per_lane(string lane, string models, string personas, string floor, string decisionCritic, string delivery, string planGate, string planCritic) + { + var resolution = await Resolver(new ScriptedPool()).ResolveAsync(EveryControl(), Route(lane), CancellationToken.None); + + resolution.Dispositions.Select(d => d.Control).ShouldBe(new[] { LaunchControls.AllowedModelIds, LaunchControls.AllowedAgentDefinitionIds, LaunchControls.AcceptanceChecks, LaunchControls.DecisionReviewMode, LaunchControls.DeliverySpec, LaunchControls.RequirePlanConfirmation, LaunchControls.PlannerReviewMode }, + customMessage: "a control the launch carried is missing from the list — that is the silent drop this resolver exists to end"); + + OutcomeOf(resolution, LaunchControls.AllowedModelIds).ToString().ShouldBe(models); + OutcomeOf(resolution, LaunchControls.AllowedAgentDefinitionIds).ToString().ShouldBe(personas); + OutcomeOf(resolution, LaunchControls.AcceptanceChecks).ToString().ShouldBe(floor); + OutcomeOf(resolution, LaunchControls.DecisionReviewMode).ToString().ShouldBe(decisionCritic); + OutcomeOf(resolution, LaunchControls.DeliverySpec).ToString().ShouldBe(delivery); + OutcomeOf(resolution, LaunchControls.RequirePlanConfirmation).ToString().ShouldBe(planGate); + OutcomeOf(resolution, LaunchControls.PlannerReviewMode).ToString().ShouldBe(planCritic); + + resolution.Dispositions.Where(d => d.Outcome is LaunchControlOutcome.NotApplicable or LaunchControlOutcome.Refused or LaunchControlOutcome.Clamped).ShouldAllBe(d => !string.IsNullOrWhiteSpace(d.Reason), "every disposition short of a plain apply says why"); + resolution.GradesOperatorFloor.ShouldBe(floor == "Applied", "the mandate reads the same advertisement the floor's disposition does"); + } + + [Fact] + public async Task A_launch_carrying_no_route_dependent_control_reports_none_and_never_consults_the_pool() + { + var pool = new ScriptedPool(); + + var resolution = await Resolver(pool).ResolveAsync(new TaskLaunchRequest { TeamId = Guid.NewGuid(), ActorUserId = Guid.NewGuid(), SurfaceKind = "chat" }, Route(TaskProjectionKinds.SingleAgent), CancellationToken.None); + + resolution.Dispositions.ShouldBeEmpty(); + resolution.ModelClamp.ShouldBeNull(); + pool.Calls.ShouldBe(0); + } + + [Theory] + [InlineData(TaskProjectionKinds.SingleAgent, true)] // the one agent IS the pin, so the launch bakes the pooled row + [InlineData(TaskProjectionKinds.PlanMapSynth, false)] // branches are held at dispatch; the synthesis keeps its model + public async Task A_pinned_model_row_outside_the_pool_is_clamped_to_the_pools_default(string lane, bool bakedAtLaunch) + { + var request = EveryControl() with { Overrides = new TaskExecutionOverrides { ModelCredentialModelId = ForeignRow } }; + + var resolution = await Resolver(new ScriptedPool(poolDefault: PoolDefault)).ResolveAsync(request, Route(lane), CancellationToken.None); + + var models = resolution.Dispositions.Single(d => d.Control == LaunchControls.AllowedModelIds); + models.Outcome.ShouldBe(LaunchControlOutcome.Clamped); + models.Reason.ShouldContain(ForeignRow.ToString(), customMessage: "the reason names the pin that did not fit"); + models.Reason.ShouldContain("pool-default", customMessage: "and the model that runs instead"); + (resolution.ModelClamp is not null).ShouldBe(bakedAtLaunch); + } + + [Fact] + public async Task A_pinned_model_name_is_looked_up_among_the_pools_rows_not_the_whole_team() + { + // Names repeat across credentials, so a pinned NAME is pooled only when a pooled ROW carries it. + var request = EveryControl() with { Overrides = new TaskExecutionOverrides { Model = "claude-sonnet" } }; + var pooled = new ScriptedPool(byName: new ModelDispatchRef { ModelId = "claude-sonnet", ModelCredentialId = Guid.NewGuid(), Provider = "Anthropic" }); + + var resolution = await Resolver(pooled).ResolveAsync(request, Route(TaskProjectionKinds.SingleAgent), CancellationToken.None); + + OutcomeOf(resolution, LaunchControls.AllowedModelIds).ShouldBe(LaunchControlOutcome.Applied); + resolution.ModelClamp.ShouldBeNull(); + pooled.LastAllowedRowIds.ShouldBe(new[] { PooledRow }, "the name lookup is bounded to the operator's pool"); + } + + [Fact] + public async Task A_pin_outside_a_pool_that_resolves_nothing_is_refused() + { + var request = EveryControl() with { Overrides = new TaskExecutionOverrides { ModelCredentialModelId = ForeignRow } }; + + var resolution = await Resolver(new ScriptedPool(poolDefault: null)).ResolveAsync(request, Route(TaskProjectionKinds.SingleAgent), CancellationToken.None); + + OutcomeOf(resolution, LaunchControls.AllowedModelIds).ShouldBe(LaunchControlOutcome.Refused); + resolution.ModelClamp.ShouldBeNull(); + } + + [Theory] + [InlineData(TaskProjectionKinds.SingleAgent)] + [InlineData(TaskProjectionKinds.PlanMapSynth)] + public async Task A_persona_the_pool_excludes_refuses_and_no_persona_is_not_applicable(string lane) + { + var excluded = await Resolver(new ScriptedPool()).ResolveAsync(EveryControl() with { Overrides = new TaskExecutionOverrides { AgentDefinitionId = Guid.NewGuid() } }, Route(lane), CancellationToken.None); + var none = await Resolver(new ScriptedPool()).ResolveAsync(EveryControl() with { Overrides = new TaskExecutionOverrides() }, Route(lane), CancellationToken.None); + + OutcomeOf(excluded, LaunchControls.AllowedAgentDefinitionIds).ShouldBe(LaunchControlOutcome.Refused, "every agent on this lane runs as the launch's persona — the operator excluded it"); + OutcomeOf(none, LaunchControls.AllowedAgentDefinitionIds).ShouldBe(LaunchControlOutcome.NotApplicable, "no agent here carries a persona for the pool to admit or refuse"); + } + + [Fact] + public async Task A_plan_map_floor_is_refused_with_the_previews_own_acceptance_verdict() + { + var resolution = await Resolver(new ScriptedPool()).ResolveAsync(EveryControl(), Route(TaskProjectionKinds.PlanMapSynth), CancellationToken.None); + + resolution.Dispositions.Single(d => d.Control == LaunchControls.AcceptanceChecks).Reason.ShouldBe(LaunchControlResolver.FloorNotGradedReason); + resolution.GradesOperatorFloor.ShouldBeFalse(); + } + + [Fact] + public async Task A_route_whose_builder_advertises_nothing_refuses_the_floor_and_holds_no_pool() + { + var resolution = await Resolver(new ScriptedPool()).ResolveAsync(EveryControl(), Route(UnadvertisedBuilder.Kind), CancellationToken.None); + + resolution.Dispositions.Single(d => d.Control == LaunchControls.AcceptanceChecks).Reason.ShouldBe(LaunchControlResolver.FloorUnadvertisedReason); + OutcomeOf(resolution, LaunchControls.AllowedModelIds).ShouldBe(LaunchControlOutcome.NotApplicable, "an unknown lane is never claimed to hold its agents to the pool"); + resolution.GradesOperatorFloor.ShouldBeFalse(); + } + + /// The pool's only two lookups the resolver may make, scripted; anything else is a resolver reaching past its contract. + private sealed class ScriptedPool(ModelDispatchRef? byName = null, ModelDispatchRef? poolDefault = null) : IModelPoolSelector + { + public int Calls { get; private set; } + public IReadOnlyList? LastAllowedRowIds { get; private set; } + + public Task ResolveDispatchAsync(Guid teamId, string modelName, IReadOnlyList? allowedRowIds, CancellationToken cancellationToken) + { + Calls++; + LastAllowedRowIds = allowedRowIds; + return Task.FromResult(byName); + } + + public Task ResolvePoolDefaultAsync(Guid teamId, IReadOnlyList allowedRowIds, CancellationToken cancellationToken) + { + Calls++; + return Task.FromResult(poolDefault); + } + + public Task SelectAsync(Guid teamId, string provider, IReadOnlyList? allowedModels, string? pinnedModel, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolveByRowIdAsync(Guid teamId, Guid modelCredentialModelId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task> ListPoolAsync(Guid teamId, IReadOnlyList? allowedRowIds, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task SelectBrainRowIdAsync(Guid teamId, IReadOnlyCollection eligibleProviders, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolvePinnedBrainRowIdAsync(Guid teamId, Guid modelCredentialModelId, IReadOnlyCollection eligibleProviders, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolveTeamDefaultProviderAsync(Guid teamId, CancellationToken cancellationToken) => throw new NotSupportedException(); + } + + /// A registered projection that advertises no operator-command adapter — the interface default. + private sealed class UnadvertisedBuilder : IWorkflowDefinitionBuilder + { + public const string Kind = "unadvertised-lane"; + public string ProjectionKind => Kind; + public WorkflowDefinition Build(TaskBuildContext context) => throw new NotSupportedException(); + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Tasks/TaskLaunchServiceQualityTierTests.cs b/backend/tests/CodeSpace.UnitTests/Tasks/TaskLaunchServiceQualityTierTests.cs index 0a45ed934..96b8229cb 100644 --- a/backend/tests/CodeSpace.UnitTests/Tasks/TaskLaunchServiceQualityTierTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Tasks/TaskLaunchServiceQualityTierTests.cs @@ -1,4 +1,9 @@ using CodeSpace.Core.Services.Tasks; +using CodeSpace.Core.Services.Tasks.Projection; +using CodeSpace.Core.Services.Tasks.Projection.Builders.PlanMapDynamic; +using CodeSpace.Core.Services.Tasks.Projection.Builders.PlanMapSynth; +using CodeSpace.Core.Services.Tasks.Projection.Builders.SingleAgent; +using CodeSpace.Core.Services.Tasks.Projection.Builders.Supervisor; using CodeSpace.Messages.Enums; using CodeSpace.Messages.Tasks; using Shouldly; @@ -7,11 +12,12 @@ namespace CodeSpace.UnitTests.Tasks; /// /// Pins P3.2's two tier-mandate choke points: (Delivery/ -/// Unattended on a SUPERVISOR-projected launch must carry an executable acceptanceChecks floor, fail-loud -/// otherwise) and 's tier-aware OutputReviewMode floor -/// (Delivery ⇒ at least Gate, Unattended ⇒ at least Improve — a MINIMUM an operator's explicit choice can only -/// raise, never lower). Both are pure, so they're unit-pinned directly here (no DB) — the integration tier proves -/// a Delivery launch without an acceptance check is rejected through the REAL ITaskLaunchService. +/// Unattended on a launch whose route GRADES an operator floor — the supervisor and the single agent both do — must +/// carry an executable acceptanceChecks floor, fail-loud otherwise) and 's +/// tier-aware OutputReviewMode floor (Delivery ⇒ at least Gate, Unattended ⇒ at least Improve — a MINIMUM an +/// operator's explicit choice can only raise, never lower). Both are pure, so they're unit-pinned directly here (no +/// DB) — the integration tier proves a Delivery launch without an acceptance check is rejected through the REAL +/// ITaskLaunchService. "Grades a floor" is read the way the launch reads it: the REAL builder's own advertisement. /// [Trait("Category", "Unit")] public class TaskLaunchServiceQualityTierTests @@ -29,45 +35,67 @@ public class TaskLaunchServiceQualityTierTests AcceptanceChecks = acceptanceChecks, }; - // ── EnsureAcceptanceMandate — Delivery/Unattended on a supervisor launch must carry an acceptance floor ── + // ── EnsureAcceptanceMandate — Delivery/Unattended on a route that grades a floor must carry one ── + + /// Whether the builder's route grades an operator floor — the SAME advertisement the launch's control resolution reads. + private static bool GradesFloor(IWorkflowDefinitionBuilder builder) => builder.OperatorAcceptance.AcceptsCommand == true; + + public static IEnumerable FloorGradingRoutesAtMandatedTiers() + { + foreach (var tier in new[] { QualityTier.Delivery, QualityTier.Unattended }) + { + yield return new object[] { tier, TaskProjectionKinds.Supervisor }; + yield return new object[] { tier, TaskProjectionKinds.SingleAgent }; + } + } + + private static IWorkflowDefinitionBuilder Builder(string projectionKind) => projectionKind switch + { + TaskProjectionKinds.Supervisor => new SupervisorDefinitionBuilder(), + TaskProjectionKinds.SingleAgent => new SingleAgentDefinitionBuilder(), + TaskProjectionKinds.PlanMapSynth => new PlanMapSynthDefinitionBuilder(), + _ => new PlanMapDynamicDefinitionBuilder(), + }; [Theory] - [InlineData(QualityTier.Delivery)] - [InlineData(QualityTier.Unattended)] - public void A_supervisor_launch_at_delivery_or_unattended_quality_without_an_acceptance_check_is_rejected(QualityTier tier) + [MemberData(nameof(FloorGradingRoutesAtMandatedTiers))] + public void A_launch_whose_route_grades_a_floor_at_delivery_or_unattended_quality_without_an_acceptance_check_is_rejected(QualityTier tier, string projectionKind) { + // Quick grades its single agent with the operator's argv exactly as Deep grades its terminal stop, so claiming + // Delivery there without one is the same unverified claim — it used to launch because the mandate asked + // "is this the supervisor?" rather than "does this route grade a floor?". var ex = Should.Throw(() => - TaskLaunchService.EnsureAcceptanceMandate(Request(tier), Route(TaskProjectionKinds.Supervisor))); + TaskLaunchService.EnsureAcceptanceMandate(Request(tier), GradesFloor(Builder(projectionKind)))); ex.Message.ShouldContain("acceptanceChecks", Case.Insensitive, "the operator needs an actionable name for the missing lever"); ex.Message.ShouldContain(tier.ToString()); } [Theory] - [InlineData(QualityTier.Delivery)] - [InlineData(QualityTier.Unattended)] - public void A_supervisor_launch_with_an_authored_acceptance_check_is_not_rejected(QualityTier tier) + [MemberData(nameof(FloorGradingRoutesAtMandatedTiers))] + public void A_launch_with_an_authored_acceptance_check_is_not_rejected(QualityTier tier, string projectionKind) { Should.NotThrow(() => - TaskLaunchService.EnsureAcceptanceMandate(Request(tier, new[] { "sh", "check.sh" }), Route(TaskProjectionKinds.Supervisor))); + TaskLaunchService.EnsureAcceptanceMandate(Request(tier, new[] { "sh", "check.sh" }), GradesFloor(Builder(projectionKind)))); } [Fact] - public void A_supervisor_launch_at_prototype_quality_or_no_tier_is_never_rejected() + public void A_launch_at_prototype_quality_or_no_tier_is_never_rejected() { - Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(QualityTier.Prototype), Route(TaskProjectionKinds.Supervisor))); - Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(tier: null), Route(TaskProjectionKinds.Supervisor))); + Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(QualityTier.Prototype), GradesFloor(Builder(TaskProjectionKinds.Supervisor)))); + Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(tier: null), GradesFloor(Builder(TaskProjectionKinds.SingleAgent)))); } [Theory] - [InlineData(QualityTier.Delivery)] - [InlineData(QualityTier.Unattended)] - public void A_non_supervisor_launch_at_delivery_or_unattended_quality_is_never_rejected(QualityTier tier) + [InlineData(QualityTier.Delivery, TaskProjectionKinds.PlanMapSynth)] + [InlineData(QualityTier.Delivery, TaskProjectionKinds.PlanMapDynamic)] + [InlineData(QualityTier.Unattended, TaskProjectionKinds.PlanMapSynth)] + [InlineData(QualityTier.Unattended, TaskProjectionKinds.PlanMapDynamic)] + public void A_plan_map_launch_at_delivery_or_unattended_quality_is_not_asked_for_a_floor_it_cannot_grade(QualityTier tier, string projectionKind) { - // AcceptanceChecks is inert on a non-supervisor projection today — this PR doesn't invent new acceptance-floor - // plumbing for single-agent/plan-map launches, so the mandate is inert there too, matching that existing shape. - Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(tier), Route(TaskProjectionKinds.SingleAgent))); - Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(tier), Route(TaskProjectionKinds.PlanMapDynamic))); + // Plan-map grades no operator floor (its items carry their own contracts), and a floor sent to it is refused + // before the mandate runs — demanding one here would make the lane unlaunchable at these tiers. + Should.NotThrow(() => TaskLaunchService.EnsureAcceptanceMandate(Request(tier), GradesFloor(Builder(projectionKind)))); } // ── BuildAgentProfile's tier-aware OutputReviewMode floor ── diff --git a/backend/tests/CodeSpace.UnitTests/Tasks/TaskRoutePreviewServiceTests.cs b/backend/tests/CodeSpace.UnitTests/Tasks/TaskRoutePreviewServiceTests.cs index 1dbbf1663..03fa839b2 100644 --- a/backend/tests/CodeSpace.UnitTests/Tasks/TaskRoutePreviewServiceTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Tasks/TaskRoutePreviewServiceTests.cs @@ -44,17 +44,23 @@ public class TaskRoutePreviewServiceTests { private static readonly JsonSerializerOptions Json = new(JsonSerializerDefaults.Web); + private static TaskProjectionRegistry Projections() => new([new SingleAgentDefinitionBuilder(), new SupervisorDefinitionBuilder(), new PlanMapSynthDefinitionBuilder()]); + private static IEffortRouter Router() => new EffortRouter( new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier() }), new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe(), new MapFanoutRecipe(), new SupervisorRecipe() }), new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + Projections()); + + /// The launch's own control resolver over the same builders. No preview here sets a model pool, so the pool is never consulted — a lookup would throw. + private static LaunchControlResolver Controls() => new(Projections(), new UnconsultedModelPool()); private static TaskRoutePreviewService Preview(IEffortRouter router) => new( new TaskLaunchSeedProviderRegistry(new ITaskLaunchSeedProvider[] { new ChatSeedProvider() }), new AllRepositoriesInTeam(), - new RoutingOnlySnapshotStore(router), new TaskProjectionRegistry([new SingleAgentDefinitionBuilder(), new SupervisorDefinitionBuilder(), new PlanMapSynthDefinitionBuilder()]), - new ModeProfileRegistry()); + new RoutingOnlySnapshotStore(router), Projections(), + new ModeProfileRegistry(), Controls()); private static TaskLaunchRequest Request(string goal, string? effort = null, string? recipe = null, RouteCaps? caps = null, string? shape = null, string? autonomy = null, string? completionMode = null) => new() { @@ -156,13 +162,49 @@ public async Task Adapter_compatibility_comes_from_the_actual_builder_and_worksp public async Task Unadvertised_projection_compatibility_stays_unknown_without_changing_the_route_identity() { var request = Request("Inspect this task", TaskEffortModes.Quick) with { RepositoryId = Guid.NewGuid() }; - var service = new TaskRoutePreviewService(new TaskLaunchSeedProviderRegistry([new ChatSeedProvider()]), new AllRepositoriesInTeam(), new RoutingOnlySnapshotStore(Router()), new TaskProjectionRegistry([]), new ModeProfileRegistry()); + var service = new TaskRoutePreviewService(new TaskLaunchSeedProviderRegistry([new ChatSeedProvider()]), new AllRepositoriesInTeam(), new RoutingOnlySnapshotStore(Router()), new TaskProjectionRegistry([]), new ModeProfileRegistry(), Controls()); var unknown = await service.PreviewAsync(request, CancellationToken.None); var known = await Preview(Router()).PreviewAsync(request, CancellationToken.None); unknown.AcceptanceCompatibility!.State.ShouldBe(TaskAcceptanceCompatibilityState.Unknown); JsonSerializer.Serialize(unknown.Route, Json).ShouldBe(JsonSerializer.Serialize(known.Route, Json)); } + // ─── Control dispositions: the preview states what the launch will do with each route-dependent control ───────── + + [Fact] + public async Task Preview_reports_the_same_control_dispositions_the_launch_resolves_for_the_same_input() + { + // An explicit Standard launch carrying an operator floor and a decision critic: plan-map grades no floor (the + // launch refuses it — with the preview's own acceptance verdict as the reason) and reviews no decisions. + var request = Request("Validate the report", TaskEffortModes.Standard) with { AcceptanceChecks = ["sh", "check.sh"], DecisionReviewMode = Messages.Enums.ReviewMode.Gate, RepositoryId = Guid.NewGuid() }; + + var preview = await Preview(Router()).PreviewAsync(request, CancellationToken.None); + var launch = await Controls().ResolveAsync(request, preview.Route, CancellationToken.None); + + preview.ControlDispositions.ShouldNotBeNull(); + JsonSerializer.Serialize(preview.ControlDispositions, Json).ShouldBe(JsonSerializer.Serialize(launch.Dispositions, Json), + customMessage: "the preview must call the launch's own resolver — a divergence here is a preview promising a launch it will not get"); + + var floor = preview.ControlDispositions!.Single(d => d.Control == LaunchControls.AcceptanceChecks); + floor.Outcome.ShouldBe(LaunchControlOutcome.Refused); + floor.Reason.ShouldBe(preview.AcceptanceCompatibility!.Detail, "the refusal names the same reason the preview's acceptance verdict gives"); + + preview.ControlDispositions!.Single(d => d.Control == LaunchControls.DecisionReviewMode).Outcome.ShouldBe(LaunchControlOutcome.NotApplicable); + } + + [Fact] + public async Task A_disposition_serializes_its_outcome_by_name_and_a_control_free_input_reports_none() + { + var preview = await Preview(Router()).PreviewAsync(Request("Fix a small typo", TaskEffortModes.Quick) with { DeliverySpec = new DeliverySpec { OpenPullRequest = true } }, CancellationToken.None); + + var wire = JsonSerializer.SerializeToElement(preview, Json).GetProperty("controlDispositions").EnumerateArray().Single(); + wire.GetProperty("control").GetString().ShouldBe("deliverySpec"); + wire.GetProperty("outcome").GetString().ShouldBe("NotApplicable", "the wire carries the outcome's name, not an ordinal a client would have to decode"); + wire.GetProperty("reason").GetString().ShouldNotBeNullOrWhiteSpace(); + + (await Preview(Router()).PreviewAsync(Request("Fix a small typo", TaskEffortModes.Quick), CancellationToken.None)).ControlDispositions.ShouldBeEmpty(); + } + // ─── Posture (arc3 item 3.2): "preview is not the run" ───────────────────────────────────────────────────── [Theory] @@ -260,6 +302,18 @@ private sealed class RoutingOnlySnapshotStore(IEffortRouter router) : ITaskRoute public Task ConsumeAsync(TaskRouteSnapshotConsumption consumption, CancellationToken cancellationToken) => throw new NotSupportedException(); } + /// A model pool no preview here may consult: none of them sets allowedModelIds, so any lookup is a resolver reaching for the pool when it has nothing to bound. + private sealed class UnconsultedModelPool : Core.Services.Agents.ModelCredentials.IModelPoolSelector + { + public Task SelectAsync(Guid teamId, string provider, IReadOnlyList? allowedModels, string? pinnedModel, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolveByRowIdAsync(Guid teamId, Guid modelCredentialModelId, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolveDispatchAsync(Guid teamId, string modelName, IReadOnlyList? allowedRowIds, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task> ListPoolAsync(Guid teamId, IReadOnlyList? allowedRowIds, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task SelectBrainRowIdAsync(Guid teamId, IReadOnlyCollection eligibleProviders, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolvePinnedBrainRowIdAsync(Guid teamId, Guid modelCredentialModelId, IReadOnlyCollection eligibleProviders, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ResolveTeamDefaultProviderAsync(Guid teamId, CancellationToken cancellationToken) => throw new NotSupportedException(); + } + /// A guard that accepts every repo — tenancy itself is proven against real Postgres in the integration tier; these tests pin the routing, not the query. private sealed class AllRepositoriesInTeam : ILaunchRepositoryScopeGuard { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs index c9e76edc1..b20421abd 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentCodeNodeTests.cs @@ -429,6 +429,43 @@ public async Task A_malformed_credentialed_model_id_fails_the_node() result.Error.ShouldContain("modelCredentialModelId"); } + [Fact] + public async Task The_allowed_model_pool_a_projection_baked_is_carried_onto_the_task() + { + var pool = new[] { Guid.NewGuid(), Guid.NewGuid() }; + var config = new Dictionary(RequiredConfig()) { ["allowedModelIds"] = JsonSerializer.SerializeToElement(pool.Select(id => id.ToString())) }; + + var result = await new AgentCodeNode().RunAsync(BuildContext(config, resume: null), CancellationToken.None); + + result.Status.ShouldBe(NodeStatus.Suspended); + JsonSerializer.Deserialize(result.SuspendUntil!.Payload, AgentJson.Options)!.AllowedModelIds.ShouldBe(pool, "dispatch holds the model to the pool the task carries"); + } + + [Fact] + public async Task An_absent_allowed_model_pool_leaves_the_task_unbounded_and_its_json_byte_identical() + { + var result = await new AgentCodeNode().RunAsync(BuildContext(RequiredConfig(), resume: null), CancellationToken.None); + + JsonSerializer.Deserialize(result.SuspendUntil!.Payload, AgentJson.Options)!.AllowedModelIds.ShouldBeNull(); + result.SuspendUntil.Payload.GetRawText().ShouldNotContain("allowedModelIds"); + } + + [Theory] + [InlineData("[\"not-a-uuid\"]")] + [InlineData("\"a-single-string\"")] + [InlineData("[42]")] + public async Task A_malformed_allowed_model_pool_fails_the_node_rather_than_run_on_a_different_bound(string raw) + { + // A pool is a bound: silently skipping an entry would run the agent on a different set of models than the + // operator allowed — and an all-skipped pool on none of them — so it fails like any malformed id. + var config = new Dictionary(RequiredConfig()) { ["allowedModelIds"] = JsonDocument.Parse(raw).RootElement.Clone() }; + + var result = await new AgentCodeNode().RunAsync(BuildContext(config, resume: null), CancellationToken.None); + + result.Status.ShouldBe(NodeStatus.Failure); + result.Error.ShouldContain("allowedModelIds"); + } + [Fact] public async Task An_unset_credentialed_model_is_omitted_from_the_staged_task_json() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterDegradeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterDegradeTests.cs index 6d29ead4f..83b4d9bab 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterDegradeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterDegradeTests.cs @@ -9,6 +9,7 @@ using CodeSpace.Core.Services.Tasks.Recipes.MapFanout; using CodeSpace.Core.Services.Tasks.Recipes.SingleAgent; using CodeSpace.Core.Services.Tasks.Recipes.Supervisor; +using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Messages.Tasks; using CodeSpace.Messages.Tasks.Effort; using Shouldly; @@ -43,7 +44,8 @@ public async Task The_degrade_is_data_driven_a_fake_recipe_and_fake_capability_d new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier() }), new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe(), new FakeGatedRecipe() }), new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset() }), - new CapabilityProbeRegistry(new ICapabilityProbe[] { new FakeProbe(FakeGatedRecipe.FakeCapability, available: false) })); + new CapabilityProbeRegistry(new ICapabilityProbe[] { new FakeProbe(FakeGatedRecipe.FakeCapability, available: false) }), + new TaskProjectionRegistry(Array.Empty())); var plan = await router.RouteAsync(Request(FakeGatedRecipe.FakeTier), CancellationToken.None); diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterGenericityTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterGenericityTests.cs index cc6640ae5..f4a256c89 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterGenericityTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterGenericityTests.cs @@ -6,6 +6,7 @@ using CodeSpace.Core.Services.Tasks.Effort.Classifiers.Heuristic; using CodeSpace.Core.Services.Tasks.Recipes; using CodeSpace.Core.Services.Tasks.Recipes.SingleAgent; +using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Messages.Tasks; using CodeSpace.Messages.Tasks.Effort; using Shouldly; @@ -60,7 +61,8 @@ public async Task Routing_a_request_pinned_to_the_fake_recipe_picks_up_its_proje new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), new FakeClassifier() }), new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe(), new FakeRecipe() }), new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new FakeBounds() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + new TaskProjectionRegistry(Array.Empty())); // RequestedEffort = the fake bounds kind so the effort-mode ≡ preset-kind convention resolves the fake caps; // RequestedRecipe = the fake recipe so its DefaultProjectionKind drives the projection. @@ -94,7 +96,8 @@ public async Task Auto_path_routes_via_the_registry_Default_classifier_not_a_nam new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), new FakeClassifier() }), new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe() }), new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + new TaskProjectionRegistry(Array.Empty())); var plan = await router.RouteAsync(Request(), CancellationToken.None); diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterTests.cs index 51842474e..8c2dc435f 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/EffortRouterTests.cs @@ -9,6 +9,7 @@ using CodeSpace.Core.Services.Tasks.Recipes.MapFanout; using CodeSpace.Core.Services.Tasks.Recipes.SingleAgent; using CodeSpace.Core.Services.Tasks.Recipes.Supervisor; +using CodeSpace.Core.Services.Tasks.Projection; using CodeSpace.Messages.Tasks; using CodeSpace.Messages.Tasks.Effort; using Shouldly; @@ -31,7 +32,8 @@ public class EffortRouterTests new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier() }), new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe(), new MapFanoutRecipe(), new SupervisorRecipe() }), new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + new TaskProjectionRegistry(Array.Empty())); private static EffortRouteRequest Request(string goal, string? requestedEffort = null, string? requestedRecipe = null, string? requestedProjection = null, RouteCaps? capsOverride = null, string? deliverableShape = null) => new() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/LlmEffortClassifierTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/LlmEffortClassifierTests.cs index 310042878..be6519322 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/LlmEffortClassifierTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/LlmEffortClassifierTests.cs @@ -8,6 +8,11 @@ using CodeSpace.Core.Services.Tasks.Effort; using CodeSpace.Core.Services.Tasks.Effort.Classifiers.Heuristic; using CodeSpace.Core.Services.Tasks.Effort.Classifiers.Llm; +using CodeSpace.Core.Services.Tasks.Projection; +using CodeSpace.Core.Services.Tasks.Projection.Builders.PlanMapDynamic; +using CodeSpace.Core.Services.Tasks.Projection.Builders.PlanMapSynth; +using CodeSpace.Core.Services.Tasks.Projection.Builders.SingleAgent; +using CodeSpace.Core.Services.Tasks.Projection.Builders.Supervisor; using CodeSpace.Core.Services.Tasks.Recipes; using CodeSpace.Core.Services.Tasks.Recipes.MapFanout; using CodeSpace.Core.Services.Tasks.Recipes.SingleAgent; @@ -171,7 +176,8 @@ public async Task The_router_auto_path_uses_the_llm_and_skips_the_confirm_card_w new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), llm }), recipes, new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + ProductionProjections()); var plan = await router.RouteAsync(Request("Add validation to the signup endpoint across the service with tests"), CancellationToken.None); @@ -194,7 +200,8 @@ public async Task The_router_auto_path_STILL_confirms_when_the_llm_is_unsure() new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), llm }), recipes, new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + ProductionProjections()); var plan = await router.RouteAsync(Request("do the thing"), CancellationToken.None); @@ -215,7 +222,8 @@ public async Task The_router_STILL_confirms_a_RISKY_task_even_when_the_model_is_ new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), llm }), recipes, new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() }), - new CapabilityProbeRegistry(Array.Empty())); + new CapabilityProbeRegistry(Array.Empty()), + ProductionProjections()); var plan = await router.RouteAsync(Request("Drop the production users table and deploy"), CancellationToken.None); @@ -225,6 +233,79 @@ public async Task The_router_STILL_confirms_a_RISKY_task_even_when_the_model_is_ plan.EffortMode.ShouldBe(TaskEffortModes.Deep, "the risky task still routes to deep"); } + [Fact] + public async Task The_router_STILL_confirms_an_AMBIGUOUS_task_even_when_the_model_is_confident() + { + // The ambiguity veto: a confident model that ALSO reports the goal under-specified must not have it routed as + // though it were understood. The signal was produced by every classifier and read by nothing, while the confirm + // exception told the operator it was stopping "this ambiguous or potentially risky task". + var plan = await LlmRouter(Reply(needsCodeChange: true, ambiguous: true, confidence: 0.9)).RouteAsync(Request("Make the dashboard better"), CancellationToken.None); + + plan.ClassifierConfidence.ShouldBe(0.9, "the model was confident"); + plan.NeedsConfirmCard.ShouldBeTrue("an ambiguous goal confirms regardless of model confidence — the gate the model cannot talk its way past"); + plan.Confirm.ShouldNotBeNull(); + } + + [Fact] + public async Task An_operator_floor_moves_a_confident_standard_route_off_the_plan_map_lane_and_says_so() + { + // Standard's default lane (plan-map-synth) grades no operator command, so an auto route that must honour an + // acceptance floor takes the policy's next matching row — the single-agent tier, whose one agent the floor grades. + // The same reply without a floor still routes plan-map: the confident-routing win is untouched. + var reply = Reply(needsCodeChange: true, crossFile: true, confidence: 0.9); + + var unfloored = await LlmRouter(reply).RouteAsync(Request(), CancellationToken.None); + var floored = await LlmRouter(reply).RouteAsync(Request() with { HasOperatorFloor = true }, CancellationToken.None); + + unfloored.ProjectionKind.ShouldBe(TaskProjectionKinds.PlanMapSynth); + floored.EffortMode.ShouldBe(TaskEffortModes.Quick, "the Standard row is set aside, and the next matching row down is the cheap catch-all"); + floored.ProjectionKind.ShouldBe(TaskProjectionKinds.SingleAgent, "an operator floor never lands an auto route on a lane that cannot grade it"); + floored.NeedsConfirmCard.ShouldBeFalse("the move is the router's own on a confident route — there is nothing new for the operator to confirm"); + floored.DegradedReason.ShouldNotBeNull("the route moved, so it says why — never silent"); + floored.DegradedReason!.ShouldContain(TaskProjectionKinds.PlanMapSynth); + } + + [Theory] + [InlineData(TaskEffortModes.Standard, null)] // the operator chose the tier + [InlineData(null, TaskRecipeKinds.MapFanout)] // the operator pinned the recipe on the auto path + public async Task An_operator_floor_never_moves_a_lane_the_operator_chose(string? effort, string? recipe) + { + // The launch refuses the floor on a chosen plan-map lane by name; rerouting it would override the operator. + var request = Request() with { HasOperatorFloor = true, RequestedEffort = effort, RequestedRecipe = recipe }; + + var plan = await LlmRouter(Reply(needsCodeChange: true, crossFile: true, confidence: 0.9)).RouteAsync(request, CancellationToken.None); + + plan.ProjectionKind.ShouldBe(TaskProjectionKinds.PlanMapSynth); + plan.DegradedReason.ShouldBeNull(); + } + + [Fact] + public async Task An_operator_floor_keeps_a_route_that_already_grades_it() + { + // Deep's supervisor lane grades the floor at its terminal stop — nothing to move. + var plan = await LlmRouter(Reply(estimatedCostTier: "high", confidence: 0.9)).RouteAsync(Request() with { HasOperatorFloor = true }, CancellationToken.None); + + plan.ProjectionKind.ShouldBe(TaskProjectionKinds.Supervisor); + plan.DegradedReason.ShouldBeNull(); + } + + /// The production router over a canned LLM reply, the real recipes + bounds presets, and the REAL projection builders — whose OperatorAcceptance advertisements are what the floor rule reads. + private static EffortRouter LlmRouter(JsonElement reply) + { + var recipes = new TaskRecipeRegistry(new ITaskRecipe[] { new SingleAgentRecipe(), new MapFanoutRecipe(), new SupervisorRecipe() }); + var llm = new LlmEffortClassifier(new FakeClients(new CannedClient(reply)), new FakeSelector(Pick()), recipes, new HeuristicEffortClassifier()); + + return new EffortRouter( + new EffortClassifierRegistry(new IEffortClassifier[] { new HeuristicEffortClassifier(), llm }), + recipes, + new BoundsPresetRegistry(new IBoundsPreset[] { new QuickBoundsPreset(), new StandardBoundsPreset(), new DeepBoundsPreset() }), + new CapabilityProbeRegistry(Array.Empty()), + ProductionProjections()); + } + + private static TaskProjectionRegistry ProductionProjections() => + new(new IWorkflowDefinitionBuilder[] { new SingleAgentDefinitionBuilder(), new PlanMapSynthDefinitionBuilder(), new PlanMapDynamicDefinitionBuilder(), new SupervisorDefinitionBuilder() }); + // ── Schema commit-contract pin ── [Fact] diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/PlanAuthorNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/PlanAuthorNodeTests.cs index 2ebb51478..0584efabb 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/PlanAuthorNodeTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/PlanAuthorNodeTests.cs @@ -102,6 +102,18 @@ public void Model_pins_and_review_mode_map_into_the_planner_request() request.ReviewerModelId.ShouldBe(reviewerRow); } + [Fact] + public void The_allowed_model_pool_maps_into_the_planner_request_and_an_absent_one_is_unbounded() + { + // The pool bounds the capability catalog the planner allocates subtasks from; each branch's agent.run is where + // it is enforced, so the node reads it defensively like every other planner knob. + var pooled = Guid.NewGuid(); + + PlanAuthorNode.BuildPlanRequest(Config($$"""{"allowedModelIds":["{{pooled}}","not-a-uuid"]}"""), Guid.NewGuid(), new PlanAuthorNode.PlanPromptParts("goal", [], "", "")).AllowedModelIds.ShouldBe(new[] { pooled }); + PlanAuthorNode.BuildPlanRequest(Config("""{"allowedModelIds":["not-a-uuid"]}"""), Guid.NewGuid(), new PlanAuthorNode.PlanPromptParts("goal", [], "", "")).AllowedModelIds.ShouldBeNull(); + PlanAuthorNode.BuildPlanRequest(Config("""{}"""), Guid.NewGuid(), new PlanAuthorNode.PlanPromptParts("goal", [], "", "")).AllowedModelIds.ShouldBeNull("no pool ⇒ the whole team pool — the catalog the planner has always seen"); + } + [Fact] public void The_launch_base_pin_maps_into_the_planner_request_and_a_blank_pin_is_omitted() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/PlanMapSynthDefinitionBuilderTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/PlanMapSynthDefinitionBuilderTests.cs index 3b1d227cc..2bf2948cd 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/PlanMapSynthDefinitionBuilderTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/PlanMapSynthDefinitionBuilderTests.cs @@ -67,6 +67,24 @@ public void The_projection_declares_the_whole_task_operator_floor_it_does_not_co nodes.ShouldAllBe(n => !n.Config.GetRawText().Contains("unique-task-oracle")); } + [Fact] + public void The_allowed_model_pool_bounds_the_planners_catalog_and_rides_every_branch() + { + // The planner allocates each subtask a model from the catalog it is shown, and the branch body is where a + // planner-authored {{item.model}} outside the pool is held to it at dispatch — both must carry the pool. + var pool = new[] { Guid.NewGuid(), Guid.NewGuid() }; + + var bounded = Builder.Build(Context() with { AllowedModelIds = pool }); + var unbounded = Builder.Build(Context()); + + var plannerConfig = bounded.Nodes.Single(n => n.Id == "planner").Config.Deserialize>()!; + PlanAuthorNode.BuildPlanRequest(plannerConfig, Guid.NewGuid(), new PlanAuthorNode.PlanPromptParts("goal", [], "", "")).AllowedModelIds.ShouldBe(pool, "the planner request the plan.author node builds from this very config carries the pool"); + bounded.Nodes.Single(n => n.Id == "agent").Config.GetProperty("allowedModelIds").EnumerateArray().Select(e => Guid.Parse(e.GetString()!)).ShouldBe(pool); + RealValidator().Validate(bounded).IsValid.ShouldBeTrue(); + + unbounded.Nodes.ShouldAllBe(n => !n.Config.GetRawText().Contains("allowedModelIds"), "no pool ⇒ no node carries the key — byte-identical"); + } + [Fact] public void Emits_the_planner_map_agent_synth_graph() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/SingleAgentDefinitionBuilderTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/SingleAgentDefinitionBuilderTests.cs index 5ffdea2bd..71cd79fbd 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/SingleAgentDefinitionBuilderTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/SingleAgentDefinitionBuilderTests.cs @@ -292,6 +292,23 @@ public void The_agent_node_carries_the_default_transient_retry() private static JsonElement AgentInputsOf(WorkflowDefinition def) => def.Nodes.Single(n => n.Id == "agent").Inputs; private static JsonElement TerminalInputsOf(WorkflowDefinition def) => def.Nodes.Single(n => n.Id == "done").Inputs; + // ── The operator's allowed model pool rides the one agent, where dispatch holds its model to the pool ── + + [Fact] + public void The_allowed_model_pool_rides_the_agent_node_and_an_unbounded_launch_is_byte_identical() + { + var pool = new[] { Guid.NewGuid(), Guid.NewGuid() }; + + var bounded = Builder.Build(Context(Seed(), profile: null) with { AllowedModelIds = pool }); + var unbounded = AgentConfigOf(Builder.Build(Context(Seed(), profile: null))); + + AgentConfigOf(bounded).GetProperty("allowedModelIds").EnumerateArray().Select(e => Guid.Parse(e.GetString()!)).ShouldBe(pool, "the agent.run carries the pool onto its AgentTask, where dispatch holds the model to it"); + RealValidator().Validate(bounded).IsValid.ShouldBeTrue(); + + unbounded.TryGetProperty("allowedModelIds", out _).ShouldBeFalse("no pool ⇒ the key is omitted"); + AgentConfigOf(Builder.Build(Context(Seed(), profile: null) with { AllowedModelIds = [] })).GetRawText().ShouldBe(unbounded.GetRawText(), "an empty pool is the whole team pool — byte-identical"); + } + // ── S5: the quick tier's operator checks floor becomes the single agent's contract ── [Fact]