Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs
Original file line number Diff line number Diff line change
Expand Up @@ -334,6 +334,8 @@
var reconciliation = await _harnessReconciler.ReconcileAsync(task, run.TeamId, cancellationToken).ConfigureAwait(false);
var harness = _harnesses.Resolve(reconciliation.HarnessKind);

task = await HoldToModelPoolAsync(owner, task, reconciliation, cancellationToken).ConfigureAwait(false);

if (reconciliation.Repaired)
{
_logger.LogWarning("AgentRun {RunId}: {Note}", agentRunId, reconciliation.Note);
Expand Down Expand Up @@ -4027,6 +4029,32 @@
}
}

/// <summary>
/// Hold a bounded run to its allowed model pool (<see cref="AgentTask.AllowedModelIds"/>): run the reconciler's pooled
/// row — its model on its own credential — and persist it, so a re-attach and every reader see the model the agent
/// actually runs; a model that had to move is named on the run's timeline. A pool that resolves nothing any more fails
/// the run rather than let the agent run outside it. An unbounded task passes through untouched (byte-identical). The
/// task is still the ORIGINAL (no injected secret env), so serializing it is safe.
/// </summary>
private async Task<AgentTask> HoldToModelPoolAsync(AgentRunOwnerToken owner, AgentTask task, HarnessReconciliation reconciliation, CancellationToken cancellationToken)
{
if (task.AllowedModelIds is not { Count: > 0 }) return task;

if (reconciliation.PooledModel is not { } pooled)
throw new InvalidOperationException(reconciliation.PoolNote);

if (reconciliation.PoolNote is { } note)
{
_logger.LogWarning("AgentRun {RunId}: {Note}", owner.RunId, note);
await _runs.AppendEventAsync(owner, new AgentEvent { Kind = AgentEventKind.Warning, Text = note }, cancellationToken).ConfigureAwait(false);
}

var held = task with { Model = pooled.ModelId, ModelCredentialId = pooled.ModelCredentialId, ModelCredentialModelId = null };
await PersistRuntimeIdentityAsync(owner, null, JsonSerializer.Serialize(held, AgentJson.Options), cancellationToken).ConfigureAwait(false);

return held;
}

/// <summary>Re-persist the run's stored task with its RESOLVED model filled, so the live projection shows what an "auto" run actually dispatches from the moment it starts (mirrors the harness-reconciliation write). The task is the ORIGINAL (no injected secret env) with only <see cref="AgentTask.Model"/> set, so serializing it is safe.</summary>
private Task PersistResolvedModelAsync(AgentRunOwnerToken owner, AgentTask taskWithModel, CancellationToken cancellationToken) => PersistRuntimeIdentityAsync(owner, null, JsonSerializer.Serialize(taskWithModel, AgentJson.Options), cancellationToken);

Expand Down Expand Up @@ -5002,10 +5030,10 @@
/// <summary>The same reconstruction from a payload that came from somewhere other than the row — an offloaded one fetched back out of the artifact store.</summary>
private static AgentEvent ReplayedEvent(AgentEventKind kind, string? text, string? dataJson)
{
if (dataJson is not { Length: > 0 } json) return new AgentEvent { Kind = kind, Text = text };

Check warning on line 5033 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / recurring jobs fire (worker host · Postgres)

Possible null reference assignment.

Check warning on line 5033 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (E2ETests · HTTP · Postgres)

Possible null reference assignment.

Check warning on line 5033 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (UnitTests)

Possible null reference assignment.

Check warning on line 5033 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (UnitTests)

Possible null reference assignment.

Check warning on line 5033 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (IntegrationTests · Postgres)

Possible null reference assignment.

try { using var doc = JsonDocument.Parse(json); return new AgentEvent { Kind = kind, Text = text, Data = doc.RootElement.Clone() }; }

Check warning on line 5035 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / recurring jobs fire (worker host · Postgres)

Possible null reference assignment.

Check warning on line 5035 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (E2ETests · HTTP · Postgres)

Possible null reference assignment.

Check warning on line 5035 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (UnitTests)

Possible null reference assignment.

Check warning on line 5035 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (UnitTests)

Possible null reference assignment.

Check warning on line 5035 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (IntegrationTests · Postgres)

Possible null reference assignment.
catch (JsonException) { return new AgentEvent { Kind = kind, Text = text }; }

Check warning on line 5036 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / recurring jobs fire (worker host · Postgres)

Possible null reference assignment.

Check warning on line 5036 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (E2ETests · HTTP · Postgres)

Possible null reference assignment.

Check warning on line 5036 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (UnitTests)

Possible null reference assignment.

Check warning on line 5036 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (UnitTests)

Possible null reference assignment.

Check warning on line 5036 in backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs

View workflow job for this annotation

GitHub Actions / dotnet test (IntegrationTests · Postgres)

Possible null reference assignment.
}

/// <summary>Ask the row, on a token of its own, whether the run actually reached a terminal state — the only honest answer to "did the landing take?" once an exception has been raised somewhere after the fenced write.</summary>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -36,15 +36,21 @@ namespace CodeSpace.Core.Services.Agents;
/// row, so a pin repair makes the whole (harness, model, credential) triple runnable. Blanking the model here would
/// drop a VALID operator choice in the common consistent case, so this layer never does; it only swaps the harness for
/// one that can drive the model's provider.</para>
///
/// <para>The ONE exception is the run's allowed model pool (<see cref="AgentTask.AllowedModelIds"/>, the model analogue of
/// the harness allow-list): a bounded task runs on a POOLED ROW — its named model's pooled row, else (no name, or a name
/// no pooled row carries, such as a planner-authored model outside the pool) the pool's default row — reported as
/// <see cref="HarnessReconciliation.PooledModel"/> for the executor to run and persist, and the harness is reconciled
/// against THAT row's provider. An unbounded task is reconciled exactly as before.</para>
/// </summary>
public interface IHarnessModelReconciler
{
/// <summary>Resolve the harness KIND to ACTUALLY run for <paramref name="task"/>: the authored one when it can drive the model's provider (from the pinned credential or, failing a pin, the named model's pool row), else a registered, run-ADMITTED harness that can (the always-runnable fallback). The caller resolves the kind to an adapter, so the registry stays the single owner of kind→adapter.</summary>
Task<HarnessReconciliation> ReconcileAsync(AgentTask task, Guid teamId, CancellationToken cancellationToken);
}

/// <summary>The harness KIND to run, whether it was REPAIRED away from the authored one, and a human-facing note for the timeline when it was.</summary>
public sealed record HarnessReconciliation(string HarnessKind, bool Repaired, string? Note);
/// <summary>The harness KIND to run, whether it was REPAIRED away from the authored one, and a human-facing note for the timeline when it was — plus, for a task bounded to an allowed model pool, the pooled row it runs on (null when unbounded, or when the pool resolves nothing any more) and a note when the authored model had to move or cannot be honoured.</summary>
public sealed record HarnessReconciliation(string HarnessKind, bool Repaired, string? Note, ModelDispatchRef? PooledModel = null, string? PoolNote = null);

public sealed class HarnessModelReconciler : IHarnessModelReconciler, IScopedDependency
{
Expand All @@ -61,11 +67,15 @@ public HarnessModelReconciler(IAgentHarnessRegistry harnesses, IModelPoolSelecto

public async Task<HarnessReconciliation> ReconcileAsync(AgentTask task, Guid teamId, CancellationToken cancellationToken)
{
var provider = await ResolveModelProviderAsync(task, teamId, cancellationToken).ConfigureAwait(false);
var pooled = await ResolvePooledModelAsync(task, teamId, cancellationToken).ConfigureAwait(false);
var poolNote = DescribePoolBound(task, pooled);

// A bounded task runs on its pooled row, so the harness follows THAT row's provider, not the authored model's.
var provider = pooled?.Provider ?? await ResolveModelProviderAsync(task, teamId, cancellationToken).ConfigureAwait(false);

// No provider to reconcile against (no pin AND no pooled model name) → return the authored kind verbatim (the
// caller's registry resolves it; a genuinely-unregistered kind surfaces there, unchanged).
if (provider is null) return new HarnessReconciliation(task.Harness, false, null);
if (provider is null) return new HarnessReconciliation(task.Harness, false, null, pooled, poolNote);

// The repair chooses from the registry CLAMPED to the run's harness allow-list (null/empty = the whole registry,
// which is every non-supervisor path and every pre-field task envelope). Without this clamp the run-time repair
Expand All @@ -75,7 +85,32 @@ public async Task<HarnessReconciliation> ReconcileAsync(AgentTask task, Guid tea
// admitted one, so the floor stays inside the list too.
var pool = AgentHarnessPool.Clamp(_harnesses.All, task.AllowedHarnessKinds);

return Reconcile(task.Harness, provider, pool, AgentHarnessDefaults.DefaultHarness);
return Reconcile(task.Harness, provider, pool, AgentHarnessDefaults.DefaultHarness) with { PooledModel = pooled, PoolNote = poolNote };
}

/// <summary>
/// The allowed-pool row a bounded task runs on: its named model's pooled row, else — no name, or a name no pooled row
/// carries — the pool's default row, ranked by the same agent-plane precedence the supervisor's pool-bound default uses
/// (names repeat across credentials, so both lookups are over the pool's ROWS). Null for an unbounded task, and when
/// nothing in the pool resolves any more.
/// </summary>
private async Task<ModelDispatchRef?> ResolvePooledModelAsync(AgentTask task, Guid teamId, CancellationToken cancellationToken)
{
if (task.AllowedModelIds is not { Count: > 0 } pool) return null;

var named = string.IsNullOrWhiteSpace(task.Model) ? null : await _modelSelector.ResolveDispatchAsync(teamId, task.Model, pool, cancellationToken).ConfigureAwait(false);

return named ?? await _modelSelector.ResolvePoolDefaultAsync(teamId, pool, cancellationToken).ConfigureAwait(false);
}

/// <summary>Why a bounded task's model did not run as authored — it was outside the pool, or nothing in the pool resolves any more. Null when unbounded, when the named model is pooled, and when no model was named (the pool's default is then simply the model it runs).</summary>
private static string? DescribePoolBound(AgentTask task, ModelDispatchRef? pooled)
{
if (task.AllowedModelIds is not { Count: > 0 }) return null;
if (pooled is null) return "None of this run's allowed models resolves to an enabled model under an active credential, so the agent cannot run inside its allowed model pool.";
if (string.IsNullOrWhiteSpace(task.Model) || string.Equals(task.Model.Trim(), pooled.ModelId, StringComparison.OrdinalIgnoreCase)) return null;

return $"Model '{task.Model}' is not in this run's allowed model pool; running the pool's default, '{pooled.ModelId}', instead.";
}

/// <summary>
Expand Down
62 changes: 51 additions & 11 deletions backend/src/CodeSpace.Core/Services/Tasks/Effort/EffortRouter.cs
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@
using CodeSpace.Core.Services.Agents;
using CodeSpace.Core.Services.Tasks.Bounds;
using CodeSpace.Core.Services.Tasks.Capabilities;
using CodeSpace.Core.Services.Tasks.Projection;
using CodeSpace.Core.Services.Tasks.Recipes;
using CodeSpace.Messages.Agents;
using CodeSpace.Messages.Tasks;
Expand All @@ -15,22 +16,25 @@ namespace CodeSpace.Core.Services.Tasks.Effort;
/// type: every branch point is a registry lookup, so a new classification / recipe / bounds / capability
/// strategy needs zero edit here (the fake-probe + fake-recipe contract test proves it). The pipeline: resolve
/// the decision (operator short-circuit vs the default classifier) → policy-decide the effort mode → resolve the
/// recipe (fail-open) → resolve the projection → DEGRADE if the recipe's required capability is unavailable →
/// resolve the bounds preset + merge any caps override → assemble the RoutePlan + a derived confirm card.
/// recipe (fail-open) → keep an auto route with an operator acceptance floor on a projection that grades it → resolve
/// the projection → DEGRADE if the recipe's required capability is unavailable → resolve the bounds preset + merge any
/// caps override → assemble the RoutePlan + a derived confirm card.
/// </summary>
public sealed class EffortRouter : IEffortRouter, IScopedDependency
{
private readonly IEffortClassifierRegistry _classifiers;
private readonly ITaskRecipeRegistry _recipes;
private readonly IBoundsPresetRegistry _bounds;
private readonly ICapabilityProbeRegistry _capabilities;
private readonly ITaskProjectionRegistry _projections;

public EffortRouter(IEffortClassifierRegistry classifiers, ITaskRecipeRegistry recipes, IBoundsPresetRegistry bounds, ICapabilityProbeRegistry capabilities)
public EffortRouter(IEffortClassifierRegistry classifiers, ITaskRecipeRegistry recipes, IBoundsPresetRegistry bounds, ICapabilityProbeRegistry capabilities, ITaskProjectionRegistry projections)
{
_classifiers = classifiers;
_recipes = recipes;
_bounds = bounds;
_capabilities = capabilities;
_projections = projections;
}

public async Task<RoutePlan> RouteAsync(EffortRouteRequest request, CancellationToken ct)
Expand All @@ -41,24 +45,60 @@ public async Task<RoutePlan> RouteAsync(EffortRouteRequest request, Cancellation

var recipe = ResolveRecipe(request, decision);

(effortMode, recipe, var floorReason) = KeepOperatorFloorGradable(request, wasAutoClassified, decision.Signals, effortMode, recipe);

var projectionKind = request.RequestedProjection ?? recipe.DefaultProjectionKind;

var (effectiveRecipe, effectiveProjection, degradedReason) = DegradeIfCapabilityUnavailable(request, recipe, projectionKind);

var (preset, caps) = ResolveCaps(request, effortMode, effectiveRecipe);

// A risky / irreversible task ALWAYS surfaces the confirm card regardless of the model's self-confidence — the
// classifier emits the risk signal, but the ROUTER (not the model's confidence) decides the human gate, so an
// over-confident model can't suppress the operator's escalation affordance on destructive work. This restores the
// pre-LLM always-confirm floor for risk while keeping the confident-routing win for ordinary tasks (model emits
// data, policy decides — the same tighten-only convention as the autonomy ceiling).
var needsConfirmCard = wasAutoClassified && (decision.Confidence < EffortPolicy.ConfirmConfidenceFloor || decision.Signals.RiskySideEffects);
var needsConfirmCard = NeedsConfirmCard(decision, wasAutoClassified);

var confirm = needsConfirmCard ? BuildConfirmCard(decision) : null;

return BuildPlan(decision, wasAutoClassified, effortMode, effectiveRecipe, effectiveProjection, preset, caps, needsConfirmCard, confirm, degradedReason);
return BuildPlan(decision, wasAutoClassified, effortMode, effectiveRecipe, effectiveProjection, preset, caps, needsConfirmCard, confirm, JoinReasons(floorReason, degradedReason));
}

/// <summary>
/// Whether an auto route must be confirmed by the operator before it runs: a confidence below the floor, a risky /
/// irreversible task, or an AMBIGUOUS one. The classifier emits the signals, but the ROUTER (not the model's
/// confidence) decides the human gate, so an over-confident model can't suppress the operator's escalation affordance
/// on destructive work, or route an under-specified goal as though it were understood — the confirm card says
/// exactly that. Model emits data, policy decides — the same tighten-only convention as the autonomy ceiling. An
/// explicit operator tier is already a decision and never confirms.
/// </summary>
private static bool NeedsConfirmCard(EffortDecision decision, bool wasAutoClassified) =>
wasAutoClassified && (decision.Confidence < EffortPolicy.ConfirmConfidenceFloor || decision.Signals.RiskySideEffects || decision.Signals.Ambiguous);

/// <summary>
/// An operator acceptance floor is a routing signal. When the AUTO path's classified shape lands on a projection whose
/// builder does not grade an operator command, re-decide the tier with every such tier set aside — the policy's next
/// matching row — and say so on the route. An explicit tier, a pinned recipe or a pinned projection is the operator's
/// own choice and stays where it is: the launch refuses that combination with its reason instead of moving it.
/// </summary>
private (string EffortMode, ITaskRecipe Recipe, string? Reason) KeepOperatorFloorGradable(EffortRouteRequest request, bool wasAutoClassified, EffortSignals signals, string effortMode, ITaskRecipe recipe)
{
if (!request.HasOperatorFloor || !wasAutoClassified || request.RequestedRecipe is not null || request.RequestedProjection is not null) return (effortMode, recipe, null);
if (GradesOperatorFloor(recipe.DefaultProjectionKind)) return (effortMode, recipe, null);

var admitted = EffortPolicy.Decide(signals, requestedEffort: null, mode => GradesOperatorFloor(_recipes.RecipeForEffort(mode).DefaultProjectionKind));
var rerouted = _recipes.RecipeForEffort(admitted);

// No tier the policy admits grades the floor either: leave the route as classified — the launch refuses it by name.
if (!GradesOperatorFloor(rerouted.DefaultProjectionKind)) return (effortMode, recipe, null);

return (admitted, rerouted, $"the operator's acceptance check needs a route that grades it, and '{recipe.DefaultProjectionKind}' does not; moved from {effortMode} to {admitted} ('{rerouted.DefaultProjectionKind}')");
}

/// <summary>Whether <paramref name="projectionKind"/>'s builder advertises that it grades an operator command — the same advertisement the route preview's acceptance verdict and the launch's floor disposition read.</summary>
private bool GradesOperatorFloor(string projectionKind) =>
_projections.TryResolve(projectionKind, out var builder) && builder.OperatorAcceptance.AcceptsCommand == true;

/// <summary>Every reason the route moved off what was asked for, in the order the moves happened — each is named, none is dropped.</summary>
private static string? JoinReasons(string? first, string? second) =>
first is null ? second : second is null ? first : $"{first}; {second}";

/// <summary>
/// When the resolved recipe DECLARES a required capability (<c>ITaskRecipe.RequiresCapability</c>) that the
/// probe registry reports unavailable, DEGRADE to the recipe's fallback (<c>DegradesToRecipe</c>, else the
Expand Down Expand Up @@ -244,7 +284,7 @@ private static string BuildHint(RouteCaps caps) =>
NeedsPlanReview = recipe.RequiresPlanReview,
WasAutoClassified = wasAutoClassified,
ClassifierConfidence = decision.Confidence,
DegradedReason = degradedReason, // set (non-null) when a capability degrade fired, null otherwise — never silent
DegradedReason = degradedReason, // set (non-null) when the route moved — an operator floor it could not grade, a capability degrade — null otherwise; never silent
Decision = decision,
Confirm = confirm,
};
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
using CodeSpace.Messages.Failures;
using CodeSpace.Messages.Tasks;

namespace CodeSpace.Core.Services.Tasks.Launch.Exceptions;

/// <summary>
/// The resolved route cannot honour a launch control the operator set, so the launch stops before any session or run
/// exists rather than run without it. Carries each refused control with its reason — the same dispositions the route
/// preview reports for this input — so every caller can say what to change.
/// </summary>
public sealed class TaskLaunchControlRefusedException : Exception, IFailure
{
private readonly IReadOnlyList<LaunchControlDisposition> _refused;

public TaskLaunchControlRefusedException(IReadOnlyList<LaunchControlDisposition> refused) : base(string.Join(" ", refused.Select(d => $"{d.Control}: {d.Reason}"))) { _refused = refused; }

public FailureKind Kind => FailureKind.Unprocessable;
public string Code => FailureCodes.TaskLaunchControlRefused;
public IReadOnlyDictionary<string, object?> Details => new Dictionary<string, object?> { ["controls"] = _refused };
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
using CodeSpace.Core.Services.Agents.ModelCredentials;
using CodeSpace.Messages.Tasks;

namespace CodeSpace.Core.Services.Tasks.Launch;

/// <summary>
/// What a resolved route does with each operator control whose effect depends on the route. ONE method for both callers:
/// the launch refuses on a Refused disposition and applies the model clamp, and the route preview reports the same
/// dispositions — so the preview can never state a disposition the launch would not reach.
/// </summary>
public interface ILaunchControlResolver
{
Task<LaunchControlResolution> ResolveAsync(TaskLaunchRequest request, RoutePlan route, CancellationToken cancellationToken);
}

/// <summary>The dispositions of one launch's route-dependent controls, whether its route grades an operator acceptance floor, and the pooled row the single agent runs on when its pinned model fell outside the allowed pool (null when nothing was clamped).</summary>
public sealed record LaunchControlResolution(IReadOnlyList<LaunchControlDisposition> Dispositions, bool GradesOperatorFloor, ModelDispatchRef? ModelClamp);
Loading
Loading