diff --git a/packages/webui/server/bootstrap.js b/packages/webui/server/bootstrap.js index 4b029849..22e7a5ed 100644 --- a/packages/webui/server/bootstrap.js +++ b/packages/webui/server/bootstrap.js @@ -40,7 +40,7 @@ import { shutdownMcodeAcpSingleton } from './lib/acp-client.js' import { installGracefulShutdown } from './lib/graceful-shutdown.js' import { init as initSettings, getPersistPath, getTokenEnabled } from './lib/settings.js' import { setTokenAuthEnabled as setAuthTokenEnabled } from './lib/auth.js' -import { pushTokenFirstRun } from './lib/state-bus.js' +import { pushTokenFirstRun, startSubagentStatusPolling } from './lib/state-bus.js' installGlobalErrorHandlers() @@ -126,6 +126,12 @@ listenWithPortFallback(server, { onListening: (boundPort) => { setServingPort(boundPort) stopTranscriptSync = startTranscriptSync() + // Slice 06 (Agent Team): start polling `local_runtime_background_tasks` + // so the parent's `→ task` tool line carries a live running badge + // while a subagent is busy. The poller is a no-op when no runtime db + // is present (e.g. a freshly-installed machine that hasn't run mcode + // yet), so we always call it; it returns early on its own. + startSubagentStatusPolling() console.log(`[webui] listening on http://${HOST}:${boundPort}`) console.log(`[webui] http layer: ${SERVER_IMPL}`) console.log(`[webui] LAN url: http://${LAN_IP}:${boundPort}`) diff --git a/packages/webui/server/lib/agent-team-detect.js b/packages/webui/server/lib/agent-team-detect.js new file mode 100644 index 00000000..a9e40b6a --- /dev/null +++ b/packages/webui/server/lib/agent-team-detect.js @@ -0,0 +1,101 @@ +// webui/server/lib/agent-team-detect.js +// Subagent detection — parse `` out of a +// tool output body and match it against the runtime db's subagent task +// rows so the parent's tool line can carry a jumpable subagent reference +// and the running badge. +// +// The runtime's parent stream does not emit a dedicated subagent event; +// per .tickets/webui-parity/06-agent-team-panel.md the only authoritative +// signal the webui has is the `` +// literal that lands inside the body of a `→ task` tool result. The +// session row in `local_runtime_sessions` and the task row in +// `local_runtime_background_tasks` both exist by the time the result +// arrives; this module is the bridge that records the toolCallId ↔ +// childSessionId pair on the cid and surfaces the live task status for +// the running badge. +// +// Every function in this module is pure (no I/O) or read-only — the +// runtime db is open `readonly: true` and the cid state is mutated +// through the public helpers in lib/state-bus.js, never directly. + +import { AGENT_TEAM_STATUS, projectTaskStatus } from "./agent-team-status.js"; +import { + findSubagentTaskByToolCallId, + hasSubagentTaskByToolCallId, +} from "./agent-team-tasks.js"; + +// The literal the engine writes inside the parent stream's tool body. +// Captured by the ticket's R8 investigation; non-greedy match keeps the +// attribute parser from gobbling a second sibling `` +// tag. `session_id` is the only attribute we care about today; if the +// engine grows more (parent_turn_id, agent_name, …) we extend the regex +// without rewriting the parser. +const TASK_RESULT_TAG = /]*?\bsession_id=["']([^"']+)["'][^>]*>/i; + +// Agent-name attribute, if the engine ever inlines it. Optional — the +// live task row is the primary source for `agentName`. +const TASK_RESULT_AGENT = /]*?\bagent=["']([^"']+)["']/i; + +/** + * Pull the child session id (and optional agent hint) out of a tool + * result body. Returns null when the body does not look like a + * `` payload. + * + * @param {string|undefined|null} body + * @returns {{ sessionId: string, agentName: string|null } | null} + */ +export function parseTaskResult(body) { + if (typeof body !== "string" || body.length === 0) return null; + const sidMatch = TASK_RESULT_TAG.exec(body); + if (!sidMatch) return null; + const sessionId = (sidMatch[1] || "").trim(); + if (!sessionId) return null; + const agentMatch = TASK_RESULT_AGENT.exec(body); + return { + sessionId, + agentName: agentMatch ? (agentMatch[1] || "").trim() || null : null, + }; +} + +/** + * True when the tool name (or the body's `` tag) indicates + * the engine just spawned a subagent. The header name is the cheapest + * signal we have — every `task` tool call the engine dispatches in the + * parent stream becomes a subagent row, so the header name alone is + * enough to start polling the runtime db for the live status. + * + * Tool-name variants observed in R8: + * - "task" — the canonical name + * - "Task" — capitalised by some renderers + * - "delegate" / "delegatetask" — used by older builds + * The match is case-insensitive and tolerant of underscores. + */ +export function isSubagentDispatch(toolName, body) { + if (typeof toolName === "string" && toolName.trim()) { + const n = toolName.trim().toLowerCase().replace(/[^a-z]/g, ""); + if (n === "task" || n === "delegate" || n === "delegatetask") return true; + } + // Header may not be present (e.g. the webui attached mid-stream and + // the very first frame is the body). The body's tag is enough. + if (typeof body === "string" && TASK_RESULT_TAG.test(body)) return true; + return false; +} + +/** + * Resolve the live subagent task for a toolCallId WITHOUT recording + * anything. Returns the projected status so the running badge can + * render without a per-render db hit on the wire. + * + * Read-only: callers MUST NOT mutate the returned object. + */ +export function readSubagentStatusForToolCall(toolCallId) { + return findSubagentTaskByToolCallId(toolCallId); +} + +/** Detect the moment a subagent is born — read-only, no side effects. */ +export function detectSubagentBirth(toolCallId) { + return hasSubagentTaskByToolCallId(toolCallId); +} + +/** Re-export the status vocabulary so consumers can `import { AGENT_TEAM_STATUS }` here. */ +export { AGENT_TEAM_STATUS, projectTaskStatus }; diff --git a/packages/webui/server/lib/agent-team-status.js b/packages/webui/server/lib/agent-team-status.js new file mode 100644 index 00000000..2e6e4335 --- /dev/null +++ b/packages/webui/server/lib/agent-team-status.js @@ -0,0 +1,145 @@ +// webui/server/lib/agent-team-status.js +// DB → UI status projection for the Agent Team panel. +// +// Why this module exists. The runtime db is the source of truth for the +// parent/child session graph, but its column vocabulary is **not** what the +// UI should render: +// +// • `local_runtime_sessions.status` is intentionally narrow: it only records +// `idle | interrupted | aborted | error`. The runtime does not currently +// flip this column while a session is actively running a turn — it stays +// `idle`. Treating that as "not running" is correct; treating it as the +// whole picture is wrong. +// +// • `local_runtime_background_tasks.status` records the *actual* run state +// of a delegated subagent (`running | succeeded | failed | canceled`). +// The session row does not, and projecting only the session column is +// how a sidebar would render a busy subagent as "idle". +// +// The TUI exposes a richer vocabulary (`failed | waiting | running | queued | +// done | stopped`) on its own projection layer; we do not import that, but we +// adopt the same shape so the contract stays greppable. Every UI consumer +// (the agent-team section of the sidebar, the running badge on a parent's +// task tool line, the task-view modal) reads the projected vocabulary below +// rather than raw db strings. +// +// This module is the ONLY place that decides the mapping. Raw values from +// the db are never passed through to the wire. New status values the +// runtime might grow land here as a single guard clause, and tests pin the +// current behavior so a future contributor cannot silently change it. + +/** + * The shape the UI renders. Ordered to match the TUI vocabulary for + * greppability; the order is NOT load-bearing for the UI but it does help + * the test reader see the mapping at a glance. + */ +export const AGENT_TEAM_STATUS = Object.freeze({ + IDLE: "idle", + QUEUED: "queued", + RUNNING: "running", + WAITING: "waiting", + DONE: "done", + STOPPED: "stopped", + FAILED: "failed", +}); + +const UI_STATUSES = new Set(Object.values(AGENT_TEAM_STATUS)); + +/** + * Map a `local_runtime_sessions.status` value to the UI vocabulary. + * + * Real measured values from this machine's db (ticket 06 R8 复核): + * • `idle` — the only state the engine writes for an active session + * • `interrupted` — the engine aborted a turn mid-flight (user stop, crash) + * • `aborted` — the engine aborted a turn cleanly (cancel) + * • `error` — the turn ended on an unrecoverable error + * + * The session row does NOT carry `running` / `done` / `failed` / `queued`. + * Those come from `local_runtime_background_tasks.status` (see + * `projectTaskStatus`). When the runtime grows a richer vocabulary the new + * values land here AND in the matching test. + */ +export function projectSessionStatus(rawStatus) { + const s = typeof rawStatus === "string" ? rawStatus.trim() : ""; + if (s === "error") return AGENT_TEAM_STATUS.FAILED; + if (s === "aborted") return AGENT_TEAM_STATUS.FAILED; + if (s === "interrupted") return AGENT_TEAM_STATUS.STOPPED; + // Default: idle covers both an actual `idle` row and an unknown value + // the runtime has not grown yet (we prefer "no claim" over "loud + // failure" for an unrecognised string). + return AGENT_TEAM_STATUS.IDLE; +} + +/** + * Map a `local_runtime_background_tasks.status` (when `kind = 'subagent'`) + * to the UI vocabulary. + * + * Measured values from this machine's db (ticket 06 R8 复核): + * • `running` — the subagent is mid-turn (the only value that + * claims "live" on the parent's tool line) + * • `succeeded` — the subagent finished cleanly + * • `failed` — the subagent ended on an error + * • `canceled` — the subagent was stopped or interrupted + * + * `canceled` lands on `stopped` (matches TUI vocabulary), not on `failed`: + * cancellation is a deliberate user action, not a fault. + */ +export function projectTaskStatus(rawStatus) { + const s = typeof rawStatus === "string" ? rawStatus.trim() : ""; + if (s === "running") return AGENT_TEAM_STATUS.RUNNING; + if (s === "succeeded") return AGENT_TEAM_STATUS.DONE; + if (s === "failed") return AGENT_TEAM_STATUS.FAILED; + if (s === "canceled") return AGENT_TEAM_STATUS.STOPPED; + // Unknown / empty — render as idle rather than risk a false "running" + // claim on an unrecognised future status. + return AGENT_TEAM_STATUS.IDLE; +} + +/** + * Compose the two projections for a session row that has a live task. + * + * The task row's `running` wins — that is the only way to display + * "running" at all, because the session column is intentionally narrow. + * When no task is supplied (a subagent row that has not been picked up by + * the runtime yet, or whose task row has been cleaned up), the session + * column's projection is the answer. + * + * @param {string|undefined|null} sessionRaw + * @param {string|undefined|null} taskRaw + * @returns {string} one of `AGENT_TEAM_STATUS` + */ +export function projectAgentStatus(sessionRaw, taskRaw) { + const fromTask = projectTaskStatus(taskRaw); + if (fromTask === AGENT_TEAM_STATUS.RUNNING) return fromTask; + // If the task is in a terminal state, prefer the task projection — a + // session column stuck at `idle` would otherwise re-paint a `done` / + // `failed` subagent as "running again". + if (taskRaw && String(taskRaw).trim() !== "") { + if ( + fromTask === AGENT_TEAM_STATUS.DONE || + fromTask === AGENT_TEAM_STATUS.FAILED || + fromTask === AGENT_TEAM_STATUS.STOPPED + ) { + return fromTask; + } + } + return projectSessionStatus(sessionRaw); +} + +/** + * True when a status is one the UI should treat as "live" (still being + * driven by the engine, not yet terminal). Used by the running badge on + * the parent's task tool line and by the live-render hook in the sidebar + * session tree. + */ +export function isLiveStatus(uiStatus) { + return ( + uiStatus === AGENT_TEAM_STATUS.RUNNING || + uiStatus === AGENT_TEAM_STATUS.WAITING + ); +} + +/** Defensive read for callers that trust nothing. */ +export function isUiStatus(value) { + return typeof value === "string" && UI_STATUSES.has(value); +} diff --git a/packages/webui/server/lib/agent-team-tasks.js b/packages/webui/server/lib/agent-team-tasks.js new file mode 100644 index 00000000..538eeb67 --- /dev/null +++ b/packages/webui/server/lib/agent-team-tasks.js @@ -0,0 +1,208 @@ +// webui/server/lib/agent-team-tasks.js +// Read-only projections over `local_runtime_background_tasks` and +// `local_runtime_task_session_bindings`, used by the Agent Team wiring. +// +// Why this lives in its own module. The runtime db is read-only from +// webui's perspective (auth forces its location, `MINIMAX_DATA_DIR` does +// not relocate it; see .tickets/webui-parity/06-agent-team-panel.md R8). +// Every helper here opens a fresh handle with `{ readonly: true, +// fileMustExist: true }` and closes it in a `finally` so a long-lived +// process never leaks a file descriptor across the mavis restart that +// happens whenever the user upgrades the engine. +// +// `projectTaskStatus` (the only function that returns a string the UI +// reads) deliberately calls into `agent-team-status.js`. Nothing in this +// module may pass a raw db string to a wire payload — that is the rule +// that pins the vocabulary and keeps future contributors from re-opening +// the leak this slice 06 closed. + +import { existsSync } from "node:fs"; +import path from "node:path"; + +import { getMcodeBetterSqlite3 } from "./sqlite-resolver.js"; +import { projectTaskStatus, AGENT_TEAM_STATUS } from "./agent-team-status.js"; + +const RESULT_PROBE_LIMIT = 50; + +// `MCODE_RUNTIME_DB` is read FRESH on every helper call (not bound at +// import time) so the test harness can swap the db path via +// `process.env.MCODE_RUNTIME_DB` without reloading config.js. This is +// the same lazy pattern sqlite-resolver.js already uses internally for +// the binding path — the resolver caches the module export, but every +// helper here opens a fresh handle against the current env, so a test +// can override MCODE_RUNTIME_DB and a process restart picks it up. +function readRuntimeDbPath() { + return process.env.MCODE_RUNTIME_DB || ""; +} + +/** + * Open a fresh read-only handle to the runtime db. + * + * Returns `null` (not throw) on any failure: the caller is the sidebar + * poll, and a missing / locked db is a normal state the UI must survive + * — not a server-side error worth a 500. + */ +function openRuntimeDb() { + const dbPath = readRuntimeDbPath(); + if (!dbPath || !existsSync(dbPath)) return null; + const Db = getMcodeBetterSqlite3(); + if (!Db) return null; + try { + return new Db(dbPath, { readonly: true, fileMustExist: true }); + } catch { + return null; + } +} + +/** + * Look up the latest `local_runtime_background_tasks` row whose + * `kind = 'subagent'` AND whose `record_json.toolCallId = toolCallId`. + * + * The runtime stamps the `toolCallId` of the parent session's `→ task` + * tool call on every subagent row it owns, so this lookup is what the + * running badge on the parent's tool line polls against. + * + * @param {string} toolCallId + * @returns {null | { taskId: string, status: string, agentName: string|null, childSessionId: string|null }} + */ +export function findSubagentTaskByToolCallId(toolCallId) { + if (!toolCallId) return null; + const db = openRuntimeDb(); + if (!db) return null; + try { + const row = db + .prepare( + `SELECT task_id, status, + json_extract(record_json, '$.toolCallId') AS tool_call_id, + json_extract(record_json, '$.metadata.agentName') AS agent_name, + json_extract(record_json, '$.metadata.childSessionId') AS child_session_id + FROM local_runtime_background_tasks + WHERE kind = 'subagent' + AND json_extract(record_json, '$.toolCallId') = ? + ORDER BY created_at_ms DESC + LIMIT 1`, + ) + .get(toolCallId); + if (!row) return null; + return { + taskId: row.task_id, + status: projectTaskStatus(row.status), + agentName: row.agent_name ?? null, + childSessionId: row.child_session_id ?? null, + }; + } catch { + return null; + } finally { + try { + db.close(); + } catch {} + } +} + +/** + * True iff the runtime db has ANY `local_runtime_background_tasks` row + * with `kind = 'subagent'` for this `toolCallId`. Cheaper than the + * full projection above — used to detect the moment a subagent is born + * (the SSE `session-tree-changed` trigger). + */ +export function hasSubagentTaskByToolCallId(toolCallId) { + if (!toolCallId) return false; + const db = openRuntimeDb(); + if (!db) return false; + try { + const row = db + .prepare( + `SELECT 1 AS ok + FROM local_runtime_background_tasks + WHERE kind = 'subagent' + AND json_extract(record_json, '$.toolCallId') = ? + LIMIT 1`, + ) + .get(toolCallId); + return Boolean(row && row.ok); + } catch { + return false; + } finally { + try { + db.close(); + } catch {} + } +} + +/** + * Subagent tasks owned by `parentSessionId`, regardless of which turn they + * landed on. + * + * Used by the tree-refresh path to confirm a new subagent row actually + * belongs to a parent the user can see (the `parent_session_id` on the + * new `local_runtime_sessions` row is the source of truth — but the task + * row's `metadata.parentSessionId` is a redundant cross-check, useful + * for sessions whose parent pointer races the db write). + * + * @returns {Array<{ toolCallId: string, status: string, agentName: string|null, childSessionId: string|null, createdAtMs: number }>} + */ +export function listSubagentTasksByParentSession(parentSessionId, { limit = 32 } = {}) { + if (!parentSessionId) return []; + const db = openRuntimeDb(); + if (!db) return []; + const cap = Math.max(1, Math.min(limit, RESULT_PROBE_LIMIT)); + try { + const rows = db + .prepare( + `SELECT task_id, status, created_at_ms, + json_extract(record_json, '$.toolCallId') AS tool_call_id, + json_extract(record_json, '$.metadata.agentName') AS agent_name, + json_extract(record_json, '$.metadata.childSessionId') AS child_session_id + FROM local_runtime_background_tasks + WHERE kind = 'subagent' + AND owner_session_id = ? + ORDER BY created_at_ms DESC + LIMIT ?`, + ) + .all(parentSessionId, cap); + return rows.map((row) => ({ + taskId: row.task_id, + status: projectTaskStatus(row.status), + toolCallId: row.tool_call_id ?? null, + agentName: row.agent_name ?? null, + childSessionId: row.child_session_id ?? null, + createdAtMs: Number(row.created_at_ms) || 0, + })); + } catch { + return []; + } finally { + try { + db.close(); + } catch {} + } +} + +/** + * Probe the db for newly-spawned subagent sessions a given parent has not + * yet been told about. Returns the toolCallId ↔ childSessionId pairs so + * the SSE `session-tree-changed` trigger can attach the jump reference. + * + * Compared against the in-memory `recentSubagents` map in `state-bus.js`, + * which tracks what the client already knows. The projection here is + * projection-only: callers merge the diff. + */ +export function diffSubagentsByParentSession(parentSessionId, knownToolCallIds) { + const known = knownToolCallIds instanceof Set ? knownToolCallIds : new Set(); + if (!parentSessionId) return []; + const tasks = listSubagentTasksByParentSession(parentSessionId, { limit: 64 }); + return tasks.filter((t) => t.toolCallId && !known.has(t.toolCallId)); +} + +// re-export the status vocabulary so a caller that imports this module +// for the db helpers does not have to chase a second import. +export { AGENT_TEAM_STATUS }; + +// CommonJS interop: some test runners prefer `default`. Keep the named +// exports authoritative. +export default { + findSubagentTaskByToolCallId, + hasSubagentTaskByToolCallId, + listSubagentTasksByParentSession, + diffSubagentsByParentSession, + AGENT_TEAM_STATUS, +}; diff --git a/packages/webui/server/lib/mcode-acp.js b/packages/webui/server/lib/mcode-acp.js index ed34c4fd..8d5c6e6c 100644 --- a/packages/webui/server/lib/mcode-acp.js +++ b/packages/webui/server/lib/mcode-acp.js @@ -8,6 +8,7 @@ import { streamUpdateLine } from "./chat-line.js"; import { createRunChat, runChatLinesFor, + recordSubagentForCid, } from "./state-bus.js"; import { bindDraftToMcodeSid, @@ -32,6 +33,11 @@ import { import { getMcodeModelLimit } from "./models.js"; import { buildPromptBlocks, promptTextFor } from "./attachments.js"; import { loadSessions, saveSessions } from "./sessions.js"; +import { + parseTaskResult, + isSubagentDispatch, + readSubagentStatusForToolCall, +} from "./agent-team-detect.js"; // runMcodeAcp / streamAcpPrompt — mcode acp protocol streaming. // @@ -678,7 +684,23 @@ export { PICK_DEFER_WINDOW_MS }; // `system` block (the transcript row the user reported as labelled // `系统`). The synthesized header is registered in `r.toolIndexById` // so subsequent updates for the same `toolCallId` insert after it. -export function applyToolUpdate(r, cs, update) { +// +// Slice 06 (Agent Team): also performs the two cross-stream wirings +// the dispatcher described in +// .tickets/webui-parity/06-agent-team-panel.md +// +// • when the tool call's name (or the body) signals a subagent +// dispatch, the body's `` tag is +// parsed and the (toolCallId, childSessionId) pair is recorded on +// the cid's `cs.recentSubagents` so the parent's tool line can +// carry a jumpable subagent reference and the sidebar can +// re-fetch its tree. +// +// • the live `local_runtime_background_tasks.status` (read-only) is +// polled right away for the just-recorded toolCallId and stashed +// on the entry's `status` field; the UI's running badge reads +// this from the snapshot. Subsequent updates refresh it. +export function applyToolUpdate(r, cs, update, ctx = {}) { const u = update || {}; if (!r.toolIndexById) r.toolIndexById = new Map(); // session-isolation/02: route tool-update writes into the runChat @@ -689,6 +711,20 @@ export function applyToolUpdate(r, cs, update) { let insertAfter = r.toolIndexById.get(u.toolCallId); if (insertAfter == null) { const name = u.title || u.name || u.toolName || "tool"; + // Slice 06 (Agent Team): emit a `##tc:` marker line + // BEFORE the `→ name` header so the chat renderer can correlate + // the tool block with the matching `recentSubagents[]` entry. + // The decoder consumes the marker (it never reaches the chat body) + // and attaches the id to the tool block; the ToolCard then uses + // it to look up the precise subagent for THIS dispatch. A parent + // session that spawns multiple subagents has one recentSubagents + // entry per toolCallId — matching by tool NAME instead would badge + // every `→ task` line with the newest child, which is wrong. + // Older sessions whose chat was written before this marker shipped + // simply lack it; the lookup falls back to the newest entry. + if (u.toolCallId) { + chat.push(`##tc:${u.toolCallId}`); + } chat.push(`→ ${name}`); insertAfter = chat.length - 1; r.toolIndexById.set(u.toolCallId, insertAfter); @@ -704,6 +740,29 @@ export function applyToolUpdate(r, cs, update) { .join("\n") : ""; + // Slice 06: subagent wiring — see header. + // The cid is optional (applyToolUpdate is also called from + // transcript-only fixtures); we record only when one is present so a + // unit test can drive the helper without standing up a client. + if (ctx.cid && typeof u.toolCallId === "string" && u.toolCallId) { + if (isSubagentDispatch(u.title || u.name || u.toolName || "", outText)) { + const parsed = parseTaskResult(outText); + if (parsed && parsed.sessionId) { + // Live task status (read-only): the engine writes the row + // BEFORE the result body arrives in the parent stream, so this + // projection is rarely empty here. `null` is a normal state for + // mid-stream attach races; recordSubagentForCid accepts it. + const live = readSubagentStatusForToolCall(u.toolCallId); + recordSubagentForCid(ctx.cid, { + toolCallId: u.toolCallId, + sessionId: parsed.sessionId, + agentName: parsed.agentName || (live && live.agentName) || null, + status: live && live.status ? live.status : null, + }); + } + } + } + const newLines = []; newLines.push(` [${status}]`); if (outText) { @@ -1176,8 +1235,20 @@ function streamAcpPrompt( // cs.chat) when in a turn. The viewing-session sees no // cross-contamination when the user switches mid-run. const tcChat = r && typeof r.chatArray === "function" ? r.chatArray() : cs.chat; + // Slice 06 (Agent Team): emit `##tc:` BEFORE the + // `→ name` header so the chat renderer can correlate this + // tool block with its `recentSubagents[]` entry by id + // (matching by tool NAME would badge every `→ task` line + // with the newest child, which is wrong for sessions that + // spawn more than one subagent). The decoder consumes the + // marker; it never appears in the rendered chat body. + if (u.toolCallId) { + tcChat.push(`##tc:${u.toolCallId}`); + } tcChat.push(line); // 记下这行在 chat 里的位置(之后 tool_update 用来在它后面插输出) + // — index points at the `→ name` line, which is the header + // the decoder attaches the toolCallId to. if (!r.toolIndexById) r.toolIndexById = new Map(); r.toolIndexById.set(u.toolCallId, tcChat.length - 1); // session-isolation/06: tool_call (and tool_update, @@ -1192,7 +1263,10 @@ function streamAcpPrompt( // result.answer). r.lastChunkKind = "tool_call"; } else if (c.kind === "tool_update" && c.update) { - applyToolUpdate(r, cs, c.update); + // Slice 06: pass `{ cid }` so applyToolUpdate can record the + // (toolCallId, childSessionId) pair for subagent dispatches + // and refresh the live task status on the recorded entry. + applyToolUpdate(r, cs, c.update, { cid }); } else if (c.kind === "plan_update" && c.update) { // plan_update event const u = c.update; diff --git a/packages/webui/server/lib/state-bus.js b/packages/webui/server/lib/state-bus.js index fd64d500..37aa9b51 100644 --- a/packages/webui/server/lib/state-bus.js +++ b/packages/webui/server/lib/state-bus.js @@ -1,7 +1,12 @@ // webui/server/lib/state-bus.js // Per-cid state + SSE channel management. -import { DEFAULT_WORKSPACE, DEFAULT_MODEL, MAX_CONCURRENT } from "./config.js"; +import { existsSync } from "node:fs"; +import { DEFAULT_WORKSPACE, DEFAULT_MODEL, MAX_CONCURRENT, MCODE_RUNTIME_DB } from "./config.js"; +import { + findSubagentTaskByToolCallId as _findSubagentTaskByToolCallId, +} from "./agent-team-tasks.js"; +import { invalidateSessionTree as _invalidateSessionTree } from "./session-tree.js"; import { isFirstRun } from "./auth.js"; import { loadSessions } from "./sessions.js"; import { @@ -109,6 +114,25 @@ export function makeClientState() { lastDeltaAt: null, tps: 0, }, + // Agent Team (slice 06): every (toolCallId → childSessionId) pair the + // current session has spawned. Surfaced on the wire so the parent's + // `→ task` tool line can carry a jumpable subagent reference and the + // running badge can match it back to the live background_tasks row. + // + // toolCallId — the parent's `→ task` tool call id (the engine + // emits it on every `tool_update` line). Stable for + // the lifetime of one task dispatch. + // sessionId — the child session id parsed from the engine's + // `` body. Looks + // like `mvs_<32 hex>`. + // agentName — best-effort hint from the live task row; used for + // badge color, not as identity. + // status — UI vocabulary (running/done/failed/stopped), + // NEVER a raw db string. + // createdAtMs — when the entry was first observed; lets the UI + // drop entries that have been terminal for a long + // time without flooding the wire. + recentSubagents: [], }; } @@ -1196,3 +1220,244 @@ export function pushAuthDecision({ requestId, approved, decidedBy }) { // broadcast — every connected tab should mirror modal close _writeAuthFrame("", frame); } + +// ============================================================ +// Agent Team (slice 06) — subagent references + tree refresh. +// +// The runtime does not emit a dedicated subagent event in the parent's +// stream: a `→ task` tool call is just another tool_call, and the +// subagent row materialises in `local_runtime_sessions` (and the matching +// row in `local_runtime_background_tasks`) without a per-subagent +// notification on the SSE channel. Three small wirings land here so the +// UI can render that lifecycle correctly: +// +// 1. pushSessionTreeChanged — broadcast a named `session-tree-changed` +// frame. The sidebar's session tree reads it and re-fetches +// `GET /api/session-tree`, picking up the new subagent row the +// runtime just wrote. Cheap: the named frame carries no payload, +// the listener decides when to re-read. +// +// 2. recordSubagentForCid — append a `(toolCallId, sessionId, …)` +// entry to the calling cid's `cs.recentSubagents`. Surfaced on the +// wire as part of every snapshot; the chat renderer reads it to +// turn the parent's `→ task` tool line into a jumpable subagent +// reference and to render the running badge. +// +// 3. pruneRecentSubagents — drop entries that have been terminal for +// longer than `RECENT_SUBAGENT_TTL_MS` so the array does not grow +// unbounded across a long-lived cid. Runs lazily from +// `recordSubagentForCid` rather than on a timer — there is no +// per-second churn to defend against. +// ============================================================ + +// 5 minutes — the chat renderer's jumpable badge only matters while the +// turn is on screen. After this, the sidebar tree is the source of truth +// (it always re-fetches from the runtime db) and the in-memory entry +// just clutters the wire. +const RECENT_SUBAGENT_TTL_MS = 5 * 60 * 1000; + +const RECENT_SUBAGENT_CAP = 32; + +/** + * Broadcast a `session-tree-changed` frame on every connected cid's SSE + * channel so the sidebar can re-read `GET /api/session-tree`. + * + * No payload — the listener decides when to re-fetch. This matches the + * convention used by `providers-updated` (see lib/sse.ts NAMED_EVENTS). + * Bypasses the coalescer: tree refreshes are sparse, never flood, and a + * dropped frame is a stale sidebar. Using the coalescer here would + * defeat the named-event contract. + */ +export function pushSessionTreeChanged() { + const frame = "event: session-tree-changed\ndata: {}\n\n"; + for (const [, res] of sseByCid) { + if (!res || res.writableEnded || res.destroyed) continue; + try { + res.write(frame); + } catch {} + } +} + +/** + * Append (or refresh) a subagent entry for `cid`. Idempotent on + * `toolCallId` — a second call with the same id updates `status` and + * `updatedAtMs` rather than appending a duplicate. + * + * Side-effect: emits a `session-tree-changed` SSE frame so the sidebar + * refreshes. The chat renderer sees the new entry on the next state + * push (which `pushStateFor` triggers implicitly via the caller's normal + * flow — if the caller wants an immediate state push, it calls + * `pushStateFor(cid)` itself). + * + * @param {string} cid + * @param {{ toolCallId: string, sessionId: string, agentName?: string|null, status?: string }} entry + */ +export function recordSubagentForCid(cid, entry) { + if (!cid) return; + if (!entry || typeof entry !== "object") return; + const toolCallId = typeof entry.toolCallId === "string" ? entry.toolCallId.trim() : ""; + const sessionId = typeof entry.sessionId === "string" ? entry.sessionId.trim() : ""; + if (!toolCallId || !sessionId) return; + const cs = clients.get(cid) || clients.get("default"); + if (!cs) return; + if (!Array.isArray(cs.recentSubagents)) cs.recentSubagents = []; + const now = Date.now(); + const status = typeof entry.status === "string" ? entry.status : null; + const existingIdx = cs.recentSubagents.findIndex((r) => r && r.toolCallId === toolCallId); + const next = { + toolCallId, + sessionId, + agentName: entry.agentName ?? null, + status, + createdAtMs: existingIdx >= 0 ? cs.recentSubagents[existingIdx].createdAtMs || now : now, + updatedAtMs: now, + }; + if (existingIdx >= 0) { + cs.recentSubagents[existingIdx] = next; + } else { + cs.recentSubagents.push(next); + } + pruneRecentSubagents(cs, now); + // Cap size — newest entries win. + if (cs.recentSubagents.length > RECENT_SUBAGENT_CAP) { + cs.recentSubagents = cs.recentSubagents.slice(-RECENT_SUBAGENT_CAP); + } + // Side-effect: invalidate the session-tree cache so the next read picks + // up the new subagent row the runtime just wrote (and broadcast the + // named event so connected tabs re-fetch). + try { + // dynamic import — lib/state-bus.js must not introduce a hard dep + // cycle through session-tree.js (which loads nothing here today, + // but the tree cache invalidation belongs with the cache owner). + invalidateSessionTreeFromStateBus(); + } catch {} + pushSessionTreeChanged(); +} + +/** Drop entries that have been terminal for longer than the TTL. */ +function pruneRecentSubagents(cs, now) { + if (!Array.isArray(cs.recentSubagents) || cs.recentSubagents.length === 0) return; + const terminal = new Set(["done", "failed", "stopped"]); + const cutoff = now - RECENT_SUBAGENT_TTL_MS; + cs.recentSubagents = cs.recentSubagents.filter((r) => { + if (!r) return false; + if (terminal.has(r.status)) { + // Use updatedAtMs so a long-running task isn't pruned just because + // its create timestamp is old; it keeps getting refreshed until it + // settles, then ages out cleanly. + return (r.updatedAtMs || 0) >= cutoff; + } + return true; + }); +} + +/** + * Indirect dependency on session-tree.js — broken out so the import is + * deferred and circular-free (state-bus → session-tree would otherwise + * form a cycle if session-tree ever decides to use state-bus). + * + * The function is intentionally tiny: it calls the cache invalidator and + * ignores any failure (the cache is best-effort; a failed invalidation + * means the next read is a stale read for up to 15s, which the sidebar + * already documents as a normal soft failure). + */ +function invalidateSessionTreeFromStateBus() { + try { + if (typeof _invalidateSessionTree === "function") { + _invalidateSessionTree(); + } + } catch { + /* best-effort */ + } +} + +// ============================================================ +// Polling — keep the live task status fresh on every recorded entry. +// +// Slice 06 dispatcher requires the parent's `→ task` tool line to show +// a running badge while the subagent is busy. The runtime does NOT push +// a per-task progress event on the parent stream; the only way to know +// "still running / done / failed / stopped" is to re-read the +// `local_runtime_background_tasks` row. We poll at a low cadence +// (default 2s) so the badge updates without flooding the wire. +// +// `refreshRecentSubagentStatuses` walks every connected cid, opens a +// single read-only db handle, and refreshes the `status` field on +// `cs.recentSubagents[].status` for any entry whose task row exists. It +// pushes an SSE state snapshot ONLY when at least one entry changed — +// a no-op tick is silent (matches the diff-gate contract in +// `_schedulePush`). The function is exposed so server.js can wire it +// into the existing periodic-job loop; tests drive it directly. +// ============================================================ + +const SUBAGENT_POLL_MS = Math.max( + 250, + Number(process.env.MCODE_WEBUI_SUBAGENT_POLL_MS) || 2000, +); + +let _subagentPollTimer = null; + +/** + * Poll every connected cid's recorded subagents and refresh the live + * status. Cheap: one read-only db handle per tick, one SQL query, no + * per-cid overhead beyond iterating `clients`. + */ +export function refreshRecentSubagentStatuses() { + if (clients.size === 0) return; + // Collect every (cid, toolCallId) pair we need to look up. Most cids + // will have zero recorded entries; bail early if so. + const probe = []; + for (const [cid, cs] of clients) { + if (!Array.isArray(cs.recentSubagents) || cs.recentSubagents.length === 0) continue; + for (const r of cs.recentSubagents) { + if (r && r.toolCallId) probe.push({ cid, cs, entry: r }); + } + } + if (probe.length === 0) return; + const dirty = new Set(); + for (const { cid, cs, entry } of probe) { + const live = _findSubagentTaskByToolCallId(entry.toolCallId); + if (!live) continue; + if (entry.status === live.status && entry.sessionId === live.childSessionId) continue; + entry.status = live.status; + if (live.childSessionId) entry.sessionId = live.childSessionId; + if (live.agentName) entry.agentName = live.agentName; + entry.updatedAtMs = Date.now(); + dirty.add(cid); + } + for (const cid of dirty) { + pushStateFor(cid); + } +} + +/** + * Start the periodic poll if it is not already running. Idempotent. + * Called from server.js's bootstrap once the SSE channel is up so a + * client that connects before any subagent was recorded pays zero cost. + */ +export function startSubagentStatusPolling() { + if (_subagentPollTimer !== null) return; + if (typeof setInterval !== "function") return; + // Refuse to poll if there is no runtime db at all — keeps the noise + // floor down for environments without an mcode install. + if (!MCODE_RUNTIME_DB || !existsSync(MCODE_RUNTIME_DB)) return; + _subagentPollTimer = setInterval(refreshRecentSubagentStatuses, SUBAGENT_POLL_MS); + if (typeof _subagentPollTimer.unref === "function") _subagentPollTimer.unref(); +} + +/** + * Stop the periodic poll. Used by tests that need deterministic + * tick boundaries; production never calls it. + */ +export function stopSubagentStatusPolling() { + if (_subagentPollTimer === null) return; + try { + clearInterval(_subagentPollTimer); + } catch {} + _subagentPollTimer = null; +} + +/** Test-only: read the current poll cadence. */ +export function getSubagentPollIntervalMs() { + return SUBAGENT_POLL_MS; +} diff --git a/packages/webui/test/lib/agent-team-detect.test.js b/packages/webui/test/lib/agent-team-detect.test.js new file mode 100644 index 00000000..6cc4dc75 --- /dev/null +++ b/packages/webui/test/lib/agent-team-detect.test.js @@ -0,0 +1,155 @@ +// webui/test/lib/agent-team-detect.test.js +// Unit tests for server/lib/agent-team-detect.js — pure helpers that +// parse the engine's `` body and detect +// when a subagent was dispatched. +// +// Coverage contract (every line is a documented claim from +// .tickets/webui-parity/06-agent-team-panel.md): +// +// 1. `parseTaskResult` extracts the child session id from the +// `` literal — the engine's +// only signal that a subagent row has been created. +// 2. The match is non-greedy and tolerates attribute ordering (other +// attributes can come before or after session_id). +// 3. Optional `agent="..."` attribute is captured when present. +// 4. `isSubagentDispatch` recognises "task" / "Task" / "delegate" / +// "delegatetask" / underscored variants — the engine has emitted +// each of these in the wild (R8 investigation). +// 5. The header name is enough to start polling; the body tag is the +// fallback when the webui attached mid-stream and the first frame +// is the result body alone. +// 6. All helpers are read-only — a regression that added a write +// would corrupt the runtime db and that is the kind of failure the +// rule was written to prevent. + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; + +import { + parseTaskResult, + isSubagentDispatch, + AGENT_TEAM_STATUS, +} from "../../server/lib/agent-team-detect.js"; + +describe("parseTaskResult — extract child session id from tool body", () => { + test("extracts session_id from a minimal tag", () => { + const out = parseTaskResult(``); + assert.ok(out); + assert.equal(out.sessionId, "mvs_deadbeef1234567890abcdef00000001"); + assert.equal(out.agentName, null); + }); + + test("tolerates single-quoted attribute values", () => { + const out = parseTaskResult(``); + assert.ok(out); + assert.equal(out.sessionId, "mvs_abc"); + }); + + test("captures the agent hint when the engine inlines one", () => { + const out = parseTaskResult( + ``, + ); + assert.ok(out); + assert.equal(out.sessionId, "mvs_aaa"); + assert.equal(out.agentName, "verifier"); + }); + + test("tolerates attribute ordering — session_id first or last", () => { + const first = parseTaskResult( + ``, + ); + const last = parseTaskResult( + ``, + ); + assert.equal(first.sessionId, "mvs_one"); + assert.equal(last.sessionId, "mvs_two"); + }); + + test("non-greedy: the first `` is what matters", () => { + // Two tags back-to-back would otherwise bleed across; the regex + // stops at the first closing `>`. + const body = ``; + const out = parseTaskResult(body); + assert.equal(out.sessionId, "mvs_first"); + }); + + test("extracts the session id even when the body has prose around the tag", () => { + const out = parseTaskResult( + `Spawned worker:\n\nDone.`, + ); + assert.ok(out); + assert.equal(out.sessionId, "mvs_in_a_paragraph"); + }); + + test("returns null when the body does not contain a ", () => { + assert.equal(parseTaskResult(""), null); + assert.equal(parseTaskResult("plain text"), null); + assert.equal(parseTaskResult(null), null); + assert.equal(parseTaskResult(undefined), null); + assert.equal(parseTaskResult(``), null); + }); + + test("returns null when session_id is empty", () => { + assert.equal(parseTaskResult(``), null); + assert.equal(parseTaskResult(``), null); + }); + + test("trims whitespace from the captured id", () => { + const out = parseTaskResult(``); + assert.ok(out); + assert.equal(out.sessionId, "mvs_padded"); + }); +}); + +describe("isSubagentDispatch — header name + body fallback", () => { + test("recognises the canonical 'task' header", () => { + assert.equal(isSubagentDispatch("task", ""), true); + }); + + test("is case-insensitive", () => { + assert.equal(isSubagentDispatch("Task", ""), true); + assert.equal(isSubagentDispatch("TASK", ""), true); + }); + + test("recognises the 'delegate' / 'delegatetask' legacy variants", () => { + assert.equal(isSubagentDispatch("delegate", ""), true); + assert.equal(isSubagentDispatch("delegate_task", ""), true); + assert.equal(isSubagentDispatch("Delegatetask", ""), true); + }); + + test("rejects unrelated tool names", () => { + assert.equal(isSubagentDispatch("read", ""), false); + assert.equal(isSubagentDispatch("bash", ""), false); + assert.equal(isSubagentDispatch("write", ""), false); + assert.equal(isSubagentDispatch("", ""), false); + }); + + test("the body tag is enough when the header name is missing", () => { + // Mid-stream attach: the first frame is the body, no `→ name` + // header yet. The presence of the tag itself is the signal. + assert.equal( + isSubagentDispatch( + "", + ``, + ), + true, + ); + }); + + test("rejects when neither header nor body tag match", () => { + assert.equal(isSubagentDispatch("read", "nothing relevant"), false); + }); + + test("treats nullish inputs as no signal, never throws", () => { + assert.equal(isSubagentDispatch(null, null), false); + assert.equal(isSubagentDispatch(undefined, undefined), false); + assert.equal(isSubagentDispatch(123, true), false); + }); +}); + +describe("re-exports — vocabulary surface", () => { + test("AGENT_TEAM_STATUS is re-exported so a single import suffices", () => { + assert.ok(AGENT_TEAM_STATUS); + assert.equal(AGENT_TEAM_STATUS.RUNNING, "running"); + }); +}); diff --git a/packages/webui/test/lib/agent-team-state-bus.test.js b/packages/webui/test/lib/agent-team-state-bus.test.js new file mode 100644 index 00000000..0d4f29a6 --- /dev/null +++ b/packages/webui/test/lib/agent-team-state-bus.test.js @@ -0,0 +1,239 @@ +// webui/test/lib/agent-team-state-bus.test.js +// Unit tests for the Agent Team wiring in lib/state-bus.js — +// pushSessionTreeChanged, recordSubagentForCid, recentSubagents +// pruning, and the subagent status polling helpers. +// +// Strategy: drive the public API only. We DO NOT mock state-bus.js +// itself (that would defeat the regression value). The runtime db read +// (agent-team-tasks.js) goes through env-controlled paths so a +// per-test fixture db can be substituted for `~/.minimax/...`. + +import { test, describe, before, after } from "node:test"; + +// Teardown: importing server/lib/state-bus.js drags in the webui +// runtime graph, which starts the resident ACP singleton child process +// during module load. The child's stdio keeps this test process's +// pipes open so `node --test` never sees the file finish: every test +// passes, zero failures, and the job is killed at the timeout. Stop +// the child in `after()` so the runner settles cleanly. +import assert from "node:assert/strict"; +import { mkdtempSync, writeFileSync, rmSync, existsSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createRequire } from "node:module"; +import { pathToFileURL } from "node:url"; + +const require = createRequire(import.meta.url); +const absPath = (rel) => + pathToFileURL(join(import.meta.dirname, "..", "..", "server", rel)).href; + +let tmpDir; +let dbPath; +let stateBus; + +before(async () => { + tmpDir = mkdtempSync(join(tmpdir(), "agent-team-state-bus-")); + dbPath = join(tmpDir, "runtime-state.sqlite"); + process.env.MCODE_WEBUI_SESSIONS_DB = join(tmpDir, "sessions.json"); + process.env.MCODE_WEBUI_UPLOAD_DIR = join(tmpDir, "uploads"); + // Resolve better-sqlite3 from the workspace — the same lookup the + // agent-team-tasks tests use. + const candidates = [ + join(process.cwd(), "node_modules", "better-sqlite3"), + join(process.cwd(), "..", "..", "node_modules", "better-sqlite3"), + join(process.cwd(), "..", "..", "..", "node_modules", "better-sqlite3"), + ]; + let bindingPath = null; + for (const c of candidates) { + try { + require(c); + bindingPath = c; + break; + } catch {} + } + if (!bindingPath) throw new Error("better-sqlite3 not found"); + process.env.MCODE_BETTER_SQLITE3 = bindingPath; + process.env.MCODE_RUNTIME_DB = dbPath; + + // Seed the runtime db with a row that the poll will refresh. + const Db = require(bindingPath); + const db = new Db(dbPath); + db.exec(` + CREATE TABLE local_runtime_background_tasks ( + task_id TEXT PRIMARY KEY, + owner_session_id TEXT NOT NULL, + kind TEXT NOT NULL, + status TEXT NOT NULL, + created_at_ms INTEGER NOT NULL, + updated_at_ms INTEGER NOT NULL, + ended_at_ms INTEGER, + record_json TEXT NOT NULL + ); + `); + db.prepare( + `INSERT INTO local_runtime_background_tasks + (task_id, owner_session_id, kind, status, created_at_ms, updated_at_ms, ended_at_ms, record_json) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + ).run( + "bg_poll_1", + "mvs_parent", + "subagent", + "running", + 1000, + 1100, + null, + JSON.stringify({ + toolCallId: "tc_poll_running", + metadata: { parentSessionId: "mvs_parent", childSessionId: "mvs_child_a", agentName: "verifier" }, + }), + ); + db.close(); + + stateBus = await import(absPath("lib/state-bus.js")); +}); + +after(async () => { + if (tmpDir) rmSync(tmpDir, { recursive: true, force: true }); + delete process.env.MCODE_BETTER_SQLITE3; + delete process.env.MCODE_RUNTIME_DB; + delete process.env.MCODE_WEBUI_SESSIONS_DB; + delete process.env.MCODE_WEBUI_UPLOAD_DIR; + // Stop the subagent poll so the unref'd interval does not keep the + // process alive after the suite settles. The test that calls + // `startSubagentStatusPolling` is intentionally leaving the timer + // running; we tear it down here. + if (stateBus && typeof stateBus.stopSubagentStatusPolling === "function") { + stateBus.stopSubagentStatusPolling(); + } + // Force the test runner to exit even if the resident ACP singleton + // (or some other module-side timer) is still holding stdio open. + // This is the same workaround as mcode-acp-note.test.js. All test + // assertions have already completed; the only remaining handle is + // a process-level one the test cannot observe. + setTimeout(() => process.exit(0), 10).unref(); +}); + +describe("pushSessionTreeChanged — named SSE broadcast", () => { + test("broadcasts the named frame to every connected cid", () => { + // Two cids, each with a fresh SSE writer (a capture object). + const written = []; + const resA = { write: (chunk) => written.push(["A", chunk]), writableEnded: false, destroyed: false }; + const resB = { write: (chunk) => written.push(["B", chunk]), writableEnded: false, destroyed: false }; + stateBus.setSseClient("cid-a", resA); + stateBus.setSseClient("cid-b", resB); + try { + stateBus.pushSessionTreeChanged(); + assert.equal(written.length, 2); + for (const entry of written) { + const chunk = entry[1]; + assert.ok( + chunk.startsWith("event: session-tree-changed\n"), + `frame must be a named event, got ${JSON.stringify(chunk)}`, + ); + } + } finally { + stateBus.endSseClient("cid-a", resA); + stateBus.endSseClient("cid-b", resB); + } + }); + + test("dead / closed SSE writers do not throw", () => { + const dead = { write: () => { throw new Error("closed"); }, writableEnded: true, destroyed: false }; + stateBus.setSseClient("cid-dead", dead); + assert.doesNotThrow(() => stateBus.pushSessionTreeChanged()); + stateBus.endSseClient("cid-dead", dead); + }); +}); + +describe("recordSubagentForCid — recentSubagents bookkeeping", () => { + test("appends an entry to a cid that has none yet", () => { + stateBus.getClient("cid-record-A"); + stateBus.recordSubagentForCid("cid-record-A", { + toolCallId: "tc_record_1", + sessionId: "mvs_record_1", + agentName: "verifier", + }); + const cs = stateBus.getClient("cid-record-A"); + assert.ok(Array.isArray(cs.recentSubagents)); + assert.equal(cs.recentSubagents.length, 1); + assert.equal(cs.recentSubagents[0].toolCallId, "tc_record_1"); + assert.equal(cs.recentSubagents[0].sessionId, "mvs_record_1"); + assert.equal(cs.recentSubagents[0].agentName, "verifier"); + }); + + test("a second record with the same toolCallId updates in place (no duplicate)", () => { + stateBus.getClient("cid-record-B"); + stateBus.recordSubagentForCid("cid-record-B", { + toolCallId: "tc_record_2", + sessionId: "mvs_record_2", + }); + stateBus.recordSubagentForCid("cid-record-B", { + toolCallId: "tc_record_2", + sessionId: "mvs_record_2", + status: "done", + }); + const cs = stateBus.getClient("cid-record-B"); + assert.equal(cs.recentSubagents.length, 1); + assert.equal(cs.recentSubagents[0].status, "done"); + }); + + test("ignores empty / nullish entries rather than writing garbage", () => { + stateBus.getClient("cid-record-C"); + stateBus.recordSubagentForCid("cid-record-C", null); + stateBus.recordSubagentForCid("cid-record-C", {}); + stateBus.recordSubagentForCid("cid-record-C", { toolCallId: "", sessionId: "" }); + stateBus.recordSubagentForCid("cid-record-C", { toolCallId: "tc_x", sessionId: "" }); + const cs = stateBus.getClient("cid-record-C"); + assert.equal(cs.recentSubagents.length, 0); + }); + + test("ignores calls with no cid (best-effort — no throw)", () => { + assert.doesNotThrow(() => + stateBus.recordSubagentForCid("", { toolCallId: "tc_x", sessionId: "mvs_x" }), + ); + }); +}); + +describe("refreshRecentSubagentStatuses — poll path", () => { + test("refreshes a recorded entry's status from the runtime db", () => { + const cid = "cid-poll-1"; + stateBus.getClient(cid); + stateBus.recordSubagentForCid(cid, { + toolCallId: "tc_poll_running", + sessionId: "mvs_child_a", + agentName: null, + status: "queued", // stale on purpose — poll must overwrite + }); + stateBus.refreshRecentSubagentStatuses(); + const cs = stateBus.getClient(cid); + const entry = cs.recentSubagents.find((r) => r.toolCallId === "tc_poll_running"); + assert.ok(entry); + assert.equal(entry.status, "running"); + assert.equal(entry.agentName, "verifier"); + }); + + test("is a no-op when no cid has recorded entries", () => { + // We can't isolate from the other tests' cids without resetting + // the module; just verify the call returns without throwing when + // there are no probes (the implementation bails early). + assert.doesNotThrow(() => stateBus.refreshRecentSubagentStatuses()); + }); +}); + +describe("subagent poll cadence — env knob", () => { + test("default cadence is 2000ms (process env override absent)", () => { + // The exported constant was bound at module load; sanity-check it. + assert.equal(stateBus.getSubagentPollIntervalMs(), 2000); + }); +}); + +describe("startSubagentStatusPolling / stopSubagentStatusPolling", () => { + test("start is idempotent; stop tears down cleanly", () => { + // Already started in production bootstrap (when the server boots), + // but the helper is idempotent so calling it again is a no-op. + assert.doesNotThrow(() => stateBus.startSubagentStatusPolling()); + assert.doesNotThrow(() => stateBus.startSubagentStatusPolling()); + assert.doesNotThrow(() => stateBus.stopSubagentStatusPolling()); + assert.doesNotThrow(() => stateBus.stopSubagentStatusPolling()); + }); +}); diff --git a/packages/webui/test/lib/agent-team-status.test.js b/packages/webui/test/lib/agent-team-status.test.js new file mode 100644 index 00000000..f9dc2327 --- /dev/null +++ b/packages/webui/test/lib/agent-team-status.test.js @@ -0,0 +1,203 @@ +// webui/test/lib/agent-team-status.test.js +// Unit tests for server/lib/agent-team-status.js — the DB → UI status +// projection that the Agent Team panel depends on. +// +// Coverage contract (every line here is a documented claim from +// .tickets/webui-parity/06-agent-team-panel.md): +// +// 1. `local_runtime_sessions.status` only carries +// `idle | interrupted | aborted | error` — measured on this machine, +// ticket 06 R8. The TUI vocabulary (failed/waiting/running/queued/ +// done/stopped) is a projection-layer vocabulary, NOT a db field. +// 2. `local_runtime_background_tasks.status` carries +// `running | succeeded | failed | canceled` for `kind = 'subagent'`. +// 3. The UI NEVER reads raw db strings — every consumer must project +// through this module. +// 4. `projectAgentStatus` must prefer a live task status over a stale +// session column (otherwise a busy subagent paints as idle). +// 5. `projectAgentStatus` must prefer a terminal task status over the +// session column (otherwise a finished subagent repaints as idle). +// 6. `isLiveStatus` returns true for `running` and `waiting` only — +// the parent tool-line running badge depends on it. +// 7. Unknown / nullish inputs fall back to `idle` rather than throwing +// (a runtime growth path cannot break the panel). + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; + +import { + AGENT_TEAM_STATUS, + projectSessionStatus, + projectTaskStatus, + projectAgentStatus, + isLiveStatus, + isUiStatus, +} from "../../server/lib/agent-team-status.js"; + +describe("projectSessionStatus — sessions.status → UI", () => { + test("idle stays idle (the default render for an active session)", () => { + assert.equal(projectSessionStatus("idle"), AGENT_TEAM_STATUS.IDLE); + }); + + test("interrupted maps to stopped (user-pressed-stop or crash)", () => { + assert.equal(projectSessionStatus("interrupted"), AGENT_TEAM_STATUS.STOPPED); + }); + + test("aborted maps to failed (engine cancelled the turn — not a fault, but terminal)", () => { + // Spec note: ticket 06 R8 maps `aborted` to `failed`. We surface that + // decision through this projection rather than carving it out at every + // consumer. Adjust this test if the contract changes. + assert.equal(projectSessionStatus("aborted"), AGENT_TEAM_STATUS.FAILED); + }); + + test("error maps to failed", () => { + assert.equal(projectSessionStatus("error"), AGENT_TEAM_STATUS.FAILED); + }); + + test("an empty / nullish / unknown string falls back to idle, never throws", () => { + assert.equal(projectSessionStatus(""), AGENT_TEAM_STATUS.IDLE); + assert.equal(projectSessionStatus(null), AGENT_TEAM_STATUS.IDLE); + assert.equal(projectSessionStatus(undefined), AGENT_TEAM_STATUS.IDLE); + assert.equal(projectSessionStatus("future-status"), AGENT_TEAM_STATUS.IDLE); + }); + + test("whitespace-only input is treated as empty (defensive parse)", () => { + assert.equal(projectSessionStatus(" "), AGENT_TEAM_STATUS.IDLE); + }); +}); + +describe("projectTaskStatus — background_tasks.status → UI", () => { + test("running is the only state that claims live on the parent tool line", () => { + assert.equal(projectTaskStatus("running"), AGENT_TEAM_STATUS.RUNNING); + }); + + test("succeeded maps to done", () => { + assert.equal(projectTaskStatus("succeeded"), AGENT_TEAM_STATUS.DONE); + }); + + test("failed maps to failed", () => { + assert.equal(projectTaskStatus("failed"), AGENT_TEAM_STATUS.FAILED); + }); + + test("canceled maps to stopped (a deliberate user action, not a fault)", () => { + assert.equal(projectTaskStatus("canceled"), AGENT_TEAM_STATUS.STOPPED); + }); + + test("an empty / unknown task status falls back to idle", () => { + assert.equal(projectTaskStatus(""), AGENT_TEAM_STATUS.IDLE); + assert.equal(projectTaskStatus(null), AGENT_TEAM_STATUS.IDLE); + assert.equal(projectTaskStatus("queued-future"), AGENT_TEAM_STATUS.IDLE); + }); +}); + +describe("projectAgentStatus — composing both projections", () => { + test("a running task beats an idle session column", () => { + // This is the load-bearing case: the parent tool line MUST show + // "running ▶" while the subagent is busy, even though the engine + // never wrote `running` into the session row. + assert.equal( + projectAgentStatus("idle", "running"), + AGENT_TEAM_STATUS.RUNNING, + ); + }); + + test("a done task wins over an idle session column (otherwise finished subagents repaint as idle)", () => { + assert.equal( + projectAgentStatus("idle", "succeeded"), + AGENT_TEAM_STATUS.DONE, + ); + }); + + test("a failed task wins over an idle session column", () => { + assert.equal( + projectAgentStatus("idle", "failed"), + AGENT_TEAM_STATUS.FAILED, + ); + }); + + test("a canceled task wins over an idle session column", () => { + assert.equal( + projectAgentStatus("idle", "canceled"), + AGENT_TEAM_STATUS.STOPPED, + ); + }); + + test("the session column still drives the answer when no task is known", () => { + // e.g. a subagent row exists in the runtime db but the engine has not + // yet written a background_tasks row for the current turn. We must + // not claim "running" out of thin air. + assert.equal( + projectAgentStatus("error", ""), + AGENT_TEAM_STATUS.FAILED, + ); + assert.equal( + projectAgentStatus("error", null), + AGENT_TEAM_STATUS.FAILED, + ); + assert.equal( + projectAgentStatus("idle", undefined), + AGENT_TEAM_STATUS.IDLE, + ); + }); + + test("when both columns are empty the answer is idle, not failed", () => { + assert.equal(projectAgentStatus("", ""), AGENT_TEAM_STATUS.IDLE); + }); + + test("an idle task does NOT silently override a non-idle session column", () => { + // `idle` is not a definitive task verdict; prefer the session column + // when the task column says nothing. + assert.equal( + projectAgentStatus("error", "idle"), + AGENT_TEAM_STATUS.FAILED, + ); + assert.equal( + projectAgentStatus("interrupted", "idle"), + AGENT_TEAM_STATUS.STOPPED, + ); + }); +}); + +describe("isLiveStatus — parent-tool-line running badge", () => { + test("running is live", () => { + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.RUNNING), true); + }); + + test("waiting is live (engine asked for user input)", () => { + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.WAITING), true); + }); + + test("done / failed / stopped / queued / idle are NOT live", () => { + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.DONE), false); + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.FAILED), false); + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.STOPPED), false); + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.QUEUED), false); + assert.equal(isLiveStatus(AGENT_TEAM_STATUS.IDLE), false); + }); + + test("an unknown string is never live", () => { + assert.equal(isLiveStatus("almost-running"), false); + assert.equal(isLiveStatus(""), false); + assert.equal(isLiveStatus(null), false); + }); +}); + +describe("isUiStatus — guard for external payloads", () => { + test("every AGENT_TEAM_STATUS value is accepted", () => { + for (const value of Object.values(AGENT_TEAM_STATUS)) { + assert.equal(isUiStatus(value), true, `${value} should be a UI status`); + } + }); + + test("raw db strings that are NOT in the UI vocabulary are rejected", () => { + // "idle" and "running" happen to overlap between the db and UI vocab, + // but the contract is: a consumer that needs a UI status must receive + // one from this module's projections. Anything else (a db string, + // a typo, an older version of the contract) must NOT pass. + assert.equal(isUiStatus("succeeded"), false); // db-only, no UI equivalent + assert.equal(isUiStatus("canceled"), false); // db-only, no UI equivalent + assert.equal(isUiStatus("interrupted"), false); // db-only + assert.equal(isUiStatus("aborted"), false); // db-only + assert.equal(isUiStatus("not-a-status"), false); // unknown + }); +}); diff --git a/packages/webui/test/lib/agent-team-tasks.test.js b/packages/webui/test/lib/agent-team-tasks.test.js new file mode 100644 index 00000000..fa2bccff --- /dev/null +++ b/packages/webui/test/lib/agent-team-tasks.test.js @@ -0,0 +1,365 @@ +// webui/test/lib/agent-team-tasks.test.js +// Unit tests for server/lib/agent-team-tasks.js — read-only projections +// over `local_runtime_background_tasks` used by the Agent Team wiring. +// +// Strategy: the tests spin up a real better-sqlite3 db in a temp file, +// seed it with a handful of rows that mirror the shape the runtime uses +// (the schema is verified in `.tickets/webui-parity/06-agent-team-panel.md` +// R8), and exercise every helper against that. This avoids both the +// "fake the whole module" antipattern (which would let a regression ship +// a SQL bug past the suite) and the "talk to the user's real db" rule +// the ticket explicitly bans. +// +// We force the resolver to point at the temp db through +// `process.env.MCODE_BETTER_SQLITE3` + `process.env.MCODE_RUNTIME_DB`, +// exactly as the codebase does in test/integration/*.test.js. The resolver +// caches the binding, so each test file must clear it (we just re-create +// a fresh module by importing a small wrapper that re-reads the env). + +import { test, describe, before, after } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtempSync, writeFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createRequire } from "node:module"; + +import { AGENT_TEAM_STATUS } from "../../server/lib/agent-team-status.js"; + +const require = createRequire(import.meta.url); + +let tmpDir; +let dbPath; +let tasksModule; + +function loadTasksModule() { + // dynamic import so the resolver picks up the test env after we set it + return import("../../server/lib/agent-team-tasks.js"); +} + +before(async () => { + tmpDir = mkdtempSync(join(tmpdir(), "agent-team-tasks-")); + dbPath = join(tmpDir, "runtime-state.sqlite"); + // Resolve better-sqlite3 from the workspace — same approach as the + // existing test/integration tests use. + const candidates = [ + join(process.cwd(), "node_modules", "better-sqlite3"), + join(process.cwd(), "..", "..", "node_modules", "better-sqlite3"), + join(process.cwd(), "..", "..", "..", "node_modules", "better-sqlite3"), + ]; + let bindingPath = null; + for (const c of candidates) { + try { + require(c); + bindingPath = c; + break; + } catch {} + } + if (!bindingPath) { + throw new Error("better-sqlite3 not found in workspace node_modules"); + } + process.env.MCODE_BETTER_SQLITE3 = bindingPath; + process.env.MCODE_RUNTIME_DB = dbPath; + + // Create the schema in the temp db. Mirrors the live schema columns + // this code touches. + const Db = require(bindingPath); + const db = new Db(dbPath); + db.exec(` + CREATE TABLE local_runtime_background_tasks ( + task_id TEXT PRIMARY KEY, + owner_session_id TEXT NOT NULL, + kind TEXT NOT NULL, + status TEXT NOT NULL, + created_at_ms INTEGER NOT NULL, + updated_at_ms INTEGER NOT NULL, + ended_at_ms INTEGER, + record_json TEXT NOT NULL + ); + `); + // Seed: a mix of subagent and bash rows, some with the same toolCallId. + const seed = [ + { + task_id: "bg_running_1", + owner_session_id: "mvs_parent", + kind: "subagent", + status: "running", + created_at_ms: 1000, + updated_at_ms: 1100, + ended_at_ms: null, + record_json: JSON.stringify({ + toolCallId: "tool_call_a", + metadata: { + parentSessionId: "mvs_parent", + childSessionId: "mvs_child_running", + agentName: "verifier", + }, + }), + }, + { + task_id: "bg_done_1", + owner_session_id: "mvs_parent", + kind: "subagent", + status: "succeeded", + created_at_ms: 2000, + updated_at_ms: 2500, + ended_at_ms: 2500, + record_json: JSON.stringify({ + toolCallId: "tool_call_b", + metadata: { + parentSessionId: "mvs_parent", + childSessionId: "mvs_child_done", + agentName: "explore", + }, + }), + }, + { + task_id: "bg_failed_1", + owner_session_id: "mvs_parent", + kind: "subagent", + status: "failed", + created_at_ms: 3000, + updated_at_ms: 3100, + ended_at_ms: 3100, + record_json: JSON.stringify({ + toolCallId: "tool_call_c", + metadata: { + parentSessionId: "mvs_parent", + childSessionId: "mvs_child_failed", + agentName: "worker", + }, + }), + }, + { + task_id: "bg_canceled_1", + owner_session_id: "mvs_parent", + kind: "subagent", + status: "canceled", + created_at_ms: 4000, + updated_at_ms: 4100, + ended_at_ms: 4100, + record_json: JSON.stringify({ + toolCallId: "tool_call_d", + metadata: { + parentSessionId: "mvs_parent", + childSessionId: "mvs_child_canceled", + agentName: "coder", + }, + }), + }, + { + task_id: "bg_bash_1", + owner_session_id: "mvs_parent", + kind: "bash", + status: "succeeded", + created_at_ms: 1500, + updated_at_ms: 1600, + ended_at_ms: 1600, + record_json: JSON.stringify({ + toolCallId: "tool_call_bash", + metadata: {}, + }), + }, + { + task_id: "bg_other_parent", + owner_session_id: "mvs_other_parent", + kind: "subagent", + status: "running", + created_at_ms: 1200, + updated_at_ms: 1200, + ended_at_ms: null, + record_json: JSON.stringify({ + toolCallId: "tool_call_other", + metadata: { + parentSessionId: "mvs_other_parent", + childSessionId: "mvs_child_other", + agentName: "verifier", + }, + }), + }, + ]; + const insert = db.prepare( + `INSERT INTO local_runtime_background_tasks + (task_id, owner_session_id, kind, status, created_at_ms, updated_at_ms, ended_at_ms, record_json) + VALUES (@task_id, @owner_session_id, @kind, @status, @created_at_ms, @updated_at_ms, @ended_at_ms, @record_json)`, + ); + for (const row of seed) insert.run(row); + db.close(); + + // Force the resolver to refresh its cache. The resolver binds better-sqlite3 + // exactly once and caches the module, so a fresh dynamic import of the + // tasks module after the env is set is enough — there is no need to + // re-import the resolver here. + tasksModule = await loadTasksModule(); +}); + +after(() => { + if (tmpDir) rmSync(tmpDir, { recursive: true, force: true }); + delete process.env.MCODE_BETTER_SQLITE3; + delete process.env.MCODE_RUNTIME_DB; +}); + +describe("findSubagentTaskByToolCallId", () => { + test("returns the projected UI status for a running subagent task", () => { + const out = tasksModule.findSubagentTaskByToolCallId("tool_call_a"); + assert.ok(out, "expected a hit for tool_call_a"); + assert.equal(out.status, AGENT_TEAM_STATUS.RUNNING); + assert.equal(out.agentName, "verifier"); + assert.equal(out.childSessionId, "mvs_child_running"); + }); + + test("maps succeeded → done", () => { + const out = tasksModule.findSubagentTaskByToolCallId("tool_call_b"); + assert.equal(out.status, AGENT_TEAM_STATUS.DONE); + assert.equal(out.agentName, "explore"); + }); + + test("maps failed → failed", () => { + const out = tasksModule.findSubagentTaskByToolCallId("tool_call_c"); + assert.equal(out.status, AGENT_TEAM_STATUS.FAILED); + assert.equal(out.agentName, "worker"); + }); + + test("maps canceled → stopped", () => { + const out = tasksModule.findSubagentTaskByToolCallId("tool_call_d"); + assert.equal(out.status, AGENT_TEAM_STATUS.STOPPED); + assert.equal(out.agentName, "coder"); + }); + + test("ignores bash rows even if their toolCallId matches", () => { + // The kind filter is load-bearing — a bash task with the same + // toolCallId as a subagent must never leak through this projection. + const out = tasksModule.findSubagentTaskByToolCallId("tool_call_bash"); + assert.equal(out, null); + }); + + test("returns null for an unknown toolCallId", () => { + assert.equal( + tasksModule.findSubagentTaskByToolCallId("tool_call_missing"), + null, + ); + }); + + test("returns null when the input is empty / nullish", () => { + assert.equal(tasksModule.findSubagentTaskByToolCallId(""), null); + assert.equal(tasksModule.findSubagentTaskByToolCallId(null), null); + assert.equal(tasksModule.findSubagentTaskByToolCallId(undefined), null); + }); + + test("every returned status is in the UI vocabulary (no raw db leakage)", () => { + // Last line of defence: a regression that drops `projectTaskStatus` + // out of the helper would silently smuggle the db vocabulary into + // the wire. The vocabularies intentionally overlap on `running` / + // `idle` / `failed`, so the contract is the membership check, not + // an inequality. + const ids = [ + "tool_call_a", + "tool_call_b", + "tool_call_c", + "tool_call_d", + ]; + const allowed = new Set(Object.values(AGENT_TEAM_STATUS)); + for (const id of ids) { + const row = tasksModule.findSubagentTaskByToolCallId(id); + assert.ok(row, `expected a hit for ${id}`); + assert.ok( + allowed.has(row.status), + `${row.status} must be one of ${[...allowed].join(", ")}`, + ); + } + }); +}); + +describe("hasSubagentTaskByToolCallId", () => { + test("true for a known subagent toolCallId", () => { + assert.equal(tasksModule.hasSubagentTaskByToolCallId("tool_call_a"), true); + }); + + test("false for a bash toolCallId with the same id space", () => { + assert.equal( + tasksModule.hasSubagentTaskByToolCallId("tool_call_bash"), + false, + ); + }); + + test("false for an unknown toolCallId", () => { + assert.equal( + tasksModule.hasSubagentTaskByToolCallId("tool_call_missing"), + false, + ); + }); + + test("false for an empty / nullish input (no spurious matches)", () => { + assert.equal(tasksModule.hasSubagentTaskByToolCallId(""), false); + assert.equal(tasksModule.hasSubagentTaskByToolCallId(null), false); + assert.equal(tasksModule.hasSubagentTaskByToolCallId(undefined), false); + }); +}); + +describe("listSubagentTasksByParentSession", () => { + test("returns only the parent's subagent rows, in descending recency", () => { + const list = tasksModule.listSubagentTasksByParentSession("mvs_parent"); + assert.equal(list.length, 4); + assert.deepEqual( + list.map((row) => row.toolCallId), + ["tool_call_d", "tool_call_c", "tool_call_b", "tool_call_a"], + ); + // Every status field has been projected, never raw. + for (const row of list) { + assert.ok(Object.values(AGENT_TEAM_STATUS).includes(row.status)); + } + }); + + test("does not leak rows from a different parent", () => { + const list = tasksModule.listSubagentTasksByParentSession("mvs_parent"); + assert.ok( + !list.some((row) => row.childSessionId === "mvs_child_other"), + "mvs_other_parent's row must not leak into mvs_parent's list", + ); + }); + + test("respects the limit cap", () => { + const list = tasksModule.listSubagentTasksByParentSession("mvs_parent", { limit: 2 }); + assert.equal(list.length, 2); + }); + + test("returns [] for an unknown parent (no throw)", () => { + assert.deepEqual( + tasksModule.listSubagentTasksByParentSession("mvs_no_such_parent"), + [], + ); + }); + + test("returns [] for an empty / nullish input", () => { + assert.deepEqual(tasksModule.listSubagentTasksByParentSession(""), []); + assert.deepEqual(tasksModule.listSubagentTasksByParentSession(null), []); + }); +}); + +describe("diffSubagentsByParentSession", () => { + test("returns the rows the caller has not seen yet", () => { + const known = new Set(["tool_call_a", "tool_call_b"]); + const diff = tasksModule.diffSubagentsByParentSession("mvs_parent", known); + assert.deepEqual( + diff.map((row) => row.toolCallId).sort(), + ["tool_call_c", "tool_call_d"], + ); + }); + + test("returns [] when every row is already known", () => { + const known = new Set([ + "tool_call_a", + "tool_call_b", + "tool_call_c", + "tool_call_d", + ]); + assert.deepEqual( + tasksModule.diffSubagentsByParentSession("mvs_parent", known), + [], + ); + }); + + test("treats a non-Set input as an empty set (every row is new)", () => { + const diff = tasksModule.diffSubagentsByParentSession("mvs_parent", null); + assert.equal(diff.length, 4); + }); +}); diff --git a/packages/webui/test/lib/mcode-acp-note.test.js b/packages/webui/test/lib/mcode-acp-note.test.js index f530d01e..852f7ae7 100644 --- a/packages/webui/test/lib/mcode-acp-note.test.js +++ b/packages/webui/test/lib/mcode-acp-note.test.js @@ -100,8 +100,12 @@ describe("applyToolUpdate — synthetic header when the tool_call never arrived title: "Read", status: "in_progress", }); - assert.deepEqual(cs.chat, ["› hi", "→ Read", " [in_progress]"]); - assert.equal(r.toolIndexById.get("tc-orphan"), 1); + // Slice 06: applyToolUpdate emits a `##tc:` marker + // immediately before the synthetic `→ name` header so the + // decoder can correlate the block with its recentSubagents[]. + // The marker is the +1 shift relative to the pre-slice-06 shape. + assert.deepEqual(cs.chat, ["› hi", "##tc:tc-orphan", "→ Read", " [in_progress]"]); + assert.equal(r.toolIndexById.get("tc-orphan"), 2); }); test("a second update for the same orphan toolCallId appends after the synthesized header", () => { @@ -114,11 +118,13 @@ describe("applyToolUpdate — synthetic header when the tool_call never arrived status: "completed", locations: [{ path: "/home/u/.agents/rule.md" }], }); - // Header is written once; subsequent bodies insert at insertAfter+1, so - // each new body's status line lands right after the header and pushes - // the prior body deeper. This matches the existing known-tool behaviour - // and pins it for the synthetic-header path. + // Header is written once (slice 06 also wrote the `##tc:` marker + // on the synthetic path; subsequent updates insert body lines + // AFTER the marker+header pair). The marker carries no UI weight — + // the decoder consumes it - so its only effect is a +1 line + // shift relative to the pre-slice-06 shape. assert.deepEqual(cs.chat, [ + "##tc:tc-1", "→ Bash", " [completed]", " @ /home/u/.agents/rule.md", @@ -132,11 +138,11 @@ describe("applyToolUpdate — synthetic header when the tool_call never arrived status: "completed", locations: [{ path: "/home/u/.agents/rule.md" }], }); - // The newly-added body landed at index 1 (right after the header). The - // prior body lines (the duplicate path, the prior "ok" output) shifted - // by two positions. - assert.equal(cs.chat[1], " [completed]"); - assert.equal(cs.chat[2], " @ /home/u/.agents/rule.md"); + // The newly-added body landed at index 2 (right after the + // marker+header pair). The prior body lines (the duplicate path, + // the prior "ok" output) shifted by two positions. + assert.equal(cs.chat[2], " [completed]"); + assert.equal(cs.chat[3], " @ /home/u/.agents/rule.md"); }); test("a known toolCallId (prior tool_call arrived) inserts after the existing header", () => { @@ -158,7 +164,12 @@ describe("applyToolUpdate — synthetic header when the tool_call never arrived const cs = { chat: [] }; const r = {}; applyToolUpdate(r, cs, { toolCallId: "tc-x", status: "completed" }); - assert.equal(cs.chat[0], "→ tool"); + // Slice 06: applyToolUpdate emits a `##tc:` marker + // immediately before the `→ name` header so the decoder can + // correlate the block with the matching recentSubagents[] entry. + // cs.chat[0] is now the marker; the header lands at cs.chat[1]. + assert.equal(cs.chat[0], "##tc:tc-x"); + assert.equal(cs.chat[1], "→ tool"); }); }); diff --git a/packages/webui/test/lib/mcode-acp-tc-marker.test.js b/packages/webui/test/lib/mcode-acp-tc-marker.test.js new file mode 100644 index 00000000..2561dd33 --- /dev/null +++ b/packages/webui/test/lib/mcode-acp-tc-marker.test.js @@ -0,0 +1,111 @@ +// webui/test/lib/mcode-acp-tc-marker.test.js +// Regression for slice 06's `##tc:` marker emission. +// +// What this locks. The marker is the contract that lets the chat +// renderer correlate a `→ name` block with the matching +// `recentSubagents[]` entry by id (matching by tool name would badge +// every `→ task` line with the newest child, which is wrong for +// sessions that spawn multiple subagents). Without the marker on +// the wire, the ToolCard falls back to the newest entry — which +// happened to be the bug the acceptance pass flagged. +// +// Coverage contract: +// 1. `applyToolUpdate` emits `##tc:` immediately BEFORE the +// `→ name` header on the synthetic-header path (no prior +// tool_call). +// 2. `applyToolUpdate` does NOT re-emit the marker on subsequent +// updates for the same toolCallId (the marker is paired with +// the header, not with every body line). +// 3. `applyToolUpdate` does NOT emit the marker when a known +// toolCallId is inserted after an existing header (the header +// was already emitted by a prior tool_call, whose marker is +// already in the chat). + +import { test, describe, after } from "node:test"; +import assert from "node:assert/strict"; + +const { + applyToolUpdate, +} = await import("../../server/lib/mcode-acp.js"); +const { getMcodeAcpClient, shutdownMcodeAcpSingleton } = await import( + "../../server/lib/acp-client.js" +); + +// Tear down the ACP singleton so the test process exits (same pattern +// as mcode-acp-note.test.js). +after(async () => { + try { + await getMcodeAcpClient(); + } catch { + /* engine never started */ + } + try { + shutdownMcodeAcpSingleton(); + } catch { + /* nothing was started */ + } + await new Promise((r) => setTimeout(r, 50)); +}); + +describe("applyToolUpdate — emits ##tc: marker for slice 06 correlation", () => { + test("emits the marker BEFORE the synthetic → name header", () => { + const cs = { chat: ["› hi"] }; + const r = {}; + applyToolUpdate(r, cs, { + toolCallId: "tc-marker-1", + title: "task", + status: "in_progress", + }); + // The marker must come immediately before the header (so the + // decoder can park it and attach to the next tool block). + assert.equal(cs.chat[0], "› hi"); + assert.equal(cs.chat[1], "##tc:tc-marker-1"); + assert.equal(cs.chat[2], "→ task"); + assert.equal(cs.chat[3], " [in_progress]"); + // toolIndexById points at the header (the marker is consumed by + // the decoder; it's not a position the server tracks separately). + assert.equal(r.toolIndexById.get("tc-marker-1"), 2); + }); + + test("does NOT re-emit the marker on subsequent updates for the same id", () => { + const cs = { chat: [] }; + const r = {}; + applyToolUpdate(r, cs, { toolCallId: "tc-reuse", title: "task", status: "in_progress" }); + applyToolUpdate(r, cs, { toolCallId: "tc-reuse", status: "completed" }); + // One marker, one header, then bodies. The marker would never + // appear again — subsequent body lines splice after the header. + const markerCount = cs.chat.filter((line) => line === "##tc:tc-reuse").length; + assert.equal(markerCount, 1, "marker must emit exactly once per dispatch"); + }); + + test("does NOT emit a marker when inserting into an existing header", () => { + // Simulate a prior tool_call that already wrote the marker. + // applyToolUpdate's insert-after-existing path must not duplicate. + const cs = { chat: ["##tc:tc-known", "→ task {}", " [in_progress]"] }; + const r = { + toolIndexById: new Map([["tc-known", 1]]), + }; + applyToolUpdate(r, cs, { toolCallId: "tc-known", status: "completed" }); + // No new marker line; the body splices after the existing header. + const markerCount = cs.chat.filter((line) => line === "##tc:tc-known").length; + assert.equal(markerCount, 1, "marker must NOT duplicate on insert-after"); + }); + + test("omits the marker when toolCallId is absent (older engine contract)", () => { + const cs = { chat: [] }; + const r = {}; + applyToolUpdate(r, cs, { + // no toolCallId on purpose — older engines do not stamp one. + title: "task", + status: "in_progress", + }); + // The marker is conditional on a toolCallId existing; without + // one, the header still lands but with no marker so the decoder + // falls back to the newest-entry lookup (lib/agent-team-lookup). + assert.equal(cs.chat[0], "→ task"); + assert.equal( + cs.chat.some((line) => line.startsWith("##tc:")), + false, + ); + }); +}); diff --git a/packages/webui/test/routes/chat-run-mirror.check.mjs b/packages/webui/test/routes/chat-run-mirror.check.mjs index 8e75e17f..27741217 100644 --- a/packages/webui/test/routes/chat-run-mirror.check.mjs +++ b/packages/webui/test/routes/chat-run-mirror.check.mjs @@ -260,10 +260,17 @@ function recordBy(id) { return sessions.loadSessions().find((s) => s && s.id === id) || null; } -// §§ marker lines carry a variable duration — filter them for chat -// assertions that pin stable lines only. +// Metadata marker lines — both the existing `§§ processed_duration=Nms` +// (turn-process disclosure) and the slice-06 `##tc:` +// marker the decoder consumes — are stripped before comparing chat +// content. The markers carry runtime-only metadata that must NEVER +// appear in user-visible chat, so a stable comparison means +// stable + marker-free. const stable = (chat) => - (chat || []).filter((line) => !String(line).startsWith("§§")); + (chat || []).filter((line) => { + const s = String(line); + return !s.startsWith("§§") && !s.startsWith("##tc:"); + }); before(async (t) => { await setupMocks(t); @@ -350,7 +357,15 @@ describe("run-mirror — mid-run switch keeps views and records isolated", () => emitToolAndAnswer(); buf = sb.runChatLinesFor(cid, sidA); assert.equal(buf[0], "▲ pondering", "message stream strips the ▲ cursor"); - assert.equal(buf[1], "→ Bash {\"cmd\":\"ls\"}"); + // Slice 06 (Agent Team): the tool_call path emits a `##tc:tc-1` + // marker immediately before the `→ Bash` header so the chat + // renderer can correlate the block with its `recentSubagents[]` + // entry. The marker is consumed by `decodeTranscript` (it never + // appears in the rendered chat) so the deepEqual checks below + // are unaffected — the `stable()` helper strips it the same + // way it strips `§§ processed_duration`. + assert.equal(buf[1], "##tc:tc-1", "toolCallId marker precedes the → name header"); + assert.equal(buf[2], "→ Bash {\"cmd\":\"ls\"}"); assert.ok(buf.includes(" [completed]")); assert.ok(buf.includes(" file.txt")); assert.match(buf[buf.length - 1], /^● part one/); @@ -381,7 +396,13 @@ describe("run-mirror — mid-run switch keeps views and records isolated", () => "owning view keeps its running indicator after switch-back", ); // base (3 persisted lines) + the 6 buffered stream lines so far - assert.equal(backSnap.chat.length, 9); + // (▲ pondering, → Bash header, status line, output line, ● line) + // — plus the slice-06 `##tc:` marker that precedes the tool + // header. The marker is consumed by the decoder and never + // appears in the rendered chat body; it only inflates the raw + // buffer length. Filtering it via `stable()` would drop it + // here too — see `stable()`'s docstring. + assert.equal(backSnap.chat.length, 10); assert.deepEqual( stable(backSnap.chat).slice(0, 3), ["› hello", "● ok", "› run A2"], diff --git a/packages/webui/webapp/components/chat.tsx b/packages/webui/webapp/components/chat.tsx index ec0783d3..044dd33e 100644 --- a/packages/webui/webapp/components/chat.tsx +++ b/packages/webui/webapp/components/chat.tsx @@ -2,8 +2,12 @@ import { useCallback, useEffect, useMemo, useRef, useState } from "react"; +import * as api from "@/lib/api"; import { renderMarkdown } from "@/lib/markdown"; import { reportActionError } from "@/lib/action-errors"; +import { findSubagentForBlock } from "@/lib/agent-team-lookup"; +import { badgeLabelAndGlyph, agentLabel } from "@/lib/i18n-agent-team"; +import { useLocale } from "@/lib/use-locale"; import { decodeTranscript, groupActivity, @@ -905,6 +909,19 @@ function ToolCard({ block, t }: { block: TranscriptBlock; t: (key: MessageKey) = const paths = block.toolPaths ?? []; const hasBody = output.length > 0 || paths.length > 0; const iconType = iconByName(block.toolName); + // Slice 06 — Agent Team: when this tool is the parent of a subagent + // dispatch, attach the live status badge + jump reference from the + // server's `recentSubagents` array. The match is by `toolCallId` + // (carried on the block by the `##tc:` marker the decoder + // consumes) so a session that spawned multiple subagents badges + // each `→ task` line with its OWN child — matching by tool NAME + // would badge every line with the newest child, which is wrong. + // See `lib/agent-team-lookup.ts#findSubagentForBlock` for the + // matching rule and its unit tests. + const store = useSessionContext(); + const { locale } = useLocale(); + const recent = store?.state?.recentSubagents; + const subagent = findSubagentForBlock(recent, block); const statusKey = block.toolStatus === "failed" @@ -913,6 +930,16 @@ function ToolCard({ block, t }: { block: TranscriptBlock; t: (key: MessageKey) = ? "tool.status.in_progress" : "tool.status.completed"; + // Subagent badge — label and glyph are resolved through i18n so + // both locales actually differ (the previous slice hardcoded English + // glyphs here, leaving the file orphaned — the acceptance fix wires + // the keys through `tAgentTeam` / `agentLabel`). + const badge = subagent ? badgeLabelAndGlyph(locale, subagent.status) : null; + const agentNameLabel = subagent ? agentLabel(locale, subagent.agentName) : null; + const subagentLabel = badge && agentNameLabel + ? `${badge.glyph} ${agentNameLabel}` + : null; + return (
+ ) : null} {block.toolArgs ? ( {block.toolArgs} diff --git a/packages/webui/webapp/components/session-tree.tsx b/packages/webui/webapp/components/session-tree.tsx index 3cd169dc..a2acb714 100644 --- a/packages/webui/webapp/components/session-tree.tsx +++ b/packages/webui/webapp/components/session-tree.tsx @@ -45,7 +45,8 @@ const SESSION_VISIBLE_LIMIT = 6; const UNTITLED: MessageKey = "sidebar.untitled"; export function SessionTree({ t }: { t: (key: MessageKey) => string }) { - const { state } = useSessionContext(); + const store = useSessionContext(); + const state = store.state; const activeId = state?.mcodeSessionId ?? null; // Live, from the SSE snapshot — unlike `session.status` in the payload below, // which the server reads from the engine's database through a 15s cache. @@ -98,6 +99,19 @@ export function SessionTree({ t }: { t: (key: MessageKey) => string }) { void refresh(true); }, [refresh, running]); + // Slice 06 — Agent Team. The server emits a `session-tree-changed` SSE + // frame the moment a subagent row lands in the runtime db. The store + // bumps `treeRevision` on every frame; we react by forcing a re-read + // past the 15s cache. Skipped on mount: the initial fetch already runs. + const sawTreeChange = useRef(false); + useEffect(() => { + if (!sawTreeChange.current) { + sawTreeChange.current = true; + return; + } + void refresh(true); + }, [refresh, store.treeRevision]); + // Open the active session's chain on first sight so "where am I" is answered // without a click. Cheap to re-run: the three updates are no-ops once open. useEffect(() => { diff --git a/packages/webui/webapp/lib/agent-team-lookup.ts b/packages/webui/webapp/lib/agent-team-lookup.ts new file mode 100644 index 00000000..4cbfd07f --- /dev/null +++ b/packages/webui/webapp/lib/agent-team-lookup.ts @@ -0,0 +1,72 @@ +// webapp/lib/agent-team-lookup.ts +// Pure lookup helper: pick the right `recentSubagents[]` entry for a tool +// block the renderer is about to badge. +// +// Why this lives in its own module. The renderer (`ToolCard`) used to +// inline the rule: take `recent[recent.length - 1]` and call it a day. +// That works for a session that spawned exactly one subagent, but a +// normal Agent Team session spawns many (one per dispatched tool_call), +// and the rule was badge-with-the-newest regardless of which block is +// being rendered — clicking an older `→ task` line then jumped to the +// wrong subagent. The dispatch brief flags this as a correctness bug. +// +// The fix is to match by `toolCallId` (the decoder attaches it to the +// block via a `##tc:` marker emitted by `applyToolUpdate`). When +// the marker is missing — older chat written before this slice shipped +// — we fall back to the newest entry with a matching agent name, so +// legacy sessions still render a usable badge. +// +// The rule is exported as a pure function so a unit test can pin both +// paths (marker present → exact match, marker absent → newest fallback) +// without a DOM, an EventSource, or a real runtime db. + +import type { RecentSubagent } from "./types"; + +/** + * The kind of tool the renderer is asking about. We accept the + * `block`-shaped input rather than the bare toolCallId so callers do + * not have to recompute the toolName gate at every site. + */ +export interface ToolBlockLike { + toolName?: string; + toolCallId?: string; +} + +/** + * Pick the `recentSubagents[]` entry that matches a given tool block. + * + * @returns the matching entry, or null when the tool name is not a + * subagent dispatch OR no entry exists. + * + * Rules: + * 1. Non-subagent tool names return null (read / bash / write never + * have a child session to jump to). + * 2. With a `toolCallId` on the block, match exactly. This is the + * primary path — it lets two `→ task` lines in the same session + * each jump to their own child. + * 3. Without a `toolCallId` (older chat, or a tool_call whose id + * never reached the marker), fall back to the newest entry — + * pragmatic for legacy sessions; the slice 06 marker is the + * durable contract going forward. + */ +export function findSubagentForBlock( + recent: readonly RecentSubagent[] | undefined, + block: ToolBlockLike, +): RecentSubagent | null { + if (!Array.isArray(recent) || recent.length === 0) return null; + const name = String(block.toolName || "").toLowerCase().replace(/[^a-z]/g, ""); + // Accept 'task' (canonical) and the legacy 'delegate' / 'delegatetask' + // variants the engine has emitted in the wild. + if (name !== "task" && name !== "delegate" && name !== "delegatetask") return null; + if (block.toolCallId) { + const match = recent.find((r) => r && r.toolCallId === block.toolCallId); + if (match) return match; + // toolCallId present but no matching recentSubagents entry: a + // mid-stream attach race where the marker landed in chat before + // the runtime wrote the background_tasks row. Fall back to the + // newest entry rather than render nothing — the run-mirror run + // is visible regardless, and the badge surfaces a hint that the + // match will resolve when the poller next tick fires. + } + return recent[recent.length - 1] || null; +} diff --git a/packages/webui/webapp/lib/i18n-agent-team.ts b/packages/webui/webapp/lib/i18n-agent-team.ts new file mode 100644 index 00000000..5d4479a9 --- /dev/null +++ b/packages/webui/webapp/lib/i18n-agent-team.ts @@ -0,0 +1,148 @@ +/** + * Bilingual strings — slice 06 (Agent Team) only. + * + * New module rather than extending `lib/i18n.ts` because that file is + * owned by slice 01 (the file-tree slice, in flight). Adding keys here + * avoids a merge conflict when both slices land. + * + * Every key MUST exist in both `en` and `zh` — the runtime check is the + * `agent-team-i18n.test.ts` lock, not just a manual review. A key + * missing in one locale silently falls back to the en value (or the key + * name itself), and the panel ships in Chinese by default, so a missing + * zh entry ships English text to a Chinese-locale user. + */ + +import type { Locale } from "./i18n"; + +const AGENT_TEAM_STRINGS = { + en: { + // Status labels for the toolCard subagent badge. The label is + // shown next to the agent name (e.g. "Running ▶ verifier") and + // doubles as the `aria-label` body. `idle` is included for the + // pre-status state the poller reports between subagent birth and + // the first background_tasks refresh. + "agentTeam.statusLabel.running": "Running", + "agentTeam.statusLabel.done": "Done", + "agentTeam.statusLabel.failed": "Failed", + "agentTeam.statusLabel.stopped": "Stopped", + "agentTeam.statusLabel.idle": "Idle", + "agentTeam.statusLabel.queued": "Queued", + // Fallback name when the runtime did not stamp an agent name + // (older chat, or a mid-stream attach race). + "agentTeam.subagentFallback": "subagent", + // Known agent team member labels. The runtime stores these as + // English tokens (`explore`, `worker`, `verifier`, `coder`) so + // the frontend translates them per-locale. A token we have not + // mapped (a future custom agent) falls back to the English token + // verbatim. + "agentTeam.agent.explore": "Explore", + "agentTeam.agent.worker": "Worker", + "agentTeam.agent.verifier": "Verifier", + "agentTeam.agent.coder": "Coder", + // Glyph prefix the badge renders before the agent label (e.g. + // "▶ Explore"). U+25B6 (▶) for running; U+2713 (✓) for done; + // U+2717 (✗) for failed; U+25A0 (■) for stopped; U+00B7 (·) for + // queued / idle. Kept separate from the label so future i18n + // (e.g. RTL) can swap just the glyph. + "agentTeam.badge.glyph.running": "\u25B6", + "agentTeam.badge.glyph.done": "\u2713", + "agentTeam.badge.glyph.failed": "\u2717", + "agentTeam.badge.glyph.stopped": "\u25A0", + "agentTeam.badge.glyph.queued": "\u00B7", + "agentTeam.badge.glyph.idle": "\u00B7", + // The badge's full tooltip / aria-label. Shown to screen readers + // and on hover; carries both the status and the session id so a + // user can paste it into the run-mirror search field if the + // click does not navigate. + "agentTeam.badge.open": "Open subagent session", + }, + zh: { + "agentTeam.statusLabel.running": "\u8FD0\u884C\u4E2D", + "agentTeam.statusLabel.done": "\u5DF2\u5B8C\u6210", + "agentTeam.statusLabel.failed": "\u5931\u8D25", + "agentTeam.statusLabel.stopped": "\u5DF2\u505C\u6B62", + "agentTeam.statusLabel.idle": "\u7A7A\u95F2", + "agentTeam.statusLabel.queued": "\u6392\u961F\u4E2D", + "agentTeam.subagentFallback": "\u5B50 agent", + "agentTeam.agent.explore": "\u63A2\u67E5\u8005", + "agentTeam.agent.worker": "\u52A9\u624B", + "agentTeam.agent.verifier": "\u9A8C\u8BC1\u8005", + "agentTeam.agent.coder": "\u7F16\u7801\u8005", + "agentTeam.badge.glyph.running": "\u25B6", + "agentTeam.badge.glyph.done": "\u2713", + "agentTeam.badge.glyph.failed": "\u2717", + "agentTeam.badge.glyph.stopped": "\u25A0", + "agentTeam.badge.glyph.queued": "\u00B7", + "agentTeam.badge.glyph.idle": "\u00B7", + "agentTeam.badge.open": "\u6253\u5F00\u5B50 agent \u4F1A\u8BDD", + }, +} as const; + +export type AgentTeamKey = keyof typeof AGENT_TEAM_STRINGS["en"]; + +/** Exported for tests. Treat as immutable — the runtime only reads. */ +export { AGENT_TEAM_STRINGS }; + +/** + * Resolve a slice-06 string for the current locale. + * + * Falls back to en when the requested locale is unknown (defensive — + * the webui only ships zh / en today, but the function should not + * throw if a future third locale slips through). Falls back to the + * raw key when the bucket is missing the entry, so a regression here + * shows the key name (e.g. "agentTeam.statusLabel.X") in the UI + * rather than rendering an empty badge. + */ +export function tAgentTeam( + locale: Locale, + key: AgentTeamKey, +): string { + const safeLocale = (locale === "zh" ? "zh" : "en") as "zh" | "en"; + const bucket = AGENT_TEAM_STRINGS[safeLocale] || AGENT_TEAM_STRINGS.en; + return bucket[key] || (AGENT_TEAM_STRINGS.en[key] ?? key); +} + +/** + * Map a server-side UI status (the AGENT_TEAM_STATUS vocabulary — see + * `server/lib/agent-team-status.js`) to the locale-resolved badge + * label and glyph pair. Returns `null` for statuses the badge does + * not render, so the caller can decide whether to render at all. + */ +export function badgeLabelAndGlyph( + locale: Locale, + uiStatus: string | null | undefined, +): { label: string; glyph: string } | null { + if (!uiStatus) return null; + if ( + uiStatus !== "running" && + uiStatus !== "done" && + uiStatus !== "failed" && + uiStatus !== "stopped" && + uiStatus !== "idle" && + uiStatus !== "queued" + ) { + return null; + } + return { + label: tAgentTeam(locale, `agentTeam.statusLabel.${uiStatus}`), + glyph: tAgentTeam(locale, `agentTeam.badge.glyph.${uiStatus}`), + }; +} + +/** + * Resolve a runtime-stored agent name token (e.g. `verifier`) to its + * locale-resolved label. Unknown tokens fall back to the English + * verbatim (so a future custom agent still renders something + * readable), then to the subagent fallback when the input is empty. + */ +export function agentLabel(locale: Locale, agentName: string | null | undefined): string { + const safeLocale = (locale === "zh" ? "zh" : "en") as "zh" | "en"; + if (typeof agentName !== "string" || !agentName) { + return tAgentTeam(locale, "agentTeam.subagentFallback"); + } + const enKey = `agentTeam.agent.${agentName}`; + if (enKey in AGENT_TEAM_STRINGS.en) { + return AGENT_TEAM_STRINGS[safeLocale][enKey as AgentTeamKey] || agentName; + } + return agentName; +} diff --git a/packages/webui/webapp/lib/sse.ts b/packages/webui/webapp/lib/sse.ts index 9e51765c..60f70fb9 100644 --- a/packages/webui/webapp/lib/sse.ts +++ b/packages/webui/webapp/lib/sse.ts @@ -28,6 +28,11 @@ export type SseAction = * refresh the management panel and re-fetch /api/models so the * composer selector shows new groups without a page reload. */ | { kind: "providers-updated"; providers: unknown[] } + /** Slice 06 — a subagent was just born or settled. The sidebar + * session tree refetches; the chat renderer's `recentSubagents` + * list is updated through the next state push. The payload is + * empty — the listener decides when to re-read. */ + | { kind: "tree-changed" } /** Keepalive; nothing to render. */ | { kind: "heartbeat" } /** A frame we recognise but intentionally do not act on. */ @@ -46,6 +51,10 @@ export const NAMED_EVENTS = [ // model selector refresh without polling. The data payload carries // the masked providers list — apiKey NEVER plaintext on this path. "providers.updated", + // Slice 06 — Agent Team. The server fires this when a subagent row + // lands in the runtime db (tool_call → background_tasks.kind = + // "subagent"), so the sidebar session tree can re-read its cache. + "session-tree-changed", "heartbeat", ] as const; @@ -102,6 +111,16 @@ export function parseSseFrame(event: string, data: string): SseAction { const providers = Array.isArray(parsed.value.providers) ? parsed.value.providers : []; return { kind: "providers-updated", providers }; } + case "session-tree-changed": { + // Slice 06: no payload — the sidebar decides when to re-fetch. + // The body is `{}` so JSON parsing is a safe no-op; malformed + // payloads are surfaced so the connection stays live. + if (data && data.trim() !== "" && data.trim() !== "{}") { + const parsed = parseJson(data); + if (!parsed.ok) return { kind: "malformed", event, detail: parsed.detail }; + } + return { kind: "tree-changed" }; + } default: return { kind: "ignored", reason: `unknown event: ${event || "(none)"}` }; } diff --git a/packages/webui/webapp/lib/store.tsx b/packages/webui/webapp/lib/store.tsx index 58dc4137..c8c20a9a 100644 --- a/packages/webui/webapp/lib/store.tsx +++ b/packages/webui/webapp/lib/store.tsx @@ -67,6 +67,16 @@ export interface StoreSnapshot { * always passes the guard. */ stateRevision: number; + /** + * Slice 06 — Agent Team. Bumped every time the server emits a + * `session-tree-changed` SSE frame. Consumers (the sidebar session + * tree, the agent-team panel) listen for the bump and re-fetch + * `GET /api/session-tree` to pick up newly-spawned subagent rows. + * Same shape as `providersRevision`: a counter is enough to trigger + * an effect; re-fetching through the typed API client keeps the + * response handling consistent across the app. + */ + treeRevision: number; } const INITIAL: StoreSnapshot = { @@ -80,6 +90,7 @@ const INITIAL: StoreSnapshot = { quotaError: null, providersRevision: 0, stateRevision: -1, + treeRevision: 0, }; let snapshot: StoreSnapshot = INITIAL; @@ -136,6 +147,12 @@ export function __testApplyAction(action: SseAction | { kind: "connected"; value case "providers-updated": setSnapshot({ providersRevision: snapshot.providersRevision + 1 }); return snapshot; + case "tree-changed": + // Slice 06: bump the revision so the sidebar session tree refetches. + // The masked payload carried by the SSE frame is NOT stored — the + // consumers re-read through the typed API client. + setSnapshot({ treeRevision: snapshot.treeRevision + 1 }); + return snapshot; case "malformed": setSnapshot({ error: `malformed ${action.event || "message"} frame` }); return snapshot; @@ -234,6 +251,12 @@ export function connect(): () => void { // across the app (and lets us drop a frame-shaped buffer). setSnapshot({ providersRevision: snapshot.providersRevision + 1 }); break; + case "tree-changed": + // Slice 06: bump the revision so the sidebar session tree + // re-fetches. The server fires this on every subagent row + // insertion (applyToolUpdate → recordSubagentForCid). + setSnapshot({ treeRevision: snapshot.treeRevision + 1 }); + break; case "malformed": setSnapshot({ error: `malformed ${action.event || "message"} frame` }); break; diff --git a/packages/webui/webapp/lib/transcript.ts b/packages/webui/webapp/lib/transcript.ts index 909d16b6..b4f489d4 100644 --- a/packages/webui/webapp/lib/transcript.ts +++ b/packages/webui/webapp/lib/transcript.ts @@ -48,6 +48,16 @@ export interface TranscriptBlock { toolOutput?: string[]; /** Tool blocks only: local paths the tool touched (the server's `@ path` lines). */ toolPaths?: string[]; + /** + * Tool blocks only: the runtime tool call id attached by the decoder + * when it sees a `##tc:` marker line immediately before the + * `→ name` header. The ToolCard uses this to look up the precise + * `recentSubagents[]` entry for THIS dispatch (matching by tool + * NAME alone would badge every `→ task` line with the newest + * child). Optional for older sessions whose chat predates the + * marker; the renderer falls back to the newest entry in that case. + */ + toolCallId?: string; /** * Assistant blocks only: total turn wall-clock duration in milliseconds, attached * by `decodeTranscript` when it encounters a `§§ processed_duration=Nms` marker @@ -96,6 +106,22 @@ const TODO_LINE = /^([✓✔○◌◯✗✘×])\s+(.+)$/; * the block stream. */ const TURN_PROCESS_LINE = /^§§\s+processed_duration=(\d+)(ms)?$/; +/** + * Slice 06 (Agent Team): server-written toolCallId marker. + * + * `server/lib/mcode-acp.js#applyToolUpdate` (and the tool_call branch of + * the stream callback) writes `##tc:` as a separate chat + * line immediately BEFORE the `→ name` header. The decoder consumes it + * and attaches the id to the following tool block, so the ToolCard can + * match the block against `recentSubagents[]` by id (NOT by tool name) + * — a session with multiple subagent dispatches would otherwise badge + * every `→ task` line with the newest child, which is wrong. + * + * Older sessions written before this marker shipped simply lack it; + * `ToolCard` falls back to the newest `recentSubagents` entry when + * `toolCallId` is missing. + */ +const TOOL_CALL_ID_LINE = /^##tc:(\S+)$/; /** Server text that is really a system notice, even under a todo glyph. */ const SYSTEM_NOTICE = /^(?:\[(?:error|warning|info|system)\]\s*)|(?:Questionnaire|requires.*(?:user input|interactive))/i; @@ -152,6 +178,13 @@ function collectContinuation( export function decodeTranscript(lines: readonly TranscriptLine[]): TranscriptBlock[] { const blocks: TranscriptBlock[] = []; let current: TranscriptBlock | null = null; + // Slice 06 (Agent Team): carry the most recently seen `##tc:` + // marker until the next tool block picks it up. The marker is + // emitted by the server on every `→ name` line whose `toolCallId` + // is known — without it the ToolCard cannot correlate a tool block + // with its `recentSubagents[]` entry by id (matching by tool name + // alone would badge every `→ task` line with the newest child). + let pendingToolCallId: string | undefined; const flush = () => { if (current) { @@ -211,6 +244,21 @@ export function decodeTranscript(lines: readonly TranscriptLine[]): TranscriptBl continue; } + // --- slice 06 toolCallId marker (`##tc:`). + // + // Consumed (it never appears in the rendered chat body) and the id + // is parked into `pendingToolCallId` until the next tool block + // opens — that block picks it up and exposes it as `toolCallId`. + // Older sessions whose chat was written before this marker shipped + // simply lack the line; `pendingToolCallId` stays undefined and + // the ToolCard falls back to the newest `recentSubagents` entry. + const tcMarker = TOOL_CALL_ID_LINE.exec(line); + if (tcMarker) { + pendingToolCallId = tcMarker[1]; + i += 1; + continue; + } + // --- block-style sections, driven by the state snapshot as well as the text if (PLAN_HEADING.test(trimmed)) { flush(); @@ -294,7 +342,13 @@ export function decodeTranscript(lines: readonly TranscriptLine[]): TranscriptBl toolArgs: (tool[2] ?? "").trim(), toolOutput: [], toolPaths: [], + // Slice 06: attach the parked toolCallId marker (cleared so the + // next tool block starts fresh — a marker that never picked up + // its block on the way through a malformed transcript is + // intentionally not retained). + ...(pendingToolCallId ? { toolCallId: pendingToolCallId } : {}), }; + pendingToolCallId = undefined; let j = i + 1; while (j < lines.length) { const body = lines[j]; diff --git a/packages/webui/webapp/lib/types.ts b/packages/webui/webapp/lib/types.ts index d71b8371..05cd0f87 100644 --- a/packages/webui/webapp/lib/types.ts +++ b/packages/webui/webapp/lib/types.ts @@ -168,6 +168,22 @@ export interface WebuiState { ask: AskState; plan: PlanState; running: RunningState; + /** + * Subagent references the current session has spawned (slice 06). + * The chat renderer reads this to attach a jumpable child-session + * link to each parent's `→ task` tool line and to render the live + * running badge; the sidebar tree always re-fetches from the runtime + * db rather than reading this array. + * + * `status` is the UI vocabulary (`running`/`done`/`failed`/`stopped`), + * never a raw db string — see `lib/agent-team-status.ts` on the + * server. The polling cadence (2s default, override via + * `MCODE_WEBUI_SUBAGENT_POLL_MS`) keeps this list fresh while a + * subagent is mid-turn. + * + * Optional in the type so legacy snapshots and fixtures stay legal. + */ + recentSubagents?: RecentSubagent[]; /** * Slash-command catalogue reported by mcode over ACP. The server's wire shape * is a dict of command groups, e.g. `{ mcode: [{ name, description }, ...] }`, @@ -197,6 +213,19 @@ export interface WebuiState { [key: string]: unknown; } +/** + * One subagent the parent session has spawned. See `WebuiState.recentSubagents`. + */ +export interface RecentSubagent { + toolCallId: string; + sessionId: string; + agentName: string | null; + /** UI vocabulary only — `running`/`done`/`failed`/`stopped`. */ + status: string | null; + createdAtMs?: number; + updatedAtMs?: number; +} + /** Named SSE events the server emits alongside the state snapshots. */ export interface AuthorizeRequest { requestId: string; diff --git a/packages/webui/webapp/test/agent-team-i18n.test.ts b/packages/webui/webapp/test/agent-team-i18n.test.ts new file mode 100644 index 00000000..b8b0e9f0 --- /dev/null +++ b/packages/webui/webapp/test/agent-team-i18n.test.ts @@ -0,0 +1,178 @@ +// webapp/test/agent-team-i18n.test.ts +// Unit tests for lib/i18n-agent-team.ts — the bilingual strings the +// Agent Team panel relies on. +// +// Coverage contract (every line is a documented acceptance fix): +// +// 1. The slice-06 i18n file MUST NOT be orphaned — every key added +// to en MUST also exist in zh, and vice versa. The acceptance +// pass returned "PASS-WITH-CONCERNS" specifically because the +// previous slice shipped this file with hardcoded English +// strings in chat.tsx and zero consumers. These tests lock the +// keys, the resolution, and the locale symmetry so the next +// contributor cannot silently regress. +// +// 2. `tAgentTeam(locale, key)` returns the locale-specific string, +// falls back to en for unknown locales, and never throws on +// missing keys. +// +// 3. `badgeLabelAndGlyph` projects the UI status to a {label, glyph} +// pair; statuses outside the UI vocabulary return null (the +// badge does not render, the ToolCard hides itself). +// +// 4. `agentLabel` maps the runtime-stored English token to the +// locale-resolved label; unknown tokens fall through to the +// fallback, never throw. + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; + +import { + tAgentTeam, + badgeLabelAndGlyph, + agentLabel, + AGENT_TEAM_STRINGS, +} from "../lib/i18n-agent-team"; + +describe("i18n-agent-team — bilingual symmetry (no orphan keys)", () => { + test("every key in en is also in zh", () => { + const enKeys = Object.keys(AGENT_TEAM_STRINGS.en); + const zhKeys = new Set(Object.keys(AGENT_TEAM_STRINGS.zh)); + for (const key of enKeys) { + assert.ok(zhKeys.has(key), `${key} missing in zh bucket`); + } + }); + + test("every key in zh is also in en (reverse direction)", () => { + const zhKeys = Object.keys(AGENT_TEAM_STRINGS.zh); + const enKeys = new Set(Object.keys(AGENT_TEAM_STRINGS.en)); + for (const key of zhKeys) { + assert.ok(enKeys.has(key), `${key} missing in en bucket`); + } + }); + + test("every localised string is non-empty", () => { + for (const locale of ["en", "zh"] as const) { + for (const [key, value] of Object.entries(AGENT_TEAM_STRINGS[locale])) { + assert.ok(typeof value === "string" && value.length > 0, `${locale}.${key} must be non-empty`); + } + } + }); + + test("the status labels are actually localised (en !== zh for status keys)", () => { + // A regression where someone adds a key in only one locale would + // ship English text to a Chinese-locale user. The exact text will + // drift, but the two locales MUST NOT agree on every status label. + const statusKeys = [ + "agentTeam.statusLabel.running", + "agentTeam.statusLabel.done", + "agentTeam.statusLabel.failed", + "agentTeam.statusLabel.stopped", + "agentTeam.statusLabel.queued", + ]; + for (const key of statusKeys) { + assert.notEqual( + AGENT_TEAM_STRINGS.en[key as keyof typeof AGENT_TEAM_STRINGS.en], + AGENT_TEAM_STRINGS.zh[key as keyof typeof AGENT_TEAM_STRINGS.zh], + `${key} must differ between en and zh`, + ); + } + }); + + test("the known agent labels are actually localised", () => { + const agentKeys = [ + "agentTeam.agent.explore", + "agentTeam.agent.worker", + "agentTeam.agent.verifier", + "agentTeam.agent.coder", + ]; + for (const key of agentKeys) { + assert.notEqual( + AGENT_TEAM_STRINGS.en[key as keyof typeof AGENT_TEAM_STRINGS.en], + AGENT_TEAM_STRINGS.zh[key as keyof typeof AGENT_TEAM_STRINGS.zh], + `${key} must differ between en and zh`, + ); + } + }); +}); + +describe("tAgentTeam — locale resolution", () => { + test("resolves to the requested locale", () => { + assert.equal(tAgentTeam("zh", "agentTeam.statusLabel.running"), "\u8FD0\u884C\u4E2D"); + assert.equal(tAgentTeam("en", "agentTeam.statusLabel.running"), "Running"); + }); + + test("falls back to en for unknown locales (defensive)", () => { + // The webui only ships zh / en today, but a future locale switcher + // could pass through an unknown value. Resolve defensively. + assert.equal( + tAgentTeam("fr" as unknown as "zh" | "en", "agentTeam.statusLabel.running"), + "Running", + ); + }); + + test("falls back to the raw key when the bucket is missing the entry (debug visibility)", () => { + // A new key added to en but missed in zh would otherwise render an + // empty badge in Chinese — show the key name instead, so a + // regression is loud in the UI rather than silently empty. + assert.equal( + tAgentTeam("zh", "agentTeam.not.a.real.key" as unknown as never), + "agentTeam.not.a.real.key", + ); + }); +}); + +describe("badgeLabelAndGlyph — projects UI status to {label, glyph}", () => { + test("returns a {label, glyph} pair for every UI status the badge renders", () => { + for (const status of ["running", "done", "failed", "stopped", "idle", "queued"]) { + const en = badgeLabelAndGlyph("en", status); + const zh = badgeLabelAndGlyph("zh", status); + assert.ok(en, `en missing for status=${status}`); + assert.ok(zh, `zh missing for status=${status}`); + assert.ok(en.label.length > 0); + assert.ok(en.glyph.length > 0); + assert.notEqual(en.label, zh.label, `en/zh labels must differ for ${status}`); + } + }); + + test("returns null for null / undefined / unknown status (no badge rendered)", () => { + assert.equal(badgeLabelAndGlyph("en", null), null); + assert.equal(badgeLabelAndGlyph("en", undefined), null); + assert.equal(badgeLabelAndGlyph("en", "queued-but-not-in-vocab"), null); + // Raw db strings must NEVER leak through — the badge contract is + // UI vocabulary only. + assert.equal(badgeLabelAndGlyph("en", "succeeded"), null); + assert.equal(badgeLabelAndGlyph("en", "canceled"), null); + }); + + test("en and zh return DIFFERENT labels (locale actually does something)", () => { + const en = badgeLabelAndGlyph("en", "running"); + const zh = badgeLabelAndGlyph("zh", "running"); + assert.ok(en && zh); + assert.notEqual(en.label, zh.label); + }); +}); + +describe("agentLabel — runtime-stored agent token → locale label", () => { + test("known tokens resolve to the locale-specific label", () => { + assert.equal(agentLabel("en", "explore"), "Explore"); + assert.equal(agentLabel("zh", "explore"), "\u63A2\u67E5\u8005"); + assert.equal(agentLabel("en", "verifier"), "Verifier"); + assert.equal(agentLabel("zh", "verifier"), "\u9A8C\u8BC1\u8005"); + }); + + test("unknown tokens fall back to the English token verbatim (forward-compat)", () => { + // A future custom agent the frontend has not been told about + // must still render a readable label rather than fall through to + // the subagent fallback (which would be misleading). + assert.equal(agentLabel("en", "future_agent"), "future_agent"); + assert.equal(agentLabel("zh", "future_agent"), "future_agent"); + }); + + test("null / empty input returns the subagent fallback label (locale-specific)", () => { + assert.equal(agentLabel("en", null), "subagent"); + assert.equal(agentLabel("zh", null), "\u5B50 agent"); + assert.equal(agentLabel("en", ""), "subagent"); + assert.equal(agentLabel("zh", undefined), "\u5B50 agent"); + }); +}); diff --git a/packages/webui/webapp/test/agent-team-lookup.test.ts b/packages/webui/webapp/test/agent-team-lookup.test.ts new file mode 100644 index 00000000..69509e63 --- /dev/null +++ b/packages/webui/webapp/test/agent-team-lookup.test.ts @@ -0,0 +1,158 @@ +// webapp/test/agent-team-lookup.test.ts +// Unit tests for lib/agent-team-lookup.ts — the pure rule that picks the +// right `recentSubagents[]` entry for a tool block. +// +// Coverage contract (every line is a documented acceptance fix): +// +// 1. Two `→ task` lines in the same session must each jump to their +// own child. The bug the previous slice shipped: every block took +// `recent[last]`, so older tool lines jumped to the newest child. +// The fix is to match by `toolCallId` (carried on the block by +// the `##tc:` marker consumed by the decoder). +// +// 2. Non-subagent tool names (read / bash / write) return null even +// when `recent` is non-empty. +// +// 3. Without a toolCallId (older chat, or a mid-stream attach race +// where the marker landed before the runtime wrote the row), +// fall back to the newest entry — pragmatic for legacy sessions. +// +// 4. An empty / nullish recent array returns null. A non-array +// recent (defensive) returns null too. + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; + +import { findSubagentForBlock, type ToolBlockLike } from "../lib/agent-team-lookup"; +import type { RecentSubagent } from "../lib/types"; + +function recent(): RecentSubagent[] { + return [ + { + toolCallId: "tc_a", + sessionId: "mvs_child_a", + agentName: "explore", + status: "done", + createdAtMs: 1000, + updatedAtMs: 1000, + }, + { + toolCallId: "tc_b", + sessionId: "mvs_child_b", + agentName: "verifier", + status: "running", + createdAtMs: 2000, + updatedAtMs: 2000, + }, + { + toolCallId: "tc_c", + sessionId: "mvs_child_c", + agentName: "worker", + status: "done", + createdAtMs: 3000, + updatedAtMs: 3000, + }, + ]; +} + +function block(opts: ToolBlockLike): ToolBlockLike { + return opts; +} + +describe("findSubagentForBlock — the multi-dispatch correctness fix", () => { + test("two → task lines each jump to their own child (the bug this fixes)", () => { + // Reproduces the exact shape the agent team panel hits on every + // multi-dispatch session. Each block asks "which subagent am I?" + // and gets its OWN entry, not the newest one. + const list = recent(); + const first = findSubagentForBlock(list, block({ toolName: "task", toolCallId: "tc_a" })); + const second = findSubagentForBlock(list, block({ toolName: "task", toolCallId: "tc_b" })); + const third = findSubagentForBlock(list, block({ toolName: "task", toolCallId: "tc_c" })); + assert.equal(first?.sessionId, "mvs_child_a"); + assert.equal(second?.sessionId, "mvs_child_b"); + assert.equal(third?.sessionId, "mvs_child_c"); + }); + + test("toolCallId matching wins over the newest entry (the regression guard)", () => { + // The pre-fix behavior took `recent[last]` regardless of which + // block was rendered. Without this test, a refactor that drops + // the toolCallId gate would silently regress — every older tool + // line would jump to the newest child. + const list = recent(); + const out = findSubagentForBlock(list, block({ toolName: "task", toolCallId: "tc_a" })); + assert.notEqual(out?.sessionId, "mvs_child_c", "must NOT take newest entry"); + }); + + test("accepts the legacy 'delegate' / 'delegatetask' tool names too", () => { + const list = recent(); + assert.equal( + findSubagentForBlock(list, block({ toolName: "delegate", toolCallId: "tc_a" }))?.sessionId, + "mvs_child_a", + ); + assert.equal( + findSubagentForBlock(list, block({ toolName: "delegatetask", toolCallId: "tc_a" }))?.sessionId, + "mvs_child_a", + ); + }); + + test("matches case-insensitively against the legacy name variants", () => { + const list = recent(); + assert.equal( + findSubagentForBlock(list, block({ toolName: "TASK", toolCallId: "tc_a" }))?.sessionId, + "mvs_child_a", + ); + }); + + test("ignores non-subagent tool names even when recent has entries", () => { + const list = recent(); + assert.equal(findSubagentForBlock(list, block({ toolName: "read", toolCallId: "tc_a" })), null); + assert.equal(findSubagentForBlock(list, block({ toolName: "bash", toolCallId: "tc_a" })), null); + assert.equal(findSubagentForBlock(list, block({ toolName: "write", toolCallId: "tc_a" })), null); + }); + + test("ignores non-subagent tool names without a toolCallId too", () => { + const list = recent(); + assert.equal(findSubagentForBlock(list, block({ toolName: "read" })), null); + }); + + test("empty / nullish recent array returns null without throwing", () => { + assert.equal(findSubagentForBlock([], block({ toolName: "task", toolCallId: "tc" })), null); + assert.equal(findSubagentForBlock(undefined, block({ toolName: "task", toolCallId: "tc" })), null); + assert.equal(findSubagentForBlock(null as unknown as readonly RecentSubagent[], block({ toolName: "task", toolCallId: "tc" })), null); + }); + + test("missing toolCallId falls back to the newest entry (legacy chat compat)", () => { + // Older chat predates the `##tc:` marker. The renderer must still + // show SOME badge rather than drop the row, so the newest entry + // is the documented fallback. This is the only path that lets the + // pre-marker chat history render a usable badge at all. + const list = recent(); + assert.equal( + findSubagentForBlock(list, block({ toolName: "task" }))?.sessionId, + "mvs_child_c", + ); + }); + + test("toolCallId present but no matching entry falls back to newest (mid-stream race)", () => { + // The runtime sometimes writes the `##tc:` marker to chat BEFORE + // it commits the `local_runtime_background_tasks` row, so the + // marker has arrived but the polling-based refresh hasn't seen the + // entry yet. Render the newest entry rather than nothing — the + // user can still click; if the match eventually lands the badge + // would change on the next tick. Pin this so a future change + // cannot silently drop the badge. + const list = recent(); + assert.equal( + findSubagentForBlock(list, block({ toolName: "task", toolCallId: "tc_unknown" }))?.sessionId, + "mvs_child_c", + ); + }); + + test("empty toolCallId string is treated as 'missing' (fall back to newest)", () => { + const list = recent(); + assert.equal( + findSubagentForBlock(list, block({ toolName: "task", toolCallId: "" }))?.sessionId, + "mvs_child_c", + ); + }); +}); diff --git a/packages/webui/webapp/test/transcript-tc-marker.test.ts b/packages/webui/webapp/test/transcript-tc-marker.test.ts new file mode 100644 index 00000000..0c812b7d --- /dev/null +++ b/packages/webui/webapp/test/transcript-tc-marker.test.ts @@ -0,0 +1,136 @@ +// webapp/test/transcript-tc-marker.test.ts +// Regression for slice 06's `##tc:` marker consumption by the +// transcript decoder. +// +// What this locks. The decoder reads chat lines and produces +// `TranscriptBlock` values for the renderer. The `toolCallId` field +// on a tool block is the only way the ToolCard can match the block +// with the right `recentSubagents[]` entry — without it, every +// `→ task` line in a multi-dispatch session would badge the newest +// child. The marker is consumed (it never appears in the chat body) +// and attached to the next tool block. +// +// Coverage contract: +// 1. `##tc:` immediately before a `→ name` header attaches the +// id to that block. +// 2. The marker is NOT emitted as a block on its own — it's a +// metadata line, consumed by the decoder. +// 3. Multiple dispatches in the same chat each carry their own id +// — proves the multi-dispatch correctness fix end-to-end on the +// decoder side. +// 4. Older chat that lacks the marker still parses cleanly; the +// block has no `toolCallId` and the lookup falls back. + +import { test, describe } from "node:test"; +import assert from "node:assert/strict"; + +import { decodeTranscript, type TranscriptBlock } from "../lib/transcript"; + +function toolBlocks(blocks: TranscriptBlock[]) { + return blocks.filter((b) => b.role === "tool"); +} + +/** Pick the first tool block and assert it exists. */ +function firstTool(blocks: TranscriptBlock[]): TranscriptBlock { + const tools = toolBlocks(blocks); + if (tools.length !== 1) { + throw new Error(`expected exactly one tool block, got ${tools.length}`); + } + return tools[0] as TranscriptBlock; +} + +describe("decodeTranscript — consumes ##tc: marker (slice 06)", () => { + test("attaches toolCallId to the next tool block", () => { + const blocks = decodeTranscript([ + "##tc:abc", + "→ task { \"agent\": \"explore\" }", + " [completed]", + ]); + const tool = firstTool(blocks); + assert.equal(tool.toolCallId, "abc"); + assert.equal(tool.toolName, "task"); + }); + + test("the marker itself is NOT emitted as a block (consumed metadata)", () => { + // A regression that emitted the marker as its own block would + // pollute the chat with `##tc:abc` rows. The decoder MUST + // consume it before the next block opens. + const blocks = decodeTranscript([ + "##tc:abc", + "→ task { }", + " [completed]", + ]); + assert.equal(blocks.length, 1, "the marker must not appear as its own block"); + assert.equal(blocks[0]?.role, "tool"); + }); + + test("two dispatches in one session each carry their own toolCallId", () => { + // The exact shape the multi-dispatch correctness fix is about. + // A session that spawned two subagents has TWO recentSubagents + // entries; without toolCallId matching, every `→ task` line + // would badge the newest. The decoder proves both blocks are + // distinct, and the lookup test (agent-team-lookup.test.ts) + // pins the matching contract. + const blocks = decodeTranscript([ + "##tc:a", + "→ task { \"agent\": \"explore\" }", + " [completed]", + "##tc:b", + "→ task { \"agent\": \"verifier\" }", + " [in_progress]", + ]); + const tools = toolBlocks(blocks); + assert.equal(tools.length, 2); + const a = tools[0]; + const b = tools[1]; + assert.ok(a && b, "test fixture expects exactly two tool blocks"); + assert.equal(a.toolCallId, "a"); + assert.equal(b.toolCallId, "b"); + // Sanity: the renderer can tell them apart. If both came out the + // same the lookup test would fail too — this is the decoder-side + // half of the same property. + assert.notEqual(a.toolCallId, b.toolCallId); + }); + + test("a marker without a following tool block is dropped, not stored", () => { + // Defensive: a stray marker (engine bug, or mid-stream attach + // race) must not pollute the block stream. The decoder parks + // it but never sees a tool block to attach to, so the next + // marker would replace it. + const blocks = decodeTranscript([ + "##tc:orphan", + "● some other block", + "##tc:abc", + "→ task { }", + ]); + const tool = firstTool(blocks); + assert.equal(tool.toolCallId, "abc"); + }); + + test("older chat without the marker still parses cleanly", () => { + // The marker is additive — older sessions (and tests written + // before slice 06) do not have it, and the lookup falls back + // to the newest entry. + const blocks = decodeTranscript([ + "→ task { }", + " [completed]", + ]); + const tool = firstTool(blocks); + assert.equal(tool.toolCallId, undefined); + assert.equal(tool.toolName, "task"); + }); + + test("a blank line between the marker and the header still attaches", () => { + // The decoder skips blank lines without consuming them as + // blocks, but the marker carries across — the parked id is + // only reset by the next tool block opening. + const blocks = decodeTranscript([ + "##tc:abc", + "", + "→ task { }", + " [completed]", + ]); + const tool = firstTool(blocks); + assert.equal(tool.toolCallId, "abc"); + }); +}); diff --git a/release/public-source.json b/release/public-source.json index 05cfada4..f99729bd 100644 --- a/release/public-source.json +++ b/release/public-source.json @@ -3378,6 +3378,9 @@ "packages/webui/server/bootstrap.js", "packages/webui/server/cleanup.js", "packages/webui/server/lib/acp-client.js", + "packages/webui/server/lib/agent-team-detect.js", + "packages/webui/server/lib/agent-team-status.js", + "packages/webui/server/lib/agent-team-tasks.js", "packages/webui/server/lib/alerts.js", "packages/webui/server/lib/attachments.js", "packages/webui/server/lib/auth.js", @@ -3479,6 +3482,10 @@ "packages/webui/test/integration/upload-limits.test.js", "packages/webui/test/lib/acp-cache.check.mjs", "packages/webui/test/lib/acp-transport-answer.test.js", + "packages/webui/test/lib/agent-team-detect.test.js", + "packages/webui/test/lib/agent-team-state-bus.test.js", + "packages/webui/test/lib/agent-team-status.test.js", + "packages/webui/test/lib/agent-team-tasks.test.js", "packages/webui/test/lib/alerts.check.mjs", "packages/webui/test/lib/auth.test.js", "packages/webui/test/lib/authorize.check.mjs", @@ -3501,6 +3508,7 @@ "packages/webui/test/lib/mavis-usage.check.mjs", "packages/webui/test/lib/mcode-acp-note.test.js", "packages/webui/test/lib/mcode-acp-ownership.check.mjs", + "packages/webui/test/lib/mcode-acp-tc-marker.test.js", "packages/webui/test/lib/mcode-exec.test.js", "packages/webui/test/lib/mcode-rpc-live-lookup.test.js", "packages/webui/test/lib/mcode-rpc.check.mjs", @@ -3607,6 +3615,7 @@ "packages/webui/webapp/components/toolbar.tsx", "packages/webui/webapp/components/workspace-picker.tsx", "packages/webui/webapp/lib/action-errors.ts", + "packages/webui/webapp/lib/agent-team-lookup.ts", "packages/webui/webapp/lib/alerts.ts", "packages/webui/webapp/lib/antd-theme.ts", "packages/webui/webapp/lib/api.ts", @@ -3614,6 +3623,7 @@ "packages/webui/webapp/lib/composer-draft.ts", "packages/webui/webapp/lib/file-preview.ts", "packages/webui/webapp/lib/files-tree.ts", + "packages/webui/webapp/lib/i18n-agent-team.ts", "packages/webui/webapp/lib/i18n.ts", "packages/webui/webapp/lib/markdown.ts", "packages/webui/webapp/lib/provider-management.ts", @@ -3635,6 +3645,8 @@ "packages/webui/webapp/styles/official-utilities.css", "packages/webui/webapp/styles/tokens.css", "packages/webui/webapp/tailwind.config.mjs", + "packages/webui/webapp/test/agent-team-i18n.test.ts", + "packages/webui/webapp/test/agent-team-lookup.test.ts", "packages/webui/webapp/test/alerts.test.ts", "packages/webui/webapp/test/api-permissions.test.ts", "packages/webui/webapp/test/attachment-drop.test.ts", @@ -3653,6 +3665,7 @@ "packages/webui/webapp/test/sse.test.ts", "packages/webui/webapp/test/store-revision.test.ts", "packages/webui/webapp/test/transcript-roundtrip.test.ts", + "packages/webui/webapp/test/transcript-tc-marker.test.ts", "packages/webui/webapp/test/transcript.test.ts", "packages/webui/webapp/test/workspace-chip.test.ts", "packages/webui/webapp/test/workspace-filter.test.ts",