diff --git a/apps/desktop/electron/main/models-dev-catalog.ts b/apps/desktop/electron/main/models-dev-catalog.ts index fbf42f28c..a8a6b310f 100644 --- a/apps/desktop/electron/main/models-dev-catalog.ts +++ b/apps/desktop/electron/main/models-dev-catalog.ts @@ -32,7 +32,7 @@ import type { ThinkingProtocol, ModelBinding, } from "@pi-desktop/shared"; -import { genericModelConfig, modelConfigWithBinding, type ModelConfig } from "@pi-desktop/agent-runtime"; +import { genericModelConfig, modelConfigWithBinding, transcriptConfigFromPi, type ModelConfig } from "@pi-desktop/agent-runtime"; import { resolveBindingLimits } from "@pi-desktop/shared"; export const MODELS_DEV_API_URL = "https://models.dev/api.json"; @@ -1424,6 +1424,16 @@ export class ModelsDevCatalog { const thinking = this.anthropicThinkingFor(input.modelId); if (thinking) baseline = { ...baseline, ...thinking }; } + // Pi owns wire capabilities, independently of models.dev's limits/prices. + // Use the original published record, never an account/relay projection. + const key = input.vendorKey?.trim().toLowerCase(); + const vendor = (key ? PI_VENDOR_ALIASES[key] ?? key : undefined) ?? this.providerKeyForRow(input); + const transport = vendor ? this.operationModels.getModel(vendor, input.modelId) : undefined; + if (transport) { + const transcript = transcriptConfigFromPi(transport); + baseline = { ...baseline, transcriptBinding: transcript.transcriptBinding, + compat: { ...baseline.compat, ...transcript.compat } }; + } if (!binding) return baseline; const limits = resolveBindingLimits(baseline, binding); return modelConfigWithBinding(limits.catalogConfig, limits.binding); diff --git a/apps/desktop/src/lib/assistant-turns.ts b/apps/desktop/src/lib/assistant-turns.ts index ec80ddcda..e47a06489 100644 --- a/apps/desktop/src/lib/assistant-turns.ts +++ b/apps/desktop/src/lib/assistant-turns.ts @@ -68,6 +68,7 @@ export function messageThinking(message: UiMessage): string { } function isVisibleMessage(message: UiMessage): boolean { + if (message.role === "system" && message.modelSystem) return false; return !( message.role === "assistant" && !(message.content || "").trim() && diff --git a/apps/desktop/src/lib/transcript-projection.ts b/apps/desktop/src/lib/transcript-projection.ts index 093bdb668..e50ffefc2 100644 --- a/apps/desktop/src/lib/transcript-projection.ts +++ b/apps/desktop/src/lib/transcript-projection.ts @@ -57,7 +57,8 @@ function shape(message: UiMessage): MessageShape { const value = { content, thinking, - visible: message.role !== "assistant" || content || thinking || Boolean(message.hostedSearch || message.error), + visible: !(message.role === "system" && message.modelSystem) && + (message.role !== "assistant" || content || thinking || Boolean(message.hostedSearch || message.error)), answer: message.parentToolCallId ? content || Boolean(message.error) : content || !thinking || Boolean(message.error), @@ -72,7 +73,7 @@ function canReplace(previous: UiMessage, next: UiMessage): boolean { previous.parentToolCallId !== next.parentToolCallId || previous.agentName !== next.agentName || previous.createdAt !== next.createdAt || previous.toolCallId !== next.toolCallId || previous.toolName !== next.toolName || - previous.hostedSearch !== next.hostedSearch + previous.hostedSearch !== next.hostedSearch || Boolean(previous.modelSystem) !== Boolean(next.modelSystem) ) return false; if (isDelegationStartTool(previous.toolName) && ( previous.toolArgs !== next.toolArgs || previous.toolResult !== next.toolResult diff --git a/apps/desktop/test/context-compaction.test.mjs b/apps/desktop/test/context-compaction.test.mjs index 0a61bf63f..1dedb2a28 100644 --- a/apps/desktop/test/context-compaction.test.mjs +++ b/apps/desktop/test/context-compaction.test.mjs @@ -123,12 +123,11 @@ test("the hard boundary is enforced by the host, with a model-side escape hatch" assert.match(runtime, /function contextFallbackReminder\(/); assert.match(runtime, /contextReminderClaimed/); assert.match(runtime, /contextFallbackReminderClaimed/); - // pi 0.86 carries the canonical system prompt in the transcript. The - // reminder is appended as a per-turn system message, not only as the legacy - // AgentContext.systemPrompt field. + // The reminder is appended as a replaceable system section so compaction + // can expire it without rewriting earlier instructions. assert.match( runtime, - /messages: \[\s*\.\.\.context\.messages,\s*\{\s*role: "system",\s*content: reminder,/, + /messages: \[\s*\.\.\.context\.messages,\s*\{\s*role: "system",\s*content: "",\s*sections: \{ \[CONTEXT_BUDGET_SECTION\]: reminder \}/, ); assert.match(hostPermissions, /"new_context"/); }); diff --git a/apps/desktop/test/models-dev-catalog.test.mjs b/apps/desktop/test/models-dev-catalog.test.mjs index cdfc969a1..7d79bf3e8 100644 --- a/apps/desktop/test/models-dev-catalog.test.mjs +++ b/apps/desktop/test/models-dev-catalog.test.mjs @@ -4,11 +4,15 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import test from "node:test"; import { fileURLToPath } from "node:url"; +import { createModels, InMemoryModelsStore } from "@earendil-works/pi-ai"; +import { builtinProviders } from "@earendil-works/pi-ai/providers/all"; +import { buildProviderModel } from "../../../packages/agent-runtime/dist/provider-binding.js"; import { apiStyleForAdapter, bindingForCustomModelInfo, catalogModelIdsMatch, modelIdsMatch, + resolveApiStyle, } from "@pi-desktop/shared"; import { MODELS_DEV_API_URL, @@ -103,6 +107,43 @@ async function loadFixtureCatalog(t, fixture = catalogFixture) { return catalog; } +test("published metadata retains exact Pi transcript transport bindings through runtime launch", async (t) => { + const pi = createModels({ modelsStore: new InMemoryModelsStore(), + authContext: { env: async () => undefined, fileExists: async () => false } }); + for (const provider of builtinProviders()) pi.setProvider(provider); + for (const [vendorKey, modelId] of [ + ["deepseek", "deepseek-flash"], ["anthropic", "claude-opus-5-5"], + ["openai", "gpt-6.1-sol"], ["openai-codex", "gpt-6.1-sol"], + ]) { + const original = pi.getModel(vendorKey, modelId); + assert.ok(original, `${vendorKey}/${modelId}`); + const catalog = await loadFixtureCatalog(t, { [vendorKey]: { + name: vendorKey, api: original.baseUrl, models: { [modelId]: { + id: modelId, limit: { context: 543_210, output: 40_000 }, cost: { input: 1.25, output: 2.5 }, + } }, + } }); + const target = { providerId: "account", vendorKey, modelId, baseUrl: original.baseUrl, + apiStyle: resolveApiStyle(original.api) }; + const config = catalog.modelConfigFor(target); + assert.equal(config.source, "models.dev"); + assert.equal(config.contextWindow, 543_210); + assert.equal(config.maxTokens, 40_000); + assert.equal(config.cost.input, 1.25); + assert.deepEqual(config.transcriptBinding, { modelId, api: original.api, baseUrl: original.baseUrl }); + const provider = { ...target, id: "account", name: "Fixture", apiKey: "", authKind: "none", + supportsReasoning: false, supportedThinkingLevels: ["off"], modelConfig: config }; + const wire = buildProviderModel(provider); + assert.equal(wire.compat.supportsMidConvoSystemMessages, true, vendorKey); + assert.equal(wire.compat.supportsMidConvoToolChanges, original.compat?.supportsMidConvoToolChanges === true); + assert.equal(wire.compat.supportsAdditionalTools, original.compat?.supportsAdditionalTools === true); + const relay = { ...target, baseUrl: "https://relay.invalid/v1" }; + assert.equal(buildProviderModel({ ...provider, ...relay, modelConfig: catalog.modelConfigFor(relay) }) + .compat.supportsMidConvoSystemMessages, false); + assert.equal(buildProviderModel({ ...provider, modelId: `${modelId}-alias` }) + .compat.supportsMidConvoSystemMessages, false); + } +}); + test("the bundled models.dev OpenAI record supplies the selected model limits", async () => { const catalogPath = fileURLToPath(new URL("../resources/models.dev/api.json", import.meta.url)); const catalog = new ModelsDevCatalog({ catalogPath }); diff --git a/apps/desktop/test/transcript-projection.test.mjs b/apps/desktop/test/transcript-projection.test.mjs index 01a3de4ad..15b30035f 100644 --- a/apps/desktop/test/transcript-projection.test.mjs +++ b/apps/desktop/test/transcript-projection.test.mjs @@ -14,6 +14,19 @@ const message = (id, role, content, extra = {}) => ({ id, role, content, createdAt: "2026-09-30T00:00:00.000Z", ...extra, }); +test("model system records remain hidden after loading and live projection updates", () => { + const state = message("state", "system", "", { modelSystem: { + version: 1, beforeMessageId: "user", messageJson: JSON.stringify({ role: "system", content: "Instructions", timestamp: 1 }), + } }); + const user = message("user", "user", "Hello"); + const notice = message("notice", "system", "Visible notice"); + let rows = [state, user, notice]; + assert.deepEqual(assertProjection(rows).visible.map((row) => row.id), ["user", "notice"]); + rows = upsertLiveSessionMessage(rows, message("answer", "assistant", "Hello back")); + assert.deepEqual(assertProjection(rows).visible.map((row) => row.id), ["user", "notice", "answer"]); + assert.equal(rows[0], state); +}); + function assertProjection(messages, compactions) { const actual = getTranscriptProjection(messages, compactions); const expected = buildTranscriptEntries(messages, compactions); diff --git a/crates/host-core/src/plugin_sessions.rs b/crates/host-core/src/plugin_sessions.rs index 3dee8f6a3..bbc982407 100644 --- a/crates/host-core/src/plugin_sessions.rs +++ b/crates/host-core/src/plugin_sessions.rs @@ -260,6 +260,7 @@ fn parse_message( nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }) } diff --git a/crates/host-core/src/sessions.rs b/crates/host-core/src/sessions.rs index 2280766ed..b86166114 100644 --- a/crates/host-core/src/sessions.rs +++ b/crates/host-core/src/sessions.rs @@ -12,6 +12,7 @@ use std::sync::{Mutex, OnceLock}; use crate::transcripts::{self, CompactionRecord, MessageRecord, RevisionRecord}; mod fork_files; +mod model_system; mod usage; pub use usage::record_usage; @@ -242,6 +243,9 @@ pub struct UiMessage { /// as an additive `hostedSearch` transcript block; no SQL migration. #[serde(default, skip_serializing_if = "Option::is_none")] pub hosted_search: Option, + /// Internal model-context state, preserved outside visible message text. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_system: Option, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -321,6 +325,9 @@ fn is_default_title(title: &str) -> bool { /// the search index row (None for tool rows, matching the FTS triggers). pub(crate) fn ui_to_record(message: &UiMessage) -> (MessageRecord, Option) { let mut meta_obj = serde_json::Map::new(); + if let Some(system) = &message.model_system { + meta_obj.insert("modelSystem".into(), system.clone()); + } if let Some(command) = &message.command { meta_obj.insert("command".into(), json!(command)); } @@ -473,6 +480,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { _ => Vec::new(), }; let meta = record.meta.unwrap_or(Value::Null); + let model_system = meta.get("modelSystem").cloned(); let command = meta .get("command") .and_then(Value::as_str) @@ -646,6 +654,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { nested_parent_tool_call_id, agent_name, hosted_search: hosted_search.clone(), + model_system: None, } } else { let content = blocks @@ -690,6 +699,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { nested_parent_tool_call_id, agent_name, hosted_search, + model_system, } } } @@ -909,6 +919,19 @@ fn clone_records_for_fork( .cloned() .unwrap_or_else(|| Uuid::new_v4().to_string()); if let Some(meta) = record.meta.as_mut().and_then(Value::as_object_mut) { + if let Some(system) = meta.get_mut("modelSystem").and_then(Value::as_object_mut) { + for key in ["beforeMessageId", "afterMessageId"] { + if let Some(new_id) = system + .get(key) + .and_then(Value::as_str) + .and_then(|id| message_ids.get(id)) + { + system.insert(key.into(), json!(new_id)); + } else { + system.remove(key); + } + } + } meta.remove("revisionRootId"); meta.remove("revisionCount"); meta.remove("activeRevision"); @@ -1915,6 +1938,7 @@ pub fn append_message( message: &UiMessage, turn_id: Option<&str>, ) -> Result<()> { + model_system::validate(message)?; let message = crate::session_collaboration::prepare_append(db, session_id, message, turn_id)?; let session_created = ensure_session_for_append(db, session_id)?; let (mut record, text) = ui_to_record(&message); @@ -3970,6 +3994,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, } } @@ -4622,6 +4647,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }; append_message(&db, &session.id, &tool, None).unwrap(); @@ -5137,6 +5163,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }; append_message(&db, &session.id, &assistant, None).unwrap(); @@ -5216,6 +5243,7 @@ mod tests { parent_tool_call_id: None, nested_parent_tool_call_id: None, agent_name: None, + model_system: None, hosted_search: Some(json!({ "status": "completed", "rounds": [ diff --git a/crates/host-core/src/sessions/model_system.rs b/crates/host-core/src/sessions/model_system.rs new file mode 100644 index 000000000..88dd3140f --- /dev/null +++ b/crates/host-core/src/sessions/model_system.rs @@ -0,0 +1,108 @@ +use super::UiMessage; +use anyhow::{anyhow, Result}; +use serde_json::Value; + +/// Internal model state uses existing transcript metadata, never visible chat text. +pub(super) fn validate(row: &UiMessage) -> Result<()> { + let Some(record) = &row.model_system else { + return Ok(()); + }; + let invalid = || anyhow!("Invalid model system record"); + if row.role != "system" || !row.content.is_empty() || record["version"] != 1 { + return Err(invalid()); + } + for key in ["beforeMessageId", "afterMessageId"] { + if record + .get(key) + .is_some_and(|value| value.as_str().is_none_or(str::is_empty)) + { + return Err(invalid()); + } + } + let message: Value = serde_json::from_str(record["messageJson"].as_str().ok_or_else(invalid)?) + .map_err(|_| invalid())?; + if message["role"] != "system" + || !message["content"].is_string() + || !message["timestamp"] + .as_f64() + .is_some_and(|value| (0.0..=8.64e15).contains(&value)) + { + return Err(invalid()); + } + if let Some(sections) = message.get("sections") { + let sections = sections.as_object().ok_or_else(invalid)?; + if sections + .values() + .any(|value| !value.is_null() && !value.is_string()) + { + return Err(invalid()); + } + } + for key in ["toolsAdded", "toolsRemoved"] { + if let Some(tools) = message.get(key) { + for tool in tools.as_array().ok_or_else(invalid)? { + if tool + .get("name") + .and_then(Value::as_str) + .is_none_or(str::is_empty) + || (key == "toolsAdded" + && (!tool["description"].is_string() || !tool["parameters"].is_object())) + { + return Err(invalid()); + } + } + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + fn row() -> UiMessage { + serde_json::from_value(json!({ + "id":"state", "role":"system", "content":"", "createdAt":"2026-10-01T00:00:00Z", + "modelSystem":{"version":1,"beforeMessageId":"user","messageJson":json!({ + "role":"system","content":"","timestamp":1, + "sections":{"skills":"catalog"},"toolsAdded":[{"name":"Read","description":"read","parameters":{}}] + }).to_string()} + })).unwrap() + } + + #[test] + fn state_roundtrips_in_canonical_metadata_and_remaps_fork_anchors() { + let row = row(); + validate(&row).unwrap(); + let (record, text) = super::super::ui_to_record(&row); + assert!(text.as_deref().unwrap_or("").is_empty()); + assert_eq!( + super::super::record_to_ui(record.clone()).model_system, + row.model_system + ); + let mut user = row.clone(); + user.id = "user".into(); + user.role = "user".into(); + user.model_system = None; + let (records, ids, _) = + super::super::clone_records_for_fork(vec![record, super::super::ui_to_record(&user).0]); + assert_eq!( + records[0].meta.as_ref().unwrap()["modelSystem"]["beforeMessageId"], + ids["user"] + ); + } + + #[test] + fn rejects_invalid_state_or_hiding_a_user_message() { + let mut row = row(); + row.role = "user".into(); + assert!(validate(&row).is_err()); + row.role = "system".into(); + row.model_system.as_mut().unwrap()["version"] = json!(2); + assert!(validate(&row).is_err()); + row.model_system.as_mut().unwrap()["version"] = json!(1); + row.model_system.as_mut().unwrap()["messageJson"] = json!("invalid JSON"); + assert!(validate(&row).is_err()); + } +} diff --git a/docs/adr/0039-plugin-skills-activation-and-devkit.md b/docs/adr/0039-plugin-skills-activation-and-devkit.md index 9a619975f..ce80c5d36 100644 --- a/docs/adr/0039-plugin-skills-activation-and-devkit.md +++ b/docs/adr/0039-plugin-skills-activation-and-devkit.md @@ -39,11 +39,12 @@ giving them the loop — scaffold, run, inspect, package. instructions.** Later text carries more weight, so a user's own instruction files keep the last word and an installed plugin can refine the built-in guidance but never the reverse. -3. **Runtime reuse keys on the catalog digest, not on bodies.** Enabling a - plugin, revoking `agent.prompt.inject` or renaming a skill changes the text - the model reads, so it retires the idle runtime rather than reusing a stale - prompt. An edit to a body needs no retirement: the `Skill` tool reads the file - at call time. +3. **Catalog changes update an idle runtime's skills section.** The chronological + system-transcript decision supersedes the original digest-based retirement: + enabling a plugin, revoking `agent.prompt.inject`, or renaming a skill appends + a section update before the next request. The digest excludes bodies; the + `Skill` tool reads the file and rechecks permissions at call time. Removing + the last catalog entry also removes the executable `Skill` declaration. 4. **Plugin authoring ships as a first-party devkit, not as a plugin.** `@pi-desktop/plugin-devkit` owns scaffold, check and pack; three surfaces share that one implementation — the `pi-plugin` CLI, the `PluginScaffold` / diff --git a/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md b/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md index 464fc3e16..617c59f03 100644 --- a/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md +++ b/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md @@ -17,7 +17,11 @@ request omitted its schema. The same mismatch occurred after a mode switch. Before each new prompt and after a mode switch, the sidecar clears its in-memory deferred activation set and restores it from the effective `buildSessionContext` -projection. A successful `ToolSearch` result contributes its `addedToolNames`; +projection. When system declarations exist, their replayed active set is the +baseline; only successful results after the last declaration can add a newly +activated tool. This prevents an earlier success from resurrecting a subsequent +removal and preserves the set across a compaction checkpoint. For older histories +without declarations, a successful `ToolSearch` result contributes its `addedToolNames`; a successful result from a deferred tool contributes that tool's name. A name is restored only when it remains in the current mode's deferred catalog. Failed, interrupted, or missing-result placeholder rows are ignored, and assistant/user diff --git a/docs/adr/README.md b/docs/adr/README.md index 4bac9199a..753634d3a 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -20,6 +20,7 @@ Each ADR includes: | ID | Title | Status | |---|---|---| +| chronological-system-transcript | [Preserve chronological model system state](chronological-system-transcript.md) | Accepted | | mcp-tool-approval-risk | [User MCP tools keep the normal approval path](mcp-tool-approval-risk.md) | Accepted | | models-dev-catalog-authority | [models.dev owns published model metadata](models-dev-catalog-authority.md) | Accepted for implementation | | pi-ai-core-0991-authority | [Pi 0.99.1 account model authority](pi-ai-core-0991-authority.md) | Superseded for chat model metadata | diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md new file mode 100644 index 000000000..1177064d9 --- /dev/null +++ b/docs/adr/chronological-system-transcript.md @@ -0,0 +1,121 @@ +# ADR: Preserve chronological model system state + +- Status: Accepted +- Date: 2026-10-01 +- Issue: #1285 +- References: [Pi PR #9548](https://github.com/earendil-works/pi/pull/9548), + [Pi model authority](pi-ai-core-0991-authority.md) +- Updates: ADR 0039 catalog refresh, ADR 0225 deferred-tool restoration + +## Context + +Moving new tool declarations to the beginning of a conversation changes its +cached prefix. Rebuilding from visible chat alone also loses the chronology of +instruction and tool changes. Pi 0.99.1 already owns tool-state diffing and +provider request projections; Desktop owns session persistence and permissions. + +## Decision + +Keep Pi's provider-neutral system messages in chronological order. Use the +existing Host-owned transcript writer and optional versioned `modelSystem` +metadata on hidden system rows, acknowledging writes before dispatch. Anchors +preserve model order when the user row was pre-persisted. Save the effective +state once at explicit compaction; restore it before the summary and retained +tail. Keep current mode/catalog/Host permissions authoritative for execution. + +Separate Desktop's runtime instructions, skill catalog and contextual guidance +into replaceable sections. The skill-loading header uses `skills`, and each +catalog entry uses `skill:`; a change or removal appends only that +entry's replacement or null tombstone. The fixed prefix prevents collisions with +other Desktop section names. Restoring a legacy aggregate `skills` section +replaces it with the header and individual entries once at continuation. +Catalog-only skill changes refresh an idle runtime; +body loading and permission checks stay at the existing Host bridge. Tool schema +changes cannot reuse an idle runtime merely because tool names are identical. + +Retain the published model/API/endpoint identity before account overrides. +Accept transcript capability opt-ins only when that identity matches the actual +request route. Let Pi adapt the request for partial or absent support; never +persist an adapter's folded projection. + +The current models.dev catalog remains authoritative for published limits, +prices and modalities. Independently copy only the five transcript transport +flags and their original binding from Pi's exact published model. Do not derive +transport support from a models.dev metadata match or an account endpoint +override. Both Pi and models.dev metadata projections can carry that binding; +unverified routes and generic records retain the conservative fallback. + +The Pi 1.0.0 dependency patch adds the missing mid-conversation system +capability to its `deepseek-flash` catalog entry. Authorized official-endpoint +experiments confirmed both preserved cache reuse and effective updated +instructions. Keep this correction in the single Pi catalog, not a parallel +Desktop allowlist; remove the hunk when an upgraded Pi catalog carries it. +The model/API/endpoint binding check still excludes aliases and relays. Native +tool-addition and tool-change flags are not enabled by this correction. + +## Stable declarations across provider transports + +Choose the declaration strategy by the bound Pi transport capabilities, not a +model-name allowlist. Responses (OpenAI/Codex) with verified system support +and `additional_tools` or client tool search, Chat Completions with verified +system/tool additions, and Pi's transcript transport retain native chronological +additions. Other transports declare the complete current catalog in deterministic +name order from the first request. In particular Anthropic's native transition +blocks still grow its request-level schemas, so they use fixed declarations. +This policy also covers compatible relays without enabling unsupported native +message roles or switching their configured API. + +Desktop currently exposes Chat Completions, Responses, Codex Responses, +Anthropic Messages, Gemini and Pi Messages bindings. This change does not add +new selectable Azure, Vertex, Bedrock or Mistral native bindings; models offered +through existing compatible endpoints follow that endpoint's adapter. + +Keep activation separate from declarations in a versioned `tool_activation` +system section. The section records active deferred names and a SHA-256 identity +of the account, model, API, endpoint, declarations and deferred-name set. Updates +append after tool results and persist through the existing Host journal and +compaction checkpoint. Before provider conversion, omit this Desktop-only +activation section and any resulting empty metadata-only message. Preserve all +other instruction sections, content and tool deltas. Providers that fold system +messages must not rewrite their leading instructions just because execution +activation changed. ToolSearch results tell the model which tools were activated; +uncertain models may search again. Canonical persisted history is not mutated. +Restoration never interprets the complete declaration +snapshot as permission to execute every tool. Successful ToolSearch results +newer than the saved activation section recover an interrupted activation. +Invalid, unknown-version or mismatched state grants no activation. Legacy +histories without this section retain their existing activation evidence rules. + +Check activation before extension hooks or Host execution, then retain all +existing mode, approval and Host restrictions. A catalog/schema/mode/account or +route change starts a new declaration epoch and invalidates prior activation; +removed tools cannot be invoked. Temporary prompt replacement must preserve the +activation metadata. Current runtime activation remains authoritative between +prompts; declarations do not re-grant revoked activation. + +Use 128 functions as a conservative shared fixed-catalog ceiling, including +DeepSeek Chat Completions' limit. If the full catalog exceeds that ceiling, or its estimated prompt/schema cost leaves less than the ordinary +retained-tail budget below the automatic compaction threshold, use the existing +on-demand path and log the fallback reason. Never truncate a catalog. Context +estimation charges the full declared catalog while fixed declarations are active. + +The first request is larger. Short conversations may cost more overall even +when later cache-hit ratios improve. Acceptance compares cold and subsequent +uncached tokens and cumulative input cost across both short and longer synthetic +conversations; no universal savings or hit-rate guarantee is made. + +## Alternatives and consequences + +Prepending a reconstructed snapshot was rejected because it changes history on +ordinary continuation. A new persistence store or adopting Pi AgentSession +would create another session owner; optional canonical metadata preserves the +current process and data boundaries without a database migration. Unknown old +histories establish their first baseline at continuation rather than inventing +past state. Downgrades remain readable but do not retain these cache guarantees. + +The added records consume transcript storage. A failed write stops dispatch as +a local preparation failure; retry uses the same row ID. Providers may still +fold unsupported updates and lose cache reuse. Even native mid-conversation +system support does not guarantee provider cache reuse. Entry-level updates +reduce new input tokens without changing instruction roles or priority. Offline tests prove ordering, +restoration and payload compatibility, not real-provider cache-hit ratios. diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index 8f3288775..85bb9845b 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -521,8 +521,8 @@ on the last assistant usage and delegates provider-message estimation to pi-ai; desktop-only rows use the existing character heuristic. With no usage anchor, the budget also computes the output-cap estimator over non-system conversation messages plus the current system prompt and active -tool schemas. System-transcript rows are metadata snapshots of that same prompt -and are excluded from this component, so the request estimate counts the prompt +tool schemas. System-transcript rows are chronological updates, already covered by that +folded prompt/tool floor, and are excluded from this component, so the request estimate counts the prompt and schemas exactly once. The budget uses the larger of this full-request estimate and the calibrated message estimate. This is a hard floor before the first calibration sample and prevents counting overhead twice after an @@ -1289,6 +1289,56 @@ payload hook keeps its own object and its return value still wins. + [optional user custom instructions] ``` +### 7.0.0 Chronological system state (issue #1285) + +The Pi agent loop owns `toolsAdded` / `toolsRemoved` declarations, including +same-name schema replacement. Desktop never moves an update ahead of the user +or ToolSearch result that preceded it. Instruction composition uses independent +`runtime`, `skills` (loading instructions), `skill:` (one catalog +entry each), and `context` sections. Refreshing an idle skill catalog appends only +changed entries, uses null to revoke removed entries, and updates the executable +catalog. Unchanged entries and loading instructions are not repeated. Restoring +an older aggregate `skills` section upgrades it once at the continuation +boundary; checkpoints retain the effective per-entry state. Bodies remain +on-demand and permission-checked. Different plugin tool schemas/permissions +invalidate idle reuse even when their names are unchanged. + +Before model dispatch, Desktop acknowledges new system records through Host's +existing transcript writer. Restart replays these provider-neutral records in +order. An old history without records receives the current declaration at its +continuation boundary; Desktop does not invent historical instructions. Explicit +compaction saves the effective system state once before summary and retained +tail; it does not replay old tool deltas or preserve removed tools. + +Mid-conversation instructions, tool additions and full tool changes are separate +capabilities. Desktop accepts catalog opt-ins only for the same published model, +wire API and effective endpoint (including path and port). Unknown models, +changed bindings and unverified relays use Pi's conservative request projection. +Partial support keeps instruction updates but folds tools as required; removals +and redefinitions use the adapter's supported fallback. Model switching never +rewrites the canonical journal. Cache savings depend on the actual provider; +unsupported routes may still rebuild the request prefix. + +Published model limits, prices and modalities remain owned by models.dev. Its +runtime projection separately carries Pi's exact published transcript capability +flags and original binding. An enriched metadata source does not grant native +support by itself, and account overrides never replace that original binding. + +The pinned Pi patch declares mid-conversation system support for +`deepseek-flash` on its published `openai-completions` binding at +`https://api.deepseek.com`. Skill changes on this binding append system updates +without rewriting the previous request prefix. This does not enable native tool +additions or changes, and does not apply to unverified gateways, aliases or APIs. +The declaration comes from the patched Pi catalog, not provider settings or a +second Desktop model catalog. + +When restoring assistant history, map the current local account ID to the Pi +model's provider identity and retain recorded model IDs. A different recorded +account or model remains distinct. Legacy rows without identity retain the +current-model fallback. Same-model Completions reasoning stays in its native +reasoning field, never appended to visible answer text because of an account +UUID/vendor-name mismatch. + ### 7.0.1 User custom system prompt files (issue #542) The `[optional user custom instructions]` layer is the pi-compatible file pair @@ -1357,10 +1407,9 @@ grammar and validated against another fails every call. ### 7.1 Active tool context and on-demand loading (D185, ADR 0048) -The sidecar builds one complete tool registry, but it does not serialize every -registered schema into every provider request. Each new user prompt starts with -the mode's core set plus any deferred tools that can be restored from successful -activation evidence still present in the effective session context: +The sidecar builds one complete tool registry. Native anchored-addition routes +declare core tools and add activated deferred tools in place. Other routes use +the fixed-declaration policy below. Both preserve these execution activation rules: - Agent: `Read`, `Bash`, `Edit`, and `Write` (matching pi's coding-agent core) - Agent: `Skill` whenever the skill catalog is non-empty (D404, ADR 0230) — the @@ -1388,20 +1437,52 @@ The sidecar activates up to four matches, records their names in the canonical schemas. Providers with native deferred-tool search receive the definitions at that load point; other providers receive the active definitions normally. -At the start of each new user prompt, the sidecar clears the in-memory deferred -activation set and rebuilds it from the effective context. Successful -`ToolSearch` results contribute their canonical `details.addedToolNames`. -For compatibility, historical `details.activated` and top-level -`addedToolNames` markers are also accepted. Successful results from deferred -tools contribute that tool's name. Only names still present in the current -mode's deferred catalog are restored. Failed rows, interrupted or -missing-result placeholders, and assistant/user prose never activate a tool. -The tool registry, host permission path, tool timeout, and workspace containment -rules remain unchanged. `ToolSearch` is local to the sidecar and does not cross -the host RPC boundary. Its activation marker is retained in the persisted tool -result, so a runtime restart or a new prompt can reuse an eligible capability -while that evidence remains in the effective context; a fresh search is still -required after the evidence is compacted away or otherwise absent. +Deferred activation is sticky for the live runtime. At restoration, recorded +system messages (including a compaction checkpoint) define the active baseline +for ordinary on-demand histories. Only successful results after the latest +system record can add activation; older results must not resurrect removed +tools. Legacy histories without system records use successful results throughout +their effective context. ToolSearch accepts canonical `details.addedToolNames` +and historical `details.activated` / top-level `addedToolNames`; successful +results from deferred tools also restore their names. Failed results, +missing-result placeholders and assistant/user prose never activate tools. +Only names in the current mode's deferred catalog are eligible. + +Select by the bound transport, not a model name. Responses with verified system +support plus `supportsAdditionalTools` or `supportsToolSearch`, Chat Completions +with verified system/tool additions, and Pi Messages retain native additions. +Other bindings, including Anthropic Messages, Gemini, ordinary Chat Completions, +older Responses/Codex models and compatible relays, declare the complete catalog +in deterministic name order on the first request. Anthropic's native tool-change +blocks still grow request-level schemas, so do not exempt them. ToolSearch changes +activation without changing the declared schemas. A visible schema does not +permit execution: inactive deferred calls are rejected before extension hooks +and the Host; activated calls still require the existing mode and Host checks. +ToolSearch remains local and never grants approval or bypasses permissions. + +Fixed declarations persist separately from activation. A version-1 +`tool_activation` section records active names and a fingerprint of the account, +model, API, endpoint, schema catalog and deferred set. Activation changes append +at the continuation boundary, and the existing system journal/checkpoint saves +both declarations and activation. Strip the activation section only from the +provider projection, dropping metadata-only empty messages but preserving all +other sections/content/tool deltas. This prevents folding APIs from moving an +activation change into the leading prompt; persisted state remains complete. +Model guidance uses successful ToolSearch results, not private activation JSON. +Restore only validated activation for a +matching fingerprint, plus successful ToolSearch results newer than that state; +never activate tools merely because the full snapshot declared them. Malformed, +unknown-version and mismatched activation state fail closed. A catalog/schema, +mode, account, model or route change creates a new epoch and requires new +activation. Removal immediately removes the tool from executable registration. + +If the full catalog exceeds 128 functions or its prompt/schema estimate cannot +leave the normal retained-tail budget below the compaction threshold, retain +on-demand declarations and emit a diagnostic explaining that ToolSearch cache +stability is not guaranteed. The 128-function ceiling is conservative across +fixed-catalog APIs. Do not truncate tools or opt unknown endpoints into native +capabilities. First-request schema overhead increases; +cache stability does not imply that short conversations become cheaper. For user-visible HTML deliverables, the default system prompt asks the agent to activate `BrowserPreview` once after creating the page or making its first diff --git a/docs/spec/03-runtime/04-data-storage.md b/docs/spec/03-runtime/04-data-storage.md index 90ecf86ba..bcb12e6c0 100644 --- a/docs/spec/03-runtime/04-data-storage.md +++ b/docs/spec/03-runtime/04-data-storage.md @@ -181,6 +181,24 @@ per message; `seq` is implied by line order: {"type":"compaction","id":"cp1","summary":"…","firstKeptMessageId":"m2","throughMessageId":"m3","tokensBefore":917000,"retainedTail":[…],"providerId":"…","modelId":"…","createdAt":"…"} ``` +Internal system-state rows use role `system`, empty visible content, and optional +`meta.modelSystem = { version: 1, messageJson, beforeMessageId?, afterMessageId? }`. +`messageJson` is validated JSON text of a Pi system message with sections and tool +schema deltas; executable functions are excluded. JSON text preserves section and schema key +order across Rust storage; parsing for validation never reserializes it. Stable row IDs make retries +idempotent. The anchors restore logical model order when a user row was already +persisted before its preceding declaration; a surviving following anchor takes +precedence, then a preceding anchor, then the record's continuation position. +Forks remap surviving anchor IDs. Normal system notices remain visible; internal +model-state rows do not produce transcript bubbles or search text. + +Compaction details may include one `systemMessageJson` checkpoint. It replaces old +system updates in the retained tail and is restored before the summary. These +optional metadata fields use the existing JSONL/SQLite index and require no +schema migration. Old sessions remain readable; their first continuation records +a new baseline. Older app versions ignore the metadata and reconstruct their +usual current prompt, so downgrade does not promise the same cache prefix. + `sessions/.inflight.json` — the assistant reply currently streaming in the session, as one `{ schema, sessionId, turnId, savedAt, message }` object that host-core replaces atomically (temp + rename) on every diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index fd9539868..1aeb81261 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -16411,6 +16411,93 @@ renderer's durable transcript reads. No real model or provider is contacted. separate release qualification. No MCP, Codemode or virtual-router migration is included. See `docs/project/pi-0991-adoption.md` for candidate evidence. +### E2E-SYSTEM-TRANSCRIPT: Ordered model state survives restart and compaction + +- Fixture: isolated Host data directory, production AgentSidecar transport and + separate Node runtime/Pi process, local SSE provider; no real provider + credentials or user's Desktop process. The harness persists completed + messages; it does not exercise Electron's UI or persistence outbox. +- Send a prompt, activate BrowserPreview through ToolSearch, and verify the next + request includes the schema. Confirm the baseline and delta are acknowledged + by Host and survive stopping/restarting both processes without duplicate rows. +- Resume the session, update a skill catalog on the same runtime, and confirm the + next request uses the new catalog while keeping a second unchanged entry. + Runtime/adapter tests verify the native delta contains only the modified entry, + and fallback models receive the complete current catalog with removals applied. + Invoke Skill through the actual sidecar + bridge with a fixture body supplied by the embedding host. Compact, restart + the sidecar, and verify current skills and active tools survive without + replaying old tool results. Remove the final skill and verify its catalog and + executable schema disappear without recreating the runtime. +- Automated: `node scripts/e2e-system-transcript.mjs` with + `PI_DESKTOP_HOST_BIN` pointing to the candidate Host binary. Requires built + shared/host-runtime/agent-runtime packages. Renderer projection tests ensure + model-state records stay hidden while ordinary system notices remain visible. +- Adapter contract tests cover exact model/API/endpoint capability binding, + partial support, unverified relay fallback, removals/redefinitions and model + switching without canonical-history mutation. Real cache-hit improvement is + a separately authorized provider experiment, not an offline-test claim. + +- Flash regression: load the pinned Pi `deepseek-flash` model, serialize its + Desktop projection, and pass skill updates/removals through the real adapter. + The official binding preserves the earlier wire prefix; changed bindings + retain the folded fallback. Covered by `flash-transcript.test.ts`. +- Authorized live Flash acceptance: use isolated development sessions with + short and long contexts. Alternate unchanged turns and single-skill updates, + confirm current Skill bodies execute, and compare actual outgoing prefixes + plus reported usage. After restart, verify no duplicate system updates and + same-model assistant reasoning remains separate from visible content. Adapter + regressions also cover legacy identities and genuine account/model changes. Cache + percentages are observations, not deterministic pass thresholds. + +### E2E-FIXED-TOOL-DECLARATIONS: Stable schemas across transports with independent activation + +- Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, + and a child-process fetch boundary redirected to local HTTP/SSE. Credentials + are dummy values and synthetic tools have no external side effects. +- Prompt, search Alpha, execute Alpha, search Beta and execute Beta. HTTP payload + tests assert identical ordered `tools` and unchanged prior message prefixes. + Directly calling a declared but inactive tool must fail before the Host. +- Restart both processes, then compact and restart again: Alpha stays activated, + Beta remains inactive despite its declaration. Activate Beta, then remove it + from the catalog and verify it cannot execute. New schema/route epochs must + not restore old grants. Host rejection and Plan guards remain effective. +- Runtime contracts additionally cover legacy migration, interrupted activation, + malformed metadata, temporary prompt replacement, deterministic catalog order, + the 128-tool boundary and insufficient-context fallback. +- Automated: `node scripts/e2e-fixed-tool-declarations.mjs` with built + shared/host-runtime/agent-runtime and `PI_DESKTOP_HOST_BIN` set to the candidate + Host binary; `fixed-tool-runtime.test.ts` and `fixed-tool-declarations.test.ts` + cover HTTP payload and policy contracts. Electron UI/outbox is outside this + fixture's scope. +- Separately authorized official Flash experiments compare several independent + on-demand/fixed sessions with the same synthetic catalog and call sequence. + Record cold requests, both activations, follow-up cache hits/misses and + cumulative input cost. Report the larger first request and possible short-chat + cost increase, alongside any longer-conversation benefit. Offline test success + alone is not evidence of provider cache behavior. + +- Overflow/system-update integration: recover a provider context overflow with + a system delta after the failed assistant message. Keep that delta, remove the + failed assistant before compaction, and reuse one visible assistant message + through successful recovery. Terminal failure and Stop retain their existing + closure behavior. Covered by the parameterized runtime overflow user-path test. + +- Cross-provider acceptance: `fixed-tool-providers.test.ts` enters the real runtime + prompt/ToolSearch/execution path and captures each Desktop-selectable Pi + adapter's actual serialized payload at `onPayload`, before network dispatch. + Cover Chat Completions, Anthropic (native system on/off), Responses and Codex + (fallback, additional tools, client tool search), Gemini and Pi Messages. + Search A, execute A, search B, execute B, then finish: all five requests retain + their top-level schema state and prior semantic message prefix. Native routes + retain deferred schema additions. Activation JSON never reaches the provider. + This proves request construction, not server cache hits or paid API acceptance. +- The local HTTP/SSE fixture additionally covers official Flash, unflagged Chat + Completions and a compatible relay. Canonical activation restoration, denial + before Host execution, mode/account/catalog invalidation and compaction remain + required. Fixed declarations increase first-request size; oversized catalogs + explicitly fall back without a cache-stability guarantee. + #### E2E-262: Transcript path:line opens and scrolls the host file viewer - **Preconditions:** Isolated Electron/Chromium, an active workspace and session, diff --git a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md index ebea6505f..d27868f8e 100644 --- a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md +++ b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md @@ -981,15 +981,26 @@ sidecar 最多激活四个匹配项,并将名称写入 canonical 模式。具有本机延迟工具搜索的提供商可在该负载点接收定义;其他 提供商通常会收到活动定义。 -每个新用户提示前都会清除延迟激活集,再从有效上下文重建。成功的 -`ToolSearch` 结果读取 canonical `details.addedToolNames`;为兼容历史 -数据,也接受 `details.activated` 和顶层 `addedToolNames`。成功的延迟 -工具结果会贡献其工具名。仅恢复当前模式延迟目录中仍存在的名称;失败、 -中断、缺少结果的占位行以及助手/用户文本不会激活工具。工具注册表、主机 -权限路径、工具超时和工作区包含规则保持不变。`ToolSearch` 是 sidecar 的 -本地工具,不跨越主机 RPC 边界。激活标记保留在持久化工具结果中,因此 -只要证据仍在有效上下文,运行时重启或新提示都可以复用能力;证据被压缩 -或消失后仍需重新搜索。 +Deferred activation remains sticky within a live runtime. Restoration uses +successful activation evidence and the current catalog; old declarations do not +re-grant tools revoked from the live activation set. For routes without verified native anchored tool additions, +full declarations and execution activation are independent: versioned +`tool_activation` sections carry the account/model/API/endpoint/catalog identity +and active names through restart and compaction. Only matching, valid state and +newer successful ToolSearch results restore activation; malformed or changed +epochs fail closed. Inactive declared tools are blocked before extension/Host +execution, and activation never bypasses mode or approval checks. The full +catalog is deterministic from the first request. More than 128 tools or an +insufficient context budget falls back to on-demand declarations with a +diagnostic, without truncation. Responses and Chat Completions bindings with verified anchored additions, and +Pi Messages, retain native incremental publication. Anthropic native tool changes +still grow request schemas and therefore use fixed declarations. Strip only the +private activation section from provider projection, preserving canonical state +and all other instructions. This also stabilizes ToolSearch on folding APIs and +compatible relays without enabling new transport capabilities. +Fixed declarations may increase total cost for short conversations. See the +English section 7.1 and the chronological-system-transcript ADR for the complete +contract. 对于用户可见的 HTML 可交付成果,默认系统提示要求代理 创建页面或创建第一个页面后激活 `BrowserPreview` 一次 diff --git a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md index 0831bdb1f..ae32e2742 100644 --- a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md @@ -9290,6 +9290,54 @@ the latest destination. These assertions measure work counts, not device FPS. - **里程碑:** M6+ - **状态:** 单测和源码契约覆盖(`update-cache.test.mjs`、`auto-update.test.mjs`);仍需 Windows 安装器/E2E 验证。 +### E2E-FIXED-TOOL-DECLARATIONS: Stable schemas across transports with independent activation + +- Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, + and a child-process fetch boundary redirected to local HTTP/SSE. Credentials + are dummy values and synthetic tools have no external side effects. +- Prompt, search Alpha, execute Alpha, search Beta and execute Beta. HTTP payload + tests assert identical ordered `tools` and unchanged prior message prefixes. + Directly calling a declared but inactive tool must fail before the Host. +- Restart both processes, then compact and restart again: Alpha stays activated, + Beta remains inactive despite its declaration. Activate Beta, then remove it + from the catalog and verify it cannot execute. New schema/route epochs must + not restore old grants. Host rejection and Plan guards remain effective. +- Runtime contracts additionally cover legacy migration, interrupted activation, + malformed metadata, temporary prompt replacement, deterministic catalog order, + the 128-tool boundary and insufficient-context fallback. +- Automated: `node scripts/e2e-fixed-tool-declarations.mjs` with built + shared/host-runtime/agent-runtime and `PI_DESKTOP_HOST_BIN` set to the candidate + Host binary; `fixed-tool-runtime.test.ts` and `fixed-tool-declarations.test.ts` + cover HTTP payload and policy contracts. Electron UI/outbox is outside this + fixture's scope. +- Separately authorized official Flash experiments compare several independent + on-demand/fixed sessions with the same synthetic catalog and call sequence. + Record cold requests, both activations, follow-up cache hits/misses and + cumulative input cost. Report the larger first request and possible short-chat + cost increase, alongside any longer-conversation benefit. Offline test success + alone is not evidence of provider cache behavior. + +- Overflow/system-update integration: recover a provider context overflow with + a system delta after the failed assistant message. Keep that delta, remove the + failed assistant before compaction, and reuse one visible assistant message + through successful recovery. Terminal failure and Stop retain their existing + closure behavior. Covered by the parameterized runtime overflow user-path test. + +- Cross-provider acceptance: `fixed-tool-providers.test.ts` enters the real runtime + prompt/ToolSearch/execution path and captures each Desktop-selectable Pi + adapter's actual serialized payload at `onPayload`, before network dispatch. + Cover Chat Completions, Anthropic (native system on/off), Responses and Codex + (fallback, additional tools, client tool search), Gemini and Pi Messages. + Search A, execute A, search B, execute B, then finish: all five requests retain + their top-level schema state and prior semantic message prefix. Native routes + retain deferred schema additions. Activation JSON never reaches the provider. + This proves request construction, not server cache hits or paid API acceptance. +- The local HTTP/SSE fixture additionally covers official Flash, unflagged Chat + Completions and a compatible relay. Canonical activation restoration, denial + before Host execution, mode/account/catalog invalidation and compaction remain + required. Fixed declarations increase first-request size; oversized catalogs + explicitly fall back without a cache-stability guarantee. + #### E2E-262:聊天 path:line 引用打开文件并滚动到目标行 - **前提:** 隔离 Electron/Chromium、活动工作区和会话、可用的随应用打包文件管理器视图,以及确定性的文件系统 IPC fixture。 diff --git a/packages/agent-runtime/src/extensions/managed-exec.test.ts b/packages/agent-runtime/src/extensions/managed-exec.test.ts index b41a6beeb..5b869d663 100644 --- a/packages/agent-runtime/src/extensions/managed-exec.test.ts +++ b/packages/agent-runtime/src/extensions/managed-exec.test.ts @@ -8,7 +8,8 @@ it("cancels an owned process tree after an explicit readiness signal", async () const root = mkdtempSync(join(tmpdir(), "pi-owned-exec-")); const ready = join(root, "ready.json"); const owner = new AbortController(); - const childSource = `require('node:fs').writeFileSync(${JSON.stringify(ready)}, JSON.stringify([process.ppid,process.pid])); setInterval(()=>{},1000);`; + // Publish readiness only after the complete PID list is visible to the reader. + const childSource = `const fs=require('node:fs'); fs.writeFileSync(${JSON.stringify(`${ready}.tmp`)}, JSON.stringify([process.ppid,process.pid])); fs.renameSync(${JSON.stringify(`${ready}.tmp`)}, ${JSON.stringify(ready)}); setInterval(()=>{},1000);`; const parentSource = `require('node:child_process').spawn(process.execPath,['-e',${JSON.stringify(childSource)}],{stdio:'inherit'}); setInterval(()=>{},1000);`; const pending = managedExec(process.execPath, ["-e", parentSource], root, owner.signal); try { diff --git a/packages/agent-runtime/src/extensions/prompt-chain.test.ts b/packages/agent-runtime/src/extensions/prompt-chain.test.ts index 1669a0b55..fa212b0c4 100644 --- a/packages/agent-runtime/src/extensions/prompt-chain.test.ts +++ b/packages/agent-runtime/src/extensions/prompt-chain.test.ts @@ -39,7 +39,10 @@ it.each([ }); runtime = new DesktopAgentRuntime({ host: { - call: async (method) => { throw new Error(`Unexpected host call: ${method}`); }, + call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + throw new Error(`Unexpected host call: ${method}`); + }, onNotification: () => () => {}, }, sessionId: "prompt-chain-fixture", diff --git a/packages/agent-runtime/src/fixed-tool-declarations.test.ts b/packages/agent-runtime/src/fixed-tool-declarations.test.ts new file mode 100644 index 000000000..451fd277c --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-declarations.test.ts @@ -0,0 +1,71 @@ +import { describe, expect, it } from "vitest"; +import type { AgentMessage, AgentTool } from "@earendil-works/pi-agent-core"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { Type, type Api, type Model } from "@earendil-works/pi-ai"; +import { toolDeclarationPolicy, toolActivationSection, restoredToolActivation, TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; +import { replaceSystemPrompt, systemTranscriptCheckpoint } from "./system-transcript.js"; +import { convertToLlm } from "./pi-runtime-messages.js"; + +const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; +const tool = (name: string): AgentTool => ({ name, label: name, description: "Fixture tool", parameters: Type.Object({}), + execute: async () => ({ content: [{ type: "text", text: "Done" }], details: {} }) }); +const deferred = new Set(["Alpha", "Beta"]); +const policy = (tools = [tool("Alpha"), tool("Beta")], overrides: Partial> = {}, prompt = "") => + toolDeclarationPolicy({ ...model, ...overrides } as Model, tools, deferred, prompt, "fixture-account"); + +describe("fixed tool declaration policy", () => { + it("persists activation without projecting it into model instructions", () => { + const current = policy(); + const messages: AgentMessage[] = [ + { role: "system" as const, content: "", timestamp: 1, toolsAdded: current.tools, + sections: { runtime: "Rules", [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set()) } }, + { role: "system" as const, content: "", timestamp: 2, + sections: { [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set(["Alpha"])) } }, + { role: "system" as const, content: "Changed instruction", timestamp: 3, + sections: { obsolete: null, [TOOL_ACTIVATION_SECTION]: null }, toolsRemoved: [{ name: "Beta" }] }, + ]; + const before = JSON.stringify(messages); + const projected = convertToLlm(messages); + expect(projected).toHaveLength(2); + expect(projected[0]).toMatchObject({ sections: { runtime: "Rules" }, toolsAdded: current.tools }); + expect(projected[1]).toMatchObject({ content: "Changed instruction", sections: { obsolete: null }, toolsRemoved: [{ name: "Beta" }] }); + expect(JSON.stringify(projected)).not.toContain(TOOL_ACTIVATION_SECTION); + expect(JSON.stringify(messages)).toEqual(before); + }); + it("has a deterministic declaration order and snapshot identity", () => { + const first = policy(); + const second = policy([tool("Beta"), tool("Alpha")]); + expect(second.key).toBe(first.key); + expect(second.tools?.map((tool) => tool.name)).toEqual(["Alpha", "Beta"]); + expect(policy([{ ...tool("Alpha"), parameters: Type.Object({ id: Type.String() }) }]).key).not.toBe(first.key); + }); + it.each([ + { id: "deepseek-pro" }, { api: "openai-responses" }, { baseUrl: "https://relay.invalid" }, + { baseUrl: "https://api.deepseek.com/v1" }, { compat: { supportsMidConvoSystemMessages: false } }, + ] as Partial>[])("stabilizes declarations without assuming transcript capabilities: %j", (overrides) => { + expect(policy(undefined, overrides).tools?.map((tool) => tool.name)).toEqual(["Alpha", "Beta"]); + expect(policy(undefined, overrides).fallback).toBeUndefined(); + }); + it("falls back without truncating catalogs beyond the provider's function limit", () => { + expect(policy(Array.from({ length: 128 }, (_, index) => tool(`Tool${index}`))).tools).toHaveLength(128); + expect(policy(Array.from({ length: 129 }, (_, index) => tool(`Tool${index}`)))).toMatchObject({ fallback: "tool-count" }); + }); + it("leaves room for retained conversation and output", () => { + expect(policy([tool("Alpha")], { contextWindow: 32000, maxTokens: 4000 }, "x".repeat(80000))) + .toMatchObject({ fallback: "context-budget" }); + }); + it("restores activation independently of full declarations, including compacted and transient prompts", () => { + const current = policy(); + const messages = [{ role: "system" as const, content: "", timestamp: 1, + toolsAdded: current.tools, sections: { runtime: "Rules", [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set(["Alpha"])) } }]; + const changed = replaceSystemPrompt(messages, "Temporary nudge"); + expect(restoredToolActivation(changed, current.key)?.active).toEqual(["Alpha"]); + expect(restoredToolActivation([systemTranscriptCheckpoint(changed)!], current.key)?.active).toEqual(["Alpha"]); + expect(restoredToolActivation(messages, "different-snapshot")?.active).toEqual([]); + expect(restoredToolActivation([], current.key)).toBeUndefined(); + }); + it.each(["invalid JSON", '{"version":2}', null])("fails closed on invalid activation metadata: %s", (value) => { + expect(restoredToolActivation([{ role: "system", content: "", timestamp: 1, + sections: { [TOOL_ACTIVATION_SECTION]: value } }], policy().key)?.active).toEqual([]); + }); +}); diff --git a/packages/agent-runtime/src/fixed-tool-declarations.ts b/packages/agent-runtime/src/fixed-tool-declarations.ts new file mode 100644 index 000000000..cf3a450fd --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-declarations.ts @@ -0,0 +1,84 @@ +import { createHash } from "node:crypto"; +import type { AgentMessage, AgentTool } from "@earendil-works/pi-agent-core"; +import { getCurrentSystemMessage, toToolDeclaration, Type, type Api, type Model } from "@earendil-works/pi-ai"; +import * as Value from "typebox/value"; +import { contextBudgetLimitsFor, automaticCompactionThresholdFor } from "./context-budget.js"; +import { estimateOutputCapInputTokens } from "./output-cap.js"; + +export const TOOL_ACTIVATION_SECTION = "tool_activation"; +const activationSchema = Type.Object({ + version: Type.Literal(1), snapshot: Type.String({ pattern: "^[a-f0-9]{64}$" }), + active: Type.Array(Type.String({ minLength: 1 }), { uniqueItems: true }), +}, { additionalProperties: false }); + +export type ToolDeclarationPolicy = { + key: string; + tools?: AgentTool[]; + fallback?: "tool-count" | "context-budget"; +}; + +/** Only these Pi transports keep new schemas out of the request's initial tools. */ +function hasAnchoredToolAdditions(model: Model): boolean { + if (model.api === "pi-messages") return true; + const compat = model.compat; + if (!compat || !("supportsMidConvoSystemMessages" in compat) + || compat.supportsMidConvoSystemMessages !== true) return false; + if (model.api === "openai-completions") { + return "supportsMidConvoToolAdditions" in compat && compat.supportsMidConvoToolAdditions === true; + } + if (["openai-responses", "openai-codex-responses"].includes(model.api)) { + return ("supportsAdditionalTools" in compat && compat.supportsAdditionalTools === true) + || ("supportsToolSearch" in compat && compat.supportsToolSearch === true); + } + // Anthropic's native tool-change blocks still grow its request-level schemas. + return false; +} + +/** Keep native anchored additions, otherwise stabilize the complete catalog. */ +export function toolDeclarationPolicy( + model: Model, tools: readonly AgentTool[], deferred: ReadonlySet, prompt: string, accountId: string, +): ToolDeclarationPolicy { + const ordered = [...tools].sort((a, b) => a.name < b.name ? -1 : a.name > b.name ? 1 : 0); + const key = createHash("sha256").update(JSON.stringify({ + account: accountId, model: model.id, api: model.api, endpoint: model.baseUrl.replace(/\/+$/, ""), + tools: ordered.map(toToolDeclaration), deferred: [...deferred].sort(), + })).digest("hex"); + if (hasAnchoredToolAdditions(model)) return { key }; + // Conservative shared ceiling, including Chat Completions' 128-function limit. + // Larger catalogs retain on-demand loading; never truncate or raise API limits. + if (ordered.length > 128) return { key, fallback: "tool-count" }; + const budget = contextBudgetLimitsFor(model); + const tokens = estimateOutputCapInputTokens({ messages: [], systemPrompt: prompt, tools: ordered }, model); + // Leave the normal retained-tail budget available to the conversation. A + // fixed catalog must not make every fresh/compacted request overflow again. + if (tokens >= automaticCompactionThresholdFor(budget) - budget.keepRecentTokens) { + return { key, fallback: "context-budget" }; + } + return { key, tools: ordered }; +} + +export function toolActivationSection(key: string, active: ReadonlySet): string { + return JSON.stringify({ version: 1, snapshot: key, active: [...active].sort() }); +} + +/** A declaration is not activation. Malformed/new-version state fails closed. */ +export function restoredToolActivation(messages: readonly AgentMessage[], key: string): { active: string[]; replayFrom: number } | undefined { + const section = getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION]; + const closed = { active: [], replayFrom: messages.length }; + if (section == null) return messages.some((message) => message.role === "system" + && TOOL_ACTIVATION_SECTION in (message.sections ?? {})) ? closed : undefined; + try { + const parsed: unknown = JSON.parse(section); + if (!Value.Check(activationSchema, parsed) || parsed.snapshot !== key) return closed; + const lastState = messages.map((message) => message.role === "system" + && TOOL_ACTIVATION_SECTION in (message.sections ?? {})).lastIndexOf(true); + return { active: parsed.active, replayFrom: lastState + 1 }; + } catch { return closed; } +} + +/** Activation is appended after the result, never inserted into old instructions. */ +export function syncToolActivation(messages: AgentMessage[], section: string): AgentMessage[] { + if (getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION] === section) return messages; + const timestamp = messages.reduce((latest, message) => Math.max(latest, message.timestamp + 1), Date.now()); + return [...messages, { role: "system", content: "", timestamp, sections: { [TOOL_ACTIVATION_SECTION]: section } }]; +} diff --git a/packages/agent-runtime/src/fixed-tool-providers.test.ts b/packages/agent-runtime/src/fixed-tool-providers.test.ts new file mode 100644 index 000000000..12f8b2439 --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-providers.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from "vitest"; +import type { Agent } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, type Api, type Model, type AssistantMessage, type StreamFunction } from "@earendil-works/pi-ai"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { runtimeFixture, pluginTools } from "./test-helpers/fixed-tool-fixture.js"; + +const base = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; +const routes: { api: Api; compat?: Model["compat"]; native?: boolean; label: string }[] = [ + { label: "Chat Completions", api: "openai-completions" }, + { label: "Claude without native transitions", api: "anthropic-messages" }, + { label: "Claude with native transitions", api: "anthropic-messages", compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolChanges: true } }, + { label: "OpenAI Responses fallback", api: "openai-responses" }, + { label: "Codex fallback", api: "openai-codex-responses" }, + { label: "Gemini", api: "google-generative-ai" }, + { label: "Kimi native additions", api: "openai-completions", native: true, compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: true } }, + { label: "OpenAI native additions", api: "openai-responses", native: true, compat: { supportsMidConvoSystemMessages: true, supportsAdditionalTools: true } }, + { label: "Codex native search", api: "openai-codex-responses", native: true, compat: { supportsMidConvoSystemMessages: true, supportsToolSearch: true } }, + { label: "Codex native additions", api: "openai-codex-responses", native: true, compat: { supportsMidConvoSystemMessages: true, supportsAdditionalTools: true } }, + { label: "Pi transcript", api: "pi-messages", native: true }, +]; +// Cache markers move with the last message; they are not model input text. +function semantic(value: unknown): unknown { + if (Array.isArray(value)) return value.map(semantic); + if (value && typeof value === "object") return Object.fromEntries(Object.entries(value) + .filter(([key]) => key !== "cache_control" && key !== "cachePoint") + .map(([key, item]) => [key, semantic(item)])); + return value; +} +function payloadView(value: unknown): { tools: unknown; system: unknown; messages: unknown[] } { + const p = value as Record; + const config = p.config as Record | undefined; + const context = p.context as Record | undefined; + return { tools: p.tools ?? p.toolConfig ?? config?.tools, + system: p.system ?? p.instructions ?? config?.systemInstruction, + messages: (p.messages ?? p.input ?? p.contents ?? context?.messages ?? []) as unknown[] }; +} + +describe("ToolSearch user path through every Desktop-selectable Pi request adapter", () => { + it.each(routes)("preserves schemas and history for $label", async ({ api, compat, native }) => { + const model = { ...base, id: "fixture-model", provider: "fixture", api, + baseUrl: "https://fixture.invalid", compat, reasoning: false } as Model; + const adapter: { stream: StreamFunction } = await import(`@earendil-works/pi-ai/api/${api}`); + // Synthetic token satisfies Codex's local parser; no real auth is read. + const key = `fixture.${Buffer.from(JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: "fixture" } })).toString("base64")}.fixture`; + const f = runtimeFixture([], pluginTools, { id: "fixture", name: "Fixture", modelId: model.id, + baseUrl: model.baseUrl, apiKey: key, authKind: "api_key", supportsReasoning: false, + supportedThinkingLevels: ["off"], modelConfig: modelConfigFromPi(model) }); + const requests: ReturnType[] = []; + const captureErrors: string[] = []; + const calls: { name: string; arguments: Record }[] = [ + { name: "ToolSearch", arguments: { query: "plugin_alpha" } }, { name: "plugin_alpha", arguments: {} }, + { name: "ToolSearch", arguments: { query: "plugin_beta" } }, { name: "plugin_beta", arguments: {} }, + ]; + const agent = (f.runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = async (resolved, context) => { + if (resolved.api !== api) captureErrors.push(`Wrong adapter: ${resolved.api}`); + let captured: unknown; + // Run the serializer and stop at the external request boundary, before + // network access or cloud credential lookup. + const probe = adapter.stream(resolved, context, { apiKey: key, maxTokens: 1024, + onPayload: (payload) => { captured = payload; throw new Error("fixture payload captured"); }, + }); + const outcome = await probe.result(); + if (captured === undefined) captureErrors.push(outcome.errorMessage ?? "No payload"); + else if (JSON.stringify(captured).includes("tool_activation")) captureErrors.push("Activation metadata leaked"); + requests.push(payloadView(semantic(captured ?? {}))); + const call = calls.shift(); + const message: AssistantMessage = { role: "assistant", api, provider: resolved.provider, model: resolved.id, + content: call ? [{ type: "toolCall", id: `call-${requests.length}`, ...call }] : [{ type: "text", text: "Done." }], + stopReason: call ? "toolUse" : "stop", timestamp: Date.now(), + usage: { input: 10, output: 2, cacheRead: 0, cacheWrite: 0, totalTokens: 12, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } } }; + const stream = createAssistantMessageEventStream(); + stream.push({ type: "done", reason: call ? "toolUse" : "stop", message }); + return stream; + }; + try { + await f.prompt(); + expect(captureErrors).toEqual([]); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual(["plugin_alpha", "plugin_beta"]); + expect(requests).toHaveLength(5); + expect(requests[0].messages.length).toBeGreaterThan(0); + if (!native) expect(JSON.stringify(requests[0].tools)).toContain("plugin_beta"); + else { + expect(JSON.stringify(requests[0].tools) ?? "").not.toContain("plugin_beta"); + expect(JSON.stringify(requests.at(-1)?.messages)).toContain("plugin_beta"); + } + for (let index = 1; index < requests.length; index++) { + expect(requests[index].tools).toEqual(requests[0].tools); + expect(requests[index].system).toEqual(requests[0].system); + const previous = requests[index - 1].messages; + expect(requests[index].messages.slice(0, previous.length)).toEqual(previous); + } + } finally { await f.runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/fixed-tool-runtime.test.ts b/packages/agent-runtime/src/fixed-tool-runtime.test.ts new file mode 100644 index 000000000..4d32bb64e --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-runtime.test.ts @@ -0,0 +1,153 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { systemTranscriptCheckpoint } from "./system-transcript.js"; +import { readSystemMessage } from "./system-transcript-journal.js"; +import type { UiMessage } from "@pi-desktop/shared"; +import { flashProvider, pluginTools, wireFixture, runtimeFixture, type Payload } from "./test-helpers/fixed-tool-fixture.js"; + +afterEach(() => vi.unstubAllGlobals()); +describe("fixed Flash declarations through runtime and HTTP/SSE", () => { + it("rejects a declared but inactive tool without invoking Host", async () => { + const wire = await wireFixture([{ name: "plugin_beta" }]); + const f = runtimeFixture(); + try { + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual([]); + expect(f.rows.find((row) => row.toolName === "plugin_beta")).toMatchObject({ isError: true }); + expect(JSON.stringify(wire.requests[1].messages)).toContain("Call ToolSearch to activate plugin_beta"); + expect(wire.requests[1].tools).toEqual(wire.requests[0].tools); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it.each([false, true])("restores only activated tools after restart (checkpoint=%s)", async (checkpoint) => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + let declared: Payload["tools"]; + try { + await f.prompt(); + expect(f.errors).toEqual([]); + declared = wire.requests.at(-1)!.tools; + history = structuredClone(f.rows); + if (checkpoint) { + const system = systemTranscriptCheckpoint(history.filter((row) => row.modelSystem) + .map((row) => readSystemMessage(row.modelSystem!.messageJson)))!; + history = [{ id: "checkpoint", role: "system", content: "", createdAt: new Date().toISOString(), + modelSystem: { version: 1, messageJson: JSON.stringify(system) } }]; + } + } finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("restored-user"); + expect(restored.errors).toEqual([]); + expect(restored.executed).toEqual(["plugin_alpha"]); + expect(resumed.requests[0].tools).toEqual(declared!); + expect([...restored.rows].reverse().find((row) => row.toolName === "plugin_beta")).toMatchObject({ isError: true }); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it.each(["schema", "removed", "route"])("does not revive activation after a %s change", async (change) => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + try { await f.prompt(); history = structuredClone(f.rows); } + finally { await f.runtime.dispose(); await wire.close(); } + const tools = change === "removed" ? pluginTools.slice(1) : change === "schema" + ? [{ ...pluginTools[0], description: "Changed schema epoch" }, pluginTools[1]] : pluginTools; + const provider = change === "route" ? { ...flashProvider(), baseUrl: "https://relay.invalid/v1" } : flashProvider(); + const resumed = await wireFixture([{ name: "plugin_alpha" }]); + const restored = runtimeFixture(history!, tools, provider); + try { + await restored.prompt("changed-user"); + expect(restored.errors).toEqual([]); + expect(restored.executed).toEqual([]); + if (change === "removed") expect(resumed.requests[0].tools.map((tool) => tool.function.name)).not.toContain("plugin_alpha"); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("still requires Host approval after ToolSearch activation", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }]); + const f = runtimeFixture([], pluginTools, flashProvider(), true); + try { + await f.prompt(); + expect(f.executed).toEqual(["plugin_alpha"]); + expect(f.rows.find((row) => row.toolName === "plugin_alpha")).toMatchObject({ isError: true }); + expect(JSON.stringify(wire.requests.at(-1)?.messages)).toContain("Permission denied"); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it("replays a successful activation if stopped before its next declaration checkpoint", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + try { + await f.prompt(); + const first = f.rows.find((row) => row.modelSystem)!; + history = structuredClone(f.rows.filter((row) => !row.modelSystem || row === first)); + } finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("resumed-user"); + expect(restored.executed).toEqual(["plugin_alpha"]); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("migrates legacy on-demand history without granting the rest of the catalog", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture([], pluginTools, { ...flashProvider(), baseUrl: "https://relay.invalid" }); + let history: UiMessage[]; + try { await f.prompt(); history = structuredClone(f.rows).map((row) => { + if (!row.modelSystem) return row; + const state = readSystemMessage(row.modelSystem.messageJson); + delete state.sections?.tool_activation; + if (state.toolsAdded) state.toolsAdded = state.toolsAdded.filter((tool) => tool.name !== "plugin_beta"); + return { ...row, modelSystem: { ...row.modelSystem, messageJson: JSON.stringify(state) } }; + }); } + finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("migrated-user"); + expect(restored.executed).toEqual(["plugin_alpha"]); + expect(resumed.requests[0].tools.map((tool) => tool.function.name)).toEqual(expect.arrayContaining(["plugin_alpha", "plugin_beta"])); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("retains the Plan execution guard for predeclared tools", async () => { + const wire = await wireFixture([{ name: "Write", args: { path: "fixture", content: "denied" } }]); + const f = runtimeFixture(); + try { + f.runtime.setMode("plan"); + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual([]); + expect(f.rows.find((row) => row.toolName === "Write")).toMatchObject({ isError: true }); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it.each([ + { name: "official Flash", provider: flashProvider() }, + { name: "unflagged Chat Completions", provider: { ...flashProvider(), modelId: "fixture-chat", modelConfig: undefined } }, + { name: "compatible relay", provider: { ...flashProvider(), baseUrl: "https://relay.invalid/v1" } }, + ])("keeps the complete request prefix for $name while searching and executing two different tools", async ({ provider }) => { + const wire = await wireFixture([ + { name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }, + { name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta" }, + ]); + const f = runtimeFixture([], pluginTools, provider); + try { + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual(["plugin_alpha", "plugin_beta"]); + expect(wire.requests).toHaveLength(5); + expect(wire.requests[0].tools.map((tool) => tool.function.name)).toEqual(expect.arrayContaining(["plugin_alpha", "plugin_beta"])); + for (let index = 1; index < wire.requests.length; index++) { + expect(wire.requests[index].tools).toEqual(wire.requests[0].tools); + const previous = wire.requests[index - 1].messages; + expect(wire.requests[index].messages.slice(0, previous.length)).toEqual(previous); + } + } finally { await f.runtime.dispose(); await wire.close(); } + }); +}); diff --git a/packages/agent-runtime/src/flash-transcript.test.ts b/packages/agent-runtime/src/flash-transcript.test.ts new file mode 100644 index 000000000..a423319cf --- /dev/null +++ b/packages/agent-runtime/src/flash-transcript.test.ts @@ -0,0 +1,144 @@ +import type { Agent } from "@earendil-works/pi-agent-core"; +import { DesktopAgentRuntime } from "./runtime.js"; +import { describe, expect, it } from "vitest"; +import { normalizeContext, type Message, type Model } from "@earendil-works/pi-ai"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { stream } from "@earendil-works/pi-ai/api/openai-completions"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { buildProviderModel, type RuntimeProviderConfig } from "./provider-binding.js"; + +function flashProvider(): RuntimeProviderConfig { + const flash = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); + if (!flash) throw new Error("Missing published Flash model"); + return { + id: "flash-fixture", name: "Flash fixture", modelId: flash.id, baseUrl: flash.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: true, supportedThinkingLevels: ["high"], + // Main sends this projection across the sidecar boundary. + modelConfig: JSON.parse(JSON.stringify(modelConfigFromPi(flash))), + }; +} + +type Payload = { messages: Array<{ role: string; content?: unknown }> }; +async function request(messages: Message[], baseUrl?: string): Promise { + const provider = flashProvider(); + const model = buildProviderModel({ ...provider, ...(baseUrl ? { baseUrl } : {}) }); + let captured: Payload | undefined; + const response = stream(model as Model<"openai-completions">, normalizeContext({ messages }), { + apiKey: "fixture", + fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + expect((await response.result()).stopReason).toBe("stop"); + if (!captured) throw new Error("No Flash request captured"); + return captured; +} + +describe("published Flash transcript compatibility", () => { + it("enables system updates independently of native tool-state capabilities", () => { + const provider = flashProvider(); + expect(provider.modelConfig?.compat?.supportsMidConvoSystemMessages).toBe(true); + expect(buildProviderModel(provider).compat).toMatchObject({ + supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false, + supportsMidConvoToolChanges: false, supportsAdditionalTools: false, supportsToolSearch: false, + }); + expect(buildProviderModel({ ...provider, baseUrl: "https://api.deepseek.com/" }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + + it.each([ + { modelId: "deepseek-flash-alias" }, + { baseUrl: "https://relay.example/v1" }, { baseUrl: "http://api.deepseek.com" }, + { baseUrl: "https://api.deepseek.com/v1" }, { baseUrl: "https://api.deepseek.com:8443" }, + { baseUrl: "https://api.deepseek.com?route=other" }, { baseUrl: "https://api.deepseek.com#fragment" }, + { baseUrl: "https://user@api.deepseek.com" }, + ])("retains conservative fallback on a changed binding: %j", (overrides) => { + expect(buildProviderModel({ ...flashProvider(), ...overrides }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: false }); + }); + + it("checks the selected wire API rather than the provider-wide style", () => { + const provider = flashProvider(); + expect(buildProviderModel({ ...provider, apiStyle: "responses" })).toMatchObject({ + api: "openai-completions", compat: { supportsMidConvoSystemMessages: true }, + }); + expect(buildProviderModel({ ...provider, modelConfig: { ...provider.modelConfig!, api: "openai-responses" } })) + .toMatchObject({ api: "openai-responses", compat: { supportsMidConvoSystemMessages: false } }); + }); + + it("requires the original Pi binding even when the endpoint and model name look official", () => { + const provider = flashProvider(); + for (const modelConfig of [undefined, + { ...provider.modelConfig!, source: "generic" as const }, + { ...provider.modelConfig!, transcriptBinding: undefined }, + ]) { + expect(buildProviderModel({ ...provider, modelConfig }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: false }); + } + }); + + it("appends skill changes and revocations without rewriting the previous wire prefix", async () => { + const baseline: Message[] = [ + { role: "system", content: "", timestamp: 1, sections: { + runtime: "Base rules", skills: "Load skills on demand", "skill:a": "Old Alpha", "skill:b": "Bravo", + } }, + { role: "user", content: "First request", timestamp: 2 }, + ]; + const changed: Message[] = [...baseline, + { role: "system", content: "", timestamp: 3, sections: { "skill:a": "New Alpha", "skill:b": null } }, + { role: "user", content: "Continue", timestamp: 4 }, + ]; + const before = await request(baseline); + const after = await request(changed); + expect(after.messages.slice(0, before.messages.length)).toEqual(before.messages); + expect(after.messages.map((message) => message.role)).toEqual(["system", "user", "system", "user"]); + expect(JSON.stringify(after.messages[2])).toContain("New Alpha"); + expect(JSON.stringify(after.messages[2])).toContain("skill:b"); + const relay = await request(changed, "https://relay.example/v1"); + expect(relay.messages.map((message) => message.role)).toEqual(["system", "user", "user"]); + expect(JSON.stringify(relay.messages)).toContain("New Alpha"); + expect(JSON.stringify(relay.messages)).not.toContain("Old Alpha"); + expect(JSON.stringify(relay.messages)).not.toContain("Bravo"); + }); +}); + + +describe("Flash reasoning after Desktop session restoration", () => { + it.each([ + { providerId: "flash-fixture", modelId: "deepseek-flash", sameModel: true }, + { providerId: undefined, modelId: undefined, sameModel: true }, + { providerId: "another-account", modelId: "deepseek-flash", sameModel: false }, + { providerId: "flash-fixture", modelId: "another-model", sameModel: false }, + ])("preserves source identity when restoring %j", async ({ providerId, modelId, sameModel }) => { + const provider = { ...flashProvider(), vendorKey: "deepseek" }; + let captured: Payload | undefined; + const runtime = new DesktopAgentRuntime({ + sessionId: "restore-flash", mode: "agent", provider, thinkingLevel: "high", + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + history: [{ + id: "old-answer", role: "assistant", content: "answer", thinking: "private plan", + providerId, modelId, status: "complete", createdAt: "2026-10-01T00:00:00.000Z", + }], + host: { call: async (): Promise => undefined as T }, onEvent: () => {}, + }); + const agent = (runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = (model, context) => stream(model as Model<"openai-completions">, context, { + apiKey: "fixture", fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + try { + await runtime.prompt("Continue", "new-user", "new-turn"); + expect(captured).toBeDefined(); + const assistant = captured!.messages.find((message) => message.role === "assistant"); + if (sameModel) { + expect(assistant).toMatchObject({ content: "answer", reasoning_content: "private plan" }); + } else { + expect(assistant?.content).toContain("private plan"); + expect(assistant).not.toMatchObject({ reasoning_content: "private plan" }); + } + } finally { await runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/mode-tool-access.test.ts b/packages/agent-runtime/src/mode-tool-access.test.ts index 64f96723e..042010e7e 100644 --- a/packages/agent-runtime/src/mode-tool-access.test.ts +++ b/packages/agent-runtime/src/mode-tool-access.test.ts @@ -80,7 +80,10 @@ it.each([ compactionSettings: { enabled: false, reserveTokens: 0, keepRecentTokens: 0 }, commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, subagents: [{ name: "worker", description: "Inspect", tools: ["Read"], prompt: "Inspect only.", source: "user" }], - host: { call: async (method) => { hostCalls.push(method); throw new Error("blocked tools must never reach the host"); }, onNotification: () => () => {} }, + host: { call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + hostCalls.push(method); throw new Error("blocked tools must never reach the host"); + }, onNotification: () => () => {} }, onEvent: (event) => { events.push(event); }, }); runtime.setMode(mode); diff --git a/packages/agent-runtime/src/model-capabilities.ts b/packages/agent-runtime/src/model-capabilities.ts index dda3a9bce..43e54643c 100644 --- a/packages/agent-runtime/src/model-capabilities.ts +++ b/packages/agent-runtime/src/model-capabilities.ts @@ -7,6 +7,7 @@ import { type ModelModality, } from "@pi-desktop/shared"; import type { ModelConfig, ThinkingCapabilitySet } from "./thinking-level.js"; +export { transcriptConfigFromPi } from "./transcript-compat.js"; export { agentThinkingLevel, @@ -24,6 +25,7 @@ export function modelConfigFromPi(model: Model): ModelConfig { ...metadata, ...(compat ? { compat: { ...compat } } : {}), source: "pi", + transcriptBinding: { modelId: model.id, api: model.api, baseUrl: model.baseUrl }, nativeCost: cost, // Desktop's historical tier schema differs; do not invent a translation. cost: { input: cost.input, output: cost.output, cacheRead: cost.cacheRead, cacheWrite: cost.cacheWrite }, diff --git a/packages/agent-runtime/src/output-cap.test.ts b/packages/agent-runtime/src/output-cap.test.ts index 0faaf69d8..234a118a5 100644 --- a/packages/agent-runtime/src/output-cap.test.ts +++ b/packages/agent-runtime/src/output-cap.test.ts @@ -352,3 +352,11 @@ describe("clampOutputToContext", () => { } }); }); + +// Transcript-only Pi requests have no legacy systemPrompt/tools fields. +it("counts instruction sections and tool declarations in transcript-only requests", () => { + const base = { messages: [{ role: "system", content: "" }] }; + const state = { messages: [{ role: "system", content: "", sections: { skills: "中文说明" }, toolsAdded: [{ name: "Read", parameters: { type: "object" } }] }] }; + expect(estimateOutputCapInputTokens(state)).toBeGreaterThan(estimateOutputCapInputTokens(base)); + expect(estimateOutputCapInputTokens(state)).toBeGreaterThan(4); +}); diff --git a/packages/agent-runtime/src/output-cap.ts b/packages/agent-runtime/src/output-cap.ts index ccc46f05f..728d05858 100644 --- a/packages/agent-runtime/src/output-cap.ts +++ b/packages/agent-runtime/src/output-cap.ts @@ -46,6 +46,9 @@ export type OutputCapContext = { export type OutputCapMessage = { role: string; content: unknown; + sections?: Record; + toolsAdded?: readonly unknown[]; + toolsRemoved?: readonly unknown[]; api?: Api; provider?: string; model?: string; @@ -224,6 +227,14 @@ export function estimateOutputCapInputTokens( message.content, replayOptionsFor(message, model), ); + if (message.role === "system") { + // Canonical transcript requests carry instruction sections and schema + // updates on the message, without a separate systemPrompt/tools field. + const state = [message.sections, message.toolsAdded, message.toolsRemoved] + .filter((value) => value !== undefined).map(stringifyForEstimate).join("\n"); + baseline += Math.ceil(state.length / CHARS_PER_TOKEN); + cjkChars += countCjkChars(state); + } baseline += estimated.baseline; cjkChars += estimated.cjkChars; } diff --git a/packages/agent-runtime/src/pi-runtime-messages.ts b/packages/agent-runtime/src/pi-runtime-messages.ts index a595da960..eace63930 100644 --- a/packages/agent-runtime/src/pi-runtime-messages.ts +++ b/packages/agent-runtime/src/pi-runtime-messages.ts @@ -1,6 +1,7 @@ import type { Message } from "@earendil-works/pi-ai"; import { hostedSearchReplayProjection } from "@earendil-works/pi-ai/utils/hosted-search"; import type { AgentMessage } from "@earendil-works/pi-agent-core"; +import { TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; export type CompactionSummaryMessage = { role: "compactionSummary"; @@ -136,6 +137,20 @@ export function convertToLlm(messages: AgentMessage[]): Message[] { })); break; case "system": + // Activation is Desktop execution metadata, not a model instruction. + // Persist it canonically, but omit it before Pi folds system messages: + // otherwise each ToolSearch rewrites the prefix on non-native APIs. + if (runtimeMessage.sections && TOOL_ACTIVATION_SECTION in runtimeMessage.sections) { + const sections = Object.fromEntries(Object.entries(runtimeMessage.sections) + .filter(([name]) => name !== TOOL_ACTIVATION_SECTION)); + if (Object.keys(sections).length || textFromContent(runtimeMessage.content) + || runtimeMessage.toolsAdded?.length || runtimeMessage.toolsRemoved?.length) { + converted.push(asProviderMessage({ ...runtimeMessage, sections })); + } + break; + } + converted.push(asProviderMessage(runtimeMessage)); + break; case "user": case "assistant": case "toolResult": diff --git a/packages/agent-runtime/src/plugin-skills-prompt.test.ts b/packages/agent-runtime/src/plugin-skills-prompt.test.ts index 9060c0f77..c2a63d4f1 100644 --- a/packages/agent-runtime/src/plugin-skills-prompt.test.ts +++ b/packages/agent-runtime/src/plugin-skills-prompt.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "vitest"; import { pluginSkillsPrompt, + pluginSkillsPromptSections, SKILL_TOOL_NAME, type PluginSkillDef, } from "./plugin-skills-prompt.js"; @@ -38,3 +39,18 @@ describe("pluginSkillsPrompt", () => { expect(prompt.split("\n").filter((line) => line.startsWith("- "))).toHaveLength(1); }); }); + +it("gives skill IDs collision-free section names and stable catalog ordering", () => { + const entries = [ + { id: "__proto__", name: "Prototype" }, { id: "runtime", name: "Reserved" }, + { id: "skill:a", name: "Colon" }, { id: "a", name: "A" }, + ]; + const sections = pluginSkillsPromptSections(entries); + expect(sections).toEqual(pluginSkillsPromptSections([...entries].reverse())); + expect(Object.keys(sections)).toEqual(Object.keys(pluginSkillsPromptSections([...entries].reverse()))); + expect(sections["skill:__proto__"]).toContain("Prototype"); + expect(sections["skill:runtime"]).toContain("Reserved"); + expect(sections["skill:skill:a"]).toContain("Colon"); + expect(sections["skill:a"]).toContain("A"); + expect(pluginSkillsPromptSections([])).toEqual({}); +}); diff --git a/packages/agent-runtime/src/plugin-skills-prompt.ts b/packages/agent-runtime/src/plugin-skills-prompt.ts index 9416ca37a..73eb161f8 100644 --- a/packages/agent-runtime/src/plugin-skills-prompt.ts +++ b/packages/agent-runtime/src/plugin-skills-prompt.ts @@ -11,16 +11,30 @@ export const SKILL_TOOL_NAME = "Skill"; * through the `Skill` tool when the model decides a skill applies, so a long * document costs nothing until it is needed. */ +const SKILLS_HEADER = [ + "# Skills", + "", + `Plugins have taught you the following skills. Each entry is a set of instructions you can load with the \`${SKILL_TOOL_NAME}\` tool by passing its exact id. When a task matches a skill's description, load the skill first and follow it; do not guess at its content. Load each skill at most once per task.`, +].join("\n"); + +export const SKILL_SECTION_PREFIX = "skill:"; + +function skillEntry(skill: PluginSkillDef): string { + const description = skill.description?.trim(); + return `- \`${skill.id}\` — ${skill.name}${description ? `: ${description}` : ""}`; +} + +/** Stable entry identities let Pi append only the changed or revoked skill. */ +export function pluginSkillsPromptSections(skills: readonly PluginSkillDef[]): Record { + if (!skills.length) return {}; + return Object.fromEntries([ + ["skills", SKILLS_HEADER], + ...[...skills].sort((a, b) => a.id.localeCompare(b.id)) + .map((skill) => [`${SKILL_SECTION_PREFIX}${skill.id}`, skillEntry(skill)]), + ]); +} + export function pluginSkillsPrompt(skills: PluginSkillDef[]): string | undefined { if (!skills.length) return undefined; - return [ - "# Skills", - "", - `Plugins have taught you the following skills. Each entry is a set of instructions you can load with the \`${SKILL_TOOL_NAME}\` tool by passing its exact id. When a task matches a skill's description, load the skill first and follow it; do not guess at its content. Load each skill at most once per task.`, - "", - ...skills.map((skill) => { - const description = skill.description?.trim(); - return `- \`${skill.id}\` — ${skill.name}${description ? `: ${description}` : ""}`; - }), - ].join("\n"); + return `${SKILLS_HEADER}\n\n${skills.map(skillEntry).join("\n")}`; } diff --git a/packages/agent-runtime/src/plugin-skills.test.ts b/packages/agent-runtime/src/plugin-skills.test.ts index d9df1a82e..d87fd7b55 100644 --- a/packages/agent-runtime/src/plugin-skills.test.ts +++ b/packages/agent-runtime/src/plugin-skills.test.ts @@ -40,3 +40,9 @@ describe("pluginSkillsDigest", () => { expect(pluginSkillsDigest([{ id: "a", name: "A", description: "" }])).toBe(without); }); }); + +it("does not confuse delimiter-containing catalog fields", () => { + expect(pluginSkillsDigest([{ id: "a", name: "b:c", description: "d" }])).not.toBe( + pluginSkillsDigest([{ id: "a", name: "b", description: "c:d" }]), + ); +}); diff --git a/packages/agent-runtime/src/plugin-skills.ts b/packages/agent-runtime/src/plugin-skills.ts index 42baab695..4126b89f6 100644 --- a/packages/agent-runtime/src/plugin-skills.ts +++ b/packages/agent-runtime/src/plugin-skills.ts @@ -20,15 +20,11 @@ export type PluginSkillDef = { /** * Stable fingerprint of a skill catalog. * - * `AgentRuntime.matches()` compares this so enabling a plugin, revoking its - * prompt permission, or editing a skill's front matter starts a fresh runtime - * instead of reusing a session whose catalog is already stale. Bodies are not - * part of it: they never enter the prompt, and the `Skill` tool reads them - * fresh from disk on every call. + * The runtime compares this before appending a skills section update. Bodies + * are excluded: the Skill tool reads them through the permission-checked host + * bridge on each invocation. */ export function pluginSkillsDigest(skills?: PluginSkillDef[]): string { if (!skills?.length) return ""; - return skills - .map((skill) => `${skill.id}:${skill.name}:${skill.description ?? ""}`) - .join("|"); + return JSON.stringify(skills.map((skill) => [skill.id, skill.name, skill.description ?? ""])); } diff --git a/packages/agent-runtime/src/provider-binding.test.ts b/packages/agent-runtime/src/provider-binding.test.ts index bc4036443..ad755e70a 100644 --- a/packages/agent-runtime/src/provider-binding.test.ts +++ b/packages/agent-runtime/src/provider-binding.test.ts @@ -254,12 +254,9 @@ describe("Anthropic runtime endpoint", () => { .result(); expect(result.stopReason).toBe("error"); - expect( - Object.hasOwn(model.compat ?? {}, "supportsMidConvoSystemMessages"), - ).toBe(false); - expect( - Object.hasOwn(model.compat ?? {}, "supportsMidConvoToolChanges"), - ).toBe(false); + expect(model.compat).toMatchObject({ + supportsMidConvoSystemMessages: false, supportsMidConvoToolChanges: false, + }); expect(request?.headers.get("anthropic-beta") ?? "").not.toMatch( /mid-conversation-tool-changes|inline-tools/, ); diff --git a/packages/agent-runtime/src/provider-binding.ts b/packages/agent-runtime/src/provider-binding.ts index 5733dd2ee..c42d42b5f 100644 --- a/packages/agent-runtime/src/provider-binding.ts +++ b/packages/agent-runtime/src/provider-binding.ts @@ -1,3 +1,4 @@ +import { transcriptCompat } from "./transcript-compat.js"; /** * Provider/model wiring shared by the session runtime and its subagents. * @@ -312,7 +313,7 @@ export function buildProviderModel( const binding = apiBindingForProviderModel(provider); const catalog = provider.modelConfig; const catalogModel = catalog - ? (({ source: _source, nativeCost, ...model }) => ({ + ? (({ source: _source, transcriptBinding: _transcriptBinding, nativeCost, ...model }) => ({ ...model, ...(nativeCost ? { cost: nativeCost } : {}), }))(catalog) @@ -386,7 +387,7 @@ export function buildProviderModel( }) === "on" ? true : undefined, - ...(compat ? { compat } : {}), + compat: { ...compat, ...transcriptCompat(catalog, provider.modelId, binding.api, baseUrl) }, ...(Object.keys(modelHeaders).length > 0 ? { headers: modelHeaders } : {}), } as Model; } diff --git a/packages/agent-runtime/src/provider-certificate-flow.test.ts b/packages/agent-runtime/src/provider-certificate-flow.test.ts index b533cb1d1..c3f0d8911 100644 --- a/packages/agent-runtime/src/provider-certificate-flow.test.ts +++ b/packages/agent-runtime/src/provider-certificate-flow.test.ts @@ -29,7 +29,10 @@ it.each(["session", "delegate"])( const runtime = kind === "session" ? new DesktopAgentRuntime({ sessionId: "certificate-session", mode: "agent", provider, thinkingLevel: "off", onEvent, - host: { call: async () => { throw new Error("Unexpected host request"); } }, + host: { call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + throw new Error("Unexpected host request"); + } }, commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, }) : undefined; try { diff --git a/packages/agent-runtime/src/provider-recovery-flow.test.ts b/packages/agent-runtime/src/provider-recovery-flow.test.ts index 81dc0ec83..23d529818 100644 --- a/packages/agent-runtime/src/provider-recovery-flow.test.ts +++ b/packages/agent-runtime/src/provider-recovery-flow.test.ts @@ -85,6 +85,7 @@ function fixture(steps: Step[]) { }, host: { async call(method: string): Promise { + if (method === "session.appendMessage") return undefined as T; if (method !== "tools.execute") throw new Error(`Unexpected host method: ${method}`); reads++; diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index 638c3497f..ac3ef5dc8 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -294,10 +294,10 @@ describe("system transcript reconstruction", () => { { id: "old-assistant", role: "assistant", content: "Earlier answer", createdAt: new Date(before - 1_000).toISOString(), status: "complete" }, ] }); const internals = runtime as any; - const prefix = internals.agent.state.messages[0]; + const prefix = internals.agent.state.messages.at(-1); expect(prefix.timestamp).toBeGreaterThanOrEqual(before); const rebuilt = internals.rebuiltAgentContext(); - expect(rebuilt.messages[0]).toBe(prefix); + expect(rebuilt.messages.at(-1)).toBe(prefix); expect(getCurrentTools(rebuilt.messages)).toEqual(rebuilt.tools.map(toToolDeclaration)); await runtime.dispose(); }); @@ -319,7 +319,7 @@ describe("system transcript reconstruction", () => { vi.spyOn(agent, "continue").mockImplementation(async () => { expect(agent.state.systemPrompt).toContain(marker); expect(agent.state.systemPrompt.split("SECTION_MARKER")).toHaveLength(2); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "SECTION_MARKER" }); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "SECTION_MARKER" }); expect(getCurrentTools(agent.state.messages)).toEqual(tools); response.timestamp = agent.state.messages[0]!.timestamp + 10; agent.state.messages.push(response as any); @@ -330,9 +330,9 @@ describe("system transcript reconstruction", () => { expect(agent.continue).toHaveBeenCalledOnce(); expect(agent.state.systemPrompt).toBe(before); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "SECTION_MARKER" }); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "SECTION_MARKER" }); expect(getCurrentTools(agent.state.messages)).toEqual(tools); - expect(estimateTranscriptTokens(agent.state.messages as any).usageTokens).toBe(0); + expect(agent.state.messages.at(-1)?.timestamp).toBeGreaterThan(response.timestamp); await runtime.dispose(); }); }); @@ -1962,7 +1962,7 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { }, ], }); - const tools = (runtime as any).agent.state.tools as Array<{ name: string }>; + const tools = (runtime as unknown as { activeTools(): Array<{ name: string }> }).activeTools(); const names = tools.map((tool) => tool.name); expect(names).toEqual([ @@ -2162,7 +2162,7 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(result.details.activated).toEqual(["BrowserPreview"]); expect(result.details.addedToolNames).toEqual(["BrowserPreview"]); expect(agent.state.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( - false, + true, ); const next = await (runtime as any).prepareNextTurn({ @@ -2184,10 +2184,8 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(next.context.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Tool deltas append new declarations; catalog order is not semantic. - expect([...getCurrentTools(next.context.messages)].sort((a, b) => a.name.localeCompare(b.name))).toEqual( - next.context.tools.map(toToolDeclaration).sort((a: { name: string }, b: { name: string }) => a.name.localeCompare(b.name)), - ); + // The full catalog is stable; preparation updates execution activation. + expect(getCurrentTools(next.context.messages).some((tool) => tool.name === "BrowserPreview")).toBe(true); await runtime.dispose(); }); @@ -2215,11 +2213,9 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(agent.state.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Tool deltas append new declarations; catalog order is not semantic. - const byName = (a: { name: string }, b: { name: string }) => a.name.localeCompare(b.name); - expect([...getCurrentTools(agent.state.messages)].sort(byName)).toEqual( - agent.state.tools.map(toToolDeclaration).sort(byName), - ); + // The full schema was already declared; activation remains sticky. + expect(getCurrentTools(agent.state.messages).some((tool) => tool.name === "BrowserPreview")).toBe(true); + await runtime.dispose(); }); }); @@ -2983,7 +2979,7 @@ describe("DesktopAgentRuntime plan transitions", () => { // The progress assistant is visible in the reused bubble but must be // removed before continue() rebuilds the model context. expect(agent.state.messages.filter((message: any) => message.role !== "system")).toHaveLength(1); - expect(agent.state.messages.at(-1)?.role).toBe("user"); + expect(agent.state.messages.filter((message: AgentMessage) => message.role !== "system").at(-1)?.role).toBe("user"); expect(agent.state.systemPrompt).toContain(""); } await handleAgentEvent({ type: "agent_start" }); @@ -3072,7 +3068,7 @@ describe("DesktopAgentRuntime plan transitions", () => { ]; } else { expect(agent.state.messages.filter((message: any) => message.role !== "system")).toHaveLength(1); - expect(agent.state.messages.at(-1)?.role).toBe("user"); + expect(agent.state.messages.filter((message: AgentMessage) => message.role !== "system").at(-1)?.role).toBe("user"); expect(agent.state.systemPrompt).toContain( attempts === 2 ? "" : "", ); @@ -4264,7 +4260,7 @@ describe("DesktopAgentRuntime assistant thinking events", () => { await runtime.dispose(); }); - it("reuses one assistant bubble across a successful overflow recovery", async () => { + it.each([false, true])("reuses one assistant bubble across overflow recovery (trailing system update: %s)", async (trailingSystem) => { const onEvent = vi.fn(); const runtime = createRuntime({ onEvent }); const agent = (runtime as any).agent; @@ -4283,8 +4279,9 @@ describe("DesktopAgentRuntime assistant thinking events", () => { stopReason: "stop", }); + const systemUpdate = { role: "system", content: "", timestamp: 3, sections: { context_budget: null } }; agent.prompt = vi.fn(async () => { - agent.state.messages = [user, failed]; + agent.state.messages = [user, failed, ...(trailingSystem ? [systemUpdate] : [])]; await handleAgentEvent({ type: "message_start", message: failed }); await handleAgentEvent({ type: "message_end", message: failed }); await handleAgentEvent({ type: "turn_end" }); @@ -4295,6 +4292,7 @@ describe("DesktopAgentRuntime assistant thinking events", () => { agent.continue = vi.fn(async () => { expect((runtime as any).overflowRecoveryInProgress).toBe(true); expect(agent.state.messages.filter((message: any) => message.role !== "system")).toEqual([user]); + if (trailingSystem) expect(agent.state.messages).toContain(systemUpdate); await handleAgentEvent({ type: "agent_start" }); await handleAgentEvent({ type: "turn_start" }); await handleAgentEvent({ @@ -5472,7 +5470,7 @@ describe("DesktopAgentRuntime compaction restore", () => { const estimate = estimateAgentContextTokens((runtime as any).agent.state.messages); expect(estimate.usageTokens).toBe(0); expect(estimate.lastUsageIndex).toBeNull(); - expect(budget.tokens).toBeGreaterThan(estimate.tokens); + expect(budget.tokens).toBeGreaterThanOrEqual(estimate.tokens); expect(budget.tokens).toBeGreaterThan(0); expect(budget.tokens).toBeLessThan(250_000); await runtime.dispose(); @@ -5581,8 +5579,15 @@ describe("DesktopAgentRuntime per-turn context protection", () => { // compaction trigger, so the model can close out before host compaction. expect(below.context.systemPrompt).toContain(""); expect(below.context.systemPrompt).toContain("new_context"); + expect(below.context.messages.at(-1)).toMatchObject({ + role: "system", + content: "", + sections: { context_budget: expect.stringContaining("") }, + }); + expect(getCurrentSystemMessage(below.context.messages)?.sections?.context_budget).toContain("new_context"); expect(stillBelow.context.systemPrompt).not.toContain(""); - // The reminder rides on the turn's context only; nothing is persisted. + // The legacy prompt baseline stays unchanged; the returned transcript + // carries the reminder section through the normal persistence path. expect((runtime as any).agent.state.systemPrompt).not.toContain( "", ); @@ -6849,17 +6854,18 @@ describe("DesktopAgentRuntime inline context compaction", () => { ); const generateCompaction = vi.spyOn(runtime as any, "generateCompaction"); const agent = (runtime as any).agent as Agent; - const prefix = { ...agent.state.messages[0], sections: { rules: "Keep checkpoint rules" } }; - agent.state.messages[0] = prefix as any; + const index = agent.state.messages.findIndex((message) => message.role === "system"); + const prefix = { ...agent.state.messages[index], sections: { rules: "Keep checkpoint rules" } }; + agent.state.messages[index] = prefix as any; await (runtime as any).prepareNextTurn(nextTurn); // The point of this family: the window is bought back without paying for a // summary, so no provider request is made at all. - expect(agent.state.messages[0]).toBe(prefix); + expect(agent.state.messages[0]).toMatchObject({ ...prefix, timestamp: expect.any(Number) }); expect(getCurrentTools(agent.state.messages)).toEqual(agent.state.tools.map(toToolDeclaration)); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "Keep checkpoint rules" }); - expect((runtime as any).rebuiltAgentContext().messages[0]).toBe(prefix); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "Keep checkpoint rules" }); + expect((runtime as any).rebuiltAgentContext().messages[0]).toMatchObject({ ...prefix, timestamp: expect.any(Number) }); expect(generateCompaction).not.toHaveBeenCalled(); const compaction = host.call.mock.calls.find( ([method]) => method === "session.appendCompaction", @@ -6876,7 +6882,7 @@ describe("DesktopAgentRuntime inline context compaction", () => { buildSessionContext((runtime as any).entriesWithCompaction()).messages.map( (message: any) => message.role, ), - ).toEqual(["compactionSummary"]); + ).toEqual(["system", "compactionSummary"]); const events = onEvent.mock.calls.map(([envelope]) => (envelope as any).event); expect(events).toContainEqual( expect.objectContaining({ @@ -6968,23 +6974,23 @@ describe("DesktopAgentRuntime plugin skills (D174)", () => { await runtime.dispose(); }); - it("does not reuse a runtime whose skill catalog changed", async () => { + it("reuses an idle runtime when the skill catalog changes", async () => { const runtime = createRuntime({ pluginSkills }); expect(runtimeMatches(runtime, { pluginSkills })).toBe(true); // Revoking agent.prompt.inject empties the catalog. - expect(runtimeMatches(runtime, { pluginSkills: [] })).toBe(false); + expect(runtimeMatches(runtime, { pluginSkills: [] })).toBe(true); expect( runtimeMatches(runtime, { pluginSkills: [...pluginSkills, { id: "demo.hello/other", name: "Other" }], }), - ).toBe(false); + ).toBe(true); // A renamed skill rewrites the catalog line the model reads. expect( runtimeMatches(runtime, { pluginSkills: [{ ...pluginSkills[0], name: "Renamed" }], }), - ).toBe(false); + ).toBe(true); await runtime.dispose(); }); @@ -9031,7 +9037,7 @@ describe("DesktopAgentRuntime deferred tool restore (#225)", () => { status: "complete", }; const hasTool = (runtime: DesktopAgentRuntime, name: string) => - (runtime as any).agent.state.tools.some((tool: any) => tool.name === name); + (runtime as unknown as { activeTools(): Array<{ name: string }> }).activeTools().some((tool) => tool.name === name); it("keeps a tool active across prompts while its ToolSearch activation is in context", async () => { const runtime = createRuntime({ history: [assistantRow, searchRow()] }); @@ -9157,31 +9163,29 @@ describe("DesktopAgentRuntime deferred tool restore (#225)", () => { (runtime as any).resetDeferredToolsForPrompt(); expect(hasTool(runtime, "BrowserPreview")).toBe(false); - (runtime as any).fullEntries.push( - ...(createRuntime({ - history: [ - assistantRow, - { - id: "tool-preview-ok", - role: "tool", - content: "", - createdAt: now(), - status: "complete", - toolName: "BrowserPreview", - toolCallId: "call-preview-ok", - toolStatus: "success", - toolArgs: {}, - toolResult: { content: [{ type: "text", text: "opened" }] }, - }, - ], - }) as any).fullEntries, - ); - (runtime as any).resetDeferredToolsForPrompt(); - expect(hasTool(runtime, "BrowserPreview")).toBe(true); + const restored = createRuntime({ + history: [ + assistantRow, + { + id: "tool-preview-ok", + role: "tool", + content: "", + createdAt: now(), + status: "complete", + toolName: "BrowserPreview", + toolCallId: "call-preview-ok", + toolStatus: "success", + toolArgs: {}, + toolResult: { content: [{ type: "text", text: "opened" }] }, + }, + ], + }); + expect(hasTool(restored, "BrowserPreview")).toBe(true); + await restored.dispose(); await runtime.dispose(); }); - it("restores the activation again after a mode round trip", async () => { + it("requires activation again after a fixed-catalog mode round trip", async () => { const runtime = createRuntime({ history: [assistantRow, searchRow()] }); (runtime as any).resetDeferredToolsForPrompt(); expect(hasTool(runtime, "BrowserPreview")).toBe(true); @@ -9189,7 +9193,7 @@ describe("DesktopAgentRuntime deferred tool restore (#225)", () => { runtime.setMode("plan"); runtime.setMode("agent"); - expect(hasTool(runtime, "BrowserPreview")).toBe(true); + expect(hasTool(runtime, "BrowserPreview")).toBe(false); await runtime.dispose(); }); }); @@ -10741,3 +10745,13 @@ describe("toolResultFromUi image restoration (issue #1073)", () => { expect(restored.content).toEqual([{ type: "text", text: expect.stringContaining("broken.png") }]); }); }); + +it("does not reuse stale plugin declarations when schema or permission metadata changes", async () => { + const plugin: PluginToolDef = { name: "plugin_fixture", description: "Inspect", parameters: { type: "object", properties: { path: { type: "string" } } } }; + const runtime = createRuntime({ pluginTools: [plugin] }); + try { + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin }] })).toBe(true); + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin, parameters: { type: "object", properties: { file: { type: "string" } } } }] })).toBe(false); + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin, planSafeActions: ["inspect"] }] })).toBe(false); + } finally { await runtime.dispose(); } +}); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 40ebe27f9..e1d3cf82c 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -1,3 +1,5 @@ +import { TOOL_ACTIVATION_SECTION, toolDeclarationPolicy, toolActivationSection, restoredToolActivation, syncToolActivation, type ToolDeclarationPolicy } from "./fixed-tool-declarations.js"; +import { orderSystemRows, SystemTranscriptJournal } from "./system-transcript-journal.js"; import { planWorkspaceRequiredResult } from "./plan-workspace-error.js"; import { accountModelStream } from "./request-usage.js"; import { modeToolDenial, retainModeToolDeclaration, withModeExecutionGuard } from "./mode-tool-access.js"; @@ -29,6 +31,7 @@ import { } from "@earendil-works/pi-agent-core"; import { isContextOverflow, + getCurrentTools, Type, type Api, type AssistantMessage, @@ -139,9 +142,12 @@ import type { } from "./pi-runtime-types.js"; import { initialSystemTranscript, + CONTEXT_BUDGET_SECTION, + syncSystemSections, + systemTranscriptCheckpoint, + removeTrailingAssistantMessages, rebuildSystemTranscript, replaceSystemPrompt, - syncSystemTools, systemPromptContent, } from "./system-transcript.js"; import { @@ -208,6 +214,7 @@ import type { CustomSystemPrompt } from "./custom-system-prompt.js"; import { projectMemoryPrompt } from "./project-memory-prompt.js"; import { pluginSkillsPrompt, + pluginSkillsPromptSections, SKILL_TOOL_NAME, type PluginSkillDef, } from "./plugin-skills-prompt.js"; @@ -1730,8 +1737,11 @@ export class DesktopAgentRuntime { private toolCatalog = new Map(); /** Tools intentionally omitted from the initial provider request. */ private deferredToolNames = new Set(); - /** Deferred tools loaded for the current user prompt. */ + /** Deferred tools activated for this runtime and declaration epoch. */ private activeDeferredToolNames = new Set(); + private declarationPolicy?: ToolDeclarationPolicy; + private trackToolActivation = false; + private activationHydrated = false; private scratchDir?: string; private projectPath?: string; private commandShell: CommandShellOption; @@ -1841,6 +1851,8 @@ export class DesktopAgentRuntime { }; private terminatingToolCalls = new Set(); private fullEntries: MessageEntry[]; + private readonly systemJournal = new SystemTranscriptJournal(); + private composedSections: Record = {}; private activeCompaction?: ContextCompactionRecord; private compactionEnabled: boolean; private readonly compactionStrategy: CompactionStrategy; @@ -1894,7 +1906,7 @@ export class DesktopAgentRuntime { this.streamSink = createStreamCoalescer(opts.onEvent); this.onEvent = (envelope) => this.streamSink.push(envelope); this.pluginTools = opts.pluginTools ?? []; - this.pluginSkills = opts.pluginSkills ?? []; + this.pluginSkills = [...(opts.pluginSkills ?? [])].sort((left, right) => left.id.localeCompare(right.id)); this.trustedExtensionSpecs = opts.trustedExtensions ?? []; this.subagents = opts.subagents ?? []; this.subagentProviders = opts.subagentProviders ?? {}; @@ -1918,7 +1930,6 @@ export class DesktopAgentRuntime { this.rebuildToolCatalog(); const model = buildProviderModel(this.provider); this.model = model; - const tools = this.activeTools(); const models = createProviderModels(this.provider, model); this.models = models; const runtimeApiKey = providerRequestKey(this.provider); @@ -1929,7 +1940,6 @@ export class DesktopAgentRuntime { this.delegationChains.hydrate( rebuildChainsFromTranscript(this.transcriptHistory), ); - const skillsPrompt = pluginSkillsPrompt(this.pluginSkills); // Parts: [0] is the product persona; [1:] are operational rules a custom // SYSTEM.md must not remove (tool guidance, delegation, scratch, skills). const defaultSystemPromptParts = [ @@ -1961,8 +1971,6 @@ Do not invent objections or turn speculative risks into blockers. Stop when the `Your scratch directory for this session is \`${formatScratchDirForShell(this.commandShell, this.scratchDir)}\` (in Bash: ${shellScratchVariable(this.commandShell)}). Store ad-hoc temporary and intermediate files there using absolute paths. Workspace writes must be task-related project files or required toolchain outputs. Scratch persists across turns and is deleted with the session.`, ] : []), - // Plugin skills (D174). - ...(skillsPrompt ? [skillsPrompt] : []), ]; // A custom SYSTEM.md replaces only the product persona line, never the // operational rules in the default parts: tool guidance, delegation @@ -1975,6 +1983,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ).trim(), ...defaultSystemPromptParts.slice(1), ].join("\n\n"); + this.refreshToolDeclarationPolicy(); + this.restoreDeferredToolsFromContext(); + const tools = this.declaredTools(); this.agent = new Agent({ streamFn: (m, context, options) => { this.setAgentActivity({ phase: "waiting-model", since: Date.now() }); @@ -2106,19 +2117,24 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // The provider's rule that a tool-call id is unique is enforced here, on // the last view before the wire: the request is the only place it can be // guaranteed for both a rebuilt context and one that grew in this process. - convertToLlm: (messages) => - alignRetainedReasoningIdentity( - convertToLlm(this.dropDuplicateToolCalls(messages)), - this.reasoningReplayIdentity(), - ), + convertToLlm: async (messages) => { + await this.systemJournal.persist(messages, this.fullEntries, async (message) => { + await this.host.call("session.appendMessage", { + sessionId: this.sessionId, turnId: this.turnId, message, + }); + }); + return alignRetainedReasoningIdentity( + convertToLlm(this.dropDuplicateToolCalls(messages)), this.reasoningReplayIdentity(), + ); + }, prepareNextTurnWithContext: (context, signal) => this.prepareNextTurn(context, signal), afterToolCall: async (context) => this.afterToolCall(context), initialState: { model, - tools, + tools: [], thinkingLevel: agentThinkingLevel(this.thinkingLevel), - messages: initialSystemTranscript(this.composeSystemPrompt(), tools, this.liveSessionContext().messages), + messages: initialSystemTranscript(this.composeSystemPrompt(), tools, this.liveSessionContext().messages, this.composedSections), }, // Plan transitions must be the only tool call in an assistant batch. // Sequential execution also makes the host-confirmed mode change visible @@ -2142,6 +2158,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the }, }); + // Avoid Pi inserting a tool-only baseline ahead of restored legacy rows. + this.setAgentTools(tools); + // pi awaits every listener, so a throw here would reject the run in // progress and, with nothing awaiting that rejection, could take the whole // sidecar down. Contain it: log with the session attached and let the @@ -2169,6 +2188,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.infiniteProviderRetry = enabled; } + /** Refresh catalog instructions without changing the running session owner. */ + setPluginSkills(skills: PluginSkillDef[]): void { + if (this.disposed) throw new Error("runtime disposed"); + if (this.getStatus().isRunning) throw new Error("cannot update skills during an active turn"); + const sorted = [...skills].sort((left, right) => left.id.localeCompare(right.id)); + if (pluginSkillsDigest(this.pluginSkills) === pluginSkillsDigest(sorted)) return; + this.pluginSkills = sorted; + this.rebuildToolCatalog(); + this.setAgentTools(this.activeTools()); + this.setAgentSystemPrompt(this.composeSystemPrompt()); + } + /** Switch the planning state on this Agent without creating another Agent. */ setMode(mode: Mode): void { if (this.disposed) throw new Error("runtime disposed"); @@ -2209,7 +2240,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the (this.agent.state as unknown as { systemPrompt: string }).systemPrompt = prompt; return; } - this.agent.state.messages = replaceSystemPrompt(this.agent.state.messages, prompt); + this.agent.state.messages = prompt === this.composedSystemPrompt + ? syncSystemSections(this.agent.state.messages, this.composedSections) + : replaceSystemPrompt(this.agent.state.messages, prompt); } private setAgentMessages(messages: AgentMessage[]): void { @@ -2217,16 +2250,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.agent.state.messages = messages; return; } - this.agent.state.messages = syncSystemTools( - rebuildSystemTranscript(this.agent.state.messages, messages), - this.agent.state.tools, + this.agent.state.messages = rebuildSystemTranscript( + this.agent.state.messages.filter((message) => message.role !== "system" || !this.systemJournal.isPersisted(message)), messages, ); } private setAgentTools(tools: AgentTool[]): void { - this.agent.state.tools = tools; - if (this.agentUsesTranscriptSystemMessages()) { - this.agent.state.messages = syncSystemTools(this.agent.state.messages, tools); + this.agent.state.tools = this.declarationPolicy?.tools ?? tools; + if (this.trackToolActivation && this.declarationPolicy) { + const section = toolActivationSection(this.declarationPolicy.key, this.activeDeferredToolNames); + this.composedSections = { ...this.composedSections, [TOOL_ACTIVATION_SECTION]: section }; + this.composedSystemPrompt = Object.values(this.composedSections).filter(Boolean).join("\n\n"); + this.agent.state.messages = syncToolActivation(this.agent.state.messages, section); } } @@ -2246,17 +2281,21 @@ Do not invent objections or turn speculative risks into blockers. Stop when the runningDelegationIds: this.runningDelegationIds(), }) : ""; - const composed = composeModeSystemPrompt( - this.mode, - [ - this.baseSystemPrompt, + this.composedSections = { + runtime: this.baseSystemPrompt, + ...(this.trackToolActivation && this.declarationPolicy ? { + [TOOL_ACTIVATION_SECTION]: toolActivationSection(this.declarationPolicy.key, this.activeDeferredToolNames), + } : {}), + ...pluginSkillsPromptSections(this.pluginSkills), + context: composeModeSystemPrompt(this.mode, [ ...(this.customSystemPrompt?.append ? [this.customSystemPrompt.append] : []), ...(optionalToolsPrompt ? [optionalToolsPrompt] : []), ...(projectPrompt ? [projectPrompt] : []), ...(memoryPrompt ? [memoryPrompt] : []), ...(resumablePrompt ? [resumablePrompt] : []), - ].join("\n\n"), - ); + ].join("\n\n")), + }; + const composed = Object.values(this.composedSections).filter(Boolean).join("\n\n"); this.composedSystemPrompt = composed; return composed; } @@ -2407,6 +2446,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private async beforeToolCall( context: BeforeToolCallContext, ): Promise { + if (this.deferredToolNames.has(context.toolCall.name) && !this.activeDeferredToolNames.has(context.toolCall.name)) { + return { block: true, reason: `Call ${TOOL_SEARCH_NAME} to activate ${context.toolCall.name} before using it. A tool declaration does not grant execution permission.` }; + } const toolCalls = (context.assistantMessage.content as Array<{ type?: string }>).filter( (block) => block.type === "toolCall", ); @@ -2469,9 +2511,8 @@ Do not invent objections or turn speculative risks into blockers. Stop when the /** True when this runtime can be reused for a prompt with the given config. */ matches(config: RuntimeMatchConfig): boolean { const requestedPluginTools = config.pluginTools ?? []; - const requestedPluginSkills = config.pluginSkills ?? []; - const current = this.pluginTools.map((t) => t.name).sort().join(","); - const next = requestedPluginTools.map((t) => t.name).sort().join(","); + const current = safeJson([...this.pluginTools].sort((a, b) => a.name.localeCompare(b.name))); + const next = safeJson([...requestedPluginTools].sort((a, b) => a.name.localeCompare(b.name))); const currentThinkingLevels = [ ...(this.provider.supportedThinkingLevels ?? ["off"]), ] @@ -2514,10 +2555,6 @@ Do not invent objections or turn speculative risks into blockers. Stop when the safeJson(config.customSystemPrompt ?? null) && (this.projectMemory ?? "") === (config.projectMemory?.trim() ?? "") && (this.projectPath ?? "") === (config.projectPath?.trim() ?? "") && - // Enabling a plugin, revoking agent.prompt.inject or renaming a skill - // changes the catalog digest, which retires the runtime and its stale - // prompt. Bodies are excluded: the Skill tool always reads them fresh. - pluginSkillsDigest(this.pluginSkills) === pluginSkillsDigest(requestedPluginSkills) && // Editing `~/.agents/subagents/*.md` must reach the next prompt. Definition // bodies are part of the `Task` tool's behavior, so unlike skills they // are compared in full. @@ -2821,14 +2858,17 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // (the nearest one above), keeping each call adjacent to its result as // the provider APIs require. let toolCarrier: AssistantMessage | undefined; - for (const m of history) { + for (const m of orderSystemRows(history)) { // Subagent rows belong to the transcript and to review, never to the // parent's model context (ADR 0062): the parent only ever saw the `Task` // report, and replaying a delegate's messages would both contradict that // and reintroduce the context cost delegation exists to avoid. if (m.parentToolCallId) continue; const timestamp = Date.parse(m.createdAt) || Date.now(); - if (m.role === "user") { + if (m.role === "system" && m.modelSystem) { + toolCarrier = undefined; + append(m.id, this.systemJournal.restore(m)); + } else if (m.role === "user") { toolCarrier = undefined; const attachments = (m.attachments ?? []).map((attachment) => runtimeAttachmentFromMessage( @@ -2875,8 +2915,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the role: "assistant", content, api, - provider: this.provider.id, - model: this.provider.modelId, + // Persisted providerId identifies a local account; Pi compares its + // vendor identity to decide whether native thinking can be replayed. + // Keep known account/model switches distinct; legacy rows have no + // source identity and retain the existing current-model fallback. + provider: !m.providerId || m.providerId === this.provider.id + ? this.model.provider : m.providerId, + model: m.modelId || this.model.id, usage: usageToPi(m.usage), stopReason: "stop", timestamp, @@ -2892,8 +2937,8 @@ Do not invent objections or turn speculative risks into blockers. Stop when the role: "assistant", content: [], api, - provider: this.provider.id, - model: this.provider.modelId, + provider: this.model.provider, + model: this.model.id, usage: usageToPi(undefined), stopReason: "toolUse", timestamp, @@ -2944,6 +2989,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ): Entry[] { const entries: Entry[] = [...this.fullEntries]; if (!checkpoint) return entries; + if (isRecord(checkpoint.details) && checkpoint.details.systemMessageJson) { + this.systemJournal.rememberCheckpoint(checkpoint.details.systemMessageJson); + } const throughIndex = entries.findIndex( (entry) => entry.id === checkpoint.throughMessageId, ); @@ -3628,9 +3676,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } /** - * Build the complete registry once, then expose only the core subset to the - * first provider request. This mirrors pi's active-tool model while keeping - * the host tool implementation and permission path unchanged. + * Build the complete registry before selecting fixed or on-demand + * declarations. Execution activation and Host permissions remain separate + * from the provider-visible schemas. */ private rebuildToolCatalog(): void { const catalog = new Map(); @@ -3675,6 +3723,25 @@ Do not invent objections or turn speculative risks into blockers. Stop when the }), ); } + if (this.model) this.refreshToolDeclarationPolicy(); + } + + private refreshToolDeclarationPolicy(): void { + const previous = this.declarationPolicy; + const policy = toolDeclarationPolicy(this.model, [...this.toolCatalog.values()], this.deferredToolNames, this.composeSystemPrompt(), this.provider.id); + if (previous && previous.key !== policy.key && this.trackToolActivation) { + this.activeDeferredToolNames.clear(); + this.activationHydrated = true; + } + this.declarationPolicy = policy; + this.trackToolActivation ||= Boolean(policy.tools || policy.fallback); + if (policy.fallback && (previous?.key !== policy.key || previous.fallback !== policy.fallback)) { + process.stderr.write(`[agent-runtime] fixed tool declarations unavailable (${policy.fallback}); using on-demand declarations; ToolSearch cache stability is not guaranteed.\n`); + } + } + + private declaredTools(): AgentTool[] { + return this.declarationPolicy?.tools ?? this.activeTools(); } private isPlanSafePluginTool(name: string): boolean { @@ -3763,7 +3830,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } return [ "# On-demand tools", - `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using one that is not in the current tool list.`, + `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using an on-demand tool that has not been activated. A visible schema is not activation. Successful ToolSearch results identify activated tools; if unsure, search again.`, ...lines, ].join("\n"); } @@ -3793,7 +3860,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the name: TOOL_SEARCH_NAME, label: "Tool Search", description: - "Find and activate an on-demand tool by exact name or capability. Use this before calling any tool listed under On-demand tools that is not already in the current tool list.", + "Find and activate an on-demand tool by exact name or capability. Use this before calling an inactive tool listed under On-demand tools, even when its schema is already visible. Activation does not bypass approval or mode restrictions.", parameters: Type.Object({ query: Type.String({ description: @@ -5356,8 +5423,26 @@ Do not invent objections or turn speculative risks into blockers. Stop when the */ private restoreDeferredToolsFromContext(): void { if (this.deferredToolNames.size === 0) return; + if (this.trackToolActivation && this.activationHydrated) return; const { messages } = this.liveSessionContext(); - for (const message of messages) { + const restored = this.declarationPolicy && restoredToolActivation(messages, this.declarationPolicy.key); + if (restored !== undefined) { + this.trackToolActivation = true; + for (const name of restored.active) { + if (this.deferredToolNames.has(name)) this.activeDeferredToolNames.add(name); + } + } + this.activationHydrated = true; + const lastSystem = restored ? restored.replayFrom - 1 : messages.map((message) => message.role).lastIndexOf("system"); + if (!restored && lastSystem >= 0) { + for (const tool of getCurrentTools(messages)) { + if (this.deferredToolNames.has(tool.name)) this.activeDeferredToolNames.add(tool.name); + } + } + // Legacy histories lack declarations. For current histories, only results + // after the last declaration can represent an activation not yet declared. + for (const message of messages.slice(lastSystem + 1)) { + if (restored && (message.role !== "toolResult" || message.toolName !== TOOL_SEARCH_NAME)) continue; if (message.role !== "toolResult" || message.isError) continue; if (isMissingToolResultPlaceholder(message.content)) continue; const names = @@ -5977,8 +6062,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // agentLoopContinue refuses a transcript ending in an assistant message, // and this one carries nothing worth resending anyway. - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const promptBefore = this.agentSystemPromptContent(); @@ -6030,8 +6114,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.overflowRecoveryAttempted = true; this.overflowRecoveryInProgress = true; try { - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const compacted = await this.runCompaction( "overflow", @@ -6089,8 +6172,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // pi-agent-core refuses `continue()` when the transcript ends in an // assistant message. The progress text is already visible in the reused // bubble, so it must not be sent back as model context. - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const promptBefore = this.agentSystemPromptContent(); @@ -6145,7 +6227,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ...(typeof this.agent.state.systemPrompt === "string" ? { systemPrompt: this.agent.state.systemPrompt } : {}), - tools: this.activeTools(), + tools: this.declaredTools(), }, this.model, ); @@ -6318,7 +6400,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private rebuiltAgentContext(): AgentContext { const messages = this.liveSessionContext().messages; - const tools = this.activeTools(); + const tools = this.declaredTools(); this.setAgentMessages(messages); this.setAgentTools(tools); return { @@ -6416,12 +6498,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the return { context }; } - /** - * Codex's two-tier `maybe_record`, as a system-prompt append for this turn - * only. Codex writes its reminders into conversation history; we have no - * channel for a synthetic message that stays out of the transcript, and the - * append is equivalent without persisting anything. - */ + /** A replaceable reminder section expires when compaction opens a new window. */ private withContextBudgetReminder( context: AgentContext, budget: ContextBudget, @@ -6439,7 +6516,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ...(systemPrompt ? { systemPrompt } : {}), messages: [ ...context.messages, - { role: "system", content: reminder, timestamp: Date.now() }, + { role: "system", content: "", sections: { [CONTEXT_BUDGET_SECTION]: reminder }, timestamp: Date.now() }, ], } as AgentContext; } @@ -6886,6 +6963,11 @@ Do not invent objections or turn speculative risks into blockers. Stop when the mustFitSafeBudget: boolean, fallback?: ContextCompactionFallback, ): Promise { + const systemMessage = systemTranscriptCheckpoint(this.agent.state.messages); + checkpoint = { + ...checkpoint, + details: { ...(isRecord(checkpoint.details) ? checkpoint.details : {}), ...(systemMessage ? { systemMessageJson: JSON.stringify(systemMessage) } : {}) }, + }; const compactedBudget = this.contextBudget( this.liveSessionContext(checkpoint).messages, ); @@ -6911,7 +6993,10 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // resetting its `claim_*` flags when the context window turns over. this.contextReminderClaimed = false; this.contextFallbackReminderClaimed = false; - this.setAgentMessages(this.liveSessionContext().messages); + this.agent.state.messages = this.liveSessionContext().messages; + for (const message of this.agent.state.messages) { + if (message.role === "system") this.systemJournal.remember(message, checkpoint.id); + } this.emit({ type: "compaction_end", reason, diff --git a/packages/agent-runtime/src/session-context.ts b/packages/agent-runtime/src/session-context.ts index bf640bb67..d930c55be 100644 --- a/packages/agent-runtime/src/session-context.ts +++ b/packages/agent-runtime/src/session-context.ts @@ -1,3 +1,4 @@ +import { readSystemMessage } from "./system-transcript-journal.js"; /** * Project pi session entries into the model context. * @@ -58,6 +59,8 @@ export function sessionEntryToContextMessages( return isContextMessage(entry.message) ? [entry.message] : []; case "compaction": return [ + ...(entry.details && typeof entry.details === "object" && "systemMessageJson" in entry.details + ? [readSystemMessage(entry.details.systemMessageJson)] : []), createCompactionSummaryMessage( entry.summary, entry.tokensBefore, @@ -70,7 +73,8 @@ export function sessionEntryToContextMessages( entry.timestamp, identity, )), - ...entry.retainedTail.filter(isContextMessage), + ...entry.retainedTail.filter((message) => isContextMessage(message) && + !(message.role === "system" && entry.details && typeof entry.details === "object" && "systemMessageJson" in entry.details)), ]; case "branch_summary": return entry.summary diff --git a/packages/agent-runtime/src/sidecar.ts b/packages/agent-runtime/src/sidecar.ts index ce7db0b7b..7bc2126dc 100644 --- a/packages/agent-runtime/src/sidecar.ts +++ b/packages/agent-runtime/src/sidecar.ts @@ -241,6 +241,7 @@ async function runtimeFor( runtimes.delete(sessionId); } if (reusable) { + reusable.setPluginSkills(pluginSkills); reusable.setCompactionSettings(params.compactionSettings); reusable.setInfiniteProviderRetry(params.infiniteProviderRetry === true); reusable.setMode(mode); @@ -260,10 +261,9 @@ async function runtimeFor( // The current prompt is sent separately below. Exclude its persisted row // before attachment hydration so it cannot consume the history byte budget. if (currentPrompt !== undefined && params.userMessageId) { - const last = restoredMessages.at(-1); - if (last?.role === "user" && last.id === params.userMessageId) { - restoredMessages = restoredMessages.slice(0, -1); - } + restoredMessages = restoredMessages.filter((message) => + message.role !== "user" || message.id !== params.userMessageId, + ); } const supportsVision = visionFromModelConfig(params.provider.modelConfig); history = await hydrateAttachmentHistory(restoredMessages, { diff --git a/packages/agent-runtime/src/system-transcript-journal.test.ts b/packages/agent-runtime/src/system-transcript-journal.test.ts new file mode 100644 index 000000000..59c3e815c --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-journal.test.ts @@ -0,0 +1,97 @@ +import { describe, expect, it, vi } from "vitest"; +import type { AgentMessage } from "@earendil-works/pi-agent-core"; +import type { MessageEntry, CompactionEntry } from "./pi-runtime-types.js"; +import type { UiMessage } from "@pi-desktop/shared"; +import { getCurrentTools, Type, type SystemMessage } from "@earendil-works/pi-ai"; +import { orderSystemRows, readSystemMessage, SystemTranscriptJournal } from "./system-transcript-journal.js"; +import { buildSessionContext } from "./session-context.js"; +import { CONTEXT_BUDGET_SECTION, syncSystemSections, systemTranscriptCheckpoint } from "./system-transcript.js"; + +const tool = { name: "Read", description: "Read", parameters: Type.Object({ path: Type.String() }) }; +const initial: SystemMessage = { role: "system", content: "", sections: { runtime: "Instructions", skills: "First skill" }, toolsAdded: [tool], timestamp: 1 }; +const user: AgentMessage = { role: "user", content: "Find files", timestamp: 2 }; +function entry(message: AgentMessage, id = "user"): MessageEntry { + return { type: "message", id, seq: 0, parentId: null, timestamp: message.timestamp, message }; +} +const userRow: UiMessage = { id: "user", role: "user", content: "Find files", createdAt: new Date(2).toISOString() }; + +describe("model system journal", () => { + it("restores the original order even when a user row was pre-persisted", async () => { + const journal = new SystemTranscriptJournal(); + const rows = [userRow]; + const entries = [entry(user)]; + const delta: SystemMessage = { role: "system", content: "", toolsRemoved: [{ name: "Read" }], timestamp: 3 }; + await journal.persist([initial, user, delta], entries, async (row) => { rows.push(row); }); + expect(rows.map((row) => row.role)).toEqual(["user", "system", "system"]); + const restored = new SystemTranscriptJournal(); + const messages = orderSystemRows(rows).map((row) => row.modelSystem ? restored.restore(row) : user); + expect(messages).toEqual([initial, user, delta]); + expect(getCurrentTools(messages)).toEqual([]); + const append = vi.fn(); + await restored.persist(messages, entries, append); + expect(append).not.toHaveBeenCalled(); + }); + + it("retries failed persistence with the same id and never installs it before acknowledgement", async () => { + const journal = new SystemTranscriptJournal(); + const entries = [entry(user)]; + const rows: UiMessage[] = []; + await expect(journal.persist([initial, user], entries, async (row) => { + rows.push(row); throw new Error("disk full"); + })).rejects.toMatchObject({ code: "LOCAL_REQUEST_ERROR", cause: new Error("disk full") }); + expect(entries).toHaveLength(1); + await journal.persist([initial, user], entries, async (row) => { rows.push(row); }); + expect(rows[0].id).toBe(rows[1].id); + expect(entries.map((item) => item.message.role)).toEqual(["system", "user"]); + }); + + it("replaces skill sections without rewriting the old prefix", () => { + const original = [initial, user]; + const changed = syncSystemSections(original, { runtime: "Instructions", skills: "Second skill" }); + expect(changed.slice(0, 2)).toEqual(original); + expect(changed.at(-1)).toMatchObject({ sections: { skills: "Second skill" } }); + expect(syncSystemSections(changed, { runtime: "Instructions", skills: "Second skill" })).toBe(changed); + expect(syncSystemSections(changed, { runtime: "Instructions", skills: "" }).at(-1)).toMatchObject({ sections: { skills: "" } }); + }); + + it("restores a compaction checkpoint once before the retained tail", async () => { + const delta: SystemMessage = { role: "system", content: "", toolsRemoved: [{ name: "Read" }], timestamp: 3 }; + const systemMessage = systemTranscriptCheckpoint([initial, user, delta])!; + const compaction: CompactionEntry = { + type: "compaction", id: "cp", seq: 1, parentId: "user", timestamp: 4, + summary: "Past work", tokensBefore: 100, retainedTail: [initial, user, delta], + details: { systemMessageJson: JSON.stringify(systemMessage) }, fromHook: false, + }; + const messages = buildSessionContext([entry(user), compaction]).messages; + expect(messages.filter((message) => message.role === "system")).toEqual([systemMessage]); + expect(messages[0]).toEqual(systemMessage); + expect(getCurrentTools(messages)).toEqual([]); + const journal = new SystemTranscriptJournal(); + journal.rememberCheckpoint(systemMessage); + const append = vi.fn(); + await journal.persist(JSON.parse(JSON.stringify(messages)), [], append); + expect(append).not.toHaveBeenCalled(); + }); + + it("expires budget reminders at compaction without dropping active instructions or tools", () => { + const checkpoint = systemTranscriptCheckpoint([initial, user, { + role: "system", content: "", sections: { [CONTEXT_BUDGET_SECTION]: "Old window is nearly full" }, timestamp: 3, + }]); + expect(checkpoint?.sections).toEqual(initial.sections); + expect(checkpoint?.toolsAdded).toEqual(initial.toolsAdded); + }); + + it("recognizes a restored checkpoint with text blocks without adding durable rows", async () => { + const blocks: SystemMessage = { ...initial, content: [{ type: "text", text: "Extension instructions" }] }; + const checkpoint = systemTranscriptCheckpoint([blocks, user])!; + const journal = new SystemTranscriptJournal(); + journal.rememberCheckpoint(JSON.stringify(checkpoint)); + const append = vi.fn(); + await journal.persist(JSON.parse(JSON.stringify([checkpoint, user])), [], append); + expect(append).not.toHaveBeenCalled(); + }); + + it.each([null, {}, { ...initial, timestamp: -1 }, { ...initial, toolsAdded: [{ name: "Read", parameters: "bad" }] }])("rejects malformed saved state", (value) => { + expect(() => readSystemMessage(value)).toThrow("Invalid persisted model system message"); + }); +}); diff --git a/packages/agent-runtime/src/system-transcript-journal.ts b/packages/agent-runtime/src/system-transcript-journal.ts new file mode 100644 index 000000000..e3f7492cd --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-journal.ts @@ -0,0 +1,118 @@ +import { randomUUID } from "node:crypto"; +import { LocalRequestError, Type, contentText, toToolDeclaration, type SystemMessage } from "@earendil-works/pi-ai"; +import type { AgentMessage } from "@earendil-works/pi-agent-core"; +import type { MessageEntry } from "./pi-runtime-types.js"; +import type { UiMessage } from "@pi-desktop/shared"; +import * as Value from "typebox/value"; + +const systemSchema = Type.Object({ + role: Type.Literal("system"), + content: Type.String(), + timestamp: Type.Number({ minimum: 0, maximum: 8.64e15 }), + sections: Type.Optional(Type.Record(Type.String(), Type.Union([Type.String(), Type.Null()]))), + toolsAdded: Type.Optional(Type.Array(Type.Object({ + name: Type.String({ minLength: 1 }), description: Type.String(), + parameters: Type.Record(Type.String(), Type.Unknown()), + constrainedSampling: Type.Optional(Type.Union([ + Type.Literal(false), + Type.Object({ type: Type.Literal("json_schema"), strict: Type.Union([Type.Literal("prefer"), Type.Literal("require")]) }), + Type.Object({ type: Type.Literal("grammar"), variants: Type.Record(Type.String(), Type.Unknown()) }), + ])), + }))), + toolsRemoved: Type.Optional(Type.Array(Type.Object({ name: Type.String({ minLength: 1 }) }))), +}); + +export function readSystemMessage(value: unknown): SystemMessage { + if (typeof value === "string") value = JSON.parse(value); + if (!Value.Check(systemSchema, value)) throw new Error("Invalid persisted model system message"); + // Tool parameter schemas and grammar variants are JSON objects at this boundary. + return value as SystemMessage; +} + +/** Reorder only internal rows, whose following user row may have arrived first. */ +export function orderSystemRows(history: readonly UiMessage[]): UiMessage[] { + const rows = history.filter((row) => !row.modelSystem); + for (const row of history) { + if (!row.modelSystem) continue; + if (row.role !== "system" || row.modelSystem.version !== 1) { + throw new Error("Invalid model system record"); + } + const before = row.modelSystem.beforeMessageId; + let index = before ? rows.findIndex((candidate) => candidate.id === before) : -1; + if (index < 0 && row.modelSystem.afterMessageId) { + const previous = rows.findIndex((candidate) => candidate.id === row.modelSystem?.afterMessageId); + if (previous >= 0) { + index = previous + 1; + while (rows[index]?.modelSystem) index++; + } + } + rows.splice(index < 0 ? rows.length : index, 0, row); + } + return rows; +} + +/** Persist provider-neutral declarations before dispatch; never persist a folded request. */ +export class SystemTranscriptJournal { + private readonly ids = new WeakMap(); + private readonly pending = new WeakMap(); + private readonly checkpoints = new Set(); + + restore(row: UiMessage): SystemMessage { + const message = readSystemMessage(row.modelSystem?.messageJson); + this.ids.set(message, row.id); + return message; + } + + isPersisted(message: AgentMessage): boolean { + return this.ids.has(message) || this.checkpoints.has(JSON.stringify(message)); + } + + async persist( + messages: readonly AgentMessage[], + entries: MessageEntry[], + append: (row: UiMessage) => Promise, + ): Promise { + for (let index = 0; index < messages.length; index++) { + const message = messages[index]; + if (message.role !== "system" || this.isPersisted(message)) continue; + const following = messages.slice(index + 1).find((candidate) => candidate.role !== "system"); + const nextEntry = following && entries.find((entry) => entry.message === following); + const preceding = messages.slice(0, index).reverse().find((candidate) => candidate.role !== "system"); + const previousEntry = preceding && entries.find((entry) => entry.message === preceding); + const id = this.pending.get(message) ?? randomUUID(); + this.pending.set(message, id); + const serialized: SystemMessage = { + ...message, content: contentText(message.content), + ...(message.toolsAdded ? { toolsAdded: message.toolsAdded.map(toToolDeclaration) } : {}), + }; + const row: UiMessage = { + id, role: "system", content: "", createdAt: new Date(message.timestamp).toISOString(), + modelSystem: { version: 1, messageJson: JSON.stringify(serialized), ...(nextEntry ? { beforeMessageId: nextEntry.id } : {}), ...(previousEntry ? { afterMessageId: previousEntry.id } : {}) }, + }; + try { + await append(row); + } catch (cause) { + // Stop before dispatch. Retrying the provider cannot repair a failed + // durable write, and the original host error may contain private paths. + throw new LocalRequestError("request-preparation", { cause }); + } + this.ids.set(message, id); + this.pending.delete(message); + const entry: MessageEntry = { + type: "message", id, seq: entries.length, parentId: null, + timestamp: message.timestamp, message, + }; + const position = nextEntry ? entries.indexOf(nextEntry) : entries.length; + entries.splice(position, 0, entry); + entries.forEach((item, seq) => { item.seq = seq; item.parentId = entries[seq - 1]?.id ?? null; }); + } + } + + rememberCheckpoint(message: unknown): void { + this.checkpoints.add(JSON.stringify(readSystemMessage(message))); + } + + remember(message: AgentMessage, id: string): void { + this.ids.set(message, id); + } +} diff --git a/packages/agent-runtime/src/system-transcript-order.test.ts b/packages/agent-runtime/src/system-transcript-order.test.ts new file mode 100644 index 000000000..65e127c11 --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-order.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from "vitest"; +import { Agent, type AgentTool, type AgentMessage } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, Type, getCurrentTools, type SystemMessage } from "@earendil-works/pi-ai"; +import { rebuildSystemTranscript } from "./system-transcript.js"; + +const read = { name: "Read", description: "Read a file", parameters: Type.Object({}) }; +const search = { name: "Search", description: "Search files", parameters: Type.Object({}) }; +const initial: SystemMessage = { role: "system", content: "Instructions", toolsAdded: [read], timestamp: 1 }; +const user: AgentMessage = { role: "user", content: "Find a file", timestamp: 2 }; +const result: AgentMessage = { + role: "toolResult", toolCallId: "search-1", toolName: "ToolSearch", + content: [{ type: "text", text: "Search activated" }], isError: false, timestamp: 3, +}; + +describe("chronological system updates", () => { + it("lets the Pi loop append schema changes and preserves the complete prefix", async () => { + const executable = (tool: typeof read): AgentTool => ({ ...tool, label: tool.name, execute: async () => ({ content: [], details: {} }) }); + const requests: AgentMessage[][] = []; + const agent = new Agent({ + initialState: { tools: [executable(read)], messages: [initial, user] }, + streamFn: async (_model, context) => { + requests.push([...context.messages]); + const stream = createAssistantMessageEventStream(); + stream.push({ type: "done", reason: "stop", message: { + role: "assistant", content: [{ type: "text", text: "Done" }], stopReason: "stop", timestamp: Date.now(), + api: "openai-completions", provider: "fixture", model: "fixture", + usage: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0, totalTokens: 2, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, + } }); + return stream; + }, + }); + await agent.continue(); + const before = [...agent.state.messages]; + const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; + agent.state.tools = [executable(replacement), executable(search)]; + await agent.prompt("Continue"); + expect(requests).toHaveLength(2); + expect(requests[1].slice(0, before.length)).toEqual(before); + expect(requests[1].slice(before.length)).toContainEqual(expect.objectContaining({ + role: "system", toolsAdded: [replacement, search], toolsRemoved: [{ name: "Read" }], + })); + expect(getCurrentTools(requests[1])).toEqual([replacement, search]); + await agent.prompt("Again"); + expect(requests[2].filter((message) => message.role === "system")).toHaveLength(2); + }); + + it("does not move a retained tool update ahead of its activation result", () => { + const update: SystemMessage = { role: "system", content: "", toolsAdded: [search], timestamp: 4 }; + const before = [initial, user, result, update]; + const after = rebuildSystemTranscript(before, [user, result]); + expect(after).toEqual(before); + expect(after[0]).toBe(initial); + expect(after.at(-1)).toBe(update); + }); +}); diff --git a/packages/agent-runtime/src/system-transcript-runtime.test.ts b/packages/agent-runtime/src/system-transcript-runtime.test.ts new file mode 100644 index 000000000..0e039d584 --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-runtime.test.ts @@ -0,0 +1,151 @@ +import { describe, expect, it } from "vitest"; +import type { Agent } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, getCurrentTools, getCurrentSystemPrompt, type AssistantMessage, type Message } from "@earendil-works/pi-ai"; +import type { UiMessage } from "@pi-desktop/shared"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { DesktopAgentRuntime, type RuntimeProviderConfig } from "./runtime.js"; + +const provider: RuntimeProviderConfig = { + id: "fixture", name: "Fixture", modelId: "fixture", apiKey: "", authKind: "none", + modelConfig: modelConfigFromPi({ ...Object.values(DEEPSEEK_MODELS)[0], id: "fixture", provider: "fixture", baseUrl: "https://fixture.invalid/v1", + compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: true } }), + baseUrl: "https://fixture.invalid/v1", supportsReasoning: false, supportedThinkingLevels: ["off"], +}; +const skill = { id: "fixture/notes", name: "First catalog", description: "Summarize notes" }; +function answer(content: AssistantMessage["content"], stopReason: "stop" | "toolUse" = "stop"): AssistantMessage { + return { + role: "assistant", api: "openai-completions", provider: "fixture", model: "fixture", + timestamp: Date.now() + 10, stopReason, content, + usage: { input: 10, output: 2, cacheRead: 0, cacheWrite: 0, totalTokens: 12, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, + }; +} +function fixture(history: UiMessage[] = [], skills = [skill]) { + const rows = [...history]; + const requests: Message[][] = []; + const errors: unknown[] = []; + let persistenceFailure: Error | undefined; + let respond = () => answer([{ type: "text", text: "Done." }]); + const runtime = new DesktopAgentRuntime({ + sessionId: "fixture", mode: "agent", provider, thinkingLevel: "off", history, pluginSkills: skills, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") { + if (persistenceFailure) throw persistenceFailure; + rows.push((params as { message: UiMessage }).message); + } + else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + const agent = (runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = async (_model, context) => { + requests.push(structuredClone(context.messages)); + const message = respond(); + const stream = createAssistantMessageEventStream(); + stream.push({ type: "start", partial: { ...message, content: [] } }); + stream.push({ type: "done", reason: message.stopReason as "stop" | "toolUse", message }); + return stream; + }; + return { + runtime, agent, rows, requests, errors, + failPersistence: () => { persistenceFailure = new Error("disk full"); }, + respond: (next: () => AssistantMessage) => { respond = next; }, + prompt: async (id: string) => { + rows.push({ id, role: "user", content: "Continue", createdAt: new Date().toISOString() }); + await runtime.prompt("Continue", id, `turn-${id}`); + }, + }; +} + +describe("system state through the Desktop user path", () => { + it("activates ToolSearch through Pi, retains its prefix, and restores the declarations after restart", async () => { + const f = fixture(); + let requests = 0; + f.respond(() => ++requests === 1 + ? answer([{ type: "toolCall", id: "search-call", name: "ToolSearch", arguments: { query: "BrowserPreview" } }], "toolUse") + : answer([{ type: "text", text: "Ready." }])); + try { + await f.prompt("user-1"); + expect(f.requests).toHaveLength(2); + const [before, after] = f.requests; + expect(after.slice(0, before.length)).toEqual(before); + expect(getCurrentTools(before).some((tool) => tool.name === "BrowserPreview")).toBe(false); + expect(getCurrentTools(after).some((tool) => tool.name === "BrowserPreview")).toBe(true); + const activation = after.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); + expect(after.slice(activation + 1)).toContainEqual(expect.objectContaining({ role: "system", toolsAdded: expect.arrayContaining([expect.objectContaining({ name: "BrowserPreview" })]) })); + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); + await f.prompt("same-runtime-user"); + expect(getCurrentTools(f.requests[2]).some((tool) => tool.name === "BrowserPreview")).toBe(true); + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); + const restored = fixture(structuredClone(f.rows)); + try { + await restored.prompt("user-2"); + const replay = restored.requests[0]; + expect(replay.filter((message) => message.role === "system")).toEqual(after.filter((message) => message.role === "system")); + const resultIndex = replay.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); + expect(resultIndex).toBeGreaterThan(0); + expect(replay.slice(resultIndex + 1)).toContainEqual(after.at(-1)); + expect(getCurrentTools(restored.requests[0]).some((tool) => tool.name === "BrowserPreview")).toBe(true); + expect(restored.rows.filter((row) => row.modelSystem)).toHaveLength(2); + } finally { await restored.runtime.dispose(); } + } finally { await f.runtime.dispose(); } + }); + + it("stops before provider dispatch when the Host cannot save system state", async () => { + const f = fixture(); + f.failPersistence(); + try { + await f.prompt("user-1"); + expect(f.requests).toHaveLength(0); + expect(f.errors).toContainEqual(expect.objectContaining({ + code: "INTERNAL", retriable: false, details: expect.objectContaining({ origin: "local", phase: "request-preparation" }), + })); + } finally { await f.runtime.dispose(); } + }); + + it("updates and revokes a skill catalog on the same runtime without rewriting previous requests", async () => { + const otherSkill = { id: "fixture/other", name: "Unchanged entry", description: "Keep this instruction" }; + const f = fixture([], [skill, otherSkill]); + try { + await f.prompt("user-1"); + const before = f.requests[0]; + f.runtime.setPluginSkills([{ ...skill, name: "Updated catalog" }, otherSkill]); + await f.prompt("user-2"); + const after = f.requests[1]; + expect(after.slice(0, before.length)).toEqual(before); + expect(getCurrentSystemPrompt(after)).toContain("Updated catalog"); + expect(getCurrentSystemPrompt(after)).not.toContain("First catalog"); + expect(after.filter((message) => message.role === "system")).toHaveLength(2); + const updates = after.filter((message) => message.role === "system"); + expect(updates.at(-1)?.sections).toEqual({ "skill:fixture/notes": "- `fixture/notes` — Updated catalog: Summarize notes" }); + expect(getCurrentSystemPrompt(after)).toContain("Unchanged entry"); + const restored = fixture(structuredClone(f.rows), [{ ...skill, name: "Updated catalog" }, otherSkill]); + try { + await restored.prompt("restored-user"); + expect(restored.requests[0].filter((message) => message.role === "system")).toEqual(updates); + } finally { await restored.runtime.dispose(); } + f.runtime.setPluginSkills([]); + await f.prompt("user-3"); + expect(getCurrentSystemPrompt(f.requests[2])).not.toContain("Updated catalog"); + // Revocation includes a section delta and Pi's executable-tool removal. + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(4); + expect(getCurrentTools(f.requests[2]).some((tool) => tool.name === "Skill")).toBe(false); + } finally { await f.runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/system-transcript.test.ts b/packages/agent-runtime/src/system-transcript.test.ts index 854809a06..f83ad2f9f 100644 --- a/packages/agent-runtime/src/system-transcript.test.ts +++ b/packages/agent-runtime/src/system-transcript.test.ts @@ -14,22 +14,24 @@ import { import { estimateContextTokens as estimateTranscriptTokens } from "@earendil-works/pi-ai/utils/estimate"; import { convertToLlm } from "./pi-runtime-messages.js"; import { + syncSystemSections, initialSystemTranscript, rebuildSystemTranscript, replaceSystemPrompt, - syncSystemTools, + systemTranscriptCheckpoint, systemPromptContent, + removeTrailingAssistantMessages, } from "./system-transcript.js"; const read: Tool = { name: "Read", description: "Read text", parameters: Type.Object({ path: Type.String() }) }; const edit: Tool = { ...read, name: "Edit", description: "Edit text" }; const initial: SystemMessage = { - role: "system", content: "Base instructions", timestamp: 1_000, - sections: { rules: "Keep these rules", obsolete: "Remove these rules" }, + role: "system", content: "", timestamp: 1_000, + sections: { runtime: "Base instructions", rules: "Keep these rules", obsolete: "Remove these rules" }, toolsAdded: [read, edit], }; const delta: SystemMessage = { - role: "system", content: "Additional instructions", timestamp: 1_500, + role: "system", content: "", timestamp: 1_500, sections: { rules: "Updated rules", obsolete: null }, toolsRemoved: [{ name: "Edit" }], }; @@ -50,13 +52,55 @@ function estimateContextTokens(messages: AgentMessage[]) { afterEach(() => vi.restoreAllMocks()); describe("system transcript helpers", () => { + it("removes failed trailing responses across cleanup deltas without changing system state", () => { + const messages = [initial, user, assistant, delta, { ...assistant, timestamp: 3_000 }]; + const cleaned = removeTrailingAssistantMessages(messages); + expect(cleaned).toEqual([initial, user, delta]); + expect(getCurrentTools(cleaned)).toEqual(getCurrentTools(messages)); + expect(messages).toHaveLength(5); + expect(removeTrailingAssistantMessages([user, assistant])).toEqual([user]); + expect(removeTrailingAssistantMessages([assistant, user, delta])).toEqual([assistant, user, delta]); + }); + it("updates only the changed skill entry and preserves unrelated sections", () => { + const previous: AgentMessage[] = [{ + role: "system", content: "", timestamp: 1, + sections: { runtime: "Base", skills: "Load skills", "skill:a": "A", "skill:b": "B", context: "Context", extension: "Keep" }, + }, assistant]; + const desired = { runtime: "Base", skills: "Load skills", "skill:a": "A updated", "skill:b": "B", context: "Context" }; + const updated = syncSystemSections(previous, desired); + expect(updated.slice(0, previous.length)).toEqual(previous); + expect(updated.at(-1)).toMatchObject({ sections: { "skill:a": "A updated" } }); + expect(getCurrentSystemMessage(updated)?.sections).toHaveProperty("extension", "Keep"); + expect(syncSystemSections(updated, desired)).toBe(updated); + const revoked = syncSystemSections(updated, { runtime: "Base", context: "Context" }); + expect(revoked.at(-1)).toMatchObject({ sections: { skills: null, "skill:a": null, "skill:b": null } }); + expect(getCurrentSystemPrompt(revoked)).not.toContain("A updated"); + expect(getCurrentSystemPrompt(revoked)).toContain("Keep"); + }); + + it("upgrades aggregate catalogs once and restores per-skill checkpoints without duplicate deltas", () => { + const legacy: AgentMessage[] = [{ role: "system", content: "", timestamp: 1, + sections: { runtime: "Base", skills: "Old full catalog", context: "Context" } }]; + const desired = { runtime: "Base", skills: "Load skills", "skill:z": "Z", context: "Context" }; + const upgraded = syncSystemSections(legacy, desired); + expect(getCurrentSystemPrompt(upgraded)).not.toContain("Old full catalog"); + const added = syncSystemSections(upgraded, { ...desired, "skill:a": "A" }); + expect(systemPromptContent(added)).toBe("Base\n\nLoad skills\n\nA\n\nZ\n\nContext"); + const restored = [systemTranscriptCheckpoint(added)!]; + expect(syncSystemSections(restored, { ...desired, "skill:a": "A" })).toBe(restored); + expect(systemPromptContent(restored)).toBe(systemPromptContent(added)); + const replaced = replaceSystemPrompt(restored, "Temporary prompt"); + expect(getCurrentSystemPrompt(replaced)).toBe("Temporary prompt"); + }); + it("replays sections and tool removals without giving an unchanged prefix a new timestamp", () => { const previous = [initial, delta, assistant, user]; const rebuilt = rebuildSystemTranscript(previous, [assistant, user]); expect(getCurrentSystemPrompt(rebuilt)).toBe(getCurrentSystemPrompt(previous)); - expect(rebuilt[0]).toMatchObject({ - content: "Base instructions\n\nAdditional instructions", timestamp: 1_500, - sections: { rules: "Updated rules" }, toolsAdded: [read], + expect(rebuilt).toEqual(previous); + expect(systemTranscriptCheckpoint(rebuilt)).toMatchObject({ + content: "", timestamp: 1_500, + sections: { runtime: "Base instructions", rules: "Updated rules" }, toolsAdded: [read], }); expect(getCurrentSystemMessage(rebuilt)?.sections).not.toHaveProperty("obsolete"); expect(getCurrentTools(rebuilt)).toEqual([read]); @@ -64,18 +108,17 @@ describe("system transcript helpers", () => { expect(rebuildSystemTranscript(rebuilt, [assistant, user])[0]).toBe(rebuilt[0]); }); - it("does not resurrect usage predating a folded system delta", () => { + it("keeps an earlier usage anchor and estimates an appended delta", () => { const newer = { ...delta, timestamp: 3_000 }; const rebuilt = rebuildSystemTranscript([initial, assistant, newer], [assistant]); - expect(rebuilt[0]?.timestamp).toBe(3_000); - expect(estimateContextTokens(rebuilt).lastUsageIndex).toBeNull(); + expect(rebuilt.at(-1)?.timestamp).toBe(3_000); + expect(estimateContextTokens(rebuilt).trailingTokens).toBeGreaterThan(0); }); it("keeps a complete recovery transcript and its deltas in place", () => { const previous = [initial, assistant, delta, user, { ...assistant, stopReason: "error" as const }]; const retained = previous.slice(0, -1); expect(rebuildSystemTranscript(previous, retained)).toBe(retained); - expect(syncSystemTools(retained, [read])).toBe(retained); }); it("preserves object identity when setting either the same content or rendered prompt", () => { @@ -84,14 +127,14 @@ describe("system transcript helpers", () => { expect(replaceSystemPrompt(messages, systemPromptContent(messages))).toBe(messages); }); - it("invalidates usage only on a real content change and preserves structured sections and tools", () => { + it("accounts for a real content change and preserves structured sections and tools", () => { vi.spyOn(Date, "now").mockReturnValue(5_000); const changed = replaceSystemPrompt([initial, delta, assistant], "Replacement instructions"); - expect(changed[0]).toMatchObject({ content: "Replacement instructions", timestamp: 5_000, sections: { rules: "Updated rules" } }); + expect(changed.at(-1)).toMatchObject({ content: "", timestamp: 5_000, sections: { runtime: "Replacement instructions" } }); expect(getCurrentTools(changed)).toEqual([read]); - expect(estimateContextTokens(changed).lastUsageIndex).toBeNull(); + expect(estimateContextTokens(changed).trailingTokens).toBeGreaterThan(0); expect(replaceSystemPrompt(changed, "Replacement instructions")).toBe(changed); - expect(initial.content).toBe("Base instructions"); + expect(initial.sections?.runtime).toBe("Base instructions"); const response = { ...assistant, timestamp: 6_000 }; expect(estimateContextTokens(rebuildSystemTranscript(changed, [assistant, response])).usageTokens).toBe(1_000); }); @@ -99,7 +142,7 @@ describe("system transcript helpers", () => { it("can clear content without clearing sections or tool declarations", () => { const changed = replaceSystemPrompt([initial, delta, assistant], ""); expect(systemPromptContent(changed)).toBe(""); - expect(getCurrentSystemMessage(changed)?.sections).toEqual({ rules: "Updated rules" }); + expect(getCurrentSystemMessage(changed)?.sections).toEqual({ runtime: "", rules: "Updated rules" }); expect(getCurrentTools(changed)).toEqual([read]); }); @@ -111,55 +154,32 @@ describe("system transcript helpers", () => { const response = { ...assistant, timestamp: 6_000 }; const restored = replaceSystemPrompt([...nudged, response], before); expect(getCurrentSystemPrompt(restored)).toBe(getCurrentSystemPrompt(messages)); - expect(getCurrentSystemMessage(restored)?.sections).toEqual({ rules: "Updated rules" }); - expect(restored[0]?.timestamp).toBeGreaterThan(response.timestamp); - expect(estimateContextTokens(restored).usageTokens).toBe(0); - }); - - it("records tool additions, removals, and same-name schema replacements with fresh semantic time", () => { - vi.spyOn(Date, "now").mockReturnValue(5_000); - const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; - const messages = [initial, assistant]; - const changed = syncSystemTools(messages, [replacement]); - expect(changed[1]).toMatchObject({ timestamp: 5_000, toolsAdded: [replacement], toolsRemoved: [{ name: "Read" }, { name: "Edit" }] }); - expect(getCurrentTools(changed)).toEqual([replacement]); - expect(estimateContextTokens(changed).usageTokens).toBe(0); - expect(syncSystemTools(changed, [replacement])).toBe(changed); - const response = { ...assistant, timestamp: 6_000 }; - expect(estimateContextTokens(rebuildSystemTranscript(changed, [assistant, response])).usageTokens).toBe(1_000); - const cleared = syncSystemTools(changed, []); - expect(getCurrentTools(rebuildSystemTranscript(cleared, [assistant]))).toEqual([]); - }); - - it("ignores executable-only tool changes instead of invalidating usage", () => { - const messages = [initial, assistant]; - const executable = { ...read, label: "Read", execute: vi.fn() }; - expect(syncSystemTools(messages, [executable, edit])).toBe(messages); - expect(estimateContextTokens(messages).usageTokens).toBe(1_000); + expect(getCurrentSystemMessage(restored)?.sections).toEqual({ runtime: "Base instructions", rules: "Updated rules" }); + expect(restored.at(-1)?.timestamp).toBeGreaterThan(response.timestamp); + expect(estimateContextTokens(restored).trailingTokens).toBeGreaterThan(0); }); it("advances semantic time even if the clock shares a millisecond or moves backwards", () => { vi.spyOn(Date, "now").mockReturnValue(2_000); - expect(replaceSystemPrompt([initial, assistant], "changed")[0]?.timestamp).toBe(2_001); + expect(replaceSystemPrompt([initial, assistant], "changed").at(-1)?.timestamp).toBe(2_001); vi.spyOn(Date, "now").mockReturnValue(500); - expect(syncSystemTools([initial, assistant], [read])[1]?.timestamp).toBe(2_001); + expect(replaceSystemPrompt([initial, assistant], "another change").at(-1)?.timestamp).toBe(2_001); }); it("initializes restored history conservatively without inventing an old timestamp", () => { vi.spyOn(Date, "now").mockReturnValue(5_000); const messages = initialSystemTranscript("Current config", [read], [assistant]); - expect(messages[0]).toMatchObject({ content: "Current config", toolsAdded: [read], timestamp: 5_000 }); - expect(estimateContextTokens(messages).usageTokens).toBe(0); + expect(messages.at(-1)).toMatchObject({ sections: { runtime: "Current config" }, toolsAdded: [read], timestamp: 5_000 }); + expect(estimateContextTokens(messages).trailingTokens).toBeGreaterThan(0); const rebuilt = rebuildSystemTranscript(messages, [assistant]); expect(rebuilt[0]).toBe(messages[0]); - expect(estimateContextTokens(rebuilt).usageTokens).toBe(0); + expect(estimateContextTokens(rebuilt).trailingTokens).toBeGreaterThan(0); }); it("preserves the empty transcript when there is no system state", () => { const messages: AgentMessage[] = []; expect(initialSystemTranscript("", [], messages)).toBe(messages); expect(rebuildSystemTranscript([], messages)).toBe(messages); - expect(syncSystemTools(messages, [])).toBe(messages); expect(replaceSystemPrompt(messages, "")).toBe(messages); }); @@ -177,13 +197,13 @@ describe("system transcript helpers", () => { return stream; }, }); - agent.state.messages = syncSystemTools(rebuildSystemTranscript(agent.state.messages, [user]), [tool]); - const prefix = agent.state.messages[0]; + agent.state.messages = rebuildSystemTranscript(agent.state.messages, [user]); + const prefix = agent.state.messages.filter((message) => message.role === "system"); await agent.continue(); expect(agent.state.errorMessage).toBeUndefined(); expect(requests).toHaveLength(1); - expect(requests[0]?.filter((message) => message.role === "system")).toEqual([prefix]); + expect(requests[0]?.filter((message) => message.role === "system")).toEqual(prefix); expect(getCurrentTools(requests[0]!)).toEqual([toToolDeclaration(tool)]); - expect(agent.state.messages.filter((message) => message.role === "system")).toEqual([prefix]); + expect(agent.state.messages.filter((message) => message.role === "system")).toEqual(prefix); }); }); diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index 28bd5028b..9ba1167b2 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -1,22 +1,22 @@ +import { TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; import type { AgentMessage } from "@earendil-works/pi-agent-core"; import { contentText, - createInitialSystemMessage, getCurrentSystemMessage, getCurrentSystemPrompt, - getCurrentTools, - getToolStateChanges, type SystemMessage, type Tool, toToolDeclaration, } from "@earendil-works/pi-ai"; +import { SKILL_SECTION_PREFIX } from "./plugin-skills-prompt.js"; + function systemMessages(messages: readonly AgentMessage[]): SystemMessage[] { return messages.filter((message): message is SystemMessage => message.role === "system"); } function nextSystemTimestamp(messages: readonly AgentMessage[]): number { - // 同一毫秒内也必须晚于旧响应;时钟回拨时只前移,不伪造早于 usage 的时间。 + // Never backdate a state change relative to the response it invalidates. return messages.reduce((timestamp, message) => Math.max(timestamp, message.timestamp + 1), Date.now()); } @@ -24,71 +24,107 @@ function currentSystemMessage(messages: readonly AgentMessage[]): SystemMessage const systems = systemMessages(messages); if (systems.length < 2) return systems[0]; const current = getCurrentSystemMessage(systems)!; - // 上游折叠器保留第一条的时间,但 sections/工具删除可能来自较晚的 delta。 - // 快照的语义时间必须覆盖所有贡献者,否则旧 usage 会被错误地重新激活。 + // The upstream fold retains the first timestamp. A checkpoint must cover + // every contributing update so old usage is not accidentally reactivated. return { ...current, timestamp: systems.reduce((timestamp, message) => Math.max(timestamp, message.timestamp), systems[0]!.timestamp), }; } -/** The unrendered content, without flattening named sections into the prompt. */ +/** Desktop-owned sections retain their order ahead of extension sections. */ +export const CONTEXT_BUDGET_SECTION = "context_budget"; + +function desktopSectionNames(sections: Record): string[] { + return ["runtime", TOOL_ACTIVATION_SECTION, "skills", ...Object.keys(sections) + .filter((name) => name.startsWith(SKILL_SECTION_PREFIX)) + .sort((a, b) => a.localeCompare(b)), "context"]; +} + export function systemPromptContent(messages: readonly AgentMessage[]): string { - return contentText(getCurrentSystemMessage(messages)?.content ?? ""); + const current = getCurrentSystemMessage(messages); + const sections = current?.sections; + if (sections && desktopSectionNames(sections).some((name) => name in sections)) { + return desktopSectionNames(sections).map((name) => sections[name]).filter(Boolean).join("\n\n"); + } + return contentText(current?.content ?? ""); +} + +export function syncSystemSections( + messages: AgentMessage[], + desired: Record, +): AgentMessage[] { + const current = getCurrentSystemMessage(messages)?.sections ?? {}; + const sections: Record = {}; + for (const name of desktopSectionNames({ ...current, ...desired })) { + const next = desired[name] ?? null; + if ((current[name] ?? null) !== next) sections[name] = next; + } + if (Object.keys(sections).length === 0) return messages; + return [...messages, { role: "system", content: "", sections, timestamp: nextSystemTimestamp(messages) }]; } export function initialSystemTranscript( prompt: string, tools: readonly Tool[], messages: AgentMessage[], + sections: Record = { runtime: prompt }, ): AgentMessage[] { - const system = createInitialSystemMessage(prompt, tools.map(toToolDeclaration)); - // 持久化历史不记录 system 状态,重启后无法证明新配置与旧请求相同。 - // 保守使用实际初始化时间;不能沿用上游初始值 0 来让历史 usage 假装有效。 - return system - ? [{ ...system, timestamp: nextSystemTimestamp(messages) }, ...messages] - : messages; + if (messages.some((message) => message.role === "system")) { + return syncSystemSections(messages, sections); + } + if (!prompt && tools.length === 0) return messages; + // Old sessions have no recorded baseline. Declare the current state at the + // continuation boundary rather than inventing past instructions or tools. + return [...messages, { + role: "system", content: "", sections, + toolsAdded: tools.map(toToolDeclaration), timestamp: nextSystemTimestamp(messages), + }]; } export function replaceSystemPrompt(messages: AgentMessage[], prompt: string): AgentMessage[] { - const current = currentSystemMessage(messages); - if (prompt === getCurrentSystemPrompt(messages) || prompt === contentText(current?.content ?? "")) { - return messages; - } - // prompt 只替换内容;命名 sections 与最终工具状态仍由上游 replay 负责。 - // recovery 读写未渲染内容,避免把 sections 再嵌入 content 造成重复。 - return [ - { ...current, role: "system", content: prompt, timestamp: nextSystemTimestamp(messages) }, - ...messages.filter((message) => message.role !== "system"), - ]; + if (prompt === getCurrentSystemPrompt(messages) || prompt === systemPromptContent(messages)) return messages; + const activation = getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION]; + return syncSystemSections(messages, { runtime: prompt, ...(activation ? { [TOOL_ACTIVATION_SECTION]: activation } : {}) }); } export function rebuildSystemTranscript( previous: readonly AgentMessage[], messages: AgentMessage[], ): AgentMessage[] { - // recovery 传入完整 live transcript 的切片,保留其位置、对象及 delta,不重复加前缀。 - if (messages.some((message) => message.role === "system")) return messages; - // durable projection 不含 system;用上游 replay 还原有效 sections 和工具集合, - // 而不是将渲染后的 systemPrompt 当成新消息。折叠不产生新的语义时间。 - const system = currentSystemMessage(previous); - return system ? [system, ...messages] : messages; + let result = messages; + for (let i = 0; i < previous.length; i++) { + const message = previous[i]; + if (message.role !== "system" || result.includes(message)) continue; + const following = previous.slice(i + 1).find((item) => result.includes(item)); + const preceding = previous.slice(0, i).reverse().find((item) => result.includes(item)); + let position = following ? result.indexOf(following) : preceding ? result.indexOf(preceding) + 1 : result.length; + if (!following) while (result[position]?.role === "system") position++; + if (result === messages) result = [...messages]; + result.splice(position, 0, message); + } + return result; } -export function syncSystemTools(messages: AgentMessage[], tools: readonly Tool[]): AgentMessage[] { - const changes = getToolStateChanges(getCurrentTools(messages), tools); - if (changes.toolsAdded.length === 0 && changes.toolsRemoved.length === 0) return messages; - // 真正的工具变化成为较新的前缀,旧 usage 因此失效。保留删除/同名替换的 delta, - // 并让声明集合与可执行 catalog 一致,避免 agent-loop 下一轮再次自动追加同一变化。 - return [ - ...systemMessages(messages), - { - role: "system", - content: "", - ...(changes.toolsAdded.length ? { toolsAdded: changes.toolsAdded } : {}), - ...(changes.toolsRemoved.length ? { toolsRemoved: changes.toolsRemoved } : {}), - timestamp: nextSystemTimestamp(messages), - }, - ...messages.filter((message) => message.role !== "system"), - ]; +/** Fold state only at an explicit compaction boundary, with semantic time. */ +export function systemTranscriptCheckpoint(messages: readonly AgentMessage[]): SystemMessage | undefined { + const current = currentSystemMessage(messages); + if (!current) return undefined; + // Use the journal's durable shape, including extension-provided text blocks. + const checkpoint: SystemMessage = { ...current, content: contentText(current.content), + ...(current.toolsAdded ? { toolsAdded: current.toolsAdded.map(toToolDeclaration) } : {}) }; + if (!checkpoint.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; + const { [CONTEXT_BUDGET_SECTION]: _expired, ...sections } = checkpoint.sections; + return { ...checkpoint, sections }; +} + +/** Recovery discards failed trailing responses even after a prompt cleanup delta. */ +export function removeTrailingAssistantMessages(messages: readonly AgentMessage[]): AgentMessage[] { + const result = [...messages]; + for (let index = result.length - 1; index >= 0; index--) { + if (result[index].role === "system") continue; + if (result[index].role !== "assistant") break; + result.splice(index, 1); + } + return result; } diff --git a/packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts b/packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts new file mode 100644 index 000000000..c48f4622d --- /dev/null +++ b/packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts @@ -0,0 +1,77 @@ +import { randomUUID } from "node:crypto"; +import { createServer } from "node:http"; +import { vi } from "vitest"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import type { UiMessage } from "@pi-desktop/shared"; +import { modelConfigFromPi } from "../model-capabilities.js"; +import { DesktopAgentRuntime, type PluginToolDef, type RuntimeProviderConfig } from "../runtime.js"; + +export function flashProvider(): RuntimeProviderConfig { + const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; + return { id: "flash-fixture", name: "Flash", modelId: model.id, baseUrl: model.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: modelConfigFromPi(model) }; +} +export const pluginTools: PluginToolDef[] = ["plugin_alpha", "plugin_beta"].map((name) => ({ + name, description: `${name} synthetic probe`, parameters: { type: "object", properties: {}, required: [] }, risk: "low", +})); +export type Payload = { tools: { function: { name: string } }[]; messages: { role: string; content?: unknown }[] }; +type Call = { name: string; args?: Record }; +export async function wireFixture(calls: Call[]) { + const requests: Payload[] = []; + const fetch = globalThis.fetch; + const server = createServer(async (req, res) => { + const chunks: Buffer[] = []; + for await (const chunk of req) chunks.push(Buffer.from(chunk)); + requests.push(JSON.parse(Buffer.concat(chunks).toString())); + const call = calls.shift(); + const delta = call ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), + type: "function", function: { name: call.name, arguments: JSON.stringify(call.args ?? {}) } }] } + : { role: "assistant", content: "Done." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + res.end(`data: ${JSON.stringify({ choices: [{ index: 0, delta, finish_reason: call ? "tool_calls" : "stop" }] })}\n\ndata: [DONE]\n\n`); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const address = server.address(); + if (!address || typeof address === "string") throw new Error("Missing HTTP address"); + vi.stubGlobal("fetch", ((_url, init) => fetch(`http://127.0.0.1:${address.port}`, init)) satisfies typeof fetch); + return { requests, close: async () => { + vi.unstubAllGlobals(); + await new Promise((resolve, reject) => server.close((error) => error ? reject(error) : resolve())); + } }; +} +export function runtimeFixture(history: UiMessage[] = [], tools = pluginTools, provider = flashProvider(), denied = false) { + const rows = structuredClone(history); + const executed: string[] = []; + const errors: unknown[] = []; + const runtime = new DesktopAgentRuntime({ + sessionId: "fixed-tools", mode: "agent", provider, thinkingLevel: "off", history: rows, pluginTools: tools, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") rows.push((params as { message: UiMessage }).message); + else if (method === "tools.execute") { + executed.push((params as { toolName: string }).toolName); + return denied ? { ok: false, denied: true, content: "Permission denied" } as T + : { ok: true, content: "Synthetic success" } as T; + } else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ id: event.toolCallId, role: "tool", content: "", + toolCallId: event.toolCallId, toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString() }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = event.isError ? "error" : "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + return { runtime, rows, executed, errors, prompt: async (id = "user-1") => { + rows.push({ id, role: "user", content: "Run the synthetic probes", createdAt: new Date().toISOString() }); + await runtime.prompt("Run the synthetic probes", id, `turn-${id}`); + } }; +} diff --git a/packages/agent-runtime/src/thinking-level.ts b/packages/agent-runtime/src/thinking-level.ts index 509bbd316..f386bbe80 100644 --- a/packages/agent-runtime/src/thinking-level.ts +++ b/packages/agent-runtime/src/thinking-level.ts @@ -23,6 +23,8 @@ export type ThinkingCapabilitySet = { */ export type ModelConfig = { source: "pi" | "models.dev" | "generic"; + /** Published identity before account endpoint or wire-model overrides. */ + transcriptBinding?: { modelId: string; api: string; baseUrl: string }; inputLimits?: Model["inputLimits"]; promptCache?: Model["promptCache"]; samplingParams?: Model["samplingParams"]; diff --git a/packages/agent-runtime/src/transcript-compat.test.ts b/packages/agent-runtime/src/transcript-compat.test.ts new file mode 100644 index 000000000..6e3df7ced --- /dev/null +++ b/packages/agent-runtime/src/transcript-compat.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it } from "vitest"; +import { genericModelConfig } from "./model-capabilities.js"; +import { buildProviderModel, type RuntimeProviderConfig } from "./provider-binding.js"; + +const baseUrl = "https://native.test/v1"; +const provider: RuntimeProviderConfig = { + id: "account", name: "Account", modelId: "exact-model", baseUrl, apiStyle: "openai_completions", + apiKey: "", authKind: "none", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: { + ...genericModelConfig("exact-model", baseUrl), source: "pi", + transcriptBinding: { modelId: "exact-model", api: "openai-completions", baseUrl }, + compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }, + }, +}; + +describe("transcript compatibility binding", () => { + it("keeps Pi transport capabilities when models.dev owns published metadata", () => { + expect(buildProviderModel({ ...provider, modelConfig: { ...provider.modelConfig!, source: "models.dev" } }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + it("keeps instruction and tool capabilities independent on the exact binding", () => { + expect(buildProviderModel(provider).compat).toMatchObject({ supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }); + expect(buildProviderModel({ ...provider, baseUrl: `${baseUrl}/` }).compat).toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + it.each([ + { baseUrl: "https://relay.test/v1" }, { baseUrl: "https://native.test/other" }, + { baseUrl: "https://native.test:8443/v1" }, { modelId: "alias" }, { apiStyle: "responses" }, + { modelConfig: { ...provider.modelConfig!, transcriptBinding: undefined } }, + ])("falls back for an unverified route or model: %j", (overrides) => { + expect(buildProviderModel({ ...provider, ...overrides }).compat).toMatchObject({ + supportsMidConvoSystemMessages: false, supportsMidConvoToolAdditions: false, + supportsMidConvoToolChanges: false, supportsAdditionalTools: false, + }); + }); +}); diff --git a/packages/agent-runtime/src/transcript-compat.ts b/packages/agent-runtime/src/transcript-compat.ts new file mode 100644 index 000000000..887e40687 --- /dev/null +++ b/packages/agent-runtime/src/transcript-compat.ts @@ -0,0 +1,39 @@ +import type { ModelConfig } from "./thinking-level.js"; +import type { Api, Model } from "@earendil-works/pi-ai"; + +const capabilities = [ + "supportsMidConvoSystemMessages", "supportsMidConvoToolAdditions", + "supportsMidConvoToolChanges", "supportsAdditionalTools", "supportsToolSearch", +] as const; + +/** Transport capabilities only; published limits and prices keep their owner. */ +export function transcriptConfigFromPi(model: Model): Pick { + return { + transcriptBinding: { modelId: model.id, api: model.api, baseUrl: model.baseUrl }, + compat: Object.fromEntries(capabilities.map((key) => [key, + model.compat !== undefined && key in model.compat && Reflect.get(model.compat, key) === true, + ])), + }; +} + +function endpoint(value: string): string | undefined { + try { + const url = new URL(value); + if (url.username || url.password || url.search || url.hash) return undefined; + return `${url.origin}${url.pathname.replace(/\/+$/, "")}`; + } catch { + return undefined; + } +} + +/** Catalog capabilities describe one model/API/endpoint, never a vendor label. */ +export function transcriptCompat( + catalog: ModelConfig | undefined, modelId: string, api: string, baseUrl: string, +): Record<(typeof capabilities)[number], boolean> { + const binding = catalog?.transcriptBinding; + const address = endpoint(baseUrl); + const verified = (catalog?.source === "pi" || catalog?.source === "models.dev") && binding?.modelId === modelId && + binding.api === api && address !== undefined && endpoint(binding.baseUrl) === address; + return Object.fromEntries(capabilities.map((key) => [key, verified && catalog?.compat?.[key] === true])) as + Record<(typeof capabilities)[number], boolean>; +} diff --git a/packages/agent-runtime/src/transcript-payload.test.ts b/packages/agent-runtime/src/transcript-payload.test.ts new file mode 100644 index 000000000..0cd77f8b2 --- /dev/null +++ b/packages/agent-runtime/src/transcript-payload.test.ts @@ -0,0 +1,87 @@ +import { expect, it } from "vitest"; +import { normalizeContext, Type, type Message, type Model } from "@earendil-works/pi-ai"; +import { stream } from "@earendil-works/pi-ai/api/openai-completions"; +import { genericModelConfig } from "./model-capabilities.js"; +import { buildProviderModel } from "./provider-binding.js"; + +const baseUrl = "https://native.invalid/v1"; +const read = { name: "Read", description: "Read", parameters: Type.Object({ path: Type.String() }) }; +const search = { ...read, name: "Search" }; +const messages: Message[] = [ + { role: "system", content: "", sections: { runtime: "Base", skills: "Old catalog" }, toolsAdded: [read], timestamp: 1 }, + { role: "user", content: "Find files", timestamp: 2 }, + { role: "system", content: "", sections: { skills: "New catalog" }, toolsAdded: [search], timestamp: 3 }, +]; +type Payload = { messages: Array<{ role: string; content?: unknown; tools?: unknown[] }>; tools?: Array<{ function: { name: string; parameters: unknown } }> }; +async function payload(instructions: boolean, additions: boolean, route = baseUrl, transcript = messages): Promise { + const model = buildProviderModel({ + id: "fixture", name: "Fixture", modelId: "exact", baseUrl: route, apiKey: "", authKind: "none", + supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: { + ...genericModelConfig("exact", baseUrl), source: "pi", + transcriptBinding: { modelId: "exact", api: "openai-completions", baseUrl }, + compat: { supportsMidConvoSystemMessages: instructions, supportsMidConvoToolAdditions: additions }, + }, + }); + let captured: Payload | undefined; + const response = stream(model as Model<"openai-completions">, normalizeContext({ messages: transcript }), { + apiKey: "fixture", + fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + expect((await response.result()).stopReason).toBe("stop"); + if (!captured) throw new Error("No payload captured"); + return captured; +} + +it("uses native additions only when both instruction and tool capabilities are verified", async () => { + const original = structuredClone(messages); + const native = await payload(true, true); + expect(native.tools?.map((tool) => tool.function.name)).toEqual(["Read"]); + expect(native.messages.slice(0, 2).map((message) => message.role)).toEqual(["system", "user"]); + expect(native.messages.slice(2).some((message) => message.tools?.length === 1)).toBe(true); + const partial = await payload(true, false); + expect(partial.tools?.map((tool) => tool.function.name)).toEqual(["Read", "Search"]); + expect(partial.messages.filter((message) => message.role === "system")).toHaveLength(2); + const relay = await payload(true, true, "https://relay.invalid/v1"); + expect(relay.messages.map((message) => message.role)).toEqual(["system", "user"]); + expect(relay.tools?.map((tool) => tool.function.name)).toEqual(["Read", "Search"]); + expect(JSON.stringify(relay.messages)).toContain("New catalog"); + expect(JSON.stringify(relay.messages)).not.toContain("Old catalog"); + expect(await payload(true, true)).toEqual(native); + expect(messages).toEqual(original); +}); + +it("folds removals and same-name replacements into the active request tool set", async () => { + const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; + const request = await payload(true, true, baseUrl, [...messages, { + role: "system", content: "", timestamp: 4, + toolsRemoved: [{ name: "Read" }, { name: "Search" }], toolsAdded: [replacement], + }]); + expect(request.tools).toEqual([expect.objectContaining({ function: expect.objectContaining({ name: "Read", parameters: replacement.parameters }) })]); + expect(request.messages.some((message) => message.tools)).toBe(false); +}); + +it("projects per-skill changes and revocations with native and fallback model support", async () => { + const catalog: Message[] = [ + { role: "system", content: "", timestamp: 1, sections: { + runtime: "Base", skills: "Load skills", "skill:a": "Alpha", "skill:b": "Bravo", "skill:c": "Charlie", + } }, + { role: "user", content: "Continue", timestamp: 2 }, + { role: "system", content: "", timestamp: 3, sections: { "skill:a": "Alpha updated", "skill:b": null } }, + ]; + const before = await payload(true, false, baseUrl, catalog.slice(0, 2)); + const native = await payload(true, false, baseUrl, catalog); + expect(native.messages.slice(0, before.messages.length)).toEqual(before.messages); + expect(JSON.stringify(native.messages.at(-1))).toContain("Alpha updated"); + expect(JSON.stringify(native.messages.at(-1))).not.toContain("Charlie"); + for (const route of [baseUrl, "https://relay.invalid/v1"]) { + const fallback = await payload(false, false, route, catalog); + expect(fallback.messages.map((message) => message.role)).toEqual(["system", "user"]); + expect(JSON.stringify(fallback.messages)).toContain("Alpha updated"); + expect(JSON.stringify(fallback.messages)).toContain("Charlie"); + expect(JSON.stringify(fallback.messages)).not.toContain("Bravo"); + } +}); diff --git a/packages/shared/src/types/messages.ts b/packages/shared/src/types/messages.ts index a7b1a88a0..4532533aa 100644 --- a/packages/shared/src/types/messages.ts +++ b/packages/shared/src/types/messages.ts @@ -83,6 +83,15 @@ export type UiMessage = { id: string; role: UiMessageRole; content: string; + /** Internal model instructions/tool declarations; never a visible chat row. */ + modelSystem?: { + version: 1; + /** The following user input can have been persisted before runtime admission. */ + beforeMessageId?: string; + afterMessageId?: string; + /** Opaque JSON preserves section and schema key order across Host storage. */ + messageJson: string; + }; /** Authenticated agent-to-agent provenance; never inferred from message text. */ sessionMessage?: SessionMessageOrigin; /** Present only on the durable user row created by a Live Voice operation. */ diff --git a/patches/@earendil-works__pi-ai@1.0.0.patch b/patches/@earendil-works__pi-ai@1.0.0.patch index c4e2774db..7b8d647aa 100644 --- a/patches/@earendil-works__pi-ai@1.0.0.patch +++ b/patches/@earendil-works__pi-ai@1.0.0.patch @@ -1613,3 +1613,10 @@ index 0000000000000000000000000000000000000000..5c63b882ae4cb618e964c5d2ce8d3c18 + return stream; + } +} +diff --git a/dist/providers/data/deepseek.json b/dist/providers/data/deepseek.json +--- a/dist/providers/data/deepseek.json ++++ b/dist/providers/data/deepseek.json +@@ -1 +1 @@ +-{"openai-completions":{"chat:deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"thinkingLevelMap":{"minimal":null,"low":"low","medium":null,"high":"high","max":"max"},"input":["text","image"],"cost":{"input":0.3,"output":1.2,"cacheRead":0.006,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true},"inputLimits":{"images":{"resize":{"maxWidth":2000,"maxHeight":2000,"maxBytes":4718592,"jpegQuality":80}}},"type":"chat"},"chat:deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"input":["text"],"cost":{"input":1.32,"output":3.96,"cacheRead":0.044,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"thinkingLevelMap":{"minimal":null,"low":null,"medium":null,"high":"high","max":"max"},"type":"chat"}}} ++{"openai-completions":{"chat:deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"thinkingLevelMap":{"minimal":null,"low":"low","medium":null,"high":"high","max":"max"},"input":["text","image"],"cost":{"input":0.3,"output":1.2,"cacheRead":0.006,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"inputLimits":{"images":{"resize":{"maxWidth":2000,"maxHeight":2000,"maxBytes":4718592,"jpegQuality":80}}},"type":"chat"},"chat:deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"input":["text"],"cost":{"input":1.32,"output":3.96,"cacheRead":0.044,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"thinkingLevelMap":{"minimal":null,"low":null,"medium":null,"high":"high","max":"max"},"type":"chat"}}} +\ No newline at end of file diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 06280cf87..89ae00089 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -200,7 +200,7 @@ overrides: patchedDependencies: '@earendil-works/pi-agent-core@1.0.0': 02de513ae53cf7f1e92d0cfc7fce07cf880d31195f5ec621d2f2197ead92a9da - '@earendil-works/pi-ai@1.0.0': 7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3 + '@earendil-works/pi-ai@1.0.0': 9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0 '@earendil-works/pi-coding-agent@1.0.0': 2d46470684b572b37bba57cf3cec31d81cf7747be163812b74563c8be8a95da3 importers: @@ -228,7 +228,7 @@ importers: devDependencies: '@earendil-works/pi-ai': specifier: 1.0.0 - version: 1.0.0(patch_hash=7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3) + version: 1.0.0(patch_hash=9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-mcp': specifier: 1.0.0 version: 1.0.0 @@ -423,7 +423,7 @@ importers: version: 1.0.0(patch_hash=02de513ae53cf7f1e92d0cfc7fce07cf880d31195f5ec621d2f2197ead92a9da)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-ai': specifier: 1.0.0 - version: 1.0.0(patch_hash=7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + version: 1.0.0(patch_hash=9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-coding-agent': specifier: 1.0.0 version: 1.0.0(patch_hash=2d46470684b572b37bba57cf3cec31d81cf7747be163812b74563c8be8a95da3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(ws@8.21.1)(zod@4.4.3) @@ -5327,7 +5327,7 @@ snapshots: '@earendil-works/pi-agent-core@1.0.0(patch_hash=02de513ae53cf7f1e92d0cfc7fce07cf880d31195f5ec621d2f2197ead92a9da)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': dependencies: - '@earendil-works/pi-ai': 1.0.0(patch_hash=7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + '@earendil-works/pi-ai': 1.0.0(patch_hash=9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) typebox: 1.3.27 transitivePeerDependencies: - '@aws-sdk/credential-provider-node' @@ -5342,7 +5342,7 @@ snapshots: - ws - zod - '@earendil-works/pi-ai@1.0.0(patch_hash=7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3)': + '@earendil-works/pi-ai@1.0.0(patch_hash=9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.124.0(zod@4.4.3) '@aws-sdk/client-bedrock-runtime': 3.1127.0 @@ -5367,7 +5367,7 @@ snapshots: - ws - zod - '@earendil-works/pi-ai@1.0.0(patch_hash=7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': + '@earendil-works/pi-ai@1.0.0(patch_hash=9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.124.0(zod@4.4.3) '@aws-sdk/client-bedrock-runtime': 3.1127.0 @@ -5400,7 +5400,7 @@ snapshots: dependencies: '@earendil-works/chord': 1.0.0 '@earendil-works/pi-agent-core': 1.0.0(patch_hash=02de513ae53cf7f1e92d0cfc7fce07cf880d31195f5ec621d2f2197ead92a9da)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) - '@earendil-works/pi-ai': 1.0.0(patch_hash=7b666ab69d8e8a12a7d02550e98a3a0020f38d7046da6dc82b237adf176475e3)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + '@earendil-works/pi-ai': 1.0.0(patch_hash=9a4038bcac99877bfda2690848abf7801f5b325d28872b15eedd1f4ca8619ba0)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(supports-color@7.2.0)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-codemode': 1.0.0 '@earendil-works/pi-mcp': 1.0.0 '@earendil-works/pi-tui': 1.0.0 diff --git a/scripts/e2e-fixed-tool-declarations.mjs b/scripts/e2e-fixed-tool-declarations.mjs new file mode 100644 index 000000000..9264a20ba --- /dev/null +++ b/scripts/e2e-fixed-tool-declarations.mjs @@ -0,0 +1,173 @@ +#!/usr/bin/env node +/** Production sidecar transport + isolated Host + local SSE provider. + * Message persistence is harness-owned; Electron's UI/outbox is not exercised. + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +import { DEEPSEEK_MODELS } from "../packages/agent-runtime/node_modules/@earendil-works/pi-ai/dist/providers/deepseek.models.js"; +import { modelConfigFromPi } from "../packages/agent-runtime/dist/model-capabilities.js"; +import { PROTOCOL_VERSION } from "../packages/shared/dist/protocol.js"; +import { withScenario } from "./e2e/fixture.mjs"; +import { resolveHostBinary } from "./e2e/host.mjs"; +import { createSession } from "./e2e/session.mjs"; + +const root = await mkdtemp(join(tmpdir(), "pi-system-state-e2e-")); +const requests = []; +const responses = []; +const server = createServer(async (req, res) => { + let body = ""; + for await (const chunk of req) body += chunk; + requests.push(JSON.parse(body)); + const tool = responses.shift(); + const delta = tool ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), type: "function", + function: { name: tool.name, arguments: JSON.stringify(tool.args) } }] } + : { role: "assistant", content: "The fixture completed." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + for (const [value, finish] of [[delta, null], [{}, tool ? "tool_calls" : "stop"]]) { + res.write(`data: ${JSON.stringify({ id: "fixture", model: "fixture", choices: [{ index: 0, delta: value, finish_reason: finish }] })}\n\n`); + } + res.end("data: [DONE]\n\n"); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; +const flash = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); +const model = process.env.PI_FIXED_TOOL_FIXTURE_ROUTE === "compatible" + ? { ...flash, id: "compatible-fixture", baseUrl: "https://fixture.invalid/v1", compat: undefined } + : flash; +const provider = { id: "fixture", name: "Fixed tools fixture", modelId: model.id, baseUrl: model.baseUrl, + modelConfig: modelConfigFromPi(model), apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; +const fetchHook = join(root, "fixture-fetch.mjs"); +await writeFile(fetchHook, `const fetch = globalThis.fetch; globalThis.fetch = (url, init) => { + if (String(url) !== ${JSON.stringify(`${model.baseUrl}/chat/completions`)}) throw new Error("Unexpected fixture endpoint"); + return fetch(${JSON.stringify(baseUrl)}, init); +};`); +let pluginTools = ["plugin_alpha", "plugin_beta"].map((name) => ({ name, + description: "Synthetic read-only marker", parameters: { type: "object", properties: {}, required: [] }, risk: "low" })); +try { + await withScenario("E2E-fixed-tool-declarations", async ({ host, workspace }) => { + const session = await createSession(host, workspace, "System state fixture"); + let writes = Promise.resolve(); + let writeError; + let stderr = ""; + const turns = new Map(); + const executedTools = []; + const toolRows = new Map(); + const persist = (message, turnId) => { + writes = writes.then(() => host.call("session.appendMessage", { sessionId: session.id, turnId, message })) + .catch((error) => { writeError ??= error; }); + }; + const startSidecar = () => { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, + args: [fileURLToPath(new URL("../packages/agent-runtime/dist/sidecar.js", import.meta.url))], + cwd: workspace, + env: { ...process.env, NODE_OPTIONS: `--import=${fetchHook}`, PI_DESKTOP_PLAN_UI_PROBE: "1", PI_DESKTOP_COMPACTION_STRATEGY: "fresh_window" }, + }, + onStderr: (chunk) => { stderr += chunk; }, + }); + sidecar.setHost({ call: host.call.bind(host), onNotification: () => () => {}, onExit: () => () => {} }); + for (const name of ["plugin_alpha", "plugin_beta"]) sidecar.setLocalTool(name, async () => { + executedTools.push(name); + return { ok: true, content: `${name} synthetic result` }; + }); + sidecar.onNotification((method, envelope) => { + if (method !== "agent.event") return; + const { event, turnId } = envelope; + const turn = turns.get(turnId); + if (event.type === "error") turn?.errors.push(event.error); + if (event.type === "message_end") persist(event.message, turnId); + if (event.type === "tool_start") toolRows.set(event.toolCallId, { + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + turn?.toolResults.push(event); + persist({ ...toolRows.get(event.toolCallId), toolResult: event.result, isError: event.isError, + toolStatus: event.isError ? "error" : "success" }, turnId); + } + if (event.type === "agent_end") turn?.resolve(); + }); + return sidecar; + }; + let sidecar = startSidecar(); + const params = () => ({ sessionId: session.id, mode: "agent", provider, thinkingLevel: "off", pluginTools, + projectPath: workspace, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + }); + const prompt = async (id, expectedErrors = 0) => { + const { turnId } = await host.call("session.beginTurn", { sessionId: session.id }); + await host.call("session.appendMessage", { sessionId: session.id, turnId, + message: { id, role: "user", content: "Continue", createdAt: new Date().toISOString() } }); + let timer; + const turn = { errors: [], toolResults: [] }; + const ended = new Promise((resolve, reject) => { + turn.resolve = resolve; + timer = setTimeout(() => reject(new Error(`Turn timed out: ${JSON.stringify(turn.errors)}\n${stderr}`)), 20_000); + }); + turns.set(turnId, turn); + try { + await sidecar.call("agent.prompt", { ...params(), turnId, userMessageId: id, content: "Continue" }); + await ended; + await writes; + if (writeError) throw writeError; + assert.deepEqual(turn.errors, [], JSON.stringify(turn.errors)); + assert.equal(turn.toolResults.filter((event) => event.isError).length, expectedErrors, JSON.stringify(turn.toolResults)); + await host.call("session.endTurn", { turnId, status: "completed", createNotification: false }); + return turn; + } finally { clearTimeout(timer); turns.delete(turnId); } + }; + const detail = async () => (await host.call("session.get", { id: session.id })).session; + try { + responses.push({ name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha", args: {} }); + await prompt("user-1"); + const declared = requests[0].tools; + assert(declared.some((tool) => tool.function.name === "plugin_beta")); + for (let index = 1; index < requests.length; index++) { + assert.deepEqual(requests[index].tools, declared); + assert.deepEqual(requests[index].messages.slice(0, requests[index - 1].messages.length), requests[index - 1].messages); + } + assert.deepEqual(executedTools, ["plugin_alpha"]); + const states = (await detail()).messages.filter((row) => row.modelSystem); + assert(states.length >= 2); + console.log("PASS: fixed declarations survive ToolSearch through production sidecar and HTTP/SSE"); + await sidecar.dispose(); + await host.stop(); await host.start(PROTOCOL_VERSION); + assert.deepEqual((await detail()).messages.filter((row) => row.modelSystem), states); + sidecar = startSidecar(); + responses.push({ name: "plugin_alpha", args: {} }, { name: "plugin_beta", args: {} }); + await prompt("user-2", 1); + assert.deepEqual(executedTools, ["plugin_alpha", "plugin_alpha"]); + assert.deepEqual(requests.at(-1).tools, declared); + console.log("PASS: Host and sidecar restart restore Alpha activation without activating declared Beta"); + await sidecar.call("agent.compact", params()); + const checkpoint = JSON.parse((await detail()).compaction.details.systemMessageJson); + assert.deepEqual(JSON.parse(checkpoint.sections.tool_activation).active, ["plugin_alpha"]); + await sidecar.dispose(); sidecar = startSidecar(); + responses.push({ name: "plugin_alpha", args: {} }, { name: "plugin_beta", args: {} }); + await prompt("user-3", 1); + assert.deepEqual(executedTools, ["plugin_alpha", "plugin_alpha", "plugin_alpha"]); + assert.deepEqual(requests.at(-1).tools, declared); + console.log("PASS: compaction and process restart preserve declarations and activation separately"); + responses.push({ name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta", args: {} }); + await prompt("user-4"); + assert.equal(executedTools.at(-1), "plugin_beta"); + assert.deepEqual(requests.at(-1).tools, declared); + pluginTools = pluginTools.slice(0, 1); + responses.push({ name: "plugin_beta", args: {} }); + await prompt("user-5", 1); + assert.equal(executedTools.length, 4); + assert(!requests.at(-1).tools.some((tool) => tool.function.name === "plugin_beta")); + console.log("PASS: changed catalog revokes Beta execution and creates a new declaration epoch"); + } finally { await sidecar.dispose(); await writes; } + }, resolveHostBinary(), root, PROTOCOL_VERSION); +} finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + await rm(root, { recursive: true, force: true }); +} diff --git a/scripts/e2e-system-transcript.mjs b/scripts/e2e-system-transcript.mjs new file mode 100644 index 000000000..a30350f0e --- /dev/null +++ b/scripts/e2e-system-transcript.mjs @@ -0,0 +1,173 @@ +#!/usr/bin/env node +/** Production sidecar transport + isolated Host + local SSE provider. + * Message persistence is harness-owned; Electron's UI/outbox is not exercised. + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, rm } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +import { PROTOCOL_VERSION } from "../packages/shared/dist/protocol.js"; +import { withScenario } from "./e2e/fixture.mjs"; +import { resolveHostBinary } from "./e2e/host.mjs"; +import { createSession } from "./e2e/session.mjs"; + +const root = await mkdtemp(join(tmpdir(), "pi-system-state-e2e-")); +const requests = []; +const responses = []; +const server = createServer(async (req, res) => { + let body = ""; + for await (const chunk of req) body += chunk; + requests.push(JSON.parse(body)); + const tool = responses.shift(); + const delta = tool ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), type: "function", + function: { name: tool.name, arguments: JSON.stringify(tool.args) } }] } + : { role: "assistant", content: "The fixture completed." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + for (const [value, finish] of [[delta, null], [{}, tool ? "tool_calls" : "stop"]]) { + res.write(`data: ${JSON.stringify({ id: "fixture", model: "fixture", choices: [{ index: 0, delta: value, finish_reason: finish }] })}\n\n`); + } + res.end("data: [DONE]\n\n"); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; +const provider = { id: "fixture", name: "Fixture", modelId: "fixture", baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; +let skills = [ + { id: "fixture/notes", name: "First skill", description: "Summarize notes" }, + { id: "fixture/unchanged", name: "Unchanged skill", description: "Preserve this catalog entry" }, +]; +try { + await withScenario("E2E-system-transcript", async ({ host, workspace }) => { + const session = await createSession(host, workspace, "System state fixture"); + let writes = Promise.resolve(); + let writeError; + let stderr = ""; + const turns = new Map(); + const executedTools = []; + const toolRows = new Map(); + const persist = (message, turnId) => { + writes = writes.then(() => host.call("session.appendMessage", { sessionId: session.id, turnId, message })) + .catch((error) => { writeError ??= error; }); + }; + const startSidecar = () => { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, + args: [fileURLToPath(new URL("../packages/agent-runtime/dist/sidecar.js", import.meta.url))], + cwd: workspace, + env: { ...process.env, PI_DESKTOP_PLAN_UI_PROBE: "1", PI_DESKTOP_COMPACTION_STRATEGY: "fresh_window" }, + }, + onStderr: (chunk) => { stderr += chunk; }, + }); + sidecar.setHost({ call: host.call.bind(host), onNotification: () => () => {}, onExit: () => () => {} }); + sidecar.setLocalTool("Skill", async ({ args }) => { + executedTools.push(args); + return { ok: true, content: "Fixture skill body: summarize the notes." }; + }); + sidecar.onNotification((method, envelope) => { + if (method !== "agent.event") return; + const { event, turnId } = envelope; + const turn = turns.get(turnId); + if (event.type === "error") turn?.errors.push(event.error); + if (event.type === "message_end") persist(event.message, turnId); + if (event.type === "tool_start") toolRows.set(event.toolCallId, { + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + turn?.toolResults.push(event); + persist({ ...toolRows.get(event.toolCallId), toolResult: event.result, isError: event.isError, + toolStatus: event.isError ? "error" : "success" }, turnId); + } + if (event.type === "agent_end") turn?.resolve(); + }); + return sidecar; + }; + let sidecar = startSidecar(); + const params = () => ({ sessionId: session.id, mode: "agent", provider, thinkingLevel: "off", pluginSkills: skills, + projectPath: workspace, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + }); + const identity = async () => (await sidecar.call("agent.testRuntimeIdentity", { sessionId: session.id })).runtimeId; + const prompt = async (id) => { + const { turnId } = await host.call("session.beginTurn", { sessionId: session.id }); + await host.call("session.appendMessage", { sessionId: session.id, turnId, + message: { id, role: "user", content: "Continue", createdAt: new Date().toISOString() } }); + let timer; + const turn = { errors: [], toolResults: [] }; + const ended = new Promise((resolve, reject) => { + turn.resolve = resolve; + timer = setTimeout(() => reject(new Error(`Turn timed out: ${JSON.stringify(turn.errors)}\n${stderr}`)), 20_000); + }); + turns.set(turnId, turn); + try { + await sidecar.call("agent.prompt", { ...params(), turnId, userMessageId: id, content: "Continue" }); + await ended; + await writes; + if (writeError) throw writeError; + assert.deepEqual(turn.errors, [], JSON.stringify(turn.errors)); + assert(turn.toolResults.every((event) => !event.isError), JSON.stringify(turn.toolResults)); + await host.call("session.endTurn", { turnId, status: "completed", createNotification: false }); + return turn; + } finally { clearTimeout(timer); turns.delete(turnId); } + }; + const detail = async () => (await host.call("session.get", { id: session.id })).session; + try { + responses.push({ name: "ToolSearch", args: { query: "BrowserPreview" } }); + await prompt("user-1"); + const saved = await detail(); + const states = saved.messages.filter((row) => row.modelSystem); + assert.equal(states.length, 2); + assert(JSON.parse(JSON.parse(states[1].modelSystem.messageJson).sections.tool_activation).active.includes("BrowserPreview")); + assert.deepEqual(requests[1].tools, requests[0].tools); + assert(requests[1].tools.some((tool) => tool.function.name === "BrowserPreview")); + console.log("PASS: sidecar ToolSearch and acknowledged Host persistence"); + await sidecar.dispose(); + await host.stop(); await host.start(PROTOCOL_VERSION); + const recovered = await detail(); + assert.deepEqual(recovered.messages.filter((row) => row.modelSystem), states); + sidecar = startSidecar(); + await prompt("user-2"); + const resumedStates = (await detail()).messages.filter((row) => row.modelSystem); + assert.equal(resumedStates.length, 2, JSON.stringify(resumedStates.slice(2).map((row) => row.modelSystem.messageJson))); + assert(requests.at(-1).tools.some((tool) => tool.function.name === "BrowserPreview")); + console.log("PASS: Host and sidecar process restart restores tools without duplicate state"); + const priorRuntime = await identity(); + skills = [{ ...skills[0], name: "Updated skill" }, skills[1]]; + responses.push({ name: "Skill", args: { id: skills[0].id } }); + await prompt("user-3"); + assert.equal(await identity(), priorRuntime, "skill catalog update must reuse the runtime"); + assert.equal(executedTools.length, 1, "Skill execution reaches the embedding host"); + assert(JSON.stringify(requests.at(-1).messages).includes("Fixture skill body")); + assert(JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("First skill")); + assert(JSON.stringify(requests.at(-1).messages).includes("Unchanged skill")); + console.log("PASS: skill catalog update reuses runtime and Skill executes across process boundary"); + await sidecar.call("agent.compact", params()); + const compacted = await detail(); + assert(JSON.parse(compacted.compaction.details.systemMessageJson).toolsAdded.some((tool) => tool.name === "BrowserPreview")); + await sidecar.dispose(); + sidecar = startSidecar(); + await prompt("user-4"); + assert(requests.at(-1).tools.some((tool) => tool.function.name === "BrowserPreview")); + assert(!requests.at(-1).messages.some((message) => message.role === "tool")); + assert(JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("First skill")); + console.log("PASS: compaction and sidecar restart preserve current skills and active tools"); + const beforeRemoval = await identity(); + skills = []; + await prompt("user-5"); + assert.equal(await identity(), beforeRemoval); + assert(!requests.at(-1).tools.some((tool) => tool.function.name === "Skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + console.log("PASS: removing the final skill removes its catalog and executable schema"); + } finally { await sidecar.dispose(); await writes; } + }, resolveHostBinary(), root, PROTOCOL_VERSION); +} finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + await rm(root, { recursive: true, force: true }); +}