From d8fd08dcd38d1a6ced2dffb464004f3776ecd7f6 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:51:29 +0800 Subject: [PATCH 1/8] fix(runtime): preserve chronological model system state Persist provider-neutral instruction and tool updates before dispatch so skill refresh, session restoration, and compaction preserve their order. Keep execution permissions authoritative and gate native updates on the original model, API, and endpoint binding. Related to #1285 --- apps/desktop/src/lib/assistant-turns.ts | 1 + apps/desktop/src/lib/transcript-projection.ts | 5 +- .../test/transcript-projection.test.mjs | 13 ++ crates/host-core/src/plugin_sessions.rs | 1 + crates/host-core/src/sessions.rs | 28 +++ crates/host-core/src/sessions/model_system.rs | 108 +++++++++++ ...039-plugin-skills-activation-and-devkit.md | 11 +- ...ed-tools-from-effective-session-context.md | 6 +- docs/adr/chronological-system-transcript.md | 63 +++++++ docs/spec/03-runtime/02-agent-runtime.md | 55 +++++- docs/spec/03-runtime/04-data-storage.md | 18 ++ docs/spec/06-delivery/04-e2e-test-plan.md | 39 ++++ .../src/extensions/prompt-chain.test.ts | 5 +- .../src/flash-transcript.test.ts | 144 +++++++++++++++ .../src/mode-tool-access.test.ts | 5 +- .../agent-runtime/src/model-capabilities.ts | 1 + packages/agent-runtime/src/output-cap.test.ts | 8 + packages/agent-runtime/src/output-cap.ts | 11 ++ .../src/plugin-skills-prompt.test.ts | 16 ++ .../agent-runtime/src/plugin-skills-prompt.ts | 34 +++- .../agent-runtime/src/plugin-skills.test.ts | 6 + packages/agent-runtime/src/plugin-skills.ts | 12 +- .../agent-runtime/src/provider-binding.ts | 5 +- .../src/provider-certificate-flow.test.ts | 5 +- .../src/provider-recovery-flow.test.ts | 1 + packages/agent-runtime/src/runtime.test.ts | 52 +++--- packages/agent-runtime/src/runtime.ts | 144 +++++++++------ packages/agent-runtime/src/session-context.ts | 6 +- packages/agent-runtime/src/sidecar.ts | 8 +- .../src/system-transcript-journal.test.ts | 86 +++++++++ .../src/system-transcript-journal.ts | 117 ++++++++++++ .../src/system-transcript-order.test.ts | 55 ++++++ .../src/system-transcript-runtime.test.ts | 144 +++++++++++++++ .../src/system-transcript.test.ts | 112 ++++++------ .../agent-runtime/src/system-transcript.ts | 122 ++++++++----- packages/agent-runtime/src/thinking-level.ts | 2 + .../src/transcript-compat.test.ts | 31 ++++ .../agent-runtime/src/transcript-compat.ts | 28 +++ .../src/transcript-payload.test.ts | 87 +++++++++ packages/shared/src/types/messages.ts | 9 + patches/@earendil-works__pi-ai@0.99.1.patch | 7 + pnpm-lock.yaml | 14 +- scripts/e2e-system-transcript.mjs | 172 ++++++++++++++++++ 43 files changed, 1580 insertions(+), 217 deletions(-) create mode 100644 crates/host-core/src/sessions/model_system.rs create mode 100644 docs/adr/chronological-system-transcript.md create mode 100644 packages/agent-runtime/src/flash-transcript.test.ts create mode 100644 packages/agent-runtime/src/system-transcript-journal.test.ts create mode 100644 packages/agent-runtime/src/system-transcript-journal.ts create mode 100644 packages/agent-runtime/src/system-transcript-order.test.ts create mode 100644 packages/agent-runtime/src/system-transcript-runtime.test.ts create mode 100644 packages/agent-runtime/src/transcript-compat.test.ts create mode 100644 packages/agent-runtime/src/transcript-compat.ts create mode 100644 packages/agent-runtime/src/transcript-payload.test.ts create mode 100644 scripts/e2e-system-transcript.mjs diff --git a/apps/desktop/src/lib/assistant-turns.ts b/apps/desktop/src/lib/assistant-turns.ts index ec80ddcda4..e47a064891 100644 --- a/apps/desktop/src/lib/assistant-turns.ts +++ b/apps/desktop/src/lib/assistant-turns.ts @@ -68,6 +68,7 @@ export function messageThinking(message: UiMessage): string { } function isVisibleMessage(message: UiMessage): boolean { + if (message.role === "system" && message.modelSystem) return false; return !( message.role === "assistant" && !(message.content || "").trim() && diff --git a/apps/desktop/src/lib/transcript-projection.ts b/apps/desktop/src/lib/transcript-projection.ts index 093bdb668c..e50ffefc20 100644 --- a/apps/desktop/src/lib/transcript-projection.ts +++ b/apps/desktop/src/lib/transcript-projection.ts @@ -57,7 +57,8 @@ function shape(message: UiMessage): MessageShape { const value = { content, thinking, - visible: message.role !== "assistant" || content || thinking || Boolean(message.hostedSearch || message.error), + visible: !(message.role === "system" && message.modelSystem) && + (message.role !== "assistant" || content || thinking || Boolean(message.hostedSearch || message.error)), answer: message.parentToolCallId ? content || Boolean(message.error) : content || !thinking || Boolean(message.error), @@ -72,7 +73,7 @@ function canReplace(previous: UiMessage, next: UiMessage): boolean { previous.parentToolCallId !== next.parentToolCallId || previous.agentName !== next.agentName || previous.createdAt !== next.createdAt || previous.toolCallId !== next.toolCallId || previous.toolName !== next.toolName || - previous.hostedSearch !== next.hostedSearch + previous.hostedSearch !== next.hostedSearch || Boolean(previous.modelSystem) !== Boolean(next.modelSystem) ) return false; if (isDelegationStartTool(previous.toolName) && ( previous.toolArgs !== next.toolArgs || previous.toolResult !== next.toolResult diff --git a/apps/desktop/test/transcript-projection.test.mjs b/apps/desktop/test/transcript-projection.test.mjs index 01a3de4adf..15b30035f8 100644 --- a/apps/desktop/test/transcript-projection.test.mjs +++ b/apps/desktop/test/transcript-projection.test.mjs @@ -14,6 +14,19 @@ const message = (id, role, content, extra = {}) => ({ id, role, content, createdAt: "2026-09-30T00:00:00.000Z", ...extra, }); +test("model system records remain hidden after loading and live projection updates", () => { + const state = message("state", "system", "", { modelSystem: { + version: 1, beforeMessageId: "user", messageJson: JSON.stringify({ role: "system", content: "Instructions", timestamp: 1 }), + } }); + const user = message("user", "user", "Hello"); + const notice = message("notice", "system", "Visible notice"); + let rows = [state, user, notice]; + assert.deepEqual(assertProjection(rows).visible.map((row) => row.id), ["user", "notice"]); + rows = upsertLiveSessionMessage(rows, message("answer", "assistant", "Hello back")); + assert.deepEqual(assertProjection(rows).visible.map((row) => row.id), ["user", "notice", "answer"]); + assert.equal(rows[0], state); +}); + function assertProjection(messages, compactions) { const actual = getTranscriptProjection(messages, compactions); const expected = buildTranscriptEntries(messages, compactions); diff --git a/crates/host-core/src/plugin_sessions.rs b/crates/host-core/src/plugin_sessions.rs index 3dee8f6a3e..bbc982407f 100644 --- a/crates/host-core/src/plugin_sessions.rs +++ b/crates/host-core/src/plugin_sessions.rs @@ -260,6 +260,7 @@ fn parse_message( nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }) } diff --git a/crates/host-core/src/sessions.rs b/crates/host-core/src/sessions.rs index 0b24782252..84c19f0196 100644 --- a/crates/host-core/src/sessions.rs +++ b/crates/host-core/src/sessions.rs @@ -12,6 +12,7 @@ use std::sync::{Mutex, OnceLock}; use crate::transcripts::{self, CompactionRecord, MessageRecord, RevisionRecord}; mod fork_files; +mod model_system; mod usage; pub use usage::record_usage; @@ -237,6 +238,9 @@ pub struct UiMessage { /// as an additive `hostedSearch` transcript block; no SQL migration. #[serde(default, skip_serializing_if = "Option::is_none")] pub hosted_search: Option, + /// Internal model-context state, preserved outside visible message text. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_system: Option, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -316,6 +320,9 @@ fn is_default_title(title: &str) -> bool { /// the search index row (None for tool rows, matching the FTS triggers). pub(crate) fn ui_to_record(message: &UiMessage) -> (MessageRecord, Option) { let mut meta_obj = serde_json::Map::new(); + if let Some(system) = &message.model_system { + meta_obj.insert("modelSystem".into(), system.clone()); + } if let Some(command) = &message.command { meta_obj.insert("command".into(), json!(command)); } @@ -465,6 +472,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { _ => Vec::new(), }; let meta = record.meta.unwrap_or(Value::Null); + let model_system = meta.get("modelSystem").cloned(); let command = meta .get("command") .and_then(Value::as_str) @@ -634,6 +642,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { nested_parent_tool_call_id, agent_name, hosted_search: hosted_search.clone(), + model_system: None, } } else { let content = blocks @@ -678,6 +687,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { nested_parent_tool_call_id, agent_name, hosted_search, + model_system, } } } @@ -897,6 +907,19 @@ fn clone_records_for_fork( .cloned() .unwrap_or_else(|| Uuid::new_v4().to_string()); if let Some(meta) = record.meta.as_mut().and_then(Value::as_object_mut) { + if let Some(system) = meta.get_mut("modelSystem").and_then(Value::as_object_mut) { + for key in ["beforeMessageId", "afterMessageId"] { + if let Some(new_id) = system + .get(key) + .and_then(Value::as_str) + .and_then(|id| message_ids.get(id)) + { + system.insert(key.into(), json!(new_id)); + } else { + system.remove(key); + } + } + } meta.remove("revisionRootId"); meta.remove("revisionCount"); meta.remove("activeRevision"); @@ -1903,6 +1926,7 @@ pub fn append_message( message: &UiMessage, turn_id: Option<&str>, ) -> Result<()> { + model_system::validate(message)?; let message = crate::session_collaboration::prepare_append(db, session_id, message, turn_id)?; let session_created = ensure_session_for_append(db, session_id)?; let (mut record, text) = ui_to_record(&message); @@ -3958,6 +3982,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, } } @@ -4610,6 +4635,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }; append_message(&db, &session.id, &tool, None).unwrap(); @@ -5124,6 +5150,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }; append_message(&db, &session.id, &assistant, None).unwrap(); @@ -5203,6 +5230,7 @@ mod tests { parent_tool_call_id: None, nested_parent_tool_call_id: None, agent_name: None, + model_system: None, hosted_search: Some(json!({ "status": "completed", "rounds": [ diff --git a/crates/host-core/src/sessions/model_system.rs b/crates/host-core/src/sessions/model_system.rs new file mode 100644 index 0000000000..88dd3140f2 --- /dev/null +++ b/crates/host-core/src/sessions/model_system.rs @@ -0,0 +1,108 @@ +use super::UiMessage; +use anyhow::{anyhow, Result}; +use serde_json::Value; + +/// Internal model state uses existing transcript metadata, never visible chat text. +pub(super) fn validate(row: &UiMessage) -> Result<()> { + let Some(record) = &row.model_system else { + return Ok(()); + }; + let invalid = || anyhow!("Invalid model system record"); + if row.role != "system" || !row.content.is_empty() || record["version"] != 1 { + return Err(invalid()); + } + for key in ["beforeMessageId", "afterMessageId"] { + if record + .get(key) + .is_some_and(|value| value.as_str().is_none_or(str::is_empty)) + { + return Err(invalid()); + } + } + let message: Value = serde_json::from_str(record["messageJson"].as_str().ok_or_else(invalid)?) + .map_err(|_| invalid())?; + if message["role"] != "system" + || !message["content"].is_string() + || !message["timestamp"] + .as_f64() + .is_some_and(|value| (0.0..=8.64e15).contains(&value)) + { + return Err(invalid()); + } + if let Some(sections) = message.get("sections") { + let sections = sections.as_object().ok_or_else(invalid)?; + if sections + .values() + .any(|value| !value.is_null() && !value.is_string()) + { + return Err(invalid()); + } + } + for key in ["toolsAdded", "toolsRemoved"] { + if let Some(tools) = message.get(key) { + for tool in tools.as_array().ok_or_else(invalid)? { + if tool + .get("name") + .and_then(Value::as_str) + .is_none_or(str::is_empty) + || (key == "toolsAdded" + && (!tool["description"].is_string() || !tool["parameters"].is_object())) + { + return Err(invalid()); + } + } + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + fn row() -> UiMessage { + serde_json::from_value(json!({ + "id":"state", "role":"system", "content":"", "createdAt":"2026-10-01T00:00:00Z", + "modelSystem":{"version":1,"beforeMessageId":"user","messageJson":json!({ + "role":"system","content":"","timestamp":1, + "sections":{"skills":"catalog"},"toolsAdded":[{"name":"Read","description":"read","parameters":{}}] + }).to_string()} + })).unwrap() + } + + #[test] + fn state_roundtrips_in_canonical_metadata_and_remaps_fork_anchors() { + let row = row(); + validate(&row).unwrap(); + let (record, text) = super::super::ui_to_record(&row); + assert!(text.as_deref().unwrap_or("").is_empty()); + assert_eq!( + super::super::record_to_ui(record.clone()).model_system, + row.model_system + ); + let mut user = row.clone(); + user.id = "user".into(); + user.role = "user".into(); + user.model_system = None; + let (records, ids, _) = + super::super::clone_records_for_fork(vec![record, super::super::ui_to_record(&user).0]); + assert_eq!( + records[0].meta.as_ref().unwrap()["modelSystem"]["beforeMessageId"], + ids["user"] + ); + } + + #[test] + fn rejects_invalid_state_or_hiding_a_user_message() { + let mut row = row(); + row.role = "user".into(); + assert!(validate(&row).is_err()); + row.role = "system".into(); + row.model_system.as_mut().unwrap()["version"] = json!(2); + assert!(validate(&row).is_err()); + row.model_system.as_mut().unwrap()["version"] = json!(1); + row.model_system.as_mut().unwrap()["messageJson"] = json!("invalid JSON"); + assert!(validate(&row).is_err()); + } +} diff --git a/docs/adr/0039-plugin-skills-activation-and-devkit.md b/docs/adr/0039-plugin-skills-activation-and-devkit.md index 9a619975f2..ce80c5d369 100644 --- a/docs/adr/0039-plugin-skills-activation-and-devkit.md +++ b/docs/adr/0039-plugin-skills-activation-and-devkit.md @@ -39,11 +39,12 @@ giving them the loop — scaffold, run, inspect, package. instructions.** Later text carries more weight, so a user's own instruction files keep the last word and an installed plugin can refine the built-in guidance but never the reverse. -3. **Runtime reuse keys on the catalog digest, not on bodies.** Enabling a - plugin, revoking `agent.prompt.inject` or renaming a skill changes the text - the model reads, so it retires the idle runtime rather than reusing a stale - prompt. An edit to a body needs no retirement: the `Skill` tool reads the file - at call time. +3. **Catalog changes update an idle runtime's skills section.** The chronological + system-transcript decision supersedes the original digest-based retirement: + enabling a plugin, revoking `agent.prompt.inject`, or renaming a skill appends + a section update before the next request. The digest excludes bodies; the + `Skill` tool reads the file and rechecks permissions at call time. Removing + the last catalog entry also removes the executable `Skill` declaration. 4. **Plugin authoring ships as a first-party devkit, not as a plugin.** `@pi-desktop/plugin-devkit` owns scaffold, check and pack; three surfaces share that one implementation — the `pi-plugin` CLI, the `PluginScaffold` / diff --git a/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md b/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md index 464fc3e162..617c59f03c 100644 --- a/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md +++ b/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md @@ -17,7 +17,11 @@ request omitted its schema. The same mismatch occurred after a mode switch. Before each new prompt and after a mode switch, the sidecar clears its in-memory deferred activation set and restores it from the effective `buildSessionContext` -projection. A successful `ToolSearch` result contributes its `addedToolNames`; +projection. When system declarations exist, their replayed active set is the +baseline; only successful results after the last declaration can add a newly +activated tool. This prevents an earlier success from resurrecting a subsequent +removal and preserves the set across a compaction checkpoint. For older histories +without declarations, a successful `ToolSearch` result contributes its `addedToolNames`; a successful result from a deferred tool contributes that tool's name. A name is restored only when it remains in the current mode's deferred catalog. Failed, interrupted, or missing-result placeholder rows are ignored, and assistant/user diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md new file mode 100644 index 0000000000..74884ee9f0 --- /dev/null +++ b/docs/adr/chronological-system-transcript.md @@ -0,0 +1,63 @@ +# ADR: Preserve chronological model system state + +- Status: Accepted +- Date: 2026-10-01 +- Issue: #1285 +- References: [Pi PR #9548](https://github.com/earendil-works/pi/pull/9548), + [Pi model authority](pi-ai-core-0991-authority.md) +- Updates: ADR 0039 catalog refresh, ADR 0225 deferred-tool restoration + +## Context + +Moving new tool declarations to the beginning of a conversation changes its +cached prefix. Rebuilding from visible chat alone also loses the chronology of +instruction and tool changes. Pi 0.99.1 already owns tool-state diffing and +provider request projections; Desktop owns session persistence and permissions. + +## Decision + +Keep Pi's provider-neutral system messages in chronological order. Use the +existing Host-owned transcript writer and optional versioned `modelSystem` +metadata on hidden system rows, acknowledging writes before dispatch. Anchors +preserve model order when the user row was pre-persisted. Save the effective +state once at explicit compaction; restore it before the summary and retained +tail. Keep current mode/catalog/Host permissions authoritative for execution. + +Separate Desktop's runtime instructions, skill catalog and contextual guidance +into replaceable sections. The skill-loading header uses `skills`, and each +catalog entry uses `skill:`; a change or removal appends only that +entry's replacement or null tombstone. The fixed prefix prevents collisions with +other Desktop section names. Restoring a legacy aggregate `skills` section +replaces it with the header and individual entries once at continuation. +Catalog-only skill changes refresh an idle runtime; +body loading and permission checks stay at the existing Host bridge. Tool schema +changes cannot reuse an idle runtime merely because tool names are identical. + +Retain the published model/API/endpoint identity before account overrides. +Accept transcript capability opt-ins only when that identity matches the actual +request route. Let Pi adapt the request for partial or absent support; never +persist an adapter's folded projection. + +The existing Pi 0.99.1 dependency patch adds the missing mid-conversation system +capability to its `deepseek-flash` catalog entry. Authorized official-endpoint +experiments confirmed both preserved cache reuse and effective updated +instructions. Keep this correction in the single Pi catalog, not a parallel +Desktop allowlist; remove the hunk when an upgraded Pi catalog carries it. +The model/API/endpoint binding check still excludes aliases and relays. Native +tool-addition and tool-change flags are not enabled by this correction. + +## Alternatives and consequences + +Prepending a reconstructed snapshot was rejected because it changes history on +ordinary continuation. A new persistence store or adopting Pi AgentSession +would create another session owner; optional canonical metadata preserves the +current process and data boundaries without a database migration. Unknown old +histories establish their first baseline at continuation rather than inventing +past state. Downgrades remain readable but do not retain these cache guarantees. + +The added records consume transcript storage. A failed write stops dispatch as +a local preparation failure; retry uses the same row ID. Providers may still +fold unsupported updates and lose cache reuse. Even native mid-conversation +system support does not guarantee provider cache reuse. Entry-level updates +reduce new input tokens without changing instruction roles or priority. Offline tests prove ordering, +restoration and payload compatibility, not real-provider cache-hit ratios. diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index 0262e800af..db1a2ee59f 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -519,8 +519,8 @@ anchors on the last assistant usage and estimates everything after it as `chars / 4`: that constant under-counts CJK text, and with no anchor left it omits system/tool overhead. The budget also computes the output-cap estimator over non-system conversation messages plus the current system prompt and active -tool schemas. System-transcript rows are metadata snapshots of that same prompt -and are excluded from this component, so the request estimate counts the prompt +tool schemas. System-transcript rows are chronological updates, already covered by that +folded prompt/tool floor, and are excluded from this component, so the request estimate counts the prompt and schemas exactly once. The budget uses the larger of this full-request estimate and the calibrated message estimate. This is a hard floor before the first calibration sample and prevents counting overhead twice after an @@ -1278,6 +1278,51 @@ payload hook keeps its own object and its return value still wins. + [optional user custom instructions] ``` +### 7.0.0 Chronological system state (issue #1285) + +The Pi agent loop owns `toolsAdded` / `toolsRemoved` declarations, including +same-name schema replacement. Desktop never moves an update ahead of the user +or ToolSearch result that preceded it. Instruction composition uses independent +`runtime`, `skills` (loading instructions), `skill:` (one catalog +entry each), and `context` sections. Refreshing an idle skill catalog appends only +changed entries, uses null to revoke removed entries, and updates the executable +catalog. Unchanged entries and loading instructions are not repeated. Restoring +an older aggregate `skills` section upgrades it once at the continuation +boundary; checkpoints retain the effective per-entry state. Bodies remain +on-demand and permission-checked. Different plugin tool schemas/permissions +invalidate idle reuse even when their names are unchanged. + +Before model dispatch, Desktop acknowledges new system records through Host's +existing transcript writer. Restart replays these provider-neutral records in +order. An old history without records receives the current declaration at its +continuation boundary; Desktop does not invent historical instructions. Explicit +compaction saves the effective system state once before summary and retained +tail; it does not replay old tool deltas or preserve removed tools. + +Mid-conversation instructions, tool additions and full tool changes are separate +capabilities. Desktop accepts catalog opt-ins only for the same published model, +wire API and effective endpoint (including path and port). Unknown models, +changed bindings and unverified relays use Pi's conservative request projection. +Partial support keeps instruction updates but folds tools as required; removals +and redefinitions use the adapter's supported fallback. Model switching never +rewrites the canonical journal. Cache savings depend on the actual provider; +unsupported routes may still rebuild the request prefix. + +The pinned Pi patch declares mid-conversation system support for +`deepseek-flash` on its published `openai-completions` binding at +`https://api.deepseek.com`. Skill changes on this binding append system updates +without rewriting the previous request prefix. This does not enable native tool +additions or changes, and does not apply to unverified gateways, aliases or APIs. +The declaration comes from the patched Pi catalog, not provider settings or a +second Desktop model catalog. + +When restoring assistant history, map the current local account ID to the Pi +model's provider identity and retain recorded model IDs. A different recorded +account or model remains distinct. Legacy rows without identity retain the +current-model fallback. Same-model Completions reasoning stays in its native +reasoning field, never appended to visible answer text because of an account +UUID/vendor-name mismatch. + ### 7.0.1 User custom system prompt files (issue #542) The `[optional user custom instructions]` layer is the pi-compatible file pair @@ -1378,7 +1423,11 @@ schemas. Providers with native deferred-tool search receive the definitions at that load point; other providers receive the active definitions normally. At the start of each new user prompt, the sidecar clears the in-memory deferred -activation set and rebuilds it from the effective context. Successful +activation set and rebuilds it from the effective context. Recorded system +messages (including a compaction checkpoint) define the active baseline. Only +successful tool results after the latest declaration can add a new activation; +older results must not resurrect a removed tool. Histories without system records +continue to use successful results throughout their effective context. `ToolSearch` results contribute their canonical `details.addedToolNames`. For compatibility, historical `details.activated` and top-level `addedToolNames` markers are also accepted. Successful results from deferred diff --git a/docs/spec/03-runtime/04-data-storage.md b/docs/spec/03-runtime/04-data-storage.md index d43eee94c6..e1722bc582 100644 --- a/docs/spec/03-runtime/04-data-storage.md +++ b/docs/spec/03-runtime/04-data-storage.md @@ -179,6 +179,24 @@ per message; `seq` is implied by line order: {"type":"compaction","id":"cp1","summary":"…","firstKeptMessageId":"m2","throughMessageId":"m3","tokensBefore":917000,"retainedTail":[…],"providerId":"…","modelId":"…","createdAt":"…"} ``` +Internal system-state rows use role `system`, empty visible content, and optional +`meta.modelSystem = { version: 1, messageJson, beforeMessageId?, afterMessageId? }`. +`messageJson` is validated JSON text of a Pi system message with sections and tool +schema deltas; executable functions are excluded. JSON text preserves section and schema key +order across Rust storage; parsing for validation never reserializes it. Stable row IDs make retries +idempotent. The anchors restore logical model order when a user row was already +persisted before its preceding declaration; a surviving following anchor takes +precedence, then a preceding anchor, then the record's continuation position. +Forks remap surviving anchor IDs. Normal system notices remain visible; internal +model-state rows do not produce transcript bubbles or search text. + +Compaction details may include one `systemMessageJson` checkpoint. It replaces old +system updates in the retained tail and is restored before the summary. These +optional metadata fields use the existing JSONL/SQLite index and require no +schema migration. Old sessions remain readable; their first continuation records +a new baseline. Older app versions ignore the metadata and reconstruct their +usual current prompt, so downgrade does not promise the same cache prefix. + `sessions/.inflight.json` — the assistant reply currently streaming in the session, as one `{ schema, sessionId, turnId, savedAt, message }` object that host-core replaces atomically (temp + rename) on every diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 00272e30d3..35e7005ecb 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -16258,3 +16258,42 @@ renderer's durable transcript reads. No real model or provider is contacted. - Installed Electron, real account/paid API and cross-version rollback are separate release qualification. No MCP, Codemode or virtual-router migration is included. See `docs/project/pi-0991-adoption.md` for candidate evidence. + +### E2E-SYSTEM-TRANSCRIPT: Ordered model state survives restart and compaction + +- Fixture: isolated Host data directory, production AgentSidecar transport and + separate Node runtime/Pi process, local SSE provider; no real provider + credentials or user's Desktop process. The harness persists completed + messages; it does not exercise Electron's UI or persistence outbox. +- Send a prompt, activate BrowserPreview through ToolSearch, and verify the next + request includes the schema. Confirm the baseline and delta are acknowledged + by Host and survive stopping/restarting both processes without duplicate rows. +- Resume the session, update a skill catalog on the same runtime, and confirm the + next request uses the new catalog while keeping a second unchanged entry. + Runtime/adapter tests verify the native delta contains only the modified entry, + and fallback models receive the complete current catalog with removals applied. + Invoke Skill through the actual sidecar + bridge with a fixture body supplied by the embedding host. Compact, restart + the sidecar, and verify current skills and active tools survive without + replaying old tool results. Remove the final skill and verify its catalog and + executable schema disappear without recreating the runtime. +- Automated: `node scripts/e2e-system-transcript.mjs` with + `PI_DESKTOP_HOST_BIN` pointing to the candidate Host binary. Requires built + shared/host-runtime/agent-runtime packages. Renderer projection tests ensure + model-state records stay hidden while ordinary system notices remain visible. +- Adapter contract tests cover exact model/API/endpoint capability binding, + partial support, unverified relay fallback, removals/redefinitions and model + switching without canonical-history mutation. Real cache-hit improvement is + a separately authorized provider experiment, not an offline-test claim. + +- Flash regression: load the pinned Pi `deepseek-flash` model, serialize its + Desktop projection, and pass skill updates/removals through the real adapter. + The official binding preserves the earlier wire prefix; changed bindings + retain the folded fallback. Covered by `flash-transcript.test.ts`. +- Authorized live Flash acceptance: use isolated development sessions with + short and long contexts. Alternate unchanged turns and single-skill updates, + confirm current Skill bodies execute, and compare actual outgoing prefixes + plus reported usage. After restart, verify no duplicate system updates and + same-model assistant reasoning remains separate from visible content. Adapter + regressions also cover legacy identities and genuine account/model changes. Cache + percentages are observations, not deterministic pass thresholds. diff --git a/packages/agent-runtime/src/extensions/prompt-chain.test.ts b/packages/agent-runtime/src/extensions/prompt-chain.test.ts index 1669a0b557..fa212b0c45 100644 --- a/packages/agent-runtime/src/extensions/prompt-chain.test.ts +++ b/packages/agent-runtime/src/extensions/prompt-chain.test.ts @@ -39,7 +39,10 @@ it.each([ }); runtime = new DesktopAgentRuntime({ host: { - call: async (method) => { throw new Error(`Unexpected host call: ${method}`); }, + call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + throw new Error(`Unexpected host call: ${method}`); + }, onNotification: () => () => {}, }, sessionId: "prompt-chain-fixture", diff --git a/packages/agent-runtime/src/flash-transcript.test.ts b/packages/agent-runtime/src/flash-transcript.test.ts new file mode 100644 index 0000000000..a423319cf0 --- /dev/null +++ b/packages/agent-runtime/src/flash-transcript.test.ts @@ -0,0 +1,144 @@ +import type { Agent } from "@earendil-works/pi-agent-core"; +import { DesktopAgentRuntime } from "./runtime.js"; +import { describe, expect, it } from "vitest"; +import { normalizeContext, type Message, type Model } from "@earendil-works/pi-ai"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { stream } from "@earendil-works/pi-ai/api/openai-completions"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { buildProviderModel, type RuntimeProviderConfig } from "./provider-binding.js"; + +function flashProvider(): RuntimeProviderConfig { + const flash = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); + if (!flash) throw new Error("Missing published Flash model"); + return { + id: "flash-fixture", name: "Flash fixture", modelId: flash.id, baseUrl: flash.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: true, supportedThinkingLevels: ["high"], + // Main sends this projection across the sidecar boundary. + modelConfig: JSON.parse(JSON.stringify(modelConfigFromPi(flash))), + }; +} + +type Payload = { messages: Array<{ role: string; content?: unknown }> }; +async function request(messages: Message[], baseUrl?: string): Promise { + const provider = flashProvider(); + const model = buildProviderModel({ ...provider, ...(baseUrl ? { baseUrl } : {}) }); + let captured: Payload | undefined; + const response = stream(model as Model<"openai-completions">, normalizeContext({ messages }), { + apiKey: "fixture", + fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + expect((await response.result()).stopReason).toBe("stop"); + if (!captured) throw new Error("No Flash request captured"); + return captured; +} + +describe("published Flash transcript compatibility", () => { + it("enables system updates independently of native tool-state capabilities", () => { + const provider = flashProvider(); + expect(provider.modelConfig?.compat?.supportsMidConvoSystemMessages).toBe(true); + expect(buildProviderModel(provider).compat).toMatchObject({ + supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false, + supportsMidConvoToolChanges: false, supportsAdditionalTools: false, supportsToolSearch: false, + }); + expect(buildProviderModel({ ...provider, baseUrl: "https://api.deepseek.com/" }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + + it.each([ + { modelId: "deepseek-flash-alias" }, + { baseUrl: "https://relay.example/v1" }, { baseUrl: "http://api.deepseek.com" }, + { baseUrl: "https://api.deepseek.com/v1" }, { baseUrl: "https://api.deepseek.com:8443" }, + { baseUrl: "https://api.deepseek.com?route=other" }, { baseUrl: "https://api.deepseek.com#fragment" }, + { baseUrl: "https://user@api.deepseek.com" }, + ])("retains conservative fallback on a changed binding: %j", (overrides) => { + expect(buildProviderModel({ ...flashProvider(), ...overrides }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: false }); + }); + + it("checks the selected wire API rather than the provider-wide style", () => { + const provider = flashProvider(); + expect(buildProviderModel({ ...provider, apiStyle: "responses" })).toMatchObject({ + api: "openai-completions", compat: { supportsMidConvoSystemMessages: true }, + }); + expect(buildProviderModel({ ...provider, modelConfig: { ...provider.modelConfig!, api: "openai-responses" } })) + .toMatchObject({ api: "openai-responses", compat: { supportsMidConvoSystemMessages: false } }); + }); + + it("requires the original Pi binding even when the endpoint and model name look official", () => { + const provider = flashProvider(); + for (const modelConfig of [undefined, + { ...provider.modelConfig!, source: "generic" as const }, + { ...provider.modelConfig!, transcriptBinding: undefined }, + ]) { + expect(buildProviderModel({ ...provider, modelConfig }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: false }); + } + }); + + it("appends skill changes and revocations without rewriting the previous wire prefix", async () => { + const baseline: Message[] = [ + { role: "system", content: "", timestamp: 1, sections: { + runtime: "Base rules", skills: "Load skills on demand", "skill:a": "Old Alpha", "skill:b": "Bravo", + } }, + { role: "user", content: "First request", timestamp: 2 }, + ]; + const changed: Message[] = [...baseline, + { role: "system", content: "", timestamp: 3, sections: { "skill:a": "New Alpha", "skill:b": null } }, + { role: "user", content: "Continue", timestamp: 4 }, + ]; + const before = await request(baseline); + const after = await request(changed); + expect(after.messages.slice(0, before.messages.length)).toEqual(before.messages); + expect(after.messages.map((message) => message.role)).toEqual(["system", "user", "system", "user"]); + expect(JSON.stringify(after.messages[2])).toContain("New Alpha"); + expect(JSON.stringify(after.messages[2])).toContain("skill:b"); + const relay = await request(changed, "https://relay.example/v1"); + expect(relay.messages.map((message) => message.role)).toEqual(["system", "user", "user"]); + expect(JSON.stringify(relay.messages)).toContain("New Alpha"); + expect(JSON.stringify(relay.messages)).not.toContain("Old Alpha"); + expect(JSON.stringify(relay.messages)).not.toContain("Bravo"); + }); +}); + + +describe("Flash reasoning after Desktop session restoration", () => { + it.each([ + { providerId: "flash-fixture", modelId: "deepseek-flash", sameModel: true }, + { providerId: undefined, modelId: undefined, sameModel: true }, + { providerId: "another-account", modelId: "deepseek-flash", sameModel: false }, + { providerId: "flash-fixture", modelId: "another-model", sameModel: false }, + ])("preserves source identity when restoring %j", async ({ providerId, modelId, sameModel }) => { + const provider = { ...flashProvider(), vendorKey: "deepseek" }; + let captured: Payload | undefined; + const runtime = new DesktopAgentRuntime({ + sessionId: "restore-flash", mode: "agent", provider, thinkingLevel: "high", + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + history: [{ + id: "old-answer", role: "assistant", content: "answer", thinking: "private plan", + providerId, modelId, status: "complete", createdAt: "2026-10-01T00:00:00.000Z", + }], + host: { call: async (): Promise => undefined as T }, onEvent: () => {}, + }); + const agent = (runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = (model, context) => stream(model as Model<"openai-completions">, context, { + apiKey: "fixture", fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + try { + await runtime.prompt("Continue", "new-user", "new-turn"); + expect(captured).toBeDefined(); + const assistant = captured!.messages.find((message) => message.role === "assistant"); + if (sameModel) { + expect(assistant).toMatchObject({ content: "answer", reasoning_content: "private plan" }); + } else { + expect(assistant?.content).toContain("private plan"); + expect(assistant).not.toMatchObject({ reasoning_content: "private plan" }); + } + } finally { await runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/mode-tool-access.test.ts b/packages/agent-runtime/src/mode-tool-access.test.ts index 64f96723e4..042010e7e2 100644 --- a/packages/agent-runtime/src/mode-tool-access.test.ts +++ b/packages/agent-runtime/src/mode-tool-access.test.ts @@ -80,7 +80,10 @@ it.each([ compactionSettings: { enabled: false, reserveTokens: 0, keepRecentTokens: 0 }, commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, subagents: [{ name: "worker", description: "Inspect", tools: ["Read"], prompt: "Inspect only.", source: "user" }], - host: { call: async (method) => { hostCalls.push(method); throw new Error("blocked tools must never reach the host"); }, onNotification: () => () => {} }, + host: { call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + hostCalls.push(method); throw new Error("blocked tools must never reach the host"); + }, onNotification: () => () => {} }, onEvent: (event) => { events.push(event); }, }); runtime.setMode(mode); diff --git a/packages/agent-runtime/src/model-capabilities.ts b/packages/agent-runtime/src/model-capabilities.ts index dda3a9bce8..2c7f3cd589 100644 --- a/packages/agent-runtime/src/model-capabilities.ts +++ b/packages/agent-runtime/src/model-capabilities.ts @@ -24,6 +24,7 @@ export function modelConfigFromPi(model: Model): ModelConfig { ...metadata, ...(compat ? { compat: { ...compat } } : {}), source: "pi", + transcriptBinding: { modelId: model.id, api: model.api, baseUrl: model.baseUrl }, nativeCost: cost, // Desktop's historical tier schema differs; do not invent a translation. cost: { input: cost.input, output: cost.output, cacheRead: cost.cacheRead, cacheWrite: cost.cacheWrite }, diff --git a/packages/agent-runtime/src/output-cap.test.ts b/packages/agent-runtime/src/output-cap.test.ts index 0faaf69d84..234a118a54 100644 --- a/packages/agent-runtime/src/output-cap.test.ts +++ b/packages/agent-runtime/src/output-cap.test.ts @@ -352,3 +352,11 @@ describe("clampOutputToContext", () => { } }); }); + +// Transcript-only Pi requests have no legacy systemPrompt/tools fields. +it("counts instruction sections and tool declarations in transcript-only requests", () => { + const base = { messages: [{ role: "system", content: "" }] }; + const state = { messages: [{ role: "system", content: "", sections: { skills: "中文说明" }, toolsAdded: [{ name: "Read", parameters: { type: "object" } }] }] }; + expect(estimateOutputCapInputTokens(state)).toBeGreaterThan(estimateOutputCapInputTokens(base)); + expect(estimateOutputCapInputTokens(state)).toBeGreaterThan(4); +}); diff --git a/packages/agent-runtime/src/output-cap.ts b/packages/agent-runtime/src/output-cap.ts index ccc46f05fb..728d058580 100644 --- a/packages/agent-runtime/src/output-cap.ts +++ b/packages/agent-runtime/src/output-cap.ts @@ -46,6 +46,9 @@ export type OutputCapContext = { export type OutputCapMessage = { role: string; content: unknown; + sections?: Record; + toolsAdded?: readonly unknown[]; + toolsRemoved?: readonly unknown[]; api?: Api; provider?: string; model?: string; @@ -224,6 +227,14 @@ export function estimateOutputCapInputTokens( message.content, replayOptionsFor(message, model), ); + if (message.role === "system") { + // Canonical transcript requests carry instruction sections and schema + // updates on the message, without a separate systemPrompt/tools field. + const state = [message.sections, message.toolsAdded, message.toolsRemoved] + .filter((value) => value !== undefined).map(stringifyForEstimate).join("\n"); + baseline += Math.ceil(state.length / CHARS_PER_TOKEN); + cjkChars += countCjkChars(state); + } baseline += estimated.baseline; cjkChars += estimated.cjkChars; } diff --git a/packages/agent-runtime/src/plugin-skills-prompt.test.ts b/packages/agent-runtime/src/plugin-skills-prompt.test.ts index 9060c0f77d..c2a63d4f14 100644 --- a/packages/agent-runtime/src/plugin-skills-prompt.test.ts +++ b/packages/agent-runtime/src/plugin-skills-prompt.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "vitest"; import { pluginSkillsPrompt, + pluginSkillsPromptSections, SKILL_TOOL_NAME, type PluginSkillDef, } from "./plugin-skills-prompt.js"; @@ -38,3 +39,18 @@ describe("pluginSkillsPrompt", () => { expect(prompt.split("\n").filter((line) => line.startsWith("- "))).toHaveLength(1); }); }); + +it("gives skill IDs collision-free section names and stable catalog ordering", () => { + const entries = [ + { id: "__proto__", name: "Prototype" }, { id: "runtime", name: "Reserved" }, + { id: "skill:a", name: "Colon" }, { id: "a", name: "A" }, + ]; + const sections = pluginSkillsPromptSections(entries); + expect(sections).toEqual(pluginSkillsPromptSections([...entries].reverse())); + expect(Object.keys(sections)).toEqual(Object.keys(pluginSkillsPromptSections([...entries].reverse()))); + expect(sections["skill:__proto__"]).toContain("Prototype"); + expect(sections["skill:runtime"]).toContain("Reserved"); + expect(sections["skill:skill:a"]).toContain("Colon"); + expect(sections["skill:a"]).toContain("A"); + expect(pluginSkillsPromptSections([])).toEqual({}); +}); diff --git a/packages/agent-runtime/src/plugin-skills-prompt.ts b/packages/agent-runtime/src/plugin-skills-prompt.ts index 9416ca37a9..73eb161f81 100644 --- a/packages/agent-runtime/src/plugin-skills-prompt.ts +++ b/packages/agent-runtime/src/plugin-skills-prompt.ts @@ -11,16 +11,30 @@ export const SKILL_TOOL_NAME = "Skill"; * through the `Skill` tool when the model decides a skill applies, so a long * document costs nothing until it is needed. */ +const SKILLS_HEADER = [ + "# Skills", + "", + `Plugins have taught you the following skills. Each entry is a set of instructions you can load with the \`${SKILL_TOOL_NAME}\` tool by passing its exact id. When a task matches a skill's description, load the skill first and follow it; do not guess at its content. Load each skill at most once per task.`, +].join("\n"); + +export const SKILL_SECTION_PREFIX = "skill:"; + +function skillEntry(skill: PluginSkillDef): string { + const description = skill.description?.trim(); + return `- \`${skill.id}\` — ${skill.name}${description ? `: ${description}` : ""}`; +} + +/** Stable entry identities let Pi append only the changed or revoked skill. */ +export function pluginSkillsPromptSections(skills: readonly PluginSkillDef[]): Record { + if (!skills.length) return {}; + return Object.fromEntries([ + ["skills", SKILLS_HEADER], + ...[...skills].sort((a, b) => a.id.localeCompare(b.id)) + .map((skill) => [`${SKILL_SECTION_PREFIX}${skill.id}`, skillEntry(skill)]), + ]); +} + export function pluginSkillsPrompt(skills: PluginSkillDef[]): string | undefined { if (!skills.length) return undefined; - return [ - "# Skills", - "", - `Plugins have taught you the following skills. Each entry is a set of instructions you can load with the \`${SKILL_TOOL_NAME}\` tool by passing its exact id. When a task matches a skill's description, load the skill first and follow it; do not guess at its content. Load each skill at most once per task.`, - "", - ...skills.map((skill) => { - const description = skill.description?.trim(); - return `- \`${skill.id}\` — ${skill.name}${description ? `: ${description}` : ""}`; - }), - ].join("\n"); + return `${SKILLS_HEADER}\n\n${skills.map(skillEntry).join("\n")}`; } diff --git a/packages/agent-runtime/src/plugin-skills.test.ts b/packages/agent-runtime/src/plugin-skills.test.ts index d9df1a82ee..d87fd7b559 100644 --- a/packages/agent-runtime/src/plugin-skills.test.ts +++ b/packages/agent-runtime/src/plugin-skills.test.ts @@ -40,3 +40,9 @@ describe("pluginSkillsDigest", () => { expect(pluginSkillsDigest([{ id: "a", name: "A", description: "" }])).toBe(without); }); }); + +it("does not confuse delimiter-containing catalog fields", () => { + expect(pluginSkillsDigest([{ id: "a", name: "b:c", description: "d" }])).not.toBe( + pluginSkillsDigest([{ id: "a", name: "b", description: "c:d" }]), + ); +}); diff --git a/packages/agent-runtime/src/plugin-skills.ts b/packages/agent-runtime/src/plugin-skills.ts index 42baab6951..4126b89f64 100644 --- a/packages/agent-runtime/src/plugin-skills.ts +++ b/packages/agent-runtime/src/plugin-skills.ts @@ -20,15 +20,11 @@ export type PluginSkillDef = { /** * Stable fingerprint of a skill catalog. * - * `AgentRuntime.matches()` compares this so enabling a plugin, revoking its - * prompt permission, or editing a skill's front matter starts a fresh runtime - * instead of reusing a session whose catalog is already stale. Bodies are not - * part of it: they never enter the prompt, and the `Skill` tool reads them - * fresh from disk on every call. + * The runtime compares this before appending a skills section update. Bodies + * are excluded: the Skill tool reads them through the permission-checked host + * bridge on each invocation. */ export function pluginSkillsDigest(skills?: PluginSkillDef[]): string { if (!skills?.length) return ""; - return skills - .map((skill) => `${skill.id}:${skill.name}:${skill.description ?? ""}`) - .join("|"); + return JSON.stringify(skills.map((skill) => [skill.id, skill.name, skill.description ?? ""])); } diff --git a/packages/agent-runtime/src/provider-binding.ts b/packages/agent-runtime/src/provider-binding.ts index 5733dd2eef..c42d42b5ff 100644 --- a/packages/agent-runtime/src/provider-binding.ts +++ b/packages/agent-runtime/src/provider-binding.ts @@ -1,3 +1,4 @@ +import { transcriptCompat } from "./transcript-compat.js"; /** * Provider/model wiring shared by the session runtime and its subagents. * @@ -312,7 +313,7 @@ export function buildProviderModel( const binding = apiBindingForProviderModel(provider); const catalog = provider.modelConfig; const catalogModel = catalog - ? (({ source: _source, nativeCost, ...model }) => ({ + ? (({ source: _source, transcriptBinding: _transcriptBinding, nativeCost, ...model }) => ({ ...model, ...(nativeCost ? { cost: nativeCost } : {}), }))(catalog) @@ -386,7 +387,7 @@ export function buildProviderModel( }) === "on" ? true : undefined, - ...(compat ? { compat } : {}), + compat: { ...compat, ...transcriptCompat(catalog, provider.modelId, binding.api, baseUrl) }, ...(Object.keys(modelHeaders).length > 0 ? { headers: modelHeaders } : {}), } as Model; } diff --git a/packages/agent-runtime/src/provider-certificate-flow.test.ts b/packages/agent-runtime/src/provider-certificate-flow.test.ts index b533cb1d1e..c3f0d89113 100644 --- a/packages/agent-runtime/src/provider-certificate-flow.test.ts +++ b/packages/agent-runtime/src/provider-certificate-flow.test.ts @@ -29,7 +29,10 @@ it.each(["session", "delegate"])( const runtime = kind === "session" ? new DesktopAgentRuntime({ sessionId: "certificate-session", mode: "agent", provider, thinkingLevel: "off", onEvent, - host: { call: async () => { throw new Error("Unexpected host request"); } }, + host: { call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + throw new Error("Unexpected host request"); + } }, commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, }) : undefined; try { diff --git a/packages/agent-runtime/src/provider-recovery-flow.test.ts b/packages/agent-runtime/src/provider-recovery-flow.test.ts index 81dc0ec832..23d5298189 100644 --- a/packages/agent-runtime/src/provider-recovery-flow.test.ts +++ b/packages/agent-runtime/src/provider-recovery-flow.test.ts @@ -85,6 +85,7 @@ function fixture(steps: Step[]) { }, host: { async call(method: string): Promise { + if (method === "session.appendMessage") return undefined as T; if (method !== "tools.execute") throw new Error(`Unexpected host method: ${method}`); reads++; diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index a511a7060c..9318de1cbf 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -293,10 +293,10 @@ describe("system transcript reconstruction", () => { { id: "old-assistant", role: "assistant", content: "Earlier answer", createdAt: new Date(before - 1_000).toISOString(), status: "complete" }, ] }); const internals = runtime as any; - const prefix = internals.agent.state.messages[0]; + const prefix = internals.agent.state.messages.at(-1); expect(prefix.timestamp).toBeGreaterThanOrEqual(before); const rebuilt = internals.rebuiltAgentContext(); - expect(rebuilt.messages[0]).toBe(prefix); + expect(rebuilt.messages.at(-1)).toBe(prefix); expect(getCurrentTools(rebuilt.messages)).toEqual(rebuilt.tools.map(toToolDeclaration)); await runtime.dispose(); }); @@ -318,7 +318,7 @@ describe("system transcript reconstruction", () => { vi.spyOn(agent, "continue").mockImplementation(async () => { expect(agent.state.systemPrompt).toContain(marker); expect(agent.state.systemPrompt.split("SECTION_MARKER")).toHaveLength(2); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "SECTION_MARKER" }); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "SECTION_MARKER" }); expect(getCurrentTools(agent.state.messages)).toEqual(tools); response.timestamp = agent.state.messages[0]!.timestamp + 10; agent.state.messages.push(response as any); @@ -329,9 +329,9 @@ describe("system transcript reconstruction", () => { expect(agent.continue).toHaveBeenCalledOnce(); expect(agent.state.systemPrompt).toBe(before); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "SECTION_MARKER" }); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "SECTION_MARKER" }); expect(getCurrentTools(agent.state.messages)).toEqual(tools); - expect(estimateTranscriptTokens(agent.state.messages as any).usageTokens).toBe(0); + expect(agent.state.messages.at(-1)?.timestamp).toBeGreaterThan(response.timestamp); await runtime.dispose(); }); }); @@ -2183,10 +2183,9 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(next.context.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Tool deltas append new declarations; catalog order is not semantic. - expect([...getCurrentTools(next.context.messages)].sort((a, b) => a.name.localeCompare(b.name))).toEqual( - next.context.tools.map(toToolDeclaration).sort((a: { name: string }, b: { name: string }) => a.name.localeCompare(b.name)), - ); + // The Pi loop declares changes immediately before conversion; preparation + // only changes the executable tool catalog. + expect(getCurrentTools(next.context.messages).some((tool) => tool.name === "BrowserPreview")).toBe(false); await runtime.dispose(); }); @@ -2856,7 +2855,7 @@ describe("DesktopAgentRuntime plan transitions", () => { // The progress assistant is visible in the reused bubble but must be // removed before continue() rebuilds the model context. expect(agent.state.messages.filter((message: any) => message.role !== "system")).toHaveLength(1); - expect(agent.state.messages.at(-1)?.role).toBe("user"); + expect(agent.state.messages.filter((message: AgentMessage) => message.role !== "system").at(-1)?.role).toBe("user"); expect(agent.state.systemPrompt).toContain(""); } await handleAgentEvent({ type: "agent_start" }); @@ -2945,7 +2944,7 @@ describe("DesktopAgentRuntime plan transitions", () => { ]; } else { expect(agent.state.messages.filter((message: any) => message.role !== "system")).toHaveLength(1); - expect(agent.state.messages.at(-1)?.role).toBe("user"); + expect(agent.state.messages.filter((message: AgentMessage) => message.role !== "system").at(-1)?.role).toBe("user"); expect(agent.state.systemPrompt).toContain( attempts === 2 ? "" : "", ); @@ -5178,7 +5177,7 @@ describe("DesktopAgentRuntime compaction restore", () => { const estimate = estimateAgentContextTokens((runtime as any).agent.state.messages); expect(estimate.usageTokens).toBe(0); expect(estimate.lastUsageIndex).toBeNull(); - expect(budget.tokens).toBeGreaterThan(estimate.tokens); + expect(budget.tokens).toBeGreaterThanOrEqual(estimate.tokens); expect(budget.tokens).toBeGreaterThan(0); expect(budget.tokens).toBeLessThan(250_000); await runtime.dispose(); @@ -6555,17 +6554,18 @@ describe("DesktopAgentRuntime inline context compaction", () => { ); const generateCompaction = vi.spyOn(runtime as any, "generateCompaction"); const agent = (runtime as any).agent as Agent; - const prefix = { ...agent.state.messages[0], sections: { rules: "Keep checkpoint rules" } }; - agent.state.messages[0] = prefix as any; + const index = agent.state.messages.findIndex((message) => message.role === "system"); + const prefix = { ...agent.state.messages[index], sections: { rules: "Keep checkpoint rules" } }; + agent.state.messages[index] = prefix as any; await (runtime as any).prepareNextTurn(nextTurn); // The point of this family: the window is bought back without paying for a // summary, so no provider request is made at all. - expect(agent.state.messages[0]).toBe(prefix); + expect(agent.state.messages[0]).toEqual(prefix); expect(getCurrentTools(agent.state.messages)).toEqual(agent.state.tools.map(toToolDeclaration)); expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "Keep checkpoint rules" }); - expect((runtime as any).rebuiltAgentContext().messages[0]).toBe(prefix); + expect((runtime as any).rebuiltAgentContext().messages[0]).toEqual(prefix); expect(generateCompaction).not.toHaveBeenCalled(); const compaction = host.call.mock.calls.find( ([method]) => method === "session.appendCompaction", @@ -6582,7 +6582,7 @@ describe("DesktopAgentRuntime inline context compaction", () => { buildSessionContext((runtime as any).entriesWithCompaction()).messages.map( (message: any) => message.role, ), - ).toEqual(["compactionSummary"]); + ).toEqual(["system", "compactionSummary"]); const events = onEvent.mock.calls.map(([envelope]) => (envelope as any).event); expect(events).toContainEqual( expect.objectContaining({ @@ -6674,23 +6674,23 @@ describe("DesktopAgentRuntime plugin skills (D174)", () => { await runtime.dispose(); }); - it("does not reuse a runtime whose skill catalog changed", async () => { + it("reuses an idle runtime when the skill catalog changes", async () => { const runtime = createRuntime({ pluginSkills }); expect(runtimeMatches(runtime, { pluginSkills })).toBe(true); // Revoking agent.prompt.inject empties the catalog. - expect(runtimeMatches(runtime, { pluginSkills: [] })).toBe(false); + expect(runtimeMatches(runtime, { pluginSkills: [] })).toBe(true); expect( runtimeMatches(runtime, { pluginSkills: [...pluginSkills, { id: "demo.hello/other", name: "Other" }], }), - ).toBe(false); + ).toBe(true); // A renamed skill rewrites the catalog line the model reads. expect( runtimeMatches(runtime, { pluginSkills: [{ ...pluginSkills[0], name: "Renamed" }], }), - ).toBe(false); + ).toBe(true); await runtime.dispose(); }); @@ -10447,3 +10447,13 @@ describe("toolResultFromUi image restoration (issue #1073)", () => { expect(restored.content).toEqual([{ type: "text", text: expect.stringContaining("broken.png") }]); }); }); + +it("does not reuse stale plugin declarations when schema or permission metadata changes", async () => { + const plugin: PluginToolDef = { name: "plugin_fixture", description: "Inspect", parameters: { type: "object", properties: { path: { type: "string" } } } }; + const runtime = createRuntime({ pluginTools: [plugin] }); + try { + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin }] })).toBe(true); + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin, parameters: { type: "object", properties: { file: { type: "string" } } } }] })).toBe(false); + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin, planSafeActions: ["inspect"] }] })).toBe(false); + } finally { await runtime.dispose(); } +}); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 4b3a475565..f7084b85dd 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -1,3 +1,4 @@ +import { orderSystemRows, SystemTranscriptJournal } from "./system-transcript-journal.js"; import { accountModelStream } from "./request-usage.js"; import { modeToolDenial, retainModeToolDeclaration, withModeExecutionGuard } from "./mode-tool-access.js"; import { restoreHostedSearchReplay } from "./hosted-search-replay.js"; @@ -40,6 +41,7 @@ import { } from "@earendil-works/pi-agent-core"; import { isContextOverflow, + getCurrentTools, Type, type Api, type AssistantMessage, @@ -136,9 +138,12 @@ import { withExplicitRequired } from "./tool-schema.js"; import { buildSessionContext } from "./session-context.js"; import { initialSystemTranscript, + CONTEXT_BUDGET_SECTION, + syncSystemSections, + systemTranscriptCheckpoint, + removeTrailingAssistantMessages, rebuildSystemTranscript, replaceSystemPrompt, - syncSystemTools, systemPromptContent, } from "./system-transcript.js"; import { @@ -205,6 +210,7 @@ import type { CustomSystemPrompt } from "./custom-system-prompt.js"; import { projectMemoryPrompt } from "./project-memory-prompt.js"; import { pluginSkillsPrompt, + pluginSkillsPromptSections, SKILL_TOOL_NAME, type PluginSkillDef, } from "./plugin-skills-prompt.js"; @@ -1794,6 +1800,8 @@ export class DesktopAgentRuntime { }; private terminatingToolCalls = new Set(); private fullEntries: MessageEntry[]; + private readonly systemJournal = new SystemTranscriptJournal(); + private composedSections: Record = {}; private activeCompaction?: ContextCompactionRecord; private compactionEnabled: boolean; private readonly compactionStrategy: CompactionStrategy; @@ -1846,7 +1854,7 @@ export class DesktopAgentRuntime { this.streamSink = createStreamCoalescer(opts.onEvent); this.onEvent = (envelope) => this.streamSink.push(envelope); this.pluginTools = opts.pluginTools ?? []; - this.pluginSkills = opts.pluginSkills ?? []; + this.pluginSkills = [...(opts.pluginSkills ?? [])].sort((left, right) => left.id.localeCompare(right.id)); this.trustedExtensionSpecs = opts.trustedExtensions ?? []; this.subagents = opts.subagents ?? []; this.subagentProviders = opts.subagentProviders ?? {}; @@ -1881,7 +1889,6 @@ export class DesktopAgentRuntime { this.delegationChains.hydrate( rebuildChainsFromTranscript(this.transcriptHistory), ); - const skillsPrompt = pluginSkillsPrompt(this.pluginSkills); // Parts: [0] is the product persona; [1:] are operational rules a custom // SYSTEM.md must not remove (tool guidance, delegation, scratch, skills). const defaultSystemPromptParts = [ @@ -1913,8 +1920,6 @@ Do not invent objections or turn speculative risks into blockers. Stop when the `Your scratch directory for this session is \`${formatScratchDirForShell(this.commandShell, this.scratchDir)}\` (in Bash: ${shellScratchVariable(this.commandShell)}). Store ad-hoc temporary and intermediate files there using absolute paths. Workspace writes must be task-related project files or required toolchain outputs. Scratch persists across turns and is deleted with the session.`, ] : []), - // Plugin skills (D174). - ...(skillsPrompt ? [skillsPrompt] : []), ]; // A custom SYSTEM.md replaces only the product persona line, never the // operational rules in the default parts: tool guidance, delegation @@ -2058,19 +2063,24 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // The provider's rule that a tool-call id is unique is enforced here, on // the last view before the wire: the request is the only place it can be // guaranteed for both a rebuilt context and one that grew in this process. - convertToLlm: (messages) => - alignRetainedReasoningIdentity( - convertToLlm(this.dropDuplicateToolCalls(messages)), - this.reasoningReplayIdentity(), - ), + convertToLlm: async (messages) => { + await this.systemJournal.persist(messages, this.fullEntries, async (message) => { + await this.host.call("session.appendMessage", { + sessionId: this.sessionId, turnId: this.turnId, message, + }); + }); + return alignRetainedReasoningIdentity( + convertToLlm(this.dropDuplicateToolCalls(messages)), this.reasoningReplayIdentity(), + ); + }, prepareNextTurnWithContext: (context, signal) => this.prepareNextTurn(context, signal), afterToolCall: async (context) => this.afterToolCall(context), initialState: { model, - tools, + tools: [], thinkingLevel: agentThinkingLevel(this.thinkingLevel), - messages: initialSystemTranscript(this.composeSystemPrompt(), tools, this.liveSessionContext().messages), + messages: initialSystemTranscript(this.composeSystemPrompt(), tools, this.liveSessionContext().messages, this.composedSections), }, // Plan transitions must be the only tool call in an assistant batch. // Sequential execution also makes the host-confirmed mode change visible @@ -2094,6 +2104,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the }, }); + // Avoid Pi inserting a tool-only baseline ahead of restored legacy rows. + this.setAgentTools(tools); + // pi awaits every listener, so a throw here would reject the run in // progress and, with nothing awaiting that rejection, could take the whole // sidecar down. Contain it: log with the session attached and let the @@ -2121,6 +2134,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.infiniteProviderRetry = enabled; } + /** Refresh catalog instructions without changing the running session owner. */ + setPluginSkills(skills: PluginSkillDef[]): void { + if (this.disposed) throw new Error("runtime disposed"); + if (this.getStatus().isRunning) throw new Error("cannot update skills during an active turn"); + const sorted = [...skills].sort((left, right) => left.id.localeCompare(right.id)); + if (pluginSkillsDigest(this.pluginSkills) === pluginSkillsDigest(sorted)) return; + this.pluginSkills = sorted; + this.rebuildToolCatalog(); + this.setAgentTools(this.activeTools()); + this.setAgentSystemPrompt(this.composeSystemPrompt()); + } + /** Switch the planning state on this Agent without creating another Agent. */ setMode(mode: Mode): void { if (this.disposed) throw new Error("runtime disposed"); @@ -2161,7 +2186,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the (this.agent.state as unknown as { systemPrompt: string }).systemPrompt = prompt; return; } - this.agent.state.messages = replaceSystemPrompt(this.agent.state.messages, prompt); + this.agent.state.messages = prompt === this.composedSystemPrompt + ? syncSystemSections(this.agent.state.messages, this.composedSections) + : replaceSystemPrompt(this.agent.state.messages, prompt); } private setAgentMessages(messages: AgentMessage[]): void { @@ -2169,17 +2196,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.agent.state.messages = messages; return; } - this.agent.state.messages = syncSystemTools( - rebuildSystemTranscript(this.agent.state.messages, messages), - this.agent.state.tools, + this.agent.state.messages = rebuildSystemTranscript( + this.agent.state.messages.filter((message) => message.role !== "system" || !this.systemJournal.isPersisted(message)), messages, ); } private setAgentTools(tools: AgentTool[]): void { this.agent.state.tools = tools; - if (this.agentUsesTranscriptSystemMessages()) { - this.agent.state.messages = syncSystemTools(this.agent.state.messages, tools); - } } private agentSystemPromptContent(): string { @@ -2198,17 +2221,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the runningDelegationIds: this.runningDelegationIds(), }) : ""; - const composed = composeModeSystemPrompt( - this.mode, - [ - this.baseSystemPrompt, + this.composedSections = { + runtime: this.baseSystemPrompt, + ...pluginSkillsPromptSections(this.pluginSkills), + context: composeModeSystemPrompt(this.mode, [ ...(this.customSystemPrompt?.append ? [this.customSystemPrompt.append] : []), ...(optionalToolsPrompt ? [optionalToolsPrompt] : []), ...(projectPrompt ? [projectPrompt] : []), ...(memoryPrompt ? [memoryPrompt] : []), ...(resumablePrompt ? [resumablePrompt] : []), - ].join("\n\n"), - ); + ].join("\n\n")), + }; + const composed = Object.values(this.composedSections).filter(Boolean).join("\n\n"); this.composedSystemPrompt = composed; return composed; } @@ -2421,9 +2445,8 @@ Do not invent objections or turn speculative risks into blockers. Stop when the /** True when this runtime can be reused for a prompt with the given config. */ matches(config: RuntimeMatchConfig): boolean { const requestedPluginTools = config.pluginTools ?? []; - const requestedPluginSkills = config.pluginSkills ?? []; - const current = this.pluginTools.map((t) => t.name).sort().join(","); - const next = requestedPluginTools.map((t) => t.name).sort().join(","); + const current = safeJson([...this.pluginTools].sort((a, b) => a.name.localeCompare(b.name))); + const next = safeJson([...requestedPluginTools].sort((a, b) => a.name.localeCompare(b.name))); const currentThinkingLevels = [ ...(this.provider.supportedThinkingLevels ?? ["off"]), ] @@ -2466,10 +2489,6 @@ Do not invent objections or turn speculative risks into blockers. Stop when the safeJson(config.customSystemPrompt ?? null) && (this.projectMemory ?? "") === (config.projectMemory?.trim() ?? "") && (this.projectPath ?? "") === (config.projectPath?.trim() ?? "") && - // Enabling a plugin, revoking agent.prompt.inject or renaming a skill - // changes the catalog digest, which retires the runtime and its stale - // prompt. Bodies are excluded: the Skill tool always reads them fresh. - pluginSkillsDigest(this.pluginSkills) === pluginSkillsDigest(requestedPluginSkills) && // Editing `~/.agents/subagents/*.md` must reach the next prompt. Definition // bodies are part of the `Task` tool's behavior, so unlike skills they // are compared in full. @@ -2770,14 +2789,17 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // (the nearest one above), keeping each call adjacent to its result as // the provider APIs require. let toolCarrier: AssistantMessage | undefined; - for (const m of history) { + for (const m of orderSystemRows(history)) { // Subagent rows belong to the transcript and to review, never to the // parent's model context (ADR 0062): the parent only ever saw the `Task` // report, and replaying a delegate's messages would both contradict that // and reintroduce the context cost delegation exists to avoid. if (m.parentToolCallId) continue; const timestamp = Date.parse(m.createdAt) || Date.now(); - if (m.role === "user") { + if (m.role === "system" && m.modelSystem) { + toolCarrier = undefined; + append(m.id, this.systemJournal.restore(m)); + } else if (m.role === "user") { toolCarrier = undefined; const attachments = (m.attachments ?? []).map((attachment) => runtimeAttachmentFromMessage( @@ -2824,8 +2846,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the role: "assistant", content, api, - provider: this.provider.id, - model: this.provider.modelId, + // Persisted providerId identifies a local account; Pi compares its + // vendor identity to decide whether native thinking can be replayed. + // Keep known account/model switches distinct; legacy rows have no + // source identity and retain the existing current-model fallback. + provider: !m.providerId || m.providerId === this.provider.id + ? this.model.provider : m.providerId, + model: m.modelId || this.model.id, usage: usageToPi(m.usage), stopReason: "stop", timestamp, @@ -2841,8 +2868,8 @@ Do not invent objections or turn speculative risks into blockers. Stop when the role: "assistant", content: [], api, - provider: this.provider.id, - model: this.provider.modelId, + provider: this.model.provider, + model: this.model.id, usage: usageToPi(undefined), stopReason: "toolUse", timestamp, @@ -2893,6 +2920,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ): Entry[] { const entries: Entry[] = [...this.fullEntries]; if (!checkpoint) return entries; + if (isRecord(checkpoint.details) && checkpoint.details.systemMessageJson) { + this.systemJournal.rememberCheckpoint(checkpoint.details.systemMessageJson); + } const throughIndex = entries.findIndex( (entry) => entry.id === checkpoint.throughMessageId, ); @@ -5306,7 +5336,15 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private restoreDeferredToolsFromContext(): void { if (this.deferredToolNames.size === 0) return; const { messages } = this.liveSessionContext(); - for (const message of messages) { + const lastSystem = messages.map((message) => message.role).lastIndexOf("system"); + if (lastSystem >= 0) { + for (const tool of getCurrentTools(messages)) { + if (this.deferredToolNames.has(tool.name)) this.activeDeferredToolNames.add(tool.name); + } + } + // Legacy histories lack declarations. For current histories, only results + // after the last declaration can represent an activation not yet declared. + for (const message of messages.slice(lastSystem + 1)) { if (message.role !== "toolResult" || message.isError) continue; if (isMissingToolResultPlaceholder(message.content)) continue; const names = @@ -5923,8 +5961,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // agentLoopContinue refuses a transcript ending in an assistant message, // and this one carries nothing worth resending anyway. - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const promptBefore = this.agentSystemPromptContent(); @@ -5974,8 +6011,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.pendingOverflow = false; this.suppressOverflowRunEnd = false; this.overflowRecoveryAttempted = true; - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const compacted = await this.runCompaction( "overflow", @@ -6031,8 +6067,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // pi-agent-core refuses `continue()` when the transcript ends in an // assistant message. The progress text is already visible in the reused // bubble, so it must not be sent back as model context. - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const promptBefore = this.agentSystemPromptContent(); @@ -6358,12 +6393,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the return { context }; } - /** - * Codex's two-tier `maybe_record`, as a system-prompt append for this turn - * only. Codex writes its reminders into conversation history; we have no - * channel for a synthetic message that stays out of the transcript, and the - * append is equivalent without persisting anything. - */ + /** A replaceable reminder section expires when compaction opens a new window. */ private withContextBudgetReminder( context: AgentContext, budget: ContextBudget, @@ -6381,7 +6411,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ...(systemPrompt ? { systemPrompt } : {}), messages: [ ...context.messages, - { role: "system", content: reminder, timestamp: Date.now() }, + { role: "system", content: "", sections: { [CONTEXT_BUDGET_SECTION]: reminder }, timestamp: Date.now() }, ], } as AgentContext; } @@ -6828,6 +6858,11 @@ Do not invent objections or turn speculative risks into blockers. Stop when the mustFitSafeBudget: boolean, fallback?: ContextCompactionFallback, ): Promise { + const systemMessage = systemTranscriptCheckpoint(this.agent.state.messages); + checkpoint = { + ...checkpoint, + details: { ...(isRecord(checkpoint.details) ? checkpoint.details : {}), ...(systemMessage ? { systemMessageJson: JSON.stringify(systemMessage) } : {}) }, + }; const compactedBudget = this.contextBudget( this.liveSessionContext(checkpoint).messages, ); @@ -6853,7 +6888,10 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // resetting its `claim_*` flags when the context window turns over. this.contextReminderClaimed = false; this.contextFallbackReminderClaimed = false; - this.setAgentMessages(this.liveSessionContext().messages); + this.agent.state.messages = this.liveSessionContext().messages; + for (const message of this.agent.state.messages) { + if (message.role === "system") this.systemJournal.remember(message, checkpoint.id); + } this.emit({ type: "compaction_end", reason, diff --git a/packages/agent-runtime/src/session-context.ts b/packages/agent-runtime/src/session-context.ts index 5aae7c8dea..22327fd355 100644 --- a/packages/agent-runtime/src/session-context.ts +++ b/packages/agent-runtime/src/session-context.ts @@ -1,3 +1,4 @@ +import { readSystemMessage } from "./system-transcript-journal.js"; /** * Project pi session entries into the model context. * @@ -59,6 +60,8 @@ export function sessionEntryToContextMessages( return isContextMessage(entry.message) ? [entry.message] : []; case "compaction": return [ + ...(entry.details && typeof entry.details === "object" && "systemMessageJson" in entry.details + ? [readSystemMessage(entry.details.systemMessageJson)] : []), createCompactionSummaryMessage( entry.summary, entry.tokensBefore, @@ -71,7 +74,8 @@ export function sessionEntryToContextMessages( entry.timestamp, identity, )), - ...entry.retainedTail.filter(isContextMessage), + ...entry.retainedTail.filter((message) => isContextMessage(message) && + !(message.role === "system" && entry.details && typeof entry.details === "object" && "systemMessageJson" in entry.details)), ]; case "branch_summary": return entry.summary diff --git a/packages/agent-runtime/src/sidecar.ts b/packages/agent-runtime/src/sidecar.ts index ce7db0b7b1..7bc2126dc1 100644 --- a/packages/agent-runtime/src/sidecar.ts +++ b/packages/agent-runtime/src/sidecar.ts @@ -241,6 +241,7 @@ async function runtimeFor( runtimes.delete(sessionId); } if (reusable) { + reusable.setPluginSkills(pluginSkills); reusable.setCompactionSettings(params.compactionSettings); reusable.setInfiniteProviderRetry(params.infiniteProviderRetry === true); reusable.setMode(mode); @@ -260,10 +261,9 @@ async function runtimeFor( // The current prompt is sent separately below. Exclude its persisted row // before attachment hydration so it cannot consume the history byte budget. if (currentPrompt !== undefined && params.userMessageId) { - const last = restoredMessages.at(-1); - if (last?.role === "user" && last.id === params.userMessageId) { - restoredMessages = restoredMessages.slice(0, -1); - } + restoredMessages = restoredMessages.filter((message) => + message.role !== "user" || message.id !== params.userMessageId, + ); } const supportsVision = visionFromModelConfig(params.provider.modelConfig); history = await hydrateAttachmentHistory(restoredMessages, { diff --git a/packages/agent-runtime/src/system-transcript-journal.test.ts b/packages/agent-runtime/src/system-transcript-journal.test.ts new file mode 100644 index 0000000000..22b9816a3c --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-journal.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from "vitest"; +import type { AgentMessage, MessageEntry, CompactionEntry } from "@earendil-works/pi-agent-core"; +import type { UiMessage } from "@pi-desktop/shared"; +import { getCurrentTools, Type, type SystemMessage } from "@earendil-works/pi-ai"; +import { orderSystemRows, readSystemMessage, SystemTranscriptJournal } from "./system-transcript-journal.js"; +import { buildSessionContext } from "./session-context.js"; +import { CONTEXT_BUDGET_SECTION, syncSystemSections, systemTranscriptCheckpoint } from "./system-transcript.js"; + +const tool = { name: "Read", description: "Read", parameters: Type.Object({ path: Type.String() }) }; +const initial: SystemMessage = { role: "system", content: "", sections: { runtime: "Instructions", skills: "First skill" }, toolsAdded: [tool], timestamp: 1 }; +const user: AgentMessage = { role: "user", content: "Find files", timestamp: 2 }; +function entry(message: AgentMessage, id = "user"): MessageEntry { + return { type: "message", id, seq: 0, parentId: null, timestamp: message.timestamp, message }; +} +const userRow: UiMessage = { id: "user", role: "user", content: "Find files", createdAt: new Date(2).toISOString() }; + +describe("model system journal", () => { + it("restores the original order even when a user row was pre-persisted", async () => { + const journal = new SystemTranscriptJournal(); + const rows = [userRow]; + const entries = [entry(user)]; + const delta: SystemMessage = { role: "system", content: "", toolsRemoved: [{ name: "Read" }], timestamp: 3 }; + await journal.persist([initial, user, delta], entries, async (row) => { rows.push(row); }); + expect(rows.map((row) => row.role)).toEqual(["user", "system", "system"]); + const restored = new SystemTranscriptJournal(); + const messages = orderSystemRows(rows).map((row) => row.modelSystem ? restored.restore(row) : user); + expect(messages).toEqual([initial, user, delta]); + expect(getCurrentTools(messages)).toEqual([]); + const append = vi.fn(); + await restored.persist(messages, entries, append); + expect(append).not.toHaveBeenCalled(); + }); + + it("retries failed persistence with the same id and never installs it before acknowledgement", async () => { + const journal = new SystemTranscriptJournal(); + const entries = [entry(user)]; + const rows: UiMessage[] = []; + await expect(journal.persist([initial, user], entries, async (row) => { + rows.push(row); throw new Error("disk full"); + })).rejects.toMatchObject({ code: "LOCAL_REQUEST_ERROR", cause: new Error("disk full") }); + expect(entries).toHaveLength(1); + await journal.persist([initial, user], entries, async (row) => { rows.push(row); }); + expect(rows[0].id).toBe(rows[1].id); + expect(entries.map((item) => item.message.role)).toEqual(["system", "user"]); + }); + + it("replaces skill sections without rewriting the old prefix", () => { + const original = [initial, user]; + const changed = syncSystemSections(original, { runtime: "Instructions", skills: "Second skill" }); + expect(changed.slice(0, 2)).toEqual(original); + expect(changed.at(-1)).toMatchObject({ sections: { skills: "Second skill" } }); + expect(syncSystemSections(changed, { runtime: "Instructions", skills: "Second skill" })).toBe(changed); + expect(syncSystemSections(changed, { runtime: "Instructions", skills: "" }).at(-1)).toMatchObject({ sections: { skills: "" } }); + }); + + it("restores a compaction checkpoint once before the retained tail", async () => { + const delta: SystemMessage = { role: "system", content: "", toolsRemoved: [{ name: "Read" }], timestamp: 3 }; + const systemMessage = systemTranscriptCheckpoint([initial, user, delta])!; + const compaction: CompactionEntry = { + type: "compaction", id: "cp", seq: 1, parentId: "user", timestamp: 4, + summary: "Past work", tokensBefore: 100, retainedTail: [initial, user, delta], + details: { systemMessageJson: JSON.stringify(systemMessage) }, fromHook: false, + }; + const messages = buildSessionContext([entry(user), compaction]).messages; + expect(messages.filter((message) => message.role === "system")).toEqual([systemMessage]); + expect(messages[0]).toEqual(systemMessage); + expect(getCurrentTools(messages)).toEqual([]); + const journal = new SystemTranscriptJournal(); + journal.rememberCheckpoint(systemMessage); + const append = vi.fn(); + await journal.persist(JSON.parse(JSON.stringify(messages)), [], append); + expect(append).not.toHaveBeenCalled(); + }); + + it("expires budget reminders at compaction without dropping active instructions or tools", () => { + const checkpoint = systemTranscriptCheckpoint([initial, user, { + role: "system", content: "", sections: { [CONTEXT_BUDGET_SECTION]: "Old window is nearly full" }, timestamp: 3, + }]); + expect(checkpoint?.sections).toEqual(initial.sections); + expect(checkpoint?.toolsAdded).toEqual(initial.toolsAdded); + }); + + it.each([null, {}, { ...initial, timestamp: -1 }, { ...initial, toolsAdded: [{ name: "Read", parameters: "bad" }] }])("rejects malformed saved state", (value) => { + expect(() => readSystemMessage(value)).toThrow("Invalid persisted model system message"); + }); +}); diff --git a/packages/agent-runtime/src/system-transcript-journal.ts b/packages/agent-runtime/src/system-transcript-journal.ts new file mode 100644 index 0000000000..bff05f19e4 --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-journal.ts @@ -0,0 +1,117 @@ +import { randomUUID } from "node:crypto"; +import { LocalRequestError, Type, contentText, toToolDeclaration, type SystemMessage } from "@earendil-works/pi-ai"; +import type { AgentMessage, MessageEntry } from "@earendil-works/pi-agent-core"; +import type { UiMessage } from "@pi-desktop/shared"; +import * as Value from "typebox/value"; + +const systemSchema = Type.Object({ + role: Type.Literal("system"), + content: Type.String(), + timestamp: Type.Number({ minimum: 0, maximum: 8.64e15 }), + sections: Type.Optional(Type.Record(Type.String(), Type.Union([Type.String(), Type.Null()]))), + toolsAdded: Type.Optional(Type.Array(Type.Object({ + name: Type.String({ minLength: 1 }), description: Type.String(), + parameters: Type.Record(Type.String(), Type.Unknown()), + constrainedSampling: Type.Optional(Type.Union([ + Type.Literal(false), + Type.Object({ type: Type.Literal("json_schema"), strict: Type.Union([Type.Literal("prefer"), Type.Literal("require")]) }), + Type.Object({ type: Type.Literal("grammar"), variants: Type.Record(Type.String(), Type.Unknown()) }), + ])), + }))), + toolsRemoved: Type.Optional(Type.Array(Type.Object({ name: Type.String({ minLength: 1 }) }))), +}); + +export function readSystemMessage(value: unknown): SystemMessage { + if (typeof value === "string") value = JSON.parse(value); + if (!Value.Check(systemSchema, value)) throw new Error("Invalid persisted model system message"); + // Tool parameter schemas and grammar variants are JSON objects at this boundary. + return value as SystemMessage; +} + +/** Reorder only internal rows, whose following user row may have arrived first. */ +export function orderSystemRows(history: readonly UiMessage[]): UiMessage[] { + const rows = history.filter((row) => !row.modelSystem); + for (const row of history) { + if (!row.modelSystem) continue; + if (row.role !== "system" || row.modelSystem.version !== 1) { + throw new Error("Invalid model system record"); + } + const before = row.modelSystem.beforeMessageId; + let index = before ? rows.findIndex((candidate) => candidate.id === before) : -1; + if (index < 0 && row.modelSystem.afterMessageId) { + const previous = rows.findIndex((candidate) => candidate.id === row.modelSystem?.afterMessageId); + if (previous >= 0) { + index = previous + 1; + while (rows[index]?.modelSystem) index++; + } + } + rows.splice(index < 0 ? rows.length : index, 0, row); + } + return rows; +} + +/** Persist provider-neutral declarations before dispatch; never persist a folded request. */ +export class SystemTranscriptJournal { + private readonly ids = new WeakMap(); + private readonly pending = new WeakMap(); + private readonly checkpoints = new Set(); + + restore(row: UiMessage): SystemMessage { + const message = readSystemMessage(row.modelSystem?.messageJson); + this.ids.set(message, row.id); + return message; + } + + isPersisted(message: AgentMessage): boolean { + return this.ids.has(message) || this.checkpoints.has(JSON.stringify(message)); + } + + async persist( + messages: readonly AgentMessage[], + entries: MessageEntry[], + append: (row: UiMessage) => Promise, + ): Promise { + for (let index = 0; index < messages.length; index++) { + const message = messages[index]; + if (message.role !== "system" || this.isPersisted(message)) continue; + const following = messages.slice(index + 1).find((candidate) => candidate.role !== "system"); + const nextEntry = following && entries.find((entry) => entry.message === following); + const preceding = messages.slice(0, index).reverse().find((candidate) => candidate.role !== "system"); + const previousEntry = preceding && entries.find((entry) => entry.message === preceding); + const id = this.pending.get(message) ?? randomUUID(); + this.pending.set(message, id); + const serialized: SystemMessage = { + ...message, content: contentText(message.content), + ...(message.toolsAdded ? { toolsAdded: message.toolsAdded.map(toToolDeclaration) } : {}), + }; + const row: UiMessage = { + id, role: "system", content: "", createdAt: new Date(message.timestamp).toISOString(), + modelSystem: { version: 1, messageJson: JSON.stringify(serialized), ...(nextEntry ? { beforeMessageId: nextEntry.id } : {}), ...(previousEntry ? { afterMessageId: previousEntry.id } : {}) }, + }; + try { + await append(row); + } catch (cause) { + // Stop before dispatch. Retrying the provider cannot repair a failed + // durable write, and the original host error may contain private paths. + throw new LocalRequestError("request-preparation", { cause }); + } + this.ids.set(message, id); + this.pending.delete(message); + const entry: MessageEntry = { + type: "message", id, seq: entries.length, parentId: null, + timestamp: message.timestamp, message, + }; + const position = nextEntry ? entries.indexOf(nextEntry) : entries.length; + entries.splice(position, 0, entry); + entries.forEach((item, seq) => { item.seq = seq; item.parentId = entries[seq - 1]?.id ?? null; }); + } + } + + rememberCheckpoint(message: unknown): void { + this.checkpoints.add(JSON.stringify(readSystemMessage(message))); + } + + remember(message: AgentMessage, id: string): void { + this.ids.set(message, id); + } +} diff --git a/packages/agent-runtime/src/system-transcript-order.test.ts b/packages/agent-runtime/src/system-transcript-order.test.ts new file mode 100644 index 0000000000..65e127c110 --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-order.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from "vitest"; +import { Agent, type AgentTool, type AgentMessage } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, Type, getCurrentTools, type SystemMessage } from "@earendil-works/pi-ai"; +import { rebuildSystemTranscript } from "./system-transcript.js"; + +const read = { name: "Read", description: "Read a file", parameters: Type.Object({}) }; +const search = { name: "Search", description: "Search files", parameters: Type.Object({}) }; +const initial: SystemMessage = { role: "system", content: "Instructions", toolsAdded: [read], timestamp: 1 }; +const user: AgentMessage = { role: "user", content: "Find a file", timestamp: 2 }; +const result: AgentMessage = { + role: "toolResult", toolCallId: "search-1", toolName: "ToolSearch", + content: [{ type: "text", text: "Search activated" }], isError: false, timestamp: 3, +}; + +describe("chronological system updates", () => { + it("lets the Pi loop append schema changes and preserves the complete prefix", async () => { + const executable = (tool: typeof read): AgentTool => ({ ...tool, label: tool.name, execute: async () => ({ content: [], details: {} }) }); + const requests: AgentMessage[][] = []; + const agent = new Agent({ + initialState: { tools: [executable(read)], messages: [initial, user] }, + streamFn: async (_model, context) => { + requests.push([...context.messages]); + const stream = createAssistantMessageEventStream(); + stream.push({ type: "done", reason: "stop", message: { + role: "assistant", content: [{ type: "text", text: "Done" }], stopReason: "stop", timestamp: Date.now(), + api: "openai-completions", provider: "fixture", model: "fixture", + usage: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0, totalTokens: 2, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, + } }); + return stream; + }, + }); + await agent.continue(); + const before = [...agent.state.messages]; + const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; + agent.state.tools = [executable(replacement), executable(search)]; + await agent.prompt("Continue"); + expect(requests).toHaveLength(2); + expect(requests[1].slice(0, before.length)).toEqual(before); + expect(requests[1].slice(before.length)).toContainEqual(expect.objectContaining({ + role: "system", toolsAdded: [replacement, search], toolsRemoved: [{ name: "Read" }], + })); + expect(getCurrentTools(requests[1])).toEqual([replacement, search]); + await agent.prompt("Again"); + expect(requests[2].filter((message) => message.role === "system")).toHaveLength(2); + }); + + it("does not move a retained tool update ahead of its activation result", () => { + const update: SystemMessage = { role: "system", content: "", toolsAdded: [search], timestamp: 4 }; + const before = [initial, user, result, update]; + const after = rebuildSystemTranscript(before, [user, result]); + expect(after).toEqual(before); + expect(after[0]).toBe(initial); + expect(after.at(-1)).toBe(update); + }); +}); diff --git a/packages/agent-runtime/src/system-transcript-runtime.test.ts b/packages/agent-runtime/src/system-transcript-runtime.test.ts new file mode 100644 index 0000000000..f48e132ddb --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-runtime.test.ts @@ -0,0 +1,144 @@ +import { describe, expect, it } from "vitest"; +import type { Agent } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, getCurrentTools, getCurrentSystemPrompt, type AssistantMessage, type Message } from "@earendil-works/pi-ai"; +import type { UiMessage } from "@pi-desktop/shared"; +import { DesktopAgentRuntime, type RuntimeProviderConfig } from "./runtime.js"; + +const provider: RuntimeProviderConfig = { + id: "fixture", name: "Fixture", modelId: "fixture", apiKey: "", authKind: "none", + baseUrl: "https://fixture.invalid/v1", supportsReasoning: false, supportedThinkingLevels: ["off"], +}; +const skill = { id: "fixture/notes", name: "First catalog", description: "Summarize notes" }; +function answer(content: AssistantMessage["content"], stopReason: "stop" | "toolUse" = "stop"): AssistantMessage { + return { + role: "assistant", api: "openai-completions", provider: "fixture", model: "fixture", + timestamp: Date.now() + 10, stopReason, content, + usage: { input: 10, output: 2, cacheRead: 0, cacheWrite: 0, totalTokens: 12, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, + }; +} +function fixture(history: UiMessage[] = [], skills = [skill]) { + const rows = [...history]; + const requests: Message[][] = []; + const errors: unknown[] = []; + let persistenceFailure: Error | undefined; + let respond = () => answer([{ type: "text", text: "Done." }]); + const runtime = new DesktopAgentRuntime({ + sessionId: "fixture", mode: "agent", provider, thinkingLevel: "off", history, pluginSkills: skills, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") { + if (persistenceFailure) throw persistenceFailure; + rows.push((params as { message: UiMessage }).message); + } + else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + const agent = (runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = async (_model, context) => { + requests.push(structuredClone(context.messages)); + const message = respond(); + const stream = createAssistantMessageEventStream(); + stream.push({ type: "start", partial: { ...message, content: [] } }); + stream.push({ type: "done", reason: message.stopReason as "stop" | "toolUse", message }); + return stream; + }; + return { + runtime, agent, rows, requests, errors, + failPersistence: () => { persistenceFailure = new Error("disk full"); }, + respond: (next: () => AssistantMessage) => { respond = next; }, + prompt: async (id: string) => { + rows.push({ id, role: "user", content: "Continue", createdAt: new Date().toISOString() }); + await runtime.prompt("Continue", id, `turn-${id}`); + }, + }; +} + +describe("system state through the Desktop user path", () => { + it("activates ToolSearch through Pi, retains its prefix, and restores the declarations after restart", async () => { + const f = fixture(); + let requests = 0; + f.respond(() => ++requests === 1 + ? answer([{ type: "toolCall", id: "search-call", name: "ToolSearch", arguments: { query: "BrowserPreview" } }], "toolUse") + : answer([{ type: "text", text: "Ready." }])); + try { + await f.prompt("user-1"); + expect(f.requests).toHaveLength(2); + const [before, after] = f.requests; + expect(after.slice(0, before.length)).toEqual(before); + expect(getCurrentTools(before).some((tool) => tool.name === "BrowserPreview")).toBe(false); + expect(getCurrentTools(after).some((tool) => tool.name === "BrowserPreview")).toBe(true); + const activation = after.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); + expect(after.slice(activation + 1)).toContainEqual(expect.objectContaining({ role: "system", toolsAdded: expect.arrayContaining([expect.objectContaining({ name: "BrowserPreview" })]) })); + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); + const restored = fixture(structuredClone(f.rows)); + try { + await restored.prompt("user-2"); + const replay = restored.requests[0]; + expect(replay.filter((message) => message.role === "system")).toEqual(after.filter((message) => message.role === "system")); + const resultIndex = replay.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); + expect(resultIndex).toBeGreaterThan(0); + expect(replay.slice(resultIndex + 1)).toContainEqual(after.at(-1)); + expect(getCurrentTools(restored.requests[0]).some((tool) => tool.name === "BrowserPreview")).toBe(true); + expect(restored.rows.filter((row) => row.modelSystem)).toHaveLength(2); + } finally { await restored.runtime.dispose(); } + } finally { await f.runtime.dispose(); } + }); + + it("stops before provider dispatch when the Host cannot save system state", async () => { + const f = fixture(); + f.failPersistence(); + try { + await f.prompt("user-1"); + expect(f.requests).toHaveLength(0); + expect(f.errors).toContainEqual(expect.objectContaining({ + code: "INTERNAL", retriable: false, details: expect.objectContaining({ origin: "local", phase: "request-preparation" }), + })); + } finally { await f.runtime.dispose(); } + }); + + it("updates and revokes a skill catalog on the same runtime without rewriting previous requests", async () => { + const otherSkill = { id: "fixture/other", name: "Unchanged entry", description: "Keep this instruction" }; + const f = fixture([], [skill, otherSkill]); + try { + await f.prompt("user-1"); + const before = f.requests[0]; + f.runtime.setPluginSkills([{ ...skill, name: "Updated catalog" }, otherSkill]); + await f.prompt("user-2"); + const after = f.requests[1]; + expect(after.slice(0, before.length)).toEqual(before); + expect(getCurrentSystemPrompt(after)).toContain("Updated catalog"); + expect(getCurrentSystemPrompt(after)).not.toContain("First catalog"); + expect(after.filter((message) => message.role === "system")).toHaveLength(2); + const updates = after.filter((message) => message.role === "system"); + expect(updates.at(-1)?.sections).toEqual({ "skill:fixture/notes": "- `fixture/notes` — Updated catalog: Summarize notes" }); + expect(getCurrentSystemPrompt(after)).toContain("Unchanged entry"); + const restored = fixture(structuredClone(f.rows), [{ ...skill, name: "Updated catalog" }, otherSkill]); + try { + await restored.prompt("restored-user"); + expect(restored.requests[0].filter((message) => message.role === "system")).toEqual(updates); + } finally { await restored.runtime.dispose(); } + f.runtime.setPluginSkills([]); + await f.prompt("user-3"); + expect(getCurrentSystemPrompt(f.requests[2])).not.toContain("Updated catalog"); + // Revocation includes a section delta and Pi's executable-tool removal. + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(4); + expect(getCurrentTools(f.requests[2]).some((tool) => tool.name === "Skill")).toBe(false); + } finally { await f.runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/system-transcript.test.ts b/packages/agent-runtime/src/system-transcript.test.ts index 0b5bf732e8..8e994a499c 100644 --- a/packages/agent-runtime/src/system-transcript.test.ts +++ b/packages/agent-runtime/src/system-transcript.test.ts @@ -13,22 +13,23 @@ import { } from "@earendil-works/pi-ai"; import { estimateContextTokens as estimateTranscriptTokens } from "@earendil-works/pi-ai/utils/estimate"; import { + syncSystemSections, initialSystemTranscript, rebuildSystemTranscript, replaceSystemPrompt, - syncSystemTools, + systemTranscriptCheckpoint, systemPromptContent, } from "./system-transcript.js"; const read: Tool = { name: "Read", description: "Read text", parameters: Type.Object({ path: Type.String() }) }; const edit: Tool = { ...read, name: "Edit", description: "Edit text" }; const initial: SystemMessage = { - role: "system", content: "Base instructions", timestamp: 1_000, - sections: { rules: "Keep these rules", obsolete: "Remove these rules" }, + role: "system", content: "", timestamp: 1_000, + sections: { runtime: "Base instructions", rules: "Keep these rules", obsolete: "Remove these rules" }, toolsAdded: [read, edit], }; const delta: SystemMessage = { - role: "system", content: "Additional instructions", timestamp: 1_500, + role: "system", content: "", timestamp: 1_500, sections: { rules: "Updated rules", obsolete: null }, toolsRemoved: [{ name: "Edit" }], }; @@ -49,13 +50,46 @@ function estimateContextTokens(messages: AgentMessage[]) { afterEach(() => vi.restoreAllMocks()); describe("system transcript helpers", () => { + it("updates only the changed skill entry and preserves unrelated sections", () => { + const previous: AgentMessage[] = [{ + role: "system", content: "", timestamp: 1, + sections: { runtime: "Base", skills: "Load skills", "skill:a": "A", "skill:b": "B", context: "Context", extension: "Keep" }, + }, assistant]; + const desired = { runtime: "Base", skills: "Load skills", "skill:a": "A updated", "skill:b": "B", context: "Context" }; + const updated = syncSystemSections(previous, desired); + expect(updated.slice(0, previous.length)).toEqual(previous); + expect(updated.at(-1)).toMatchObject({ sections: { "skill:a": "A updated" } }); + expect(getCurrentSystemMessage(updated)?.sections).toHaveProperty("extension", "Keep"); + expect(syncSystemSections(updated, desired)).toBe(updated); + const revoked = syncSystemSections(updated, { runtime: "Base", context: "Context" }); + expect(revoked.at(-1)).toMatchObject({ sections: { skills: null, "skill:a": null, "skill:b": null } }); + expect(getCurrentSystemPrompt(revoked)).not.toContain("A updated"); + expect(getCurrentSystemPrompt(revoked)).toContain("Keep"); + }); + + it("upgrades aggregate catalogs once and restores per-skill checkpoints without duplicate deltas", () => { + const legacy: AgentMessage[] = [{ role: "system", content: "", timestamp: 1, + sections: { runtime: "Base", skills: "Old full catalog", context: "Context" } }]; + const desired = { runtime: "Base", skills: "Load skills", "skill:z": "Z", context: "Context" }; + const upgraded = syncSystemSections(legacy, desired); + expect(getCurrentSystemPrompt(upgraded)).not.toContain("Old full catalog"); + const added = syncSystemSections(upgraded, { ...desired, "skill:a": "A" }); + expect(systemPromptContent(added)).toBe("Base\n\nLoad skills\n\nA\n\nZ\n\nContext"); + const restored = [systemTranscriptCheckpoint(added)!]; + expect(syncSystemSections(restored, { ...desired, "skill:a": "A" })).toBe(restored); + expect(systemPromptContent(restored)).toBe(systemPromptContent(added)); + const replaced = replaceSystemPrompt(restored, "Temporary prompt"); + expect(getCurrentSystemPrompt(replaced)).toBe("Temporary prompt"); + }); + it("replays sections and tool removals without giving an unchanged prefix a new timestamp", () => { const previous = [initial, delta, assistant, user]; const rebuilt = rebuildSystemTranscript(previous, [assistant, user]); expect(getCurrentSystemPrompt(rebuilt)).toBe(getCurrentSystemPrompt(previous)); - expect(rebuilt[0]).toMatchObject({ - content: "Base instructions\n\nAdditional instructions", timestamp: 1_500, - sections: { rules: "Updated rules" }, toolsAdded: [read], + expect(rebuilt).toEqual(previous); + expect(systemTranscriptCheckpoint(rebuilt)).toMatchObject({ + content: "", timestamp: 1_500, + sections: { runtime: "Base instructions", rules: "Updated rules" }, toolsAdded: [read], }); expect(getCurrentSystemMessage(rebuilt)?.sections).not.toHaveProperty("obsolete"); expect(getCurrentTools(rebuilt)).toEqual([read]); @@ -63,18 +97,17 @@ describe("system transcript helpers", () => { expect(rebuildSystemTranscript(rebuilt, [assistant, user])[0]).toBe(rebuilt[0]); }); - it("does not resurrect usage predating a folded system delta", () => { + it("keeps an earlier usage anchor and estimates an appended delta", () => { const newer = { ...delta, timestamp: 3_000 }; const rebuilt = rebuildSystemTranscript([initial, assistant, newer], [assistant]); - expect(rebuilt[0]?.timestamp).toBe(3_000); - expect(estimateContextTokens(rebuilt).lastUsageIndex).toBeNull(); + expect(rebuilt.at(-1)?.timestamp).toBe(3_000); + expect(estimateContextTokens(rebuilt).trailingTokens).toBeGreaterThan(0); }); it("keeps a complete recovery transcript and its deltas in place", () => { const previous = [initial, assistant, delta, user, { ...assistant, stopReason: "error" as const }]; const retained = previous.slice(0, -1); expect(rebuildSystemTranscript(previous, retained)).toBe(retained); - expect(syncSystemTools(retained, [read])).toBe(retained); }); it("preserves object identity when setting either the same content or rendered prompt", () => { @@ -83,14 +116,14 @@ describe("system transcript helpers", () => { expect(replaceSystemPrompt(messages, systemPromptContent(messages))).toBe(messages); }); - it("invalidates usage only on a real content change and preserves structured sections and tools", () => { + it("accounts for a real content change and preserves structured sections and tools", () => { vi.spyOn(Date, "now").mockReturnValue(5_000); const changed = replaceSystemPrompt([initial, delta, assistant], "Replacement instructions"); - expect(changed[0]).toMatchObject({ content: "Replacement instructions", timestamp: 5_000, sections: { rules: "Updated rules" } }); + expect(changed.at(-1)).toMatchObject({ content: "", timestamp: 5_000, sections: { runtime: "Replacement instructions" } }); expect(getCurrentTools(changed)).toEqual([read]); - expect(estimateContextTokens(changed).lastUsageIndex).toBeNull(); + expect(estimateContextTokens(changed).trailingTokens).toBeGreaterThan(0); expect(replaceSystemPrompt(changed, "Replacement instructions")).toBe(changed); - expect(initial.content).toBe("Base instructions"); + expect(initial.sections?.runtime).toBe("Base instructions"); const response = { ...assistant, timestamp: 6_000 }; expect(estimateContextTokens(rebuildSystemTranscript(changed, [assistant, response])).usageTokens).toBe(1_000); }); @@ -98,7 +131,7 @@ describe("system transcript helpers", () => { it("can clear content without clearing sections or tool declarations", () => { const changed = replaceSystemPrompt([initial, delta, assistant], ""); expect(systemPromptContent(changed)).toBe(""); - expect(getCurrentSystemMessage(changed)?.sections).toEqual({ rules: "Updated rules" }); + expect(getCurrentSystemMessage(changed)?.sections).toEqual({ runtime: "", rules: "Updated rules" }); expect(getCurrentTools(changed)).toEqual([read]); }); @@ -110,55 +143,32 @@ describe("system transcript helpers", () => { const response = { ...assistant, timestamp: 6_000 }; const restored = replaceSystemPrompt([...nudged, response], before); expect(getCurrentSystemPrompt(restored)).toBe(getCurrentSystemPrompt(messages)); - expect(getCurrentSystemMessage(restored)?.sections).toEqual({ rules: "Updated rules" }); - expect(restored[0]?.timestamp).toBeGreaterThan(response.timestamp); - expect(estimateContextTokens(restored).usageTokens).toBe(0); - }); - - it("records tool additions, removals, and same-name schema replacements with fresh semantic time", () => { - vi.spyOn(Date, "now").mockReturnValue(5_000); - const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; - const messages = [initial, assistant]; - const changed = syncSystemTools(messages, [replacement]); - expect(changed[1]).toMatchObject({ timestamp: 5_000, toolsAdded: [replacement], toolsRemoved: [{ name: "Read" }, { name: "Edit" }] }); - expect(getCurrentTools(changed)).toEqual([replacement]); - expect(estimateContextTokens(changed).usageTokens).toBe(0); - expect(syncSystemTools(changed, [replacement])).toBe(changed); - const response = { ...assistant, timestamp: 6_000 }; - expect(estimateContextTokens(rebuildSystemTranscript(changed, [assistant, response])).usageTokens).toBe(1_000); - const cleared = syncSystemTools(changed, []); - expect(getCurrentTools(rebuildSystemTranscript(cleared, [assistant]))).toEqual([]); - }); - - it("ignores executable-only tool changes instead of invalidating usage", () => { - const messages = [initial, assistant]; - const executable = { ...read, label: "Read", execute: vi.fn() }; - expect(syncSystemTools(messages, [executable, edit])).toBe(messages); - expect(estimateContextTokens(messages).usageTokens).toBe(1_000); + expect(getCurrentSystemMessage(restored)?.sections).toEqual({ runtime: "Base instructions", rules: "Updated rules" }); + expect(restored.at(-1)?.timestamp).toBeGreaterThan(response.timestamp); + expect(estimateContextTokens(restored).trailingTokens).toBeGreaterThan(0); }); it("advances semantic time even if the clock shares a millisecond or moves backwards", () => { vi.spyOn(Date, "now").mockReturnValue(2_000); - expect(replaceSystemPrompt([initial, assistant], "changed")[0]?.timestamp).toBe(2_001); + expect(replaceSystemPrompt([initial, assistant], "changed").at(-1)?.timestamp).toBe(2_001); vi.spyOn(Date, "now").mockReturnValue(500); - expect(syncSystemTools([initial, assistant], [read])[1]?.timestamp).toBe(2_001); + expect(replaceSystemPrompt([initial, assistant], "another change").at(-1)?.timestamp).toBe(2_001); }); it("initializes restored history conservatively without inventing an old timestamp", () => { vi.spyOn(Date, "now").mockReturnValue(5_000); const messages = initialSystemTranscript("Current config", [read], [assistant]); - expect(messages[0]).toMatchObject({ content: "Current config", toolsAdded: [read], timestamp: 5_000 }); - expect(estimateContextTokens(messages).usageTokens).toBe(0); + expect(messages.at(-1)).toMatchObject({ sections: { runtime: "Current config" }, toolsAdded: [read], timestamp: 5_000 }); + expect(estimateContextTokens(messages).trailingTokens).toBeGreaterThan(0); const rebuilt = rebuildSystemTranscript(messages, [assistant]); expect(rebuilt[0]).toBe(messages[0]); - expect(estimateContextTokens(rebuilt).usageTokens).toBe(0); + expect(estimateContextTokens(rebuilt).trailingTokens).toBeGreaterThan(0); }); it("preserves the empty transcript when there is no system state", () => { const messages: AgentMessage[] = []; expect(initialSystemTranscript("", [], messages)).toBe(messages); expect(rebuildSystemTranscript([], messages)).toBe(messages); - expect(syncSystemTools(messages, [])).toBe(messages); expect(replaceSystemPrompt(messages, "")).toBe(messages); }); @@ -176,13 +186,13 @@ describe("system transcript helpers", () => { return stream; }, }); - agent.state.messages = syncSystemTools(rebuildSystemTranscript(agent.state.messages, [user]), [tool]); - const prefix = agent.state.messages[0]; + agent.state.messages = rebuildSystemTranscript(agent.state.messages, [user]); + const prefix = agent.state.messages.filter((message) => message.role === "system"); await agent.continue(); expect(agent.state.errorMessage).toBeUndefined(); expect(requests).toHaveLength(1); - expect(requests[0]?.filter((message) => message.role === "system")).toEqual([prefix]); + expect(requests[0]?.filter((message) => message.role === "system")).toEqual(prefix); expect(getCurrentTools(requests[0]!)).toEqual([toToolDeclaration(tool)]); - expect(agent.state.messages.filter((message) => message.role === "system")).toEqual([prefix]); + expect(agent.state.messages.filter((message) => message.role === "system")).toEqual(prefix); }); }); diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index 28bd5028b7..a5bac9363b 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -1,22 +1,21 @@ import type { AgentMessage } from "@earendil-works/pi-agent-core"; import { contentText, - createInitialSystemMessage, getCurrentSystemMessage, getCurrentSystemPrompt, - getCurrentTools, - getToolStateChanges, type SystemMessage, type Tool, toToolDeclaration, } from "@earendil-works/pi-ai"; +import { SKILL_SECTION_PREFIX } from "./plugin-skills-prompt.js"; + function systemMessages(messages: readonly AgentMessage[]): SystemMessage[] { return messages.filter((message): message is SystemMessage => message.role === "system"); } function nextSystemTimestamp(messages: readonly AgentMessage[]): number { - // 同一毫秒内也必须晚于旧响应;时钟回拨时只前移,不伪造早于 usage 的时间。 + // Never backdate a state change relative to the response it invalidates. return messages.reduce((timestamp, message) => Math.max(timestamp, message.timestamp + 1), Date.now()); } @@ -24,71 +23,102 @@ function currentSystemMessage(messages: readonly AgentMessage[]): SystemMessage const systems = systemMessages(messages); if (systems.length < 2) return systems[0]; const current = getCurrentSystemMessage(systems)!; - // 上游折叠器保留第一条的时间,但 sections/工具删除可能来自较晚的 delta。 - // 快照的语义时间必须覆盖所有贡献者,否则旧 usage 会被错误地重新激活。 + // The upstream fold retains the first timestamp. A checkpoint must cover + // every contributing update so old usage is not accidentally reactivated. return { ...current, timestamp: systems.reduce((timestamp, message) => Math.max(timestamp, message.timestamp), systems[0]!.timestamp), }; } -/** The unrendered content, without flattening named sections into the prompt. */ +/** Desktop-owned sections retain their order ahead of extension sections. */ +export const CONTEXT_BUDGET_SECTION = "context_budget"; + +function desktopSectionNames(sections: Record): string[] { + return ["runtime", "skills", ...Object.keys(sections) + .filter((name) => name.startsWith(SKILL_SECTION_PREFIX)) + .sort((a, b) => a.localeCompare(b)), "context"]; +} + export function systemPromptContent(messages: readonly AgentMessage[]): string { - return contentText(getCurrentSystemMessage(messages)?.content ?? ""); + const current = getCurrentSystemMessage(messages); + const sections = current?.sections; + if (sections && desktopSectionNames(sections).some((name) => name in sections)) { + return desktopSectionNames(sections).map((name) => sections[name]).filter(Boolean).join("\n\n"); + } + return contentText(current?.content ?? ""); +} + +export function syncSystemSections( + messages: AgentMessage[], + desired: Record, +): AgentMessage[] { + const current = getCurrentSystemMessage(messages)?.sections ?? {}; + const sections: Record = {}; + for (const name of desktopSectionNames({ ...current, ...desired })) { + const next = desired[name] ?? null; + if ((current[name] ?? null) !== next) sections[name] = next; + } + if (Object.keys(sections).length === 0) return messages; + return [...messages, { role: "system", content: "", sections, timestamp: nextSystemTimestamp(messages) }]; } export function initialSystemTranscript( prompt: string, tools: readonly Tool[], messages: AgentMessage[], + sections: Record = { runtime: prompt }, ): AgentMessage[] { - const system = createInitialSystemMessage(prompt, tools.map(toToolDeclaration)); - // 持久化历史不记录 system 状态,重启后无法证明新配置与旧请求相同。 - // 保守使用实际初始化时间;不能沿用上游初始值 0 来让历史 usage 假装有效。 - return system - ? [{ ...system, timestamp: nextSystemTimestamp(messages) }, ...messages] - : messages; + if (messages.some((message) => message.role === "system")) { + return syncSystemSections(messages, sections); + } + if (!prompt && tools.length === 0) return messages; + // Old sessions have no recorded baseline. Declare the current state at the + // continuation boundary rather than inventing past instructions or tools. + return [...messages, { + role: "system", content: "", sections, + toolsAdded: tools.map(toToolDeclaration), timestamp: nextSystemTimestamp(messages), + }]; } export function replaceSystemPrompt(messages: AgentMessage[], prompt: string): AgentMessage[] { - const current = currentSystemMessage(messages); - if (prompt === getCurrentSystemPrompt(messages) || prompt === contentText(current?.content ?? "")) { - return messages; - } - // prompt 只替换内容;命名 sections 与最终工具状态仍由上游 replay 负责。 - // recovery 读写未渲染内容,避免把 sections 再嵌入 content 造成重复。 - return [ - { ...current, role: "system", content: prompt, timestamp: nextSystemTimestamp(messages) }, - ...messages.filter((message) => message.role !== "system"), - ]; + if (prompt === getCurrentSystemPrompt(messages) || prompt === systemPromptContent(messages)) return messages; + return syncSystemSections(messages, { runtime: prompt }); } export function rebuildSystemTranscript( previous: readonly AgentMessage[], messages: AgentMessage[], ): AgentMessage[] { - // recovery 传入完整 live transcript 的切片,保留其位置、对象及 delta,不重复加前缀。 - if (messages.some((message) => message.role === "system")) return messages; - // durable projection 不含 system;用上游 replay 还原有效 sections 和工具集合, - // 而不是将渲染后的 systemPrompt 当成新消息。折叠不产生新的语义时间。 - const system = currentSystemMessage(previous); - return system ? [system, ...messages] : messages; + let result = messages; + for (let i = 0; i < previous.length; i++) { + const message = previous[i]; + if (message.role !== "system" || result.includes(message)) continue; + const following = previous.slice(i + 1).find((item) => result.includes(item)); + const preceding = previous.slice(0, i).reverse().find((item) => result.includes(item)); + let position = following ? result.indexOf(following) : preceding ? result.indexOf(preceding) + 1 : result.length; + if (!following) while (result[position]?.role === "system") position++; + if (result === messages) result = [...messages]; + result.splice(position, 0, message); + } + return result; +} + +/** Fold state only at an explicit compaction boundary, with semantic time. */ +export function systemTranscriptCheckpoint(messages: readonly AgentMessage[]): SystemMessage | undefined { + const checkpoint = currentSystemMessage(messages); + if (!checkpoint?.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; + const { [CONTEXT_BUDGET_SECTION]: _expired, ...sections } = checkpoint.sections; + return { ...checkpoint, sections }; } -export function syncSystemTools(messages: AgentMessage[], tools: readonly Tool[]): AgentMessage[] { - const changes = getToolStateChanges(getCurrentTools(messages), tools); - if (changes.toolsAdded.length === 0 && changes.toolsRemoved.length === 0) return messages; - // 真正的工具变化成为较新的前缀,旧 usage 因此失效。保留删除/同名替换的 delta, - // 并让声明集合与可执行 catalog 一致,避免 agent-loop 下一轮再次自动追加同一变化。 - return [ - ...systemMessages(messages), - { - role: "system", - content: "", - ...(changes.toolsAdded.length ? { toolsAdded: changes.toolsAdded } : {}), - ...(changes.toolsRemoved.length ? { toolsRemoved: changes.toolsRemoved } : {}), - timestamp: nextSystemTimestamp(messages), - }, - ...messages.filter((message) => message.role !== "system"), - ]; +/** Recovery discards failed trailing responses even after a prompt cleanup delta. */ +export function removeTrailingAssistantMessages(messages: readonly AgentMessage[]): AgentMessage[] { + const result = [...messages]; + for (let index = result.length - 1; index >= 0; index--) { + if (result[index].role === "system") continue; + if (result[index].role !== "assistant") break; + result.splice(index, 1); + } + return result; } diff --git a/packages/agent-runtime/src/thinking-level.ts b/packages/agent-runtime/src/thinking-level.ts index 509bbd316b..f386bbe808 100644 --- a/packages/agent-runtime/src/thinking-level.ts +++ b/packages/agent-runtime/src/thinking-level.ts @@ -23,6 +23,8 @@ export type ThinkingCapabilitySet = { */ export type ModelConfig = { source: "pi" | "models.dev" | "generic"; + /** Published identity before account endpoint or wire-model overrides. */ + transcriptBinding?: { modelId: string; api: string; baseUrl: string }; inputLimits?: Model["inputLimits"]; promptCache?: Model["promptCache"]; samplingParams?: Model["samplingParams"]; diff --git a/packages/agent-runtime/src/transcript-compat.test.ts b/packages/agent-runtime/src/transcript-compat.test.ts new file mode 100644 index 0000000000..86fb4c2c6e --- /dev/null +++ b/packages/agent-runtime/src/transcript-compat.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from "vitest"; +import { genericModelConfig } from "./model-capabilities.js"; +import { buildProviderModel, type RuntimeProviderConfig } from "./provider-binding.js"; + +const baseUrl = "https://native.test/v1"; +const provider: RuntimeProviderConfig = { + id: "account", name: "Account", modelId: "exact-model", baseUrl, apiStyle: "openai_completions", + apiKey: "", authKind: "none", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: { + ...genericModelConfig("exact-model", baseUrl), source: "pi", + transcriptBinding: { modelId: "exact-model", api: "openai-completions", baseUrl }, + compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }, + }, +}; + +describe("transcript compatibility binding", () => { + it("keeps instruction and tool capabilities independent on the exact binding", () => { + expect(buildProviderModel(provider).compat).toMatchObject({ supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }); + expect(buildProviderModel({ ...provider, baseUrl: `${baseUrl}/` }).compat).toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + it.each([ + { baseUrl: "https://relay.test/v1" }, { baseUrl: "https://native.test/other" }, + { baseUrl: "https://native.test:8443/v1" }, { modelId: "alias" }, { apiStyle: "responses" }, + { modelConfig: { ...provider.modelConfig!, transcriptBinding: undefined } }, + ])("falls back for an unverified route or model: %j", (overrides) => { + expect(buildProviderModel({ ...provider, ...overrides }).compat).toMatchObject({ + supportsMidConvoSystemMessages: false, supportsMidConvoToolAdditions: false, + supportsMidConvoToolChanges: false, supportsAdditionalTools: false, + }); + }); +}); diff --git a/packages/agent-runtime/src/transcript-compat.ts b/packages/agent-runtime/src/transcript-compat.ts new file mode 100644 index 0000000000..365e2f6da9 --- /dev/null +++ b/packages/agent-runtime/src/transcript-compat.ts @@ -0,0 +1,28 @@ +import type { ModelConfig } from "./thinking-level.js"; + +const capabilities = [ + "supportsMidConvoSystemMessages", "supportsMidConvoToolAdditions", + "supportsMidConvoToolChanges", "supportsAdditionalTools", "supportsToolSearch", +] as const; + +function endpoint(value: string): string | undefined { + try { + const url = new URL(value); + if (url.username || url.password || url.search || url.hash) return undefined; + return `${url.origin}${url.pathname.replace(/\/+$/, "")}`; + } catch { + return undefined; + } +} + +/** Catalog capabilities describe one model/API/endpoint, never a vendor label. */ +export function transcriptCompat( + catalog: ModelConfig | undefined, modelId: string, api: string, baseUrl: string, +): Record<(typeof capabilities)[number], boolean> { + const binding = catalog?.transcriptBinding; + const address = endpoint(baseUrl); + const verified = catalog?.source === "pi" && binding?.modelId === modelId && + binding.api === api && address !== undefined && endpoint(binding.baseUrl) === address; + return Object.fromEntries(capabilities.map((key) => [key, verified && catalog?.compat?.[key] === true])) as + Record<(typeof capabilities)[number], boolean>; +} diff --git a/packages/agent-runtime/src/transcript-payload.test.ts b/packages/agent-runtime/src/transcript-payload.test.ts new file mode 100644 index 0000000000..0cd77f8b2d --- /dev/null +++ b/packages/agent-runtime/src/transcript-payload.test.ts @@ -0,0 +1,87 @@ +import { expect, it } from "vitest"; +import { normalizeContext, Type, type Message, type Model } from "@earendil-works/pi-ai"; +import { stream } from "@earendil-works/pi-ai/api/openai-completions"; +import { genericModelConfig } from "./model-capabilities.js"; +import { buildProviderModel } from "./provider-binding.js"; + +const baseUrl = "https://native.invalid/v1"; +const read = { name: "Read", description: "Read", parameters: Type.Object({ path: Type.String() }) }; +const search = { ...read, name: "Search" }; +const messages: Message[] = [ + { role: "system", content: "", sections: { runtime: "Base", skills: "Old catalog" }, toolsAdded: [read], timestamp: 1 }, + { role: "user", content: "Find files", timestamp: 2 }, + { role: "system", content: "", sections: { skills: "New catalog" }, toolsAdded: [search], timestamp: 3 }, +]; +type Payload = { messages: Array<{ role: string; content?: unknown; tools?: unknown[] }>; tools?: Array<{ function: { name: string; parameters: unknown } }> }; +async function payload(instructions: boolean, additions: boolean, route = baseUrl, transcript = messages): Promise { + const model = buildProviderModel({ + id: "fixture", name: "Fixture", modelId: "exact", baseUrl: route, apiKey: "", authKind: "none", + supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: { + ...genericModelConfig("exact", baseUrl), source: "pi", + transcriptBinding: { modelId: "exact", api: "openai-completions", baseUrl }, + compat: { supportsMidConvoSystemMessages: instructions, supportsMidConvoToolAdditions: additions }, + }, + }); + let captured: Payload | undefined; + const response = stream(model as Model<"openai-completions">, normalizeContext({ messages: transcript }), { + apiKey: "fixture", + fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + expect((await response.result()).stopReason).toBe("stop"); + if (!captured) throw new Error("No payload captured"); + return captured; +} + +it("uses native additions only when both instruction and tool capabilities are verified", async () => { + const original = structuredClone(messages); + const native = await payload(true, true); + expect(native.tools?.map((tool) => tool.function.name)).toEqual(["Read"]); + expect(native.messages.slice(0, 2).map((message) => message.role)).toEqual(["system", "user"]); + expect(native.messages.slice(2).some((message) => message.tools?.length === 1)).toBe(true); + const partial = await payload(true, false); + expect(partial.tools?.map((tool) => tool.function.name)).toEqual(["Read", "Search"]); + expect(partial.messages.filter((message) => message.role === "system")).toHaveLength(2); + const relay = await payload(true, true, "https://relay.invalid/v1"); + expect(relay.messages.map((message) => message.role)).toEqual(["system", "user"]); + expect(relay.tools?.map((tool) => tool.function.name)).toEqual(["Read", "Search"]); + expect(JSON.stringify(relay.messages)).toContain("New catalog"); + expect(JSON.stringify(relay.messages)).not.toContain("Old catalog"); + expect(await payload(true, true)).toEqual(native); + expect(messages).toEqual(original); +}); + +it("folds removals and same-name replacements into the active request tool set", async () => { + const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; + const request = await payload(true, true, baseUrl, [...messages, { + role: "system", content: "", timestamp: 4, + toolsRemoved: [{ name: "Read" }, { name: "Search" }], toolsAdded: [replacement], + }]); + expect(request.tools).toEqual([expect.objectContaining({ function: expect.objectContaining({ name: "Read", parameters: replacement.parameters }) })]); + expect(request.messages.some((message) => message.tools)).toBe(false); +}); + +it("projects per-skill changes and revocations with native and fallback model support", async () => { + const catalog: Message[] = [ + { role: "system", content: "", timestamp: 1, sections: { + runtime: "Base", skills: "Load skills", "skill:a": "Alpha", "skill:b": "Bravo", "skill:c": "Charlie", + } }, + { role: "user", content: "Continue", timestamp: 2 }, + { role: "system", content: "", timestamp: 3, sections: { "skill:a": "Alpha updated", "skill:b": null } }, + ]; + const before = await payload(true, false, baseUrl, catalog.slice(0, 2)); + const native = await payload(true, false, baseUrl, catalog); + expect(native.messages.slice(0, before.messages.length)).toEqual(before.messages); + expect(JSON.stringify(native.messages.at(-1))).toContain("Alpha updated"); + expect(JSON.stringify(native.messages.at(-1))).not.toContain("Charlie"); + for (const route of [baseUrl, "https://relay.invalid/v1"]) { + const fallback = await payload(false, false, route, catalog); + expect(fallback.messages.map((message) => message.role)).toEqual(["system", "user"]); + expect(JSON.stringify(fallback.messages)).toContain("Alpha updated"); + expect(JSON.stringify(fallback.messages)).toContain("Charlie"); + expect(JSON.stringify(fallback.messages)).not.toContain("Bravo"); + } +}); diff --git a/packages/shared/src/types/messages.ts b/packages/shared/src/types/messages.ts index 361a476c97..f34738a168 100644 --- a/packages/shared/src/types/messages.ts +++ b/packages/shared/src/types/messages.ts @@ -70,6 +70,15 @@ export type UiMessage = { id: string; role: UiMessageRole; content: string; + /** Internal model instructions/tool declarations; never a visible chat row. */ + modelSystem?: { + version: 1; + /** The following user input can have been persisted before runtime admission. */ + beforeMessageId?: string; + afterMessageId?: string; + /** Opaque JSON preserves section and schema key order across Host storage. */ + messageJson: string; + }; /** Authenticated agent-to-agent provenance; never inferred from message text. */ sessionMessage?: SessionMessageOrigin; /** Present only on the durable user row created by a Live Voice operation. */ diff --git a/patches/@earendil-works__pi-ai@0.99.1.patch b/patches/@earendil-works__pi-ai@0.99.1.patch index df5da3d0dc..a6db0ceffa 100644 --- a/patches/@earendil-works__pi-ai@0.99.1.patch +++ b/patches/@earendil-works__pi-ai@0.99.1.patch @@ -1613,3 +1613,10 @@ index 000000000..5c63b882a + return stream; + } +} +diff --git a/dist/providers/data/deepseek.json b/dist/providers/data/deepseek.json +--- a/dist/providers/data/deepseek.json ++++ b/dist/providers/data/deepseek.json +@@ -1 +1 @@ +-{"openai-completions":{"chat:deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"thinkingLevelMap":{"minimal":null,"low":"low","medium":null,"high":"high","max":"max"},"input":["text","image"],"cost":{"input":0.3,"output":1.2,"cacheRead":0.006,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true},"inputLimits":{"images":{"resize":{"maxWidth":2000,"maxHeight":2000,"maxBytes":4718592,"jpegQuality":80}}},"type":"chat"},"chat:deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"input":["text"],"cost":{"input":1.32,"output":3.96,"cacheRead":0.044,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"thinkingLevelMap":{"minimal":null,"low":null,"medium":null,"high":"high","max":"max"},"type":"chat"}}} ++{"openai-completions":{"chat:deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"thinkingLevelMap":{"minimal":null,"low":"low","medium":null,"high":"high","max":"max"},"input":["text","image"],"cost":{"input":0.3,"output":1.2,"cacheRead":0.006,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"inputLimits":{"images":{"resize":{"maxWidth":2000,"maxHeight":2000,"maxBytes":4718592,"jpegQuality":80}}},"type":"chat"},"chat:deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"input":["text"],"cost":{"input":1.32,"output":3.96,"cacheRead":0.044,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"thinkingLevelMap":{"minimal":null,"low":null,"medium":null,"high":"high","max":"max"},"type":"chat"}}} +\ No newline at end of file diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 4010b26e91..b984698f6e 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -178,7 +178,7 @@ patchedDependencies: hash: b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7 path: patches/@earendil-works__pi-agent-core@0.99.1.patch '@earendil-works/pi-ai@0.99.1': - hash: c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4 + hash: d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4 path: patches/@earendil-works__pi-ai@0.99.1.patch '@earendil-works/pi-coding-agent@0.99.1': hash: 0e05051778ef4ffca9bac7ec68cae64f1d659d4b8640eeeb1a121e0f620a6abe @@ -209,7 +209,7 @@ importers: devDependencies: '@earendil-works/pi-ai': specifier: 0.99.1 - version: 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3) + version: 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-mcp': specifier: 0.99.1 version: 0.99.1 @@ -404,7 +404,7 @@ importers: version: 0.99.1(patch_hash=b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-ai': specifier: 0.99.1 - version: 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + version: 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-coding-agent': specifier: 0.99.1 version: 0.99.1(patch_hash=0e05051778ef4ffca9bac7ec68cae64f1d659d4b8640eeeb1a121e0f620a6abe)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(ws@8.21.1)(zod@4.4.3) @@ -5310,7 +5310,7 @@ snapshots: '@earendil-works/pi-agent-core@0.99.1(patch_hash=b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': dependencies: '@earendil-works/chord': 0.99.1 - '@earendil-works/pi-ai': 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + '@earendil-works/pi-ai': 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-telemetry': 0.99.1 diff: 8.0.4 ignore: 7.0.8 @@ -5329,7 +5329,7 @@ snapshots: - ws - zod - '@earendil-works/pi-ai@0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3)': + '@earendil-works/pi-ai@0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.124.0(zod@4.4.3) '@aws-sdk/client-bedrock-runtime': 3.1127.0 @@ -5354,7 +5354,7 @@ snapshots: - ws - zod - '@earendil-works/pi-ai@0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': + '@earendil-works/pi-ai@0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.124.0(zod@4.4.3) '@aws-sdk/client-bedrock-runtime': 3.1127.0 @@ -5387,7 +5387,7 @@ snapshots: dependencies: '@earendil-works/chord': 0.99.1 '@earendil-works/pi-agent-core': 0.99.1(patch_hash=b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) - '@earendil-works/pi-ai': 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + '@earendil-works/pi-ai': 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-codemode': 0.99.1 '@earendil-works/pi-mcp': 0.99.1 '@earendil-works/pi-tui': 0.99.1 diff --git a/scripts/e2e-system-transcript.mjs b/scripts/e2e-system-transcript.mjs new file mode 100644 index 0000000000..6d4139152a --- /dev/null +++ b/scripts/e2e-system-transcript.mjs @@ -0,0 +1,172 @@ +#!/usr/bin/env node +/** Production sidecar transport + isolated Host + local SSE provider. + * Message persistence is harness-owned; Electron's UI/outbox is not exercised. + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, rm } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +import { PROTOCOL_VERSION } from "../packages/shared/dist/protocol.js"; +import { withScenario } from "./e2e/fixture.mjs"; +import { resolveHostBinary } from "./e2e/host.mjs"; +import { createSession } from "./e2e/session.mjs"; + +const root = await mkdtemp(join(tmpdir(), "pi-system-state-e2e-")); +const requests = []; +const responses = []; +const server = createServer(async (req, res) => { + let body = ""; + for await (const chunk of req) body += chunk; + requests.push(JSON.parse(body)); + const tool = responses.shift(); + const delta = tool ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), type: "function", + function: { name: tool.name, arguments: JSON.stringify(tool.args) } }] } + : { role: "assistant", content: "The fixture completed." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + for (const [value, finish] of [[delta, null], [{}, tool ? "tool_calls" : "stop"]]) { + res.write(`data: ${JSON.stringify({ id: "fixture", model: "fixture", choices: [{ index: 0, delta: value, finish_reason: finish }] })}\n\n`); + } + res.end("data: [DONE]\n\n"); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; +const provider = { id: "fixture", name: "Fixture", modelId: "fixture", baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; +let skills = [ + { id: "fixture/notes", name: "First skill", description: "Summarize notes" }, + { id: "fixture/unchanged", name: "Unchanged skill", description: "Preserve this catalog entry" }, +]; +try { + await withScenario("E2E-system-transcript", async ({ host, workspace }) => { + const session = await createSession(host, workspace, "System state fixture"); + let writes = Promise.resolve(); + let writeError; + let stderr = ""; + const turns = new Map(); + const executedTools = []; + const toolRows = new Map(); + const persist = (message, turnId) => { + writes = writes.then(() => host.call("session.appendMessage", { sessionId: session.id, turnId, message })) + .catch((error) => { writeError ??= error; }); + }; + const startSidecar = () => { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, + args: [fileURLToPath(new URL("../packages/agent-runtime/dist/sidecar.js", import.meta.url))], + cwd: workspace, + env: { ...process.env, PI_DESKTOP_PLAN_UI_PROBE: "1", PI_DESKTOP_COMPACTION_STRATEGY: "fresh_window" }, + }, + onStderr: (chunk) => { stderr += chunk; }, + }); + sidecar.setHost({ call: host.call.bind(host), onNotification: () => () => {}, onExit: () => () => {} }); + sidecar.setLocalTool("Skill", async ({ args }) => { + executedTools.push(args); + return { ok: true, content: "Fixture skill body: summarize the notes." }; + }); + sidecar.onNotification((method, envelope) => { + if (method !== "agent.event") return; + const { event, turnId } = envelope; + const turn = turns.get(turnId); + if (event.type === "error") turn?.errors.push(event.error); + if (event.type === "message_end") persist(event.message, turnId); + if (event.type === "tool_start") toolRows.set(event.toolCallId, { + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + turn?.toolResults.push(event); + persist({ ...toolRows.get(event.toolCallId), toolResult: event.result, isError: event.isError, + toolStatus: event.isError ? "error" : "success" }, turnId); + } + if (event.type === "agent_end") turn?.resolve(); + }); + return sidecar; + }; + let sidecar = startSidecar(); + const params = () => ({ sessionId: session.id, mode: "agent", provider, thinkingLevel: "off", pluginSkills: skills, + projectPath: workspace, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + }); + const identity = async () => (await sidecar.call("agent.testRuntimeIdentity", { sessionId: session.id })).runtimeId; + const prompt = async (id) => { + const { turnId } = await host.call("session.beginTurn", { sessionId: session.id }); + await host.call("session.appendMessage", { sessionId: session.id, turnId, + message: { id, role: "user", content: "Continue", createdAt: new Date().toISOString() } }); + let timer; + const turn = { errors: [], toolResults: [] }; + const ended = new Promise((resolve, reject) => { + turn.resolve = resolve; + timer = setTimeout(() => reject(new Error(`Turn timed out: ${JSON.stringify(turn.errors)}\n${stderr}`)), 20_000); + }); + turns.set(turnId, turn); + try { + await sidecar.call("agent.prompt", { ...params(), turnId, userMessageId: id, content: "Continue" }); + await ended; + await writes; + if (writeError) throw writeError; + assert.deepEqual(turn.errors, [], JSON.stringify(turn.errors)); + assert(turn.toolResults.every((event) => !event.isError), JSON.stringify(turn.toolResults)); + await host.call("session.endTurn", { turnId, status: "completed", createNotification: false }); + return turn; + } finally { clearTimeout(timer); turns.delete(turnId); } + }; + const detail = async () => (await host.call("session.get", { id: session.id })).session; + try { + responses.push({ name: "ToolSearch", args: { query: "BrowserPreview" } }); + await prompt("user-1"); + const saved = await detail(); + const states = saved.messages.filter((row) => row.modelSystem); + assert.equal(states.length, 2); + assert(JSON.parse(states[1].modelSystem.messageJson).toolsAdded.some((tool) => tool.name === "BrowserPreview")); + assert(requests[1].tools.some((tool) => tool.function.name === "BrowserPreview")); + console.log("PASS: sidecar ToolSearch and acknowledged Host persistence"); + await sidecar.dispose(); + await host.stop(); await host.start(PROTOCOL_VERSION); + const recovered = await detail(); + assert.deepEqual(recovered.messages.filter((row) => row.modelSystem), states); + sidecar = startSidecar(); + await prompt("user-2"); + const resumedStates = (await detail()).messages.filter((row) => row.modelSystem); + assert.equal(resumedStates.length, 2, JSON.stringify(resumedStates.slice(2).map((row) => row.modelSystem.messageJson))); + assert(requests.at(-1).tools.some((tool) => tool.function.name === "BrowserPreview")); + console.log("PASS: Host and sidecar process restart restores tools without duplicate state"); + const priorRuntime = await identity(); + skills = [{ ...skills[0], name: "Updated skill" }, skills[1]]; + responses.push({ name: "Skill", args: { id: skills[0].id } }); + await prompt("user-3"); + assert.equal(await identity(), priorRuntime, "skill catalog update must reuse the runtime"); + assert.equal(executedTools.length, 1, "Skill execution reaches the embedding host"); + assert(JSON.stringify(requests.at(-1).messages).includes("Fixture skill body")); + assert(JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("First skill")); + assert(JSON.stringify(requests.at(-1).messages).includes("Unchanged skill")); + console.log("PASS: skill catalog update reuses runtime and Skill executes across process boundary"); + await sidecar.call("agent.compact", params()); + const compacted = await detail(); + assert(JSON.parse(compacted.compaction.details.systemMessageJson).toolsAdded.some((tool) => tool.name === "BrowserPreview")); + await sidecar.dispose(); + sidecar = startSidecar(); + await prompt("user-4"); + assert(requests.at(-1).tools.some((tool) => tool.function.name === "BrowserPreview")); + assert(!requests.at(-1).messages.some((message) => message.role === "tool")); + assert(JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("First skill")); + console.log("PASS: compaction and sidecar restart preserve current skills and active tools"); + const beforeRemoval = await identity(); + skills = []; + await prompt("user-5"); + assert.equal(await identity(), beforeRemoval); + assert(!requests.at(-1).tools.some((tool) => tool.function.name === "Skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + console.log("PASS: removing the final skill removes its catalog and executable schema"); + } finally { await sidecar.dispose(); await writes; } + }, resolveHostBinary(), root, PROTOCOL_VERSION); +} finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + await rm(root, { recursive: true, force: true }); +} From 022cb0a8ffecbbe1ad5d06db5f5e5ed08c3545ee Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:02:12 +0800 Subject: [PATCH 2/8] fix(runtime): preserve transport bindings through catalog enrichment Keep Pi's exact transport capabilities independent of models.dev-owned limits and prices so catalog enrichment cannot silently disable updates. Normalize compaction checkpoints to the durable journal shape, including extension-provided text blocks, and cover recovery cleanup boundaries. Related to #1285 --- .../electron/main/models-dev-catalog.ts | 12 +++++- apps/desktop/test/models-dev-catalog.test.mjs | 41 +++++++++++++++++++ docs/adr/chronological-system-transcript.md | 7 ++++ docs/spec/03-runtime/02-agent-runtime.md | 5 +++ .../agent-runtime/src/model-capabilities.ts | 1 + .../src/provider-binding.test.ts | 9 ++-- packages/agent-runtime/src/runtime.test.ts | 9 ++-- .../src/system-transcript-journal.test.ts | 10 +++++ .../src/system-transcript-runtime.test.ts | 3 ++ .../src/system-transcript.test.ts | 10 +++++ .../agent-runtime/src/system-transcript.ts | 8 +++- .../src/transcript-compat.test.ts | 4 ++ .../agent-runtime/src/transcript-compat.ts | 13 +++++- 13 files changed, 117 insertions(+), 15 deletions(-) diff --git a/apps/desktop/electron/main/models-dev-catalog.ts b/apps/desktop/electron/main/models-dev-catalog.ts index fbf42f28ce..a8a6b310f0 100644 --- a/apps/desktop/electron/main/models-dev-catalog.ts +++ b/apps/desktop/electron/main/models-dev-catalog.ts @@ -32,7 +32,7 @@ import type { ThinkingProtocol, ModelBinding, } from "@pi-desktop/shared"; -import { genericModelConfig, modelConfigWithBinding, type ModelConfig } from "@pi-desktop/agent-runtime"; +import { genericModelConfig, modelConfigWithBinding, transcriptConfigFromPi, type ModelConfig } from "@pi-desktop/agent-runtime"; import { resolveBindingLimits } from "@pi-desktop/shared"; export const MODELS_DEV_API_URL = "https://models.dev/api.json"; @@ -1424,6 +1424,16 @@ export class ModelsDevCatalog { const thinking = this.anthropicThinkingFor(input.modelId); if (thinking) baseline = { ...baseline, ...thinking }; } + // Pi owns wire capabilities, independently of models.dev's limits/prices. + // Use the original published record, never an account/relay projection. + const key = input.vendorKey?.trim().toLowerCase(); + const vendor = (key ? PI_VENDOR_ALIASES[key] ?? key : undefined) ?? this.providerKeyForRow(input); + const transport = vendor ? this.operationModels.getModel(vendor, input.modelId) : undefined; + if (transport) { + const transcript = transcriptConfigFromPi(transport); + baseline = { ...baseline, transcriptBinding: transcript.transcriptBinding, + compat: { ...baseline.compat, ...transcript.compat } }; + } if (!binding) return baseline; const limits = resolveBindingLimits(baseline, binding); return modelConfigWithBinding(limits.catalogConfig, limits.binding); diff --git a/apps/desktop/test/models-dev-catalog.test.mjs b/apps/desktop/test/models-dev-catalog.test.mjs index cdfc969a1b..7d79bf3e82 100644 --- a/apps/desktop/test/models-dev-catalog.test.mjs +++ b/apps/desktop/test/models-dev-catalog.test.mjs @@ -4,11 +4,15 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import test from "node:test"; import { fileURLToPath } from "node:url"; +import { createModels, InMemoryModelsStore } from "@earendil-works/pi-ai"; +import { builtinProviders } from "@earendil-works/pi-ai/providers/all"; +import { buildProviderModel } from "../../../packages/agent-runtime/dist/provider-binding.js"; import { apiStyleForAdapter, bindingForCustomModelInfo, catalogModelIdsMatch, modelIdsMatch, + resolveApiStyle, } from "@pi-desktop/shared"; import { MODELS_DEV_API_URL, @@ -103,6 +107,43 @@ async function loadFixtureCatalog(t, fixture = catalogFixture) { return catalog; } +test("published metadata retains exact Pi transcript transport bindings through runtime launch", async (t) => { + const pi = createModels({ modelsStore: new InMemoryModelsStore(), + authContext: { env: async () => undefined, fileExists: async () => false } }); + for (const provider of builtinProviders()) pi.setProvider(provider); + for (const [vendorKey, modelId] of [ + ["deepseek", "deepseek-flash"], ["anthropic", "claude-opus-5-5"], + ["openai", "gpt-6.1-sol"], ["openai-codex", "gpt-6.1-sol"], + ]) { + const original = pi.getModel(vendorKey, modelId); + assert.ok(original, `${vendorKey}/${modelId}`); + const catalog = await loadFixtureCatalog(t, { [vendorKey]: { + name: vendorKey, api: original.baseUrl, models: { [modelId]: { + id: modelId, limit: { context: 543_210, output: 40_000 }, cost: { input: 1.25, output: 2.5 }, + } }, + } }); + const target = { providerId: "account", vendorKey, modelId, baseUrl: original.baseUrl, + apiStyle: resolveApiStyle(original.api) }; + const config = catalog.modelConfigFor(target); + assert.equal(config.source, "models.dev"); + assert.equal(config.contextWindow, 543_210); + assert.equal(config.maxTokens, 40_000); + assert.equal(config.cost.input, 1.25); + assert.deepEqual(config.transcriptBinding, { modelId, api: original.api, baseUrl: original.baseUrl }); + const provider = { ...target, id: "account", name: "Fixture", apiKey: "", authKind: "none", + supportsReasoning: false, supportedThinkingLevels: ["off"], modelConfig: config }; + const wire = buildProviderModel(provider); + assert.equal(wire.compat.supportsMidConvoSystemMessages, true, vendorKey); + assert.equal(wire.compat.supportsMidConvoToolChanges, original.compat?.supportsMidConvoToolChanges === true); + assert.equal(wire.compat.supportsAdditionalTools, original.compat?.supportsAdditionalTools === true); + const relay = { ...target, baseUrl: "https://relay.invalid/v1" }; + assert.equal(buildProviderModel({ ...provider, ...relay, modelConfig: catalog.modelConfigFor(relay) }) + .compat.supportsMidConvoSystemMessages, false); + assert.equal(buildProviderModel({ ...provider, modelId: `${modelId}-alias` }) + .compat.supportsMidConvoSystemMessages, false); + } +}); + test("the bundled models.dev OpenAI record supplies the selected model limits", async () => { const catalogPath = fileURLToPath(new URL("../resources/models.dev/api.json", import.meta.url)); const catalog = new ModelsDevCatalog({ catalogPath }); diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md index 74884ee9f0..c1a4840033 100644 --- a/docs/adr/chronological-system-transcript.md +++ b/docs/adr/chronological-system-transcript.md @@ -38,6 +38,13 @@ Accept transcript capability opt-ins only when that identity matches the actual request route. Let Pi adapt the request for partial or absent support; never persist an adapter's folded projection. +The current models.dev catalog remains authoritative for published limits, +prices and modalities. Independently copy only the five transcript transport +flags and their original binding from Pi's exact published model. Do not derive +transport support from a models.dev metadata match or an account endpoint +override. Both Pi and models.dev metadata projections can carry that binding; +unverified routes and generic records retain the conservative fallback. + The existing Pi 0.99.1 dependency patch adds the missing mid-conversation system capability to its `deepseek-flash` catalog entry. Authorized official-endpoint experiments confirmed both preserved cache reuse and effective updated diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index db1a2ee59f..44506236bd 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -1308,6 +1308,11 @@ and redefinitions use the adapter's supported fallback. Model switching never rewrites the canonical journal. Cache savings depend on the actual provider; unsupported routes may still rebuild the request prefix. +Published model limits, prices and modalities remain owned by models.dev. Its +runtime projection separately carries Pi's exact published transcript capability +flags and original binding. An enriched metadata source does not grant native +support by itself, and account overrides never replace that original binding. + The pinned Pi patch declares mid-conversation system support for `deepseek-flash` on its published `openai-completions` binding at `https://api.deepseek.com`. Skill changes on this binding append system updates diff --git a/packages/agent-runtime/src/model-capabilities.ts b/packages/agent-runtime/src/model-capabilities.ts index 2c7f3cd589..43e54643cb 100644 --- a/packages/agent-runtime/src/model-capabilities.ts +++ b/packages/agent-runtime/src/model-capabilities.ts @@ -7,6 +7,7 @@ import { type ModelModality, } from "@pi-desktop/shared"; import type { ModelConfig, ThinkingCapabilitySet } from "./thinking-level.js"; +export { transcriptConfigFromPi } from "./transcript-compat.js"; export { agentThinkingLevel, diff --git a/packages/agent-runtime/src/provider-binding.test.ts b/packages/agent-runtime/src/provider-binding.test.ts index bc40364434..ad755e70ac 100644 --- a/packages/agent-runtime/src/provider-binding.test.ts +++ b/packages/agent-runtime/src/provider-binding.test.ts @@ -254,12 +254,9 @@ describe("Anthropic runtime endpoint", () => { .result(); expect(result.stopReason).toBe("error"); - expect( - Object.hasOwn(model.compat ?? {}, "supportsMidConvoSystemMessages"), - ).toBe(false); - expect( - Object.hasOwn(model.compat ?? {}, "supportsMidConvoToolChanges"), - ).toBe(false); + expect(model.compat).toMatchObject({ + supportsMidConvoSystemMessages: false, supportsMidConvoToolChanges: false, + }); expect(request?.headers.get("anthropic-beta") ?? "").not.toMatch( /mid-conversation-tool-changes|inline-tools/, ); diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index 9318de1cbf..689bab439c 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -2213,11 +2213,10 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(agent.state.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Tool deltas append new declarations; catalog order is not semantic. - const byName = (a: { name: string }, b: { name: string }) => a.name.localeCompare(b.name); - expect([...getCurrentTools(agent.state.messages)].sort(byName)).toEqual( - agent.state.tools.map(toToolDeclaration).sort(byName), - ); + // Reset retains executability; Pi declares it at the next dispatch, not + // while this test calls preparation helpers outside the agent loop. + expect(getCurrentTools(agent.state.messages).some((tool) => tool.name === "BrowserPreview")).toBe(false); + await runtime.dispose(); }); }); diff --git a/packages/agent-runtime/src/system-transcript-journal.test.ts b/packages/agent-runtime/src/system-transcript-journal.test.ts index 22b9816a3c..3fc43d2eef 100644 --- a/packages/agent-runtime/src/system-transcript-journal.test.ts +++ b/packages/agent-runtime/src/system-transcript-journal.test.ts @@ -80,6 +80,16 @@ describe("model system journal", () => { expect(checkpoint?.toolsAdded).toEqual(initial.toolsAdded); }); + it("recognizes a restored checkpoint with text blocks without adding durable rows", async () => { + const blocks: SystemMessage = { ...initial, content: [{ type: "text", text: "Extension instructions" }] }; + const checkpoint = systemTranscriptCheckpoint([blocks, user])!; + const journal = new SystemTranscriptJournal(); + journal.rememberCheckpoint(JSON.stringify(checkpoint)); + const append = vi.fn(); + await journal.persist(JSON.parse(JSON.stringify([checkpoint, user])), [], append); + expect(append).not.toHaveBeenCalled(); + }); + it.each([null, {}, { ...initial, timestamp: -1 }, { ...initial, toolsAdded: [{ name: "Read", parameters: "bad" }] }])("rejects malformed saved state", (value) => { expect(() => readSystemMessage(value)).toThrow("Invalid persisted model system message"); }); diff --git a/packages/agent-runtime/src/system-transcript-runtime.test.ts b/packages/agent-runtime/src/system-transcript-runtime.test.ts index f48e132ddb..15ced3662a 100644 --- a/packages/agent-runtime/src/system-transcript-runtime.test.ts +++ b/packages/agent-runtime/src/system-transcript-runtime.test.ts @@ -86,6 +86,9 @@ describe("system state through the Desktop user path", () => { const activation = after.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); expect(after.slice(activation + 1)).toContainEqual(expect.objectContaining({ role: "system", toolsAdded: expect.arrayContaining([expect.objectContaining({ name: "BrowserPreview" })]) })); expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); + await f.prompt("same-runtime-user"); + expect(getCurrentTools(f.requests[2]).some((tool) => tool.name === "BrowserPreview")).toBe(true); + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); const restored = fixture(structuredClone(f.rows)); try { await restored.prompt("user-2"); diff --git a/packages/agent-runtime/src/system-transcript.test.ts b/packages/agent-runtime/src/system-transcript.test.ts index 8e994a499c..3d4ec5c1d3 100644 --- a/packages/agent-runtime/src/system-transcript.test.ts +++ b/packages/agent-runtime/src/system-transcript.test.ts @@ -19,6 +19,7 @@ import { replaceSystemPrompt, systemTranscriptCheckpoint, systemPromptContent, + removeTrailingAssistantMessages, } from "./system-transcript.js"; const read: Tool = { name: "Read", description: "Read text", parameters: Type.Object({ path: Type.String() }) }; @@ -50,6 +51,15 @@ function estimateContextTokens(messages: AgentMessage[]) { afterEach(() => vi.restoreAllMocks()); describe("system transcript helpers", () => { + it("removes failed trailing responses across cleanup deltas without changing system state", () => { + const messages = [initial, user, assistant, delta, { ...assistant, timestamp: 3_000 }]; + const cleaned = removeTrailingAssistantMessages(messages); + expect(cleaned).toEqual([initial, user, delta]); + expect(getCurrentTools(cleaned)).toEqual(getCurrentTools(messages)); + expect(messages).toHaveLength(5); + expect(removeTrailingAssistantMessages([user, assistant])).toEqual([user]); + expect(removeTrailingAssistantMessages([assistant, user, delta])).toEqual([assistant, user, delta]); + }); it("updates only the changed skill entry and preserves unrelated sections", () => { const previous: AgentMessage[] = [{ role: "system", content: "", timestamp: 1, diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index a5bac9363b..f3e3470337 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -106,8 +106,12 @@ export function rebuildSystemTranscript( /** Fold state only at an explicit compaction boundary, with semantic time. */ export function systemTranscriptCheckpoint(messages: readonly AgentMessage[]): SystemMessage | undefined { - const checkpoint = currentSystemMessage(messages); - if (!checkpoint?.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; + const current = currentSystemMessage(messages); + if (!current) return undefined; + // Use the journal's durable shape, including extension-provided text blocks. + const checkpoint: SystemMessage = { ...current, content: contentText(current.content), + ...(current.toolsAdded ? { toolsAdded: current.toolsAdded.map(toToolDeclaration) } : {}) }; + if (!checkpoint.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; const { [CONTEXT_BUDGET_SECTION]: _expired, ...sections } = checkpoint.sections; return { ...checkpoint, sections }; } diff --git a/packages/agent-runtime/src/transcript-compat.test.ts b/packages/agent-runtime/src/transcript-compat.test.ts index 86fb4c2c6e..6e3df7cede 100644 --- a/packages/agent-runtime/src/transcript-compat.test.ts +++ b/packages/agent-runtime/src/transcript-compat.test.ts @@ -14,6 +14,10 @@ const provider: RuntimeProviderConfig = { }; describe("transcript compatibility binding", () => { + it("keeps Pi transport capabilities when models.dev owns published metadata", () => { + expect(buildProviderModel({ ...provider, modelConfig: { ...provider.modelConfig!, source: "models.dev" } }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: true }); + }); it("keeps instruction and tool capabilities independent on the exact binding", () => { expect(buildProviderModel(provider).compat).toMatchObject({ supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }); expect(buildProviderModel({ ...provider, baseUrl: `${baseUrl}/` }).compat).toMatchObject({ supportsMidConvoSystemMessages: true }); diff --git a/packages/agent-runtime/src/transcript-compat.ts b/packages/agent-runtime/src/transcript-compat.ts index 365e2f6da9..887e406877 100644 --- a/packages/agent-runtime/src/transcript-compat.ts +++ b/packages/agent-runtime/src/transcript-compat.ts @@ -1,10 +1,21 @@ import type { ModelConfig } from "./thinking-level.js"; +import type { Api, Model } from "@earendil-works/pi-ai"; const capabilities = [ "supportsMidConvoSystemMessages", "supportsMidConvoToolAdditions", "supportsMidConvoToolChanges", "supportsAdditionalTools", "supportsToolSearch", ] as const; +/** Transport capabilities only; published limits and prices keep their owner. */ +export function transcriptConfigFromPi(model: Model): Pick { + return { + transcriptBinding: { modelId: model.id, api: model.api, baseUrl: model.baseUrl }, + compat: Object.fromEntries(capabilities.map((key) => [key, + model.compat !== undefined && key in model.compat && Reflect.get(model.compat, key) === true, + ])), + }; +} + function endpoint(value: string): string | undefined { try { const url = new URL(value); @@ -21,7 +32,7 @@ export function transcriptCompat( ): Record<(typeof capabilities)[number], boolean> { const binding = catalog?.transcriptBinding; const address = endpoint(baseUrl); - const verified = catalog?.source === "pi" && binding?.modelId === modelId && + const verified = (catalog?.source === "pi" || catalog?.source === "models.dev") && binding?.modelId === modelId && binding.api === api && address !== undefined && endpoint(binding.baseUrl) === address; return Object.fromEntries(capabilities.map((key) => [key, verified && catalog?.compat?.[key] === true])) as Record<(typeof capabilities)[number], boolean>; From 09f2cdbb7e0c624cce236e761544926e704ccabb Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:31:23 +0800 Subject: [PATCH 3/8] docs(adr): index chronological system transcript decision Register the system transcript decision in the ADR index so the documentation integrity check no longer aborts the workspace test run. --- docs/adr/README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/adr/README.md b/docs/adr/README.md index 4bac9199ab..753634d3a3 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -20,6 +20,7 @@ Each ADR includes: | ID | Title | Status | |---|---|---| +| chronological-system-transcript | [Preserve chronological model system state](chronological-system-transcript.md) | Accepted | | mcp-tool-approval-risk | [User MCP tools keep the normal approval path](mcp-tool-approval-risk.md) | Accepted | | models-dev-catalog-authority | [models.dev owns published model metadata](models-dev-catalog-authority.md) | Accepted for implementation | | pi-ai-core-0991-authority | [Pi 0.99.1 account model authority](pi-ai-core-0991-authority.md) | Superseded for chat model metadata | From b622cb975bc1b6c8eed343a74ce531e462e78ee3 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:35:22 +0800 Subject: [PATCH 4/8] test(runtime): align compaction reminder checks with system sections The reminder now travels in a replaceable transcript section so it can expire after compaction. Update the old source guard and assert that Pi folds the actual reminder content into the current system state. --- apps/desktop/test/context-compaction.test.mjs | 7 +++---- packages/agent-runtime/src/runtime.test.ts | 9 ++++++++- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/apps/desktop/test/context-compaction.test.mjs b/apps/desktop/test/context-compaction.test.mjs index d57835f90b..9080321c58 100644 --- a/apps/desktop/test/context-compaction.test.mjs +++ b/apps/desktop/test/context-compaction.test.mjs @@ -114,12 +114,11 @@ test("the hard boundary is enforced by the host, with a model-side escape hatch" assert.match(runtime, /function contextFallbackReminder\(/); assert.match(runtime, /contextReminderClaimed/); assert.match(runtime, /contextFallbackReminderClaimed/); - // pi 0.86 carries the canonical system prompt in the transcript. The - // reminder is appended as a per-turn system message, not only as the legacy - // AgentContext.systemPrompt field. + // The reminder is appended as a replaceable system section so compaction + // can expire it without rewriting earlier instructions. assert.match( runtime, - /messages: \[\s*\.\.\.context\.messages,\s*\{\s*role: "system",\s*content: reminder,/, + /messages: \[\s*\.\.\.context\.messages,\s*\{\s*role: "system",\s*content: "",\s*sections: \{ \[CONTEXT_BUDGET_SECTION\]: reminder \}/, ); assert.match(hostPermissions, /"new_context"/); }); diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index 689bab439c..61e2d83e13 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -5285,8 +5285,15 @@ describe("DesktopAgentRuntime per-turn context protection", () => { // compaction trigger, so the model can close out before host compaction. expect(below.context.systemPrompt).toContain(""); expect(below.context.systemPrompt).toContain("new_context"); + expect(below.context.messages.at(-1)).toMatchObject({ + role: "system", + content: "", + sections: { context_budget: expect.stringContaining("") }, + }); + expect(getCurrentSystemMessage(below.context.messages)?.sections?.context_budget).toContain("new_context"); expect(stillBelow.context.systemPrompt).not.toContain(""); - // The reminder rides on the turn's context only; nothing is persisted. + // The legacy prompt baseline stays unchanged; the returned transcript + // carries the reminder section through the normal persistence path. expect((runtime as any).agent.state.systemPrompt).not.toContain( "", ); From 907594acb86096458cd149d4f7123ad5e5bf9ff7 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 15:12:22 +0800 Subject: [PATCH 5/8] fix(runtime): keep Flash tool declarations stable across searches Declare the verified Flash catalog once per account and schema epoch so ToolSearch activation does not invalidate the earlier request prefix. Persist activation independently and reject inactive tools before Host execution, preserving existing mode and approval restrictions. Cover HTTP payloads, recovery, compaction, revoked tools and fallback limits, with paired live Flash measurements documenting the larger cold request and the short-conversation cost tradeoff. refs #1285 --- docs/adr/chronological-system-transcript.md | 39 +++- docs/spec/03-runtime/02-agent-runtime.md | 62 +++-- docs/spec/06-delivery/04-e2e-test-plan.md | 27 +++ .../zh-CN/spec/03-runtime/02-agent-runtime.md | 24 +- .../spec/06-delivery/04-e2e-test-plan.md | 27 +++ .../src/fixed-tool-declarations.test.ts | 52 +++++ .../src/fixed-tool-declarations.ts | 69 ++++++ .../src/fixed-tool-runtime.test.ts | 216 ++++++++++++++++++ packages/agent-runtime/src/runtime.ts | 71 +++++- .../agent-runtime/src/system-transcript.ts | 6 +- scripts/e2e-fixed-tool-declarations.mjs | 170 ++++++++++++++ 11 files changed, 717 insertions(+), 46 deletions(-) create mode 100644 packages/agent-runtime/src/fixed-tool-declarations.test.ts create mode 100644 packages/agent-runtime/src/fixed-tool-declarations.ts create mode 100644 packages/agent-runtime/src/fixed-tool-runtime.test.ts create mode 100644 scripts/e2e-fixed-tool-declarations.mjs diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md index c1a4840033..659ec00a28 100644 --- a/docs/adr/chronological-system-transcript.md +++ b/docs/adr/chronological-system-transcript.md @@ -45,7 +45,7 @@ transport support from a models.dev metadata match or an account endpoint override. Both Pi and models.dev metadata projections can carry that binding; unverified routes and generic records retain the conservative fallback. -The existing Pi 0.99.1 dependency patch adds the missing mid-conversation system +The Pi 1.0.0 dependency patch adds the missing mid-conversation system capability to its `deepseek-flash` catalog entry. Authorized official-endpoint experiments confirmed both preserved cache reuse and effective updated instructions. Keep this correction in the single Pi catalog, not a parallel @@ -53,6 +53,43 @@ Desktop allowlist; remove the hunk when an upgraded Pi catalog carries it. The model/API/endpoint binding check still excludes aliases and relays. Native tool-addition and tool-change flags are not enabled by this correction. +## Fixed declarations for the verified Flash route + +For the exact official `deepseek-flash` / `openai-completions` binding with +verified chronological system support, declare the complete current tool catalog +in deterministic name order from the first request. ToolSearch changes execution +activation only. This is a Desktop declaration policy, not an additional Pi +transport capability or an endpoint switch. Other bindings keep on-demand +schema publication; native tool-state flags alone do not prove cache stability. + +Keep activation separate from declarations in a versioned `tool_activation` +system section. The section records active deferred names and a SHA-256 identity +of the account, model, API, endpoint, declarations and deferred-name set. Updates +append after tool results and persist through the existing Host journal and +compaction checkpoint. Restoration never interprets the complete declaration +snapshot as permission to execute every tool. Successful ToolSearch results +newer than the saved activation section recover an interrupted activation. +Invalid, unknown-version or mismatched state grants no activation. Legacy +histories without this section retain their existing activation evidence rules. + +Check activation before extension hooks or Host execution, then retain all +existing mode, approval and Host restrictions. A catalog/schema/mode/account or +route change starts a new declaration epoch and invalidates prior activation; +removed tools cannot be invoked. Temporary prompt replacement must preserve the +activation metadata. Current runtime activation remains authoritative between +prompts; declarations do not re-grant revoked activation. + +DeepSeek limits a request to 128 functions. If the full catalog exceeds that +limit, or its estimated prompt/schema cost leaves less than the ordinary +retained-tail budget below the automatic compaction threshold, use the existing +on-demand path and log the fallback reason. Never truncate a catalog. Context +estimation charges the full declared catalog while fixed declarations are active. + +The first request is larger. Short conversations may cost more overall even +when later cache-hit ratios improve. Acceptance compares cold and subsequent +uncached tokens and cumulative input cost across both short and longer synthetic +conversations; no universal savings or hit-rate guarantee is made. + ## Alternatives and consequences Prepending a reconstructed snapshot was rejected because it changes history on diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index 08346ab684..15bcd08d4a 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -1393,10 +1393,10 @@ grammar and validated against another fails every call. ### 7.1 Active tool context and on-demand loading (D185, ADR 0048) -The sidecar builds one complete tool registry, but it does not serialize every -registered schema into every provider request. Each new user prompt starts with -the mode's core set plus any deferred tools that can be restored from successful -activation evidence still present in the effective session context: +The sidecar builds one complete tool registry. By default, each provider request +declares the mode's core set plus activated deferred tools. The verified Flash +binding uses the fixed-declaration policy below, while preserving the same +execution activation rules: - Agent: `Read`, `Bash`, `Edit`, and `Write` (matching pi's coding-agent core) - Agent: `Skill` whenever the skill catalog is non-empty (D404, ADR 0230) — the @@ -1424,24 +1424,42 @@ The sidecar activates up to four matches, records their names in the canonical schemas. Providers with native deferred-tool search receive the definitions at that load point; other providers receive the active definitions normally. -At the start of each new user prompt, the sidecar clears the in-memory deferred -activation set and rebuilds it from the effective context. Recorded system -messages (including a compaction checkpoint) define the active baseline. Only -successful tool results after the latest declaration can add a new activation; -older results must not resurrect a removed tool. Histories without system records -continue to use successful results throughout their effective context. -`ToolSearch` results contribute their canonical `details.addedToolNames`. -For compatibility, historical `details.activated` and top-level -`addedToolNames` markers are also accepted. Successful results from deferred -tools contribute that tool's name. Only names still present in the current -mode's deferred catalog are restored. Failed rows, interrupted or -missing-result placeholders, and assistant/user prose never activate a tool. -The tool registry, host permission path, tool timeout, and workspace containment -rules remain unchanged. `ToolSearch` is local to the sidecar and does not cross -the host RPC boundary. Its activation marker is retained in the persisted tool -result, so a runtime restart or a new prompt can reuse an eligible capability -while that evidence remains in the effective context; a fresh search is still -required after the evidence is compacted away or otherwise absent. +Deferred activation is sticky for the live runtime. At restoration, recorded +system messages (including a compaction checkpoint) define the active baseline +for ordinary on-demand histories. Only successful results after the latest +system record can add activation; older results must not resurrect removed +tools. Legacy histories without system records use successful results throughout +their effective context. ToolSearch accepts canonical `details.addedToolNames` +and historical `details.activated` / top-level `addedToolNames`; successful +results from deferred tools also restore their names. Failed results, +missing-result placeholders and assistant/user prose never activate tools. +Only names in the current mode's deferred catalog are eligible. + +For the exact official `deepseek-flash` Chat Completions binding with verified +mid-conversation system support, the runtime instead declares the complete +catalog in deterministic name order on the first request. ToolSearch changes +activation without changing the declared schemas. A visible schema does not +permit execution: inactive deferred calls are rejected before extension hooks +and the Host; activated calls still require the existing mode and Host checks. +ToolSearch remains local and never grants approval or bypasses permissions. + +Fixed declarations persist separately from activation. A version-1 +`tool_activation` section records active names and a fingerprint of the account, +model, API, endpoint, schema catalog and deferred set. Activation changes append +at the continuation boundary, and the existing system journal/checkpoint saves +both declarations and activation. Restore only validated activation for a +matching fingerprint, plus successful ToolSearch results newer than that state; +never activate tools merely because the full snapshot declared them. Malformed, +unknown-version and mismatched activation state fail closed. A catalog/schema, +mode, account, model or route change creates a new epoch and requires new +activation. Removal immediately removes the tool from executable registration. + +If the full catalog exceeds 128 functions or its prompt/schema estimate cannot +leave the normal retained-tail budget below the compaction threshold, retain +on-demand declarations and emit a diagnostic explaining that ToolSearch cache +stability is not guaranteed. Do not truncate tools. Other models and unverified +routes retain the existing Pi projection. First-request schema overhead increases; +cache stability does not imply that short conversations become cheaper. For user-visible HTML deliverables, the default system prompt asks the agent to activate `BrowserPreview` once after creating the page or making its first diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 708281ee43..1407badc29 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -16318,3 +16318,30 @@ renderer's durable transcript reads. No real model or provider is contacted. same-model assistant reasoning remains separate from visible content. Adapter regressions also cover legacy identities and genuine account/model changes. Cache percentages are observations, not deterministic pass thresholds. + +### E2E-FIXED-TOOL-DECLARATIONS: Stable Flash schemas with independent activation + +- Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, + and a child-process fetch boundary redirected to local HTTP/SSE. Credentials + are dummy values and synthetic tools have no external side effects. +- Prompt, search Alpha, execute Alpha, search Beta and execute Beta. HTTP payload + tests assert identical ordered `tools` and unchanged prior message prefixes. + Directly calling a declared but inactive tool must fail before the Host. +- Restart both processes, then compact and restart again: Alpha stays activated, + Beta remains inactive despite its declaration. Activate Beta, then remove it + from the catalog and verify it cannot execute. New schema/route epochs must + not restore old grants. Host rejection and Plan guards remain effective. +- Runtime contracts additionally cover legacy migration, interrupted activation, + malformed metadata, temporary prompt replacement, deterministic catalog order, + the 128-tool boundary and insufficient-context fallback. +- Automated: `node scripts/e2e-fixed-tool-declarations.mjs` with built + shared/host-runtime/agent-runtime and `PI_DESKTOP_HOST_BIN` set to the candidate + Host binary; `fixed-tool-runtime.test.ts` and `fixed-tool-declarations.test.ts` + cover HTTP payload and policy contracts. Electron UI/outbox is outside this + fixture's scope. +- Separately authorized official Flash experiments compare several independent + on-demand/fixed sessions with the same synthetic catalog and call sequence. + Record cold requests, both activations, follow-up cache hits/misses and + cumulative input cost. Report the larger first request and possible short-chat + cost increase, alongside any longer-conversation benefit. Offline test success + alone is not evidence of provider cache behavior. diff --git a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md index 820a0d5b39..97ec5d4abc 100644 --- a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md +++ b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md @@ -978,15 +978,21 @@ sidecar 最多激活四个匹配项,并将名称写入 canonical 模式。具有本机延迟工具搜索的提供商可在该负载点接收定义;其他 提供商通常会收到活动定义。 -每个新用户提示前都会清除延迟激活集,再从有效上下文重建。成功的 -`ToolSearch` 结果读取 canonical `details.addedToolNames`;为兼容历史 -数据,也接受 `details.activated` 和顶层 `addedToolNames`。成功的延迟 -工具结果会贡献其工具名。仅恢复当前模式延迟目录中仍存在的名称;失败、 -中断、缺少结果的占位行以及助手/用户文本不会激活工具。工具注册表、主机 -权限路径、工具超时和工作区包含规则保持不变。`ToolSearch` 是 sidecar 的 -本地工具,不跨越主机 RPC 边界。激活标记保留在持久化工具结果中,因此 -只要证据仍在有效上下文,运行时重启或新提示都可以复用能力;证据被压缩 -或消失后仍需重新搜索。 +Deferred activation remains sticky within a live runtime. Restoration uses +successful activation evidence and the current catalog; old declarations do not +re-grant tools revoked from the live activation set. For official bound Flash, +full declarations and execution activation are independent: versioned +`tool_activation` sections carry the account/model/API/endpoint/catalog identity +and active names through restart and compaction. Only matching, valid state and +newer successful ToolSearch results restore activation; malformed or changed +epochs fail closed. Inactive declared tools are blocked before extension/Host +execution, and activation never bypasses mode or approval checks. The full +catalog is deterministic from the first request. More than 128 tools or an +insufficient context budget falls back to on-demand declarations with a +diagnostic, without truncation. Other bindings retain their existing projection. +Fixed declarations may increase total cost for short conversations. See the +English section 7.1 and the chronological-system-transcript ADR for the complete +contract. 对于用户可见的 HTML 可交付成果,默认系统提示要求代理 创建页面或创建第一个页面后激活 `BrowserPreview` 一次 diff --git a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md index 4e8caff873..c48a694986 100644 --- a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md @@ -9219,3 +9219,30 @@ the latest destination. These assertions measure work counts, not device FPS. - **验收:** 缓存路径、迁移和清理单测通过;Windows task-candidate 验证应覆盖更新源传输、安装器交接和文件系统行为,且不连接真实发布源。 - **里程碑:** M6+ - **状态:** 单测和源码契约覆盖(`update-cache.test.mjs`、`auto-update.test.mjs`);仍需 Windows 安装器/E2E 验证。 + +### E2E-FIXED-TOOL-DECLARATIONS: Stable Flash schemas with independent activation + +- Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, + and a child-process fetch boundary redirected to local HTTP/SSE. Credentials + are dummy values and synthetic tools have no external side effects. +- Prompt, search Alpha, execute Alpha, search Beta and execute Beta. HTTP payload + tests assert identical ordered `tools` and unchanged prior message prefixes. + Directly calling a declared but inactive tool must fail before the Host. +- Restart both processes, then compact and restart again: Alpha stays activated, + Beta remains inactive despite its declaration. Activate Beta, then remove it + from the catalog and verify it cannot execute. New schema/route epochs must + not restore old grants. Host rejection and Plan guards remain effective. +- Runtime contracts additionally cover legacy migration, interrupted activation, + malformed metadata, temporary prompt replacement, deterministic catalog order, + the 128-tool boundary and insufficient-context fallback. +- Automated: `node scripts/e2e-fixed-tool-declarations.mjs` with built + shared/host-runtime/agent-runtime and `PI_DESKTOP_HOST_BIN` set to the candidate + Host binary; `fixed-tool-runtime.test.ts` and `fixed-tool-declarations.test.ts` + cover HTTP payload and policy contracts. Electron UI/outbox is outside this + fixture's scope. +- Separately authorized official Flash experiments compare several independent + on-demand/fixed sessions with the same synthetic catalog and call sequence. + Record cold requests, both activations, follow-up cache hits/misses and + cumulative input cost. Report the larger first request and possible short-chat + cost increase, alongside any longer-conversation benefit. Offline test success + alone is not evidence of provider cache behavior. diff --git a/packages/agent-runtime/src/fixed-tool-declarations.test.ts b/packages/agent-runtime/src/fixed-tool-declarations.test.ts new file mode 100644 index 0000000000..8b54969b9a --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-declarations.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "vitest"; +import type { AgentTool } from "@earendil-works/pi-agent-core"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { Type, type Api, type Model } from "@earendil-works/pi-ai"; +import { toolDeclarationPolicy, toolActivationSection, restoredToolActivation, TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; +import { replaceSystemPrompt, systemTranscriptCheckpoint } from "./system-transcript.js"; + +const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; +const tool = (name: string): AgentTool => ({ name, label: name, description: "Fixture tool", parameters: Type.Object({}), + execute: async () => ({ content: [{ type: "text", text: "Done" }], details: {} }) }); +const deferred = new Set(["Alpha", "Beta"]); +const policy = (tools = [tool("Alpha"), tool("Beta")], overrides: Partial> = {}, prompt = "") => + toolDeclarationPolicy({ ...model, ...overrides } as Model, tools, deferred, prompt, "fixture-account"); + +describe("fixed tool declaration policy", () => { + it("has a deterministic declaration order and snapshot identity", () => { + const first = policy(); + const second = policy([tool("Beta"), tool("Alpha")]); + expect(second.key).toBe(first.key); + expect(second.tools?.map((tool) => tool.name)).toEqual(["Alpha", "Beta"]); + expect(policy([{ ...tool("Alpha"), parameters: Type.Object({ id: Type.String() }) }]).key).not.toBe(first.key); + }); + it.each([ + { id: "deepseek-pro" }, { api: "openai-responses" }, { baseUrl: "https://relay.invalid" }, + { baseUrl: "https://api.deepseek.com/v1" }, { compat: { supportsMidConvoSystemMessages: false } }, + ] as Partial>[])("does not enable unverified bindings: %j", (overrides) => { + expect(policy(undefined, overrides).tools).toBeUndefined(); + expect(policy(undefined, overrides).fallback).toBeUndefined(); + }); + it("falls back without truncating catalogs beyond the provider's function limit", () => { + expect(policy(Array.from({ length: 128 }, (_, index) => tool(`Tool${index}`))).tools).toHaveLength(128); + expect(policy(Array.from({ length: 129 }, (_, index) => tool(`Tool${index}`)))).toMatchObject({ fallback: "tool-count" }); + }); + it("leaves room for retained conversation and output", () => { + expect(policy([tool("Alpha")], { contextWindow: 32000, maxTokens: 4000 }, "x".repeat(80000))) + .toMatchObject({ fallback: "context-budget" }); + }); + it("restores activation independently of full declarations, including compacted and transient prompts", () => { + const current = policy(); + const messages = [{ role: "system" as const, content: "", timestamp: 1, + toolsAdded: current.tools, sections: { runtime: "Rules", [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set(["Alpha"])) } }]; + const changed = replaceSystemPrompt(messages, "Temporary nudge"); + expect(restoredToolActivation(changed, current.key)?.active).toEqual(["Alpha"]); + expect(restoredToolActivation([systemTranscriptCheckpoint(changed)!], current.key)?.active).toEqual(["Alpha"]); + expect(restoredToolActivation(messages, "different-snapshot")?.active).toEqual([]); + expect(restoredToolActivation([], current.key)).toBeUndefined(); + }); + it.each(["invalid JSON", '{"version":2}', null])("fails closed on invalid activation metadata: %s", (value) => { + expect(restoredToolActivation([{ role: "system", content: "", timestamp: 1, + sections: { [TOOL_ACTIVATION_SECTION]: value } }], policy().key)?.active).toEqual([]); + }); +}); diff --git a/packages/agent-runtime/src/fixed-tool-declarations.ts b/packages/agent-runtime/src/fixed-tool-declarations.ts new file mode 100644 index 0000000000..721f39663e --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-declarations.ts @@ -0,0 +1,69 @@ +import { createHash } from "node:crypto"; +import type { AgentMessage, AgentTool } from "@earendil-works/pi-agent-core"; +import { getCurrentSystemMessage, toToolDeclaration, Type, type Api, type Model } from "@earendil-works/pi-ai"; +import * as Value from "typebox/value"; +import { contextBudgetLimitsFor, automaticCompactionThresholdFor } from "./context-budget.js"; +import { estimateOutputCapInputTokens } from "./output-cap.js"; + +export const TOOL_ACTIVATION_SECTION = "tool_activation"; +const activationSchema = Type.Object({ + version: Type.Literal(1), snapshot: Type.String({ pattern: "^[a-f0-9]{64}$" }), + active: Type.Array(Type.String({ minLength: 1 }), { uniqueItems: true }), +}, { additionalProperties: false }); + +export type ToolDeclarationPolicy = { + key: string; + tools?: AgentTool[]; + fallback?: "tool-count" | "context-budget"; +}; + +/** The verified Flash route has chronological system updates, but no tool deltas. */ +export function toolDeclarationPolicy( + model: Model, tools: readonly AgentTool[], deferred: ReadonlySet, prompt: string, accountId: string, +): ToolDeclarationPolicy { + const ordered = [...tools].sort((a, b) => a.name < b.name ? -1 : a.name > b.name ? 1 : 0); + const key = createHash("sha256").update(JSON.stringify({ + account: accountId, model: model.id, api: model.api, endpoint: model.baseUrl.replace(/\/+$/, ""), + tools: ordered.map(toToolDeclaration), deferred: [...deferred].sort(), + })).digest("hex"); + if (model.id !== "deepseek-flash" || model.api !== "openai-completions" + || model.baseUrl.replace(/\/+$/, "") !== "https://api.deepseek.com" + || !model.compat || !("supportsMidConvoSystemMessages" in model.compat) + || model.compat.supportsMidConvoSystemMessages !== true) return { key }; + // DeepSeek Chat Completions permits at most 128 functions. Never truncate. + if (ordered.length > 128) return { key, fallback: "tool-count" }; + const budget = contextBudgetLimitsFor(model); + const tokens = estimateOutputCapInputTokens({ messages: [], systemPrompt: prompt, tools: ordered }, model); + // Leave the normal retained-tail budget available to the conversation. A + // fixed catalog must not make every fresh/compacted request overflow again. + if (tokens >= automaticCompactionThresholdFor(budget) - budget.keepRecentTokens) { + return { key, fallback: "context-budget" }; + } + return { key, tools: ordered }; +} + +export function toolActivationSection(key: string, active: ReadonlySet): string { + return JSON.stringify({ version: 1, snapshot: key, active: [...active].sort() }); +} + +/** A declaration is not activation. Malformed/new-version state fails closed. */ +export function restoredToolActivation(messages: readonly AgentMessage[], key: string): { active: string[]; replayFrom: number } | undefined { + const section = getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION]; + const closed = { active: [], replayFrom: messages.length }; + if (section == null) return messages.some((message) => message.role === "system" + && TOOL_ACTIVATION_SECTION in (message.sections ?? {})) ? closed : undefined; + try { + const parsed: unknown = JSON.parse(section); + if (!Value.Check(activationSchema, parsed) || parsed.snapshot !== key) return closed; + const lastState = messages.map((message) => message.role === "system" + && TOOL_ACTIVATION_SECTION in (message.sections ?? {})).lastIndexOf(true); + return { active: parsed.active, replayFrom: lastState + 1 }; + } catch { return closed; } +} + +/** Activation is appended after the result, never inserted into old instructions. */ +export function syncToolActivation(messages: AgentMessage[], section: string): AgentMessage[] { + if (getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION] === section) return messages; + const timestamp = messages.reduce((latest, message) => Math.max(latest, message.timestamp + 1), Date.now()); + return [...messages, { role: "system", content: "", timestamp, sections: { [TOOL_ACTIVATION_SECTION]: section } }]; +} diff --git a/packages/agent-runtime/src/fixed-tool-runtime.test.ts b/packages/agent-runtime/src/fixed-tool-runtime.test.ts new file mode 100644 index 0000000000..2c86736cea --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-runtime.test.ts @@ -0,0 +1,216 @@ +import { randomUUID } from "node:crypto"; +import { createServer } from "node:http"; +import { afterEach, describe, expect, it, vi } from "vitest"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { systemTranscriptCheckpoint } from "./system-transcript.js"; +import { readSystemMessage } from "./system-transcript-journal.js"; +import type { UiMessage } from "@pi-desktop/shared"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { DesktopAgentRuntime, type PluginToolDef, type RuntimeProviderConfig } from "./runtime.js"; + +function flashProvider(): RuntimeProviderConfig { + const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; + return { id: "flash-fixture", name: "Flash", modelId: model.id, baseUrl: model.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: modelConfigFromPi(model) }; +} +const pluginTools: PluginToolDef[] = ["plugin_alpha", "plugin_beta"].map((name) => ({ + name, description: `${name} synthetic probe`, parameters: { type: "object", properties: {}, required: [] }, risk: "low", +})); +type Payload = { tools: { function: { name: string } }[]; messages: { role: string; content?: unknown }[] }; +type Call = { name: string; args?: Record }; +async function wireFixture(calls: Call[]) { + const requests: Payload[] = []; + const fetch = globalThis.fetch; + const server = createServer(async (req, res) => { + const chunks: Buffer[] = []; + for await (const chunk of req) chunks.push(Buffer.from(chunk)); + requests.push(JSON.parse(Buffer.concat(chunks).toString())); + const call = calls.shift(); + const delta = call ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), + type: "function", function: { name: call.name, arguments: JSON.stringify(call.args ?? {}) } }] } + : { role: "assistant", content: "Done." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + res.end(`data: ${JSON.stringify({ choices: [{ index: 0, delta, finish_reason: call ? "tool_calls" : "stop" }] })}\n\ndata: [DONE]\n\n`); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const address = server.address(); + if (!address || typeof address === "string") throw new Error("Missing HTTP address"); + vi.stubGlobal("fetch", ((_url, init) => fetch(`http://127.0.0.1:${address.port}`, init)) satisfies typeof fetch); + return { requests, close: async () => { + vi.unstubAllGlobals(); + await new Promise((resolve, reject) => server.close((error) => error ? reject(error) : resolve())); + } }; +} +function runtimeFixture(history: UiMessage[] = [], tools = pluginTools, provider = flashProvider(), denied = false) { + const rows = structuredClone(history); + const executed: string[] = []; + const errors: unknown[] = []; + const runtime = new DesktopAgentRuntime({ + sessionId: "fixed-tools", mode: "agent", provider, thinkingLevel: "off", history: rows, pluginTools: tools, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") rows.push((params as { message: UiMessage }).message); + else if (method === "tools.execute") { + executed.push((params as { toolName: string }).toolName); + return denied ? { ok: false, denied: true, content: "Permission denied" } as T + : { ok: true, content: "Synthetic success" } as T; + } else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ id: event.toolCallId, role: "tool", content: "", + toolCallId: event.toolCallId, toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString() }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = event.isError ? "error" : "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + return { runtime, rows, executed, errors, prompt: async (id = "user-1") => { + rows.push({ id, role: "user", content: "Run the synthetic probes", createdAt: new Date().toISOString() }); + await runtime.prompt("Run the synthetic probes", id, `turn-${id}`); + } }; +} +afterEach(() => vi.unstubAllGlobals()); +describe("fixed Flash declarations through runtime and HTTP/SSE", () => { + it("rejects a declared but inactive tool without invoking Host", async () => { + const wire = await wireFixture([{ name: "plugin_beta" }]); + const f = runtimeFixture(); + try { + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual([]); + expect(f.rows.find((row) => row.toolName === "plugin_beta")).toMatchObject({ isError: true }); + expect(JSON.stringify(wire.requests[1].messages)).toContain("Call ToolSearch to activate plugin_beta"); + expect(wire.requests[1].tools).toEqual(wire.requests[0].tools); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it.each([false, true])("restores only activated tools after restart (checkpoint=%s)", async (checkpoint) => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + let declared: Payload["tools"]; + try { + await f.prompt(); + expect(f.errors).toEqual([]); + declared = wire.requests.at(-1)!.tools; + history = structuredClone(f.rows); + if (checkpoint) { + const system = systemTranscriptCheckpoint(history.filter((row) => row.modelSystem) + .map((row) => readSystemMessage(row.modelSystem!.messageJson)))!; + history = [{ id: "checkpoint", role: "system", content: "", createdAt: new Date().toISOString(), + modelSystem: { version: 1, messageJson: JSON.stringify(system) } }]; + } + } finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("restored-user"); + expect(restored.errors).toEqual([]); + expect(restored.executed).toEqual(["plugin_alpha"]); + expect(resumed.requests[0].tools).toEqual(declared!); + expect([...restored.rows].reverse().find((row) => row.toolName === "plugin_beta")).toMatchObject({ isError: true }); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it.each(["schema", "removed", "route"])("does not revive activation after a %s change", async (change) => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + try { await f.prompt(); history = structuredClone(f.rows); } + finally { await f.runtime.dispose(); await wire.close(); } + const tools = change === "removed" ? pluginTools.slice(1) : change === "schema" + ? [{ ...pluginTools[0], description: "Changed schema epoch" }, pluginTools[1]] : pluginTools; + const provider = change === "route" ? { ...flashProvider(), baseUrl: "https://relay.invalid/v1" } : flashProvider(); + const resumed = await wireFixture([{ name: "plugin_alpha" }]); + const restored = runtimeFixture(history!, tools, provider); + try { + await restored.prompt("changed-user"); + expect(restored.errors).toEqual([]); + expect(restored.executed).toEqual([]); + if (change === "removed") expect(resumed.requests[0].tools.map((tool) => tool.function.name)).not.toContain("plugin_alpha"); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("still requires Host approval after ToolSearch activation", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }]); + const f = runtimeFixture([], pluginTools, flashProvider(), true); + try { + await f.prompt(); + expect(f.executed).toEqual(["plugin_alpha"]); + expect(f.rows.find((row) => row.toolName === "plugin_alpha")).toMatchObject({ isError: true }); + expect(JSON.stringify(wire.requests.at(-1)?.messages)).toContain("Permission denied"); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it("replays a successful activation if stopped before its next declaration checkpoint", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + try { + await f.prompt(); + const first = f.rows.find((row) => row.modelSystem)!; + history = structuredClone(f.rows.filter((row) => !row.modelSystem || row === first)); + } finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("resumed-user"); + expect(restored.executed).toEqual(["plugin_alpha"]); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("migrates legacy on-demand history without granting the rest of the catalog", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture([], pluginTools, { ...flashProvider(), baseUrl: "https://relay.invalid" }); + let history: UiMessage[]; + try { await f.prompt(); history = structuredClone(f.rows); } + finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("migrated-user"); + expect(restored.executed).toEqual(["plugin_alpha"]); + expect(resumed.requests[0].tools.map((tool) => tool.function.name)).toEqual(expect.arrayContaining(["plugin_alpha", "plugin_beta"])); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("retains the Plan execution guard for predeclared tools", async () => { + const wire = await wireFixture([{ name: "Write", args: { path: "fixture", content: "denied" } }]); + const f = runtimeFixture(); + try { + f.runtime.setMode("plan"); + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual([]); + expect(f.rows.find((row) => row.toolName === "Write")).toMatchObject({ isError: true }); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it("keeps the complete request prefix while searching and executing two different tools", async () => { + const wire = await wireFixture([ + { name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }, + { name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta" }, + ]); + const f = runtimeFixture(); + try { + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual(["plugin_alpha", "plugin_beta"]); + expect(wire.requests).toHaveLength(5); + expect(wire.requests[0].tools.map((tool) => tool.function.name)).toEqual(expect.arrayContaining(["plugin_alpha", "plugin_beta"])); + for (let index = 1; index < wire.requests.length; index++) { + expect(wire.requests[index].tools).toEqual(wire.requests[0].tools); + const previous = wire.requests[index - 1].messages; + expect(wire.requests[index].messages.slice(0, previous.length)).toEqual(previous); + } + } finally { await f.runtime.dispose(); await wire.close(); } + }); +}); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index aa13794048..fbb689fd4a 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -1,3 +1,4 @@ +import { TOOL_ACTIVATION_SECTION, toolDeclarationPolicy, toolActivationSection, restoredToolActivation, syncToolActivation, type ToolDeclarationPolicy } from "./fixed-tool-declarations.js"; import { orderSystemRows, SystemTranscriptJournal } from "./system-transcript-journal.js"; import { accountModelStream } from "./request-usage.js"; import { modeToolDenial, retainModeToolDeclaration, withModeExecutionGuard } from "./mode-tool-access.js"; @@ -1691,8 +1692,11 @@ export class DesktopAgentRuntime { private toolCatalog = new Map(); /** Tools intentionally omitted from the initial provider request. */ private deferredToolNames = new Set(); - /** Deferred tools loaded for the current user prompt. */ + /** Deferred tools activated for this runtime and declaration epoch. */ private activeDeferredToolNames = new Set(); + private declarationPolicy?: ToolDeclarationPolicy; + private trackToolActivation = false; + private activationHydrated = false; private scratchDir?: string; private projectPath?: string; private commandShell: CommandShellOption; @@ -1880,7 +1884,6 @@ export class DesktopAgentRuntime { this.rebuildToolCatalog(); const model = buildProviderModel(this.provider); this.model = model; - const tools = this.activeTools(); const models = createProviderModels(this.provider, model); this.models = models; const runtimeApiKey = providerRequestKey(this.provider); @@ -1934,6 +1937,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ).trim(), ...defaultSystemPromptParts.slice(1), ].join("\n\n"); + this.refreshToolDeclarationPolicy(); + this.restoreDeferredToolsFromContext(); + const tools = this.declaredTools(); this.agent = new Agent({ streamFn: (m, context, options) => { this.setAgentActivity({ phase: "waiting-model", since: Date.now() }); @@ -2204,7 +2210,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } private setAgentTools(tools: AgentTool[]): void { - this.agent.state.tools = tools; + this.agent.state.tools = this.declarationPolicy?.tools ?? tools; + if (this.trackToolActivation && this.declarationPolicy) { + const section = toolActivationSection(this.declarationPolicy.key, this.activeDeferredToolNames); + this.composedSections = { ...this.composedSections, [TOOL_ACTIVATION_SECTION]: section }; + this.composedSystemPrompt = Object.values(this.composedSections).filter(Boolean).join("\n\n"); + this.agent.state.messages = syncToolActivation(this.agent.state.messages, section); + } } private agentSystemPromptContent(): string { @@ -2225,6 +2237,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the : ""; this.composedSections = { runtime: this.baseSystemPrompt, + ...(this.trackToolActivation && this.declarationPolicy ? { + [TOOL_ACTIVATION_SECTION]: toolActivationSection(this.declarationPolicy.key, this.activeDeferredToolNames), + } : {}), ...pluginSkillsPromptSections(this.pluginSkills), context: composeModeSystemPrompt(this.mode, [ ...(this.customSystemPrompt?.append ? [this.customSystemPrompt.append] : []), @@ -2385,6 +2400,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private async beforeToolCall( context: BeforeToolCallContext, ): Promise { + if (this.deferredToolNames.has(context.toolCall.name) && !this.activeDeferredToolNames.has(context.toolCall.name)) { + return { block: true, reason: `Call ${TOOL_SEARCH_NAME} to activate ${context.toolCall.name} before using it. A tool declaration does not grant execution permission.` }; + } const toolCalls = (context.assistantMessage.content as Array<{ type?: string }>).filter( (block) => block.type === "toolCall", ); @@ -3612,9 +3630,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } /** - * Build the complete registry once, then expose only the core subset to the - * first provider request. This mirrors pi's active-tool model while keeping - * the host tool implementation and permission path unchanged. + * Build the complete registry before selecting fixed or on-demand + * declarations. Execution activation and Host permissions remain separate + * from the provider-visible schemas. */ private rebuildToolCatalog(): void { const catalog = new Map(); @@ -3659,6 +3677,25 @@ Do not invent objections or turn speculative risks into blockers. Stop when the }), ); } + if (this.model) this.refreshToolDeclarationPolicy(); + } + + private refreshToolDeclarationPolicy(): void { + const previous = this.declarationPolicy; + const policy = toolDeclarationPolicy(this.model, [...this.toolCatalog.values()], this.deferredToolNames, this.composeSystemPrompt(), this.provider.id); + if (previous && previous.key !== policy.key && this.trackToolActivation) { + this.activeDeferredToolNames.clear(); + this.activationHydrated = true; + } + this.declarationPolicy = policy; + this.trackToolActivation ||= Boolean(policy.tools || policy.fallback); + if (policy.fallback && (previous?.key !== policy.key || previous.fallback !== policy.fallback)) { + process.stderr.write(`[agent-runtime] fixed tool declarations unavailable (${policy.fallback}); using on-demand declarations; ToolSearch cache stability is not guaranteed.\n`); + } + } + + private declaredTools(): AgentTool[] { + return this.declarationPolicy?.tools ?? this.activeTools(); } private isPlanSafePluginTool(name: string): boolean { @@ -3747,7 +3784,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } return [ "# On-demand tools", - `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using one that is not in the current tool list.`, + `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using an on-demand tool that has not been activated. A visible schema is not activation; the tool_activation section, when present, records active names.`, ...lines, ].join("\n"); } @@ -3777,7 +3814,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the name: TOOL_SEARCH_NAME, label: "Tool Search", description: - "Find and activate an on-demand tool by exact name or capability. Use this before calling any tool listed under On-demand tools that is not already in the current tool list.", + "Find and activate an on-demand tool by exact name or capability. Use this before calling an inactive tool listed under On-demand tools, even when its schema is already visible. Activation does not bypass approval or mode restrictions.", parameters: Type.Object({ query: Type.String({ description: @@ -5340,9 +5377,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the */ private restoreDeferredToolsFromContext(): void { if (this.deferredToolNames.size === 0) return; + if (this.trackToolActivation && this.activationHydrated) return; const { messages } = this.liveSessionContext(); - const lastSystem = messages.map((message) => message.role).lastIndexOf("system"); - if (lastSystem >= 0) { + const restored = this.declarationPolicy && restoredToolActivation(messages, this.declarationPolicy.key); + if (restored !== undefined) { + this.trackToolActivation = true; + for (const name of restored.active) { + if (this.deferredToolNames.has(name)) this.activeDeferredToolNames.add(name); + } + } + this.activationHydrated = true; + const lastSystem = restored ? restored.replayFrom - 1 : messages.map((message) => message.role).lastIndexOf("system"); + if (!restored && lastSystem >= 0) { for (const tool of getCurrentTools(messages)) { if (this.deferredToolNames.has(tool.name)) this.activeDeferredToolNames.add(tool.name); } @@ -5350,6 +5396,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // Legacy histories lack declarations. For current histories, only results // after the last declaration can represent an activation not yet declared. for (const message of messages.slice(lastSystem + 1)) { + if (restored && (message.role !== "toolResult" || message.toolName !== TOOL_SEARCH_NAME)) continue; if (message.role !== "toolResult" || message.isError) continue; if (isMissingToolResultPlaceholder(message.content)) continue; const names = @@ -6127,7 +6174,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ...(typeof this.agent.state.systemPrompt === "string" ? { systemPrompt: this.agent.state.systemPrompt } : {}), - tools: this.activeTools(), + tools: this.declaredTools(), }, this.model, ); @@ -6300,7 +6347,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private rebuiltAgentContext(): AgentContext { const messages = this.liveSessionContext().messages; - const tools = this.activeTools(); + const tools = this.declaredTools(); this.setAgentMessages(messages); this.setAgentTools(tools); return { diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index f3e3470337..9ba1167b22 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -1,3 +1,4 @@ +import { TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; import type { AgentMessage } from "@earendil-works/pi-agent-core"; import { contentText, @@ -35,7 +36,7 @@ function currentSystemMessage(messages: readonly AgentMessage[]): SystemMessage export const CONTEXT_BUDGET_SECTION = "context_budget"; function desktopSectionNames(sections: Record): string[] { - return ["runtime", "skills", ...Object.keys(sections) + return ["runtime", TOOL_ACTIVATION_SECTION, "skills", ...Object.keys(sections) .filter((name) => name.startsWith(SKILL_SECTION_PREFIX)) .sort((a, b) => a.localeCompare(b)), "context"]; } @@ -83,7 +84,8 @@ export function initialSystemTranscript( export function replaceSystemPrompt(messages: AgentMessage[], prompt: string): AgentMessage[] { if (prompt === getCurrentSystemPrompt(messages) || prompt === systemPromptContent(messages)) return messages; - return syncSystemSections(messages, { runtime: prompt }); + const activation = getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION]; + return syncSystemSections(messages, { runtime: prompt, ...(activation ? { [TOOL_ACTIVATION_SECTION]: activation } : {}) }); } export function rebuildSystemTranscript( diff --git a/scripts/e2e-fixed-tool-declarations.mjs b/scripts/e2e-fixed-tool-declarations.mjs new file mode 100644 index 0000000000..c44d32ef58 --- /dev/null +++ b/scripts/e2e-fixed-tool-declarations.mjs @@ -0,0 +1,170 @@ +#!/usr/bin/env node +/** Production sidecar transport + isolated Host + local SSE provider. + * Message persistence is harness-owned; Electron's UI/outbox is not exercised. + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +import { DEEPSEEK_MODELS } from "../packages/agent-runtime/node_modules/@earendil-works/pi-ai/dist/providers/deepseek.models.js"; +import { modelConfigFromPi } from "../packages/agent-runtime/dist/model-capabilities.js"; +import { PROTOCOL_VERSION } from "../packages/shared/dist/protocol.js"; +import { withScenario } from "./e2e/fixture.mjs"; +import { resolveHostBinary } from "./e2e/host.mjs"; +import { createSession } from "./e2e/session.mjs"; + +const root = await mkdtemp(join(tmpdir(), "pi-system-state-e2e-")); +const requests = []; +const responses = []; +const server = createServer(async (req, res) => { + let body = ""; + for await (const chunk of req) body += chunk; + requests.push(JSON.parse(body)); + const tool = responses.shift(); + const delta = tool ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), type: "function", + function: { name: tool.name, arguments: JSON.stringify(tool.args) } }] } + : { role: "assistant", content: "The fixture completed." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + for (const [value, finish] of [[delta, null], [{}, tool ? "tool_calls" : "stop"]]) { + res.write(`data: ${JSON.stringify({ id: "fixture", model: "fixture", choices: [{ index: 0, delta: value, finish_reason: finish }] })}\n\n`); + } + res.end("data: [DONE]\n\n"); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; +const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); +const provider = { id: "fixture", name: "Flash fixture", modelId: model.id, baseUrl: model.baseUrl, + modelConfig: modelConfigFromPi(model), apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; +const fetchHook = join(root, "fixture-fetch.mjs"); +await writeFile(fetchHook, `const fetch = globalThis.fetch; globalThis.fetch = (url, init) => { + if (String(url) !== "https://api.deepseek.com/chat/completions") throw new Error("Unexpected fixture endpoint"); + return fetch(${JSON.stringify(baseUrl)}, init); +};`); +let pluginTools = ["plugin_alpha", "plugin_beta"].map((name) => ({ name, + description: "Synthetic read-only marker", parameters: { type: "object", properties: {}, required: [] }, risk: "low" })); +try { + await withScenario("E2E-fixed-tool-declarations", async ({ host, workspace }) => { + const session = await createSession(host, workspace, "System state fixture"); + let writes = Promise.resolve(); + let writeError; + let stderr = ""; + const turns = new Map(); + const executedTools = []; + const toolRows = new Map(); + const persist = (message, turnId) => { + writes = writes.then(() => host.call("session.appendMessage", { sessionId: session.id, turnId, message })) + .catch((error) => { writeError ??= error; }); + }; + const startSidecar = () => { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, + args: [fileURLToPath(new URL("../packages/agent-runtime/dist/sidecar.js", import.meta.url))], + cwd: workspace, + env: { ...process.env, NODE_OPTIONS: `--import=${fetchHook}`, PI_DESKTOP_PLAN_UI_PROBE: "1", PI_DESKTOP_COMPACTION_STRATEGY: "fresh_window" }, + }, + onStderr: (chunk) => { stderr += chunk; }, + }); + sidecar.setHost({ call: host.call.bind(host), onNotification: () => () => {}, onExit: () => () => {} }); + for (const name of ["plugin_alpha", "plugin_beta"]) sidecar.setLocalTool(name, async () => { + executedTools.push(name); + return { ok: true, content: `${name} synthetic result` }; + }); + sidecar.onNotification((method, envelope) => { + if (method !== "agent.event") return; + const { event, turnId } = envelope; + const turn = turns.get(turnId); + if (event.type === "error") turn?.errors.push(event.error); + if (event.type === "message_end") persist(event.message, turnId); + if (event.type === "tool_start") toolRows.set(event.toolCallId, { + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + turn?.toolResults.push(event); + persist({ ...toolRows.get(event.toolCallId), toolResult: event.result, isError: event.isError, + toolStatus: event.isError ? "error" : "success" }, turnId); + } + if (event.type === "agent_end") turn?.resolve(); + }); + return sidecar; + }; + let sidecar = startSidecar(); + const params = () => ({ sessionId: session.id, mode: "agent", provider, thinkingLevel: "off", pluginTools, + projectPath: workspace, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + }); + const prompt = async (id, expectedErrors = 0) => { + const { turnId } = await host.call("session.beginTurn", { sessionId: session.id }); + await host.call("session.appendMessage", { sessionId: session.id, turnId, + message: { id, role: "user", content: "Continue", createdAt: new Date().toISOString() } }); + let timer; + const turn = { errors: [], toolResults: [] }; + const ended = new Promise((resolve, reject) => { + turn.resolve = resolve; + timer = setTimeout(() => reject(new Error(`Turn timed out: ${JSON.stringify(turn.errors)}\n${stderr}`)), 20_000); + }); + turns.set(turnId, turn); + try { + await sidecar.call("agent.prompt", { ...params(), turnId, userMessageId: id, content: "Continue" }); + await ended; + await writes; + if (writeError) throw writeError; + assert.deepEqual(turn.errors, [], JSON.stringify(turn.errors)); + assert.equal(turn.toolResults.filter((event) => event.isError).length, expectedErrors, JSON.stringify(turn.toolResults)); + await host.call("session.endTurn", { turnId, status: "completed", createNotification: false }); + return turn; + } finally { clearTimeout(timer); turns.delete(turnId); } + }; + const detail = async () => (await host.call("session.get", { id: session.id })).session; + try { + responses.push({ name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha", args: {} }); + await prompt("user-1"); + const declared = requests[0].tools; + assert(declared.some((tool) => tool.function.name === "plugin_beta")); + for (let index = 1; index < requests.length; index++) { + assert.deepEqual(requests[index].tools, declared); + assert.deepEqual(requests[index].messages.slice(0, requests[index - 1].messages.length), requests[index - 1].messages); + } + assert.deepEqual(executedTools, ["plugin_alpha"]); + const states = (await detail()).messages.filter((row) => row.modelSystem); + assert(states.length >= 2); + console.log("PASS: fixed declarations survive ToolSearch through production sidecar and HTTP/SSE"); + await sidecar.dispose(); + await host.stop(); await host.start(PROTOCOL_VERSION); + assert.deepEqual((await detail()).messages.filter((row) => row.modelSystem), states); + sidecar = startSidecar(); + responses.push({ name: "plugin_alpha", args: {} }, { name: "plugin_beta", args: {} }); + await prompt("user-2", 1); + assert.deepEqual(executedTools, ["plugin_alpha", "plugin_alpha"]); + assert.deepEqual(requests.at(-1).tools, declared); + console.log("PASS: Host and sidecar restart restore Alpha activation without activating declared Beta"); + await sidecar.call("agent.compact", params()); + const checkpoint = JSON.parse((await detail()).compaction.details.systemMessageJson); + assert.deepEqual(JSON.parse(checkpoint.sections.tool_activation).active, ["plugin_alpha"]); + await sidecar.dispose(); sidecar = startSidecar(); + responses.push({ name: "plugin_alpha", args: {} }, { name: "plugin_beta", args: {} }); + await prompt("user-3", 1); + assert.deepEqual(executedTools, ["plugin_alpha", "plugin_alpha", "plugin_alpha"]); + assert.deepEqual(requests.at(-1).tools, declared); + console.log("PASS: compaction and process restart preserve declarations and activation separately"); + responses.push({ name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta", args: {} }); + await prompt("user-4"); + assert.equal(executedTools.at(-1), "plugin_beta"); + assert.deepEqual(requests.at(-1).tools, declared); + pluginTools = pluginTools.slice(0, 1); + responses.push({ name: "plugin_beta", args: {} }); + await prompt("user-5", 1); + assert.equal(executedTools.length, 4); + assert(!requests.at(-1).tools.some((tool) => tool.function.name === "plugin_beta")); + console.log("PASS: changed catalog revokes Beta execution and creates a new declaration epoch"); + } finally { await sidecar.dispose(); await writes; } + }, resolveHostBinary(), root, PROTOCOL_VERSION); +} finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + await rm(root, { recursive: true, force: true }); +} From 16c518953beb47048bd903d61f9003094867bd0b Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 23:09:10 +0800 Subject: [PATCH 6/8] fix(runtime): stabilize ToolSearch across provider transports Select fixed declarations by transport capabilities instead of a Flash allowlist, retaining native anchored additions where Pi supports them. Keep execution activation private to Desktop so folding adapters do not rewrite their leading instructions after each search. Exercise all Desktop-selectable adapters through the runtime tool loop, with compatible-route HTTP and process coverage and independent activation restoration and permission checks. --- docs/adr/chronological-system-transcript.md | 36 ++++--- docs/spec/03-runtime/02-agent-runtime.md | 29 ++++-- docs/spec/06-delivery/04-e2e-test-plan.md | 17 +++- .../zh-CN/spec/03-runtime/02-agent-runtime.md | 9 +- .../spec/06-delivery/04-e2e-test-plan.md | 17 +++- .../src/fixed-tool-declarations.test.ts | 25 ++++- .../src/fixed-tool-declarations.ts | 27 +++-- .../src/fixed-tool-providers.test.ts | 98 +++++++++++++++++++ .../src/fixed-tool-runtime.test.ts | 91 +++-------------- .../agent-runtime/src/pi-runtime-messages.ts | 15 +++ packages/agent-runtime/src/runtime.test.ts | 66 ++++++------- packages/agent-runtime/src/runtime.ts | 2 +- .../src/system-transcript-runtime.test.ts | 4 + .../src/test-helpers/fixed-tool-fixture.ts | 77 +++++++++++++++ scripts/e2e-fixed-tool-declarations.mjs | 9 +- 15 files changed, 372 insertions(+), 150 deletions(-) create mode 100644 packages/agent-runtime/src/fixed-tool-providers.test.ts create mode 100644 packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md index 659ec00a28..1177064d94 100644 --- a/docs/adr/chronological-system-transcript.md +++ b/docs/adr/chronological-system-transcript.md @@ -53,20 +53,34 @@ Desktop allowlist; remove the hunk when an upgraded Pi catalog carries it. The model/API/endpoint binding check still excludes aliases and relays. Native tool-addition and tool-change flags are not enabled by this correction. -## Fixed declarations for the verified Flash route - -For the exact official `deepseek-flash` / `openai-completions` binding with -verified chronological system support, declare the complete current tool catalog -in deterministic name order from the first request. ToolSearch changes execution -activation only. This is a Desktop declaration policy, not an additional Pi -transport capability or an endpoint switch. Other bindings keep on-demand -schema publication; native tool-state flags alone do not prove cache stability. +## Stable declarations across provider transports + +Choose the declaration strategy by the bound Pi transport capabilities, not a +model-name allowlist. Responses (OpenAI/Codex) with verified system support +and `additional_tools` or client tool search, Chat Completions with verified +system/tool additions, and Pi's transcript transport retain native chronological +additions. Other transports declare the complete current catalog in deterministic +name order from the first request. In particular Anthropic's native transition +blocks still grow its request-level schemas, so they use fixed declarations. +This policy also covers compatible relays without enabling unsupported native +message roles or switching their configured API. + +Desktop currently exposes Chat Completions, Responses, Codex Responses, +Anthropic Messages, Gemini and Pi Messages bindings. This change does not add +new selectable Azure, Vertex, Bedrock or Mistral native bindings; models offered +through existing compatible endpoints follow that endpoint's adapter. Keep activation separate from declarations in a versioned `tool_activation` system section. The section records active deferred names and a SHA-256 identity of the account, model, API, endpoint, declarations and deferred-name set. Updates append after tool results and persist through the existing Host journal and -compaction checkpoint. Restoration never interprets the complete declaration +compaction checkpoint. Before provider conversion, omit this Desktop-only +activation section and any resulting empty metadata-only message. Preserve all +other instruction sections, content and tool deltas. Providers that fold system +messages must not rewrite their leading instructions just because execution +activation changed. ToolSearch results tell the model which tools were activated; +uncertain models may search again. Canonical persisted history is not mutated. +Restoration never interprets the complete declaration snapshot as permission to execute every tool. Successful ToolSearch results newer than the saved activation section recover an interrupted activation. Invalid, unknown-version or mismatched state grants no activation. Legacy @@ -79,8 +93,8 @@ removed tools cannot be invoked. Temporary prompt replacement must preserve the activation metadata. Current runtime activation remains authoritative between prompts; declarations do not re-grant revoked activation. -DeepSeek limits a request to 128 functions. If the full catalog exceeds that -limit, or its estimated prompt/schema cost leaves less than the ordinary +Use 128 functions as a conservative shared fixed-catalog ceiling, including +DeepSeek Chat Completions' limit. If the full catalog exceeds that ceiling, or its estimated prompt/schema cost leaves less than the ordinary retained-tail budget below the automatic compaction threshold, use the existing on-demand path and log the fallback reason. Never truncate a catalog. Context estimation charges the full declared catalog while fixed declarations are active. diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index ccc989a632..85bb9845b8 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -1407,10 +1407,9 @@ grammar and validated against another fails every call. ### 7.1 Active tool context and on-demand loading (D185, ADR 0048) -The sidecar builds one complete tool registry. By default, each provider request -declares the mode's core set plus activated deferred tools. The verified Flash -binding uses the fixed-declaration policy below, while preserving the same -execution activation rules: +The sidecar builds one complete tool registry. Native anchored-addition routes +declare core tools and add activated deferred tools in place. Other routes use +the fixed-declaration policy below. Both preserve these execution activation rules: - Agent: `Read`, `Bash`, `Edit`, and `Write` (matching pi's coding-agent core) - Agent: `Skill` whenever the skill catalog is non-empty (D404, ADR 0230) — the @@ -1449,9 +1448,13 @@ results from deferred tools also restore their names. Failed results, missing-result placeholders and assistant/user prose never activate tools. Only names in the current mode's deferred catalog are eligible. -For the exact official `deepseek-flash` Chat Completions binding with verified -mid-conversation system support, the runtime instead declares the complete -catalog in deterministic name order on the first request. ToolSearch changes +Select by the bound transport, not a model name. Responses with verified system +support plus `supportsAdditionalTools` or `supportsToolSearch`, Chat Completions +with verified system/tool additions, and Pi Messages retain native additions. +Other bindings, including Anthropic Messages, Gemini, ordinary Chat Completions, +older Responses/Codex models and compatible relays, declare the complete catalog +in deterministic name order on the first request. Anthropic's native tool-change +blocks still grow request-level schemas, so do not exempt them. ToolSearch changes activation without changing the declared schemas. A visible schema does not permit execution: inactive deferred calls are rejected before extension hooks and the Host; activated calls still require the existing mode and Host checks. @@ -1461,7 +1464,12 @@ Fixed declarations persist separately from activation. A version-1 `tool_activation` section records active names and a fingerprint of the account, model, API, endpoint, schema catalog and deferred set. Activation changes append at the continuation boundary, and the existing system journal/checkpoint saves -both declarations and activation. Restore only validated activation for a +both declarations and activation. Strip the activation section only from the +provider projection, dropping metadata-only empty messages but preserving all +other sections/content/tool deltas. This prevents folding APIs from moving an +activation change into the leading prompt; persisted state remains complete. +Model guidance uses successful ToolSearch results, not private activation JSON. +Restore only validated activation for a matching fingerprint, plus successful ToolSearch results newer than that state; never activate tools merely because the full snapshot declared them. Malformed, unknown-version and mismatched activation state fail closed. A catalog/schema, @@ -1471,8 +1479,9 @@ activation. Removal immediately removes the tool from executable registration. If the full catalog exceeds 128 functions or its prompt/schema estimate cannot leave the normal retained-tail budget below the compaction threshold, retain on-demand declarations and emit a diagnostic explaining that ToolSearch cache -stability is not guaranteed. Do not truncate tools. Other models and unverified -routes retain the existing Pi projection. First-request schema overhead increases; +stability is not guaranteed. The 128-function ceiling is conservative across +fixed-catalog APIs. Do not truncate tools or opt unknown endpoints into native +capabilities. First-request schema overhead increases; cache stability does not imply that short conversations become cheaper. For user-visible HTML deliverables, the default system prompt asks the agent to diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index e7e8d853a1..93a5b7dac0 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -16400,7 +16400,7 @@ renderer's durable transcript reads. No real model or provider is contacted. regressions also cover legacy identities and genuine account/model changes. Cache percentages are observations, not deterministic pass thresholds. -### E2E-FIXED-TOOL-DECLARATIONS: Stable Flash schemas with independent activation +### E2E-FIXED-TOOL-DECLARATIONS: Stable schemas across transports with independent activation - Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, and a child-process fetch boundary redirected to local HTTP/SSE. Credentials @@ -16432,3 +16432,18 @@ renderer's durable transcript reads. No real model or provider is contacted. failed assistant before compaction, and reuse one visible assistant message through successful recovery. Terminal failure and Stop retain their existing closure behavior. Covered by the parameterized runtime overflow user-path test. + +- Cross-provider acceptance: `fixed-tool-providers.test.ts` enters the real runtime + prompt/ToolSearch/execution path and captures each Desktop-selectable Pi + adapter's actual serialized payload at `onPayload`, before network dispatch. + Cover Chat Completions, Anthropic (native system on/off), Responses and Codex + (fallback, additional tools, client tool search), Gemini and Pi Messages. + Search A, execute A, search B, execute B, then finish: all five requests retain + their top-level schema state and prior semantic message prefix. Native routes + retain deferred schema additions. Activation JSON never reaches the provider. + This proves request construction, not server cache hits or paid API acceptance. +- The local HTTP/SSE fixture additionally covers official Flash, unflagged Chat + Completions and a compatible relay. Canonical activation restoration, denial + before Host execution, mode/account/catalog invalidation and compaction remain + required. Fixed declarations increase first-request size; oversized catalogs + explicitly fall back without a cache-stability guarantee. diff --git a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md index 63fdcd288c..d27868f8e6 100644 --- a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md +++ b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md @@ -983,7 +983,7 @@ sidecar 最多激活四个匹配项,并将名称写入 canonical Deferred activation remains sticky within a live runtime. Restoration uses successful activation evidence and the current catalog; old declarations do not -re-grant tools revoked from the live activation set. For official bound Flash, +re-grant tools revoked from the live activation set. For routes without verified native anchored tool additions, full declarations and execution activation are independent: versioned `tool_activation` sections carry the account/model/API/endpoint/catalog identity and active names through restart and compaction. Only matching, valid state and @@ -992,7 +992,12 @@ epochs fail closed. Inactive declared tools are blocked before extension/Host execution, and activation never bypasses mode or approval checks. The full catalog is deterministic from the first request. More than 128 tools or an insufficient context budget falls back to on-demand declarations with a -diagnostic, without truncation. Other bindings retain their existing projection. +diagnostic, without truncation. Responses and Chat Completions bindings with verified anchored additions, and +Pi Messages, retain native incremental publication. Anthropic native tool changes +still grow request schemas and therefore use fixed declarations. Strip only the +private activation section from provider projection, preserving canonical state +and all other instructions. This also stabilizes ToolSearch on folding APIs and +compatible relays without enabling new transport capabilities. Fixed declarations may increase total cost for short conversations. See the English section 7.1 and the chronological-system-transcript ADR for the complete contract. diff --git a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md index 056d9d57f9..67a003f7d2 100644 --- a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md @@ -9260,7 +9260,7 @@ the latest destination. These assertions measure work counts, not device FPS. - **里程碑:** M6+ - **状态:** 单测和源码契约覆盖(`update-cache.test.mjs`、`auto-update.test.mjs`);仍需 Windows 安装器/E2E 验证。 -### E2E-FIXED-TOOL-DECLARATIONS: Stable Flash schemas with independent activation +### E2E-FIXED-TOOL-DECLARATIONS: Stable schemas across transports with independent activation - Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, and a child-process fetch boundary redirected to local HTTP/SSE. Credentials @@ -9292,3 +9292,18 @@ the latest destination. These assertions measure work counts, not device FPS. failed assistant before compaction, and reuse one visible assistant message through successful recovery. Terminal failure and Stop retain their existing closure behavior. Covered by the parameterized runtime overflow user-path test. + +- Cross-provider acceptance: `fixed-tool-providers.test.ts` enters the real runtime + prompt/ToolSearch/execution path and captures each Desktop-selectable Pi + adapter's actual serialized payload at `onPayload`, before network dispatch. + Cover Chat Completions, Anthropic (native system on/off), Responses and Codex + (fallback, additional tools, client tool search), Gemini and Pi Messages. + Search A, execute A, search B, execute B, then finish: all five requests retain + their top-level schema state and prior semantic message prefix. Native routes + retain deferred schema additions. Activation JSON never reaches the provider. + This proves request construction, not server cache hits or paid API acceptance. +- The local HTTP/SSE fixture additionally covers official Flash, unflagged Chat + Completions and a compatible relay. Canonical activation restoration, denial + before Host execution, mode/account/catalog invalidation and compaction remain + required. Fixed declarations increase first-request size; oversized catalogs + explicitly fall back without a cache-stability guarantee. diff --git a/packages/agent-runtime/src/fixed-tool-declarations.test.ts b/packages/agent-runtime/src/fixed-tool-declarations.test.ts index 8b54969b9a..451fd277ca 100644 --- a/packages/agent-runtime/src/fixed-tool-declarations.test.ts +++ b/packages/agent-runtime/src/fixed-tool-declarations.test.ts @@ -1,9 +1,10 @@ import { describe, expect, it } from "vitest"; -import type { AgentTool } from "@earendil-works/pi-agent-core"; +import type { AgentMessage, AgentTool } from "@earendil-works/pi-agent-core"; import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; import { Type, type Api, type Model } from "@earendil-works/pi-ai"; import { toolDeclarationPolicy, toolActivationSection, restoredToolActivation, TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; import { replaceSystemPrompt, systemTranscriptCheckpoint } from "./system-transcript.js"; +import { convertToLlm } from "./pi-runtime-messages.js"; const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; const tool = (name: string): AgentTool => ({ name, label: name, description: "Fixture tool", parameters: Type.Object({}), @@ -13,6 +14,24 @@ const policy = (tools = [tool("Alpha"), tool("Beta")], overrides: Partial, tools, deferred, prompt, "fixture-account"); describe("fixed tool declaration policy", () => { + it("persists activation without projecting it into model instructions", () => { + const current = policy(); + const messages: AgentMessage[] = [ + { role: "system" as const, content: "", timestamp: 1, toolsAdded: current.tools, + sections: { runtime: "Rules", [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set()) } }, + { role: "system" as const, content: "", timestamp: 2, + sections: { [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set(["Alpha"])) } }, + { role: "system" as const, content: "Changed instruction", timestamp: 3, + sections: { obsolete: null, [TOOL_ACTIVATION_SECTION]: null }, toolsRemoved: [{ name: "Beta" }] }, + ]; + const before = JSON.stringify(messages); + const projected = convertToLlm(messages); + expect(projected).toHaveLength(2); + expect(projected[0]).toMatchObject({ sections: { runtime: "Rules" }, toolsAdded: current.tools }); + expect(projected[1]).toMatchObject({ content: "Changed instruction", sections: { obsolete: null }, toolsRemoved: [{ name: "Beta" }] }); + expect(JSON.stringify(projected)).not.toContain(TOOL_ACTIVATION_SECTION); + expect(JSON.stringify(messages)).toEqual(before); + }); it("has a deterministic declaration order and snapshot identity", () => { const first = policy(); const second = policy([tool("Beta"), tool("Alpha")]); @@ -23,8 +42,8 @@ describe("fixed tool declaration policy", () => { it.each([ { id: "deepseek-pro" }, { api: "openai-responses" }, { baseUrl: "https://relay.invalid" }, { baseUrl: "https://api.deepseek.com/v1" }, { compat: { supportsMidConvoSystemMessages: false } }, - ] as Partial>[])("does not enable unverified bindings: %j", (overrides) => { - expect(policy(undefined, overrides).tools).toBeUndefined(); + ] as Partial>[])("stabilizes declarations without assuming transcript capabilities: %j", (overrides) => { + expect(policy(undefined, overrides).tools?.map((tool) => tool.name)).toEqual(["Alpha", "Beta"]); expect(policy(undefined, overrides).fallback).toBeUndefined(); }); it("falls back without truncating catalogs beyond the provider's function limit", () => { diff --git a/packages/agent-runtime/src/fixed-tool-declarations.ts b/packages/agent-runtime/src/fixed-tool-declarations.ts index 721f39663e..cf3a450fd1 100644 --- a/packages/agent-runtime/src/fixed-tool-declarations.ts +++ b/packages/agent-runtime/src/fixed-tool-declarations.ts @@ -17,7 +17,24 @@ export type ToolDeclarationPolicy = { fallback?: "tool-count" | "context-budget"; }; -/** The verified Flash route has chronological system updates, but no tool deltas. */ +/** Only these Pi transports keep new schemas out of the request's initial tools. */ +function hasAnchoredToolAdditions(model: Model): boolean { + if (model.api === "pi-messages") return true; + const compat = model.compat; + if (!compat || !("supportsMidConvoSystemMessages" in compat) + || compat.supportsMidConvoSystemMessages !== true) return false; + if (model.api === "openai-completions") { + return "supportsMidConvoToolAdditions" in compat && compat.supportsMidConvoToolAdditions === true; + } + if (["openai-responses", "openai-codex-responses"].includes(model.api)) { + return ("supportsAdditionalTools" in compat && compat.supportsAdditionalTools === true) + || ("supportsToolSearch" in compat && compat.supportsToolSearch === true); + } + // Anthropic's native tool-change blocks still grow its request-level schemas. + return false; +} + +/** Keep native anchored additions, otherwise stabilize the complete catalog. */ export function toolDeclarationPolicy( model: Model, tools: readonly AgentTool[], deferred: ReadonlySet, prompt: string, accountId: string, ): ToolDeclarationPolicy { @@ -26,11 +43,9 @@ export function toolDeclarationPolicy( account: accountId, model: model.id, api: model.api, endpoint: model.baseUrl.replace(/\/+$/, ""), tools: ordered.map(toToolDeclaration), deferred: [...deferred].sort(), })).digest("hex"); - if (model.id !== "deepseek-flash" || model.api !== "openai-completions" - || model.baseUrl.replace(/\/+$/, "") !== "https://api.deepseek.com" - || !model.compat || !("supportsMidConvoSystemMessages" in model.compat) - || model.compat.supportsMidConvoSystemMessages !== true) return { key }; - // DeepSeek Chat Completions permits at most 128 functions. Never truncate. + if (hasAnchoredToolAdditions(model)) return { key }; + // Conservative shared ceiling, including Chat Completions' 128-function limit. + // Larger catalogs retain on-demand loading; never truncate or raise API limits. if (ordered.length > 128) return { key, fallback: "tool-count" }; const budget = contextBudgetLimitsFor(model); const tokens = estimateOutputCapInputTokens({ messages: [], systemPrompt: prompt, tools: ordered }, model); diff --git a/packages/agent-runtime/src/fixed-tool-providers.test.ts b/packages/agent-runtime/src/fixed-tool-providers.test.ts new file mode 100644 index 0000000000..12f8b24395 --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-providers.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from "vitest"; +import type { Agent } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, type Api, type Model, type AssistantMessage, type StreamFunction } from "@earendil-works/pi-ai"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { runtimeFixture, pluginTools } from "./test-helpers/fixed-tool-fixture.js"; + +const base = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; +const routes: { api: Api; compat?: Model["compat"]; native?: boolean; label: string }[] = [ + { label: "Chat Completions", api: "openai-completions" }, + { label: "Claude without native transitions", api: "anthropic-messages" }, + { label: "Claude with native transitions", api: "anthropic-messages", compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolChanges: true } }, + { label: "OpenAI Responses fallback", api: "openai-responses" }, + { label: "Codex fallback", api: "openai-codex-responses" }, + { label: "Gemini", api: "google-generative-ai" }, + { label: "Kimi native additions", api: "openai-completions", native: true, compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: true } }, + { label: "OpenAI native additions", api: "openai-responses", native: true, compat: { supportsMidConvoSystemMessages: true, supportsAdditionalTools: true } }, + { label: "Codex native search", api: "openai-codex-responses", native: true, compat: { supportsMidConvoSystemMessages: true, supportsToolSearch: true } }, + { label: "Codex native additions", api: "openai-codex-responses", native: true, compat: { supportsMidConvoSystemMessages: true, supportsAdditionalTools: true } }, + { label: "Pi transcript", api: "pi-messages", native: true }, +]; +// Cache markers move with the last message; they are not model input text. +function semantic(value: unknown): unknown { + if (Array.isArray(value)) return value.map(semantic); + if (value && typeof value === "object") return Object.fromEntries(Object.entries(value) + .filter(([key]) => key !== "cache_control" && key !== "cachePoint") + .map(([key, item]) => [key, semantic(item)])); + return value; +} +function payloadView(value: unknown): { tools: unknown; system: unknown; messages: unknown[] } { + const p = value as Record; + const config = p.config as Record | undefined; + const context = p.context as Record | undefined; + return { tools: p.tools ?? p.toolConfig ?? config?.tools, + system: p.system ?? p.instructions ?? config?.systemInstruction, + messages: (p.messages ?? p.input ?? p.contents ?? context?.messages ?? []) as unknown[] }; +} + +describe("ToolSearch user path through every Desktop-selectable Pi request adapter", () => { + it.each(routes)("preserves schemas and history for $label", async ({ api, compat, native }) => { + const model = { ...base, id: "fixture-model", provider: "fixture", api, + baseUrl: "https://fixture.invalid", compat, reasoning: false } as Model; + const adapter: { stream: StreamFunction } = await import(`@earendil-works/pi-ai/api/${api}`); + // Synthetic token satisfies Codex's local parser; no real auth is read. + const key = `fixture.${Buffer.from(JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: "fixture" } })).toString("base64")}.fixture`; + const f = runtimeFixture([], pluginTools, { id: "fixture", name: "Fixture", modelId: model.id, + baseUrl: model.baseUrl, apiKey: key, authKind: "api_key", supportsReasoning: false, + supportedThinkingLevels: ["off"], modelConfig: modelConfigFromPi(model) }); + const requests: ReturnType[] = []; + const captureErrors: string[] = []; + const calls: { name: string; arguments: Record }[] = [ + { name: "ToolSearch", arguments: { query: "plugin_alpha" } }, { name: "plugin_alpha", arguments: {} }, + { name: "ToolSearch", arguments: { query: "plugin_beta" } }, { name: "plugin_beta", arguments: {} }, + ]; + const agent = (f.runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = async (resolved, context) => { + if (resolved.api !== api) captureErrors.push(`Wrong adapter: ${resolved.api}`); + let captured: unknown; + // Run the serializer and stop at the external request boundary, before + // network access or cloud credential lookup. + const probe = adapter.stream(resolved, context, { apiKey: key, maxTokens: 1024, + onPayload: (payload) => { captured = payload; throw new Error("fixture payload captured"); }, + }); + const outcome = await probe.result(); + if (captured === undefined) captureErrors.push(outcome.errorMessage ?? "No payload"); + else if (JSON.stringify(captured).includes("tool_activation")) captureErrors.push("Activation metadata leaked"); + requests.push(payloadView(semantic(captured ?? {}))); + const call = calls.shift(); + const message: AssistantMessage = { role: "assistant", api, provider: resolved.provider, model: resolved.id, + content: call ? [{ type: "toolCall", id: `call-${requests.length}`, ...call }] : [{ type: "text", text: "Done." }], + stopReason: call ? "toolUse" : "stop", timestamp: Date.now(), + usage: { input: 10, output: 2, cacheRead: 0, cacheWrite: 0, totalTokens: 12, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } } }; + const stream = createAssistantMessageEventStream(); + stream.push({ type: "done", reason: call ? "toolUse" : "stop", message }); + return stream; + }; + try { + await f.prompt(); + expect(captureErrors).toEqual([]); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual(["plugin_alpha", "plugin_beta"]); + expect(requests).toHaveLength(5); + expect(requests[0].messages.length).toBeGreaterThan(0); + if (!native) expect(JSON.stringify(requests[0].tools)).toContain("plugin_beta"); + else { + expect(JSON.stringify(requests[0].tools) ?? "").not.toContain("plugin_beta"); + expect(JSON.stringify(requests.at(-1)?.messages)).toContain("plugin_beta"); + } + for (let index = 1; index < requests.length; index++) { + expect(requests[index].tools).toEqual(requests[0].tools); + expect(requests[index].system).toEqual(requests[0].system); + const previous = requests[index - 1].messages; + expect(requests[index].messages.slice(0, previous.length)).toEqual(previous); + } + } finally { await f.runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/fixed-tool-runtime.test.ts b/packages/agent-runtime/src/fixed-tool-runtime.test.ts index 2c86736cea..4d32bb64ec 100644 --- a/packages/agent-runtime/src/fixed-tool-runtime.test.ts +++ b/packages/agent-runtime/src/fixed-tool-runtime.test.ts @@ -1,82 +1,9 @@ -import { randomUUID } from "node:crypto"; -import { createServer } from "node:http"; import { afterEach, describe, expect, it, vi } from "vitest"; -import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; import { systemTranscriptCheckpoint } from "./system-transcript.js"; import { readSystemMessage } from "./system-transcript-journal.js"; import type { UiMessage } from "@pi-desktop/shared"; -import { modelConfigFromPi } from "./model-capabilities.js"; -import { DesktopAgentRuntime, type PluginToolDef, type RuntimeProviderConfig } from "./runtime.js"; +import { flashProvider, pluginTools, wireFixture, runtimeFixture, type Payload } from "./test-helpers/fixed-tool-fixture.js"; -function flashProvider(): RuntimeProviderConfig { - const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; - return { id: "flash-fixture", name: "Flash", modelId: model.id, baseUrl: model.baseUrl, - apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"], - modelConfig: modelConfigFromPi(model) }; -} -const pluginTools: PluginToolDef[] = ["plugin_alpha", "plugin_beta"].map((name) => ({ - name, description: `${name} synthetic probe`, parameters: { type: "object", properties: {}, required: [] }, risk: "low", -})); -type Payload = { tools: { function: { name: string } }[]; messages: { role: string; content?: unknown }[] }; -type Call = { name: string; args?: Record }; -async function wireFixture(calls: Call[]) { - const requests: Payload[] = []; - const fetch = globalThis.fetch; - const server = createServer(async (req, res) => { - const chunks: Buffer[] = []; - for await (const chunk of req) chunks.push(Buffer.from(chunk)); - requests.push(JSON.parse(Buffer.concat(chunks).toString())); - const call = calls.shift(); - const delta = call ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), - type: "function", function: { name: call.name, arguments: JSON.stringify(call.args ?? {}) } }] } - : { role: "assistant", content: "Done." }; - res.writeHead(200, { "content-type": "text/event-stream" }); - res.end(`data: ${JSON.stringify({ choices: [{ index: 0, delta, finish_reason: call ? "tool_calls" : "stop" }] })}\n\ndata: [DONE]\n\n`); - }); - await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); - const address = server.address(); - if (!address || typeof address === "string") throw new Error("Missing HTTP address"); - vi.stubGlobal("fetch", ((_url, init) => fetch(`http://127.0.0.1:${address.port}`, init)) satisfies typeof fetch); - return { requests, close: async () => { - vi.unstubAllGlobals(); - await new Promise((resolve, reject) => server.close((error) => error ? reject(error) : resolve())); - } }; -} -function runtimeFixture(history: UiMessage[] = [], tools = pluginTools, provider = flashProvider(), denied = false) { - const rows = structuredClone(history); - const executed: string[] = []; - const errors: unknown[] = []; - const runtime = new DesktopAgentRuntime({ - sessionId: "fixed-tools", mode: "agent", provider, thinkingLevel: "off", history: rows, pluginTools: tools, - commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, - host: { call: async (method: string, params?: unknown): Promise => { - if (method === "session.appendMessage") rows.push((params as { message: UiMessage }).message); - else if (method === "tools.execute") { - executed.push((params as { toolName: string }).toolName); - return denied ? { ok: false, denied: true, content: "Permission denied" } as T - : { ok: true, content: "Synthetic success" } as T; - } else throw new Error(`Unexpected host method: ${method}`); - return undefined as T; - } }, - onEvent: ({ event }) => { - if (event.type === "error") errors.push(event.error); - if (event.type === "tool_start") rows.push({ id: event.toolCallId, role: "tool", content: "", - toolCallId: event.toolCallId, toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString() }); - if (event.type === "tool_end") { - const row = rows.find((row) => row.id === event.toolCallId)!; - row.toolResult = event.result; row.isError = event.isError; row.toolStatus = event.isError ? "error" : "success"; - } - if (event.type === "message_end") { - const index = rows.findIndex((row) => row.id === event.message.id); - if (index < 0) rows.push(event.message); else rows[index] = event.message; - } - }, - }); - return { runtime, rows, executed, errors, prompt: async (id = "user-1") => { - rows.push({ id, role: "user", content: "Run the synthetic probes", createdAt: new Date().toISOString() }); - await runtime.prompt("Run the synthetic probes", id, `turn-${id}`); - } }; -} afterEach(() => vi.unstubAllGlobals()); describe("fixed Flash declarations through runtime and HTTP/SSE", () => { it("rejects a declared but inactive tool without invoking Host", async () => { @@ -171,7 +98,13 @@ describe("fixed Flash declarations through runtime and HTTP/SSE", () => { const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); const f = runtimeFixture([], pluginTools, { ...flashProvider(), baseUrl: "https://relay.invalid" }); let history: UiMessage[]; - try { await f.prompt(); history = structuredClone(f.rows); } + try { await f.prompt(); history = structuredClone(f.rows).map((row) => { + if (!row.modelSystem) return row; + const state = readSystemMessage(row.modelSystem.messageJson); + delete state.sections?.tool_activation; + if (state.toolsAdded) state.toolsAdded = state.toolsAdded.filter((tool) => tool.name !== "plugin_beta"); + return { ...row, modelSystem: { ...row.modelSystem, messageJson: JSON.stringify(state) } }; + }); } finally { await f.runtime.dispose(); await wire.close(); } const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); const restored = runtimeFixture(history!); @@ -194,12 +127,16 @@ describe("fixed Flash declarations through runtime and HTTP/SSE", () => { } finally { await f.runtime.dispose(); await wire.close(); } }); - it("keeps the complete request prefix while searching and executing two different tools", async () => { + it.each([ + { name: "official Flash", provider: flashProvider() }, + { name: "unflagged Chat Completions", provider: { ...flashProvider(), modelId: "fixture-chat", modelConfig: undefined } }, + { name: "compatible relay", provider: { ...flashProvider(), baseUrl: "https://relay.invalid/v1" } }, + ])("keeps the complete request prefix for $name while searching and executing two different tools", async ({ provider }) => { const wire = await wireFixture([ { name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }, { name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta" }, ]); - const f = runtimeFixture(); + const f = runtimeFixture([], pluginTools, provider); try { await f.prompt(); expect(f.errors).toEqual([]); diff --git a/packages/agent-runtime/src/pi-runtime-messages.ts b/packages/agent-runtime/src/pi-runtime-messages.ts index a595da9603..eace63930a 100644 --- a/packages/agent-runtime/src/pi-runtime-messages.ts +++ b/packages/agent-runtime/src/pi-runtime-messages.ts @@ -1,6 +1,7 @@ import type { Message } from "@earendil-works/pi-ai"; import { hostedSearchReplayProjection } from "@earendil-works/pi-ai/utils/hosted-search"; import type { AgentMessage } from "@earendil-works/pi-agent-core"; +import { TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; export type CompactionSummaryMessage = { role: "compactionSummary"; @@ -136,6 +137,20 @@ export function convertToLlm(messages: AgentMessage[]): Message[] { })); break; case "system": + // Activation is Desktop execution metadata, not a model instruction. + // Persist it canonically, but omit it before Pi folds system messages: + // otherwise each ToolSearch rewrites the prefix on non-native APIs. + if (runtimeMessage.sections && TOOL_ACTIVATION_SECTION in runtimeMessage.sections) { + const sections = Object.fromEntries(Object.entries(runtimeMessage.sections) + .filter(([name]) => name !== TOOL_ACTIVATION_SECTION)); + if (Object.keys(sections).length || textFromContent(runtimeMessage.content) + || runtimeMessage.toolsAdded?.length || runtimeMessage.toolsRemoved?.length) { + converted.push(asProviderMessage({ ...runtimeMessage, sections })); + } + break; + } + converted.push(asProviderMessage(runtimeMessage)); + break; case "user": case "assistant": case "toolResult": diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index f4d2dec13b..ac3ef5dc8f 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -1962,7 +1962,7 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { }, ], }); - const tools = (runtime as any).agent.state.tools as Array<{ name: string }>; + const tools = (runtime as unknown as { activeTools(): Array<{ name: string }> }).activeTools(); const names = tools.map((tool) => tool.name); expect(names).toEqual([ @@ -2162,7 +2162,7 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(result.details.activated).toEqual(["BrowserPreview"]); expect(result.details.addedToolNames).toEqual(["BrowserPreview"]); expect(agent.state.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( - false, + true, ); const next = await (runtime as any).prepareNextTurn({ @@ -2184,9 +2184,8 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(next.context.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // The Pi loop declares changes immediately before conversion; preparation - // only changes the executable tool catalog. - expect(getCurrentTools(next.context.messages).some((tool) => tool.name === "BrowserPreview")).toBe(false); + // The full catalog is stable; preparation updates execution activation. + expect(getCurrentTools(next.context.messages).some((tool) => tool.name === "BrowserPreview")).toBe(true); await runtime.dispose(); }); @@ -2214,9 +2213,8 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(agent.state.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Reset retains executability; Pi declares it at the next dispatch, not - // while this test calls preparation helpers outside the agent loop. - expect(getCurrentTools(agent.state.messages).some((tool) => tool.name === "BrowserPreview")).toBe(false); + // The full schema was already declared; activation remains sticky. + expect(getCurrentTools(agent.state.messages).some((tool) => tool.name === "BrowserPreview")).toBe(true); await runtime.dispose(); }); }); @@ -6864,10 +6862,10 @@ describe("DesktopAgentRuntime inline context compaction", () => { // The point of this family: the window is bought back without paying for a // summary, so no provider request is made at all. - expect(agent.state.messages[0]).toEqual(prefix); + expect(agent.state.messages[0]).toMatchObject({ ...prefix, timestamp: expect.any(Number) }); expect(getCurrentTools(agent.state.messages)).toEqual(agent.state.tools.map(toToolDeclaration)); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "Keep checkpoint rules" }); - expect((runtime as any).rebuiltAgentContext().messages[0]).toEqual(prefix); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "Keep checkpoint rules" }); + expect((runtime as any).rebuiltAgentContext().messages[0]).toMatchObject({ ...prefix, timestamp: expect.any(Number) }); expect(generateCompaction).not.toHaveBeenCalled(); const compaction = host.call.mock.calls.find( ([method]) => method === "session.appendCompaction", @@ -9039,7 +9037,7 @@ describe("DesktopAgentRuntime deferred tool restore (#225)", () => { status: "complete", }; const hasTool = (runtime: DesktopAgentRuntime, name: string) => - (runtime as any).agent.state.tools.some((tool: any) => tool.name === name); + (runtime as unknown as { activeTools(): Array<{ name: string }> }).activeTools().some((tool) => tool.name === name); it("keeps a tool active across prompts while its ToolSearch activation is in context", async () => { const runtime = createRuntime({ history: [assistantRow, searchRow()] }); @@ -9165,31 +9163,29 @@ describe("DesktopAgentRuntime deferred tool restore (#225)", () => { (runtime as any).resetDeferredToolsForPrompt(); expect(hasTool(runtime, "BrowserPreview")).toBe(false); - (runtime as any).fullEntries.push( - ...(createRuntime({ - history: [ - assistantRow, - { - id: "tool-preview-ok", - role: "tool", - content: "", - createdAt: now(), - status: "complete", - toolName: "BrowserPreview", - toolCallId: "call-preview-ok", - toolStatus: "success", - toolArgs: {}, - toolResult: { content: [{ type: "text", text: "opened" }] }, - }, - ], - }) as any).fullEntries, - ); - (runtime as any).resetDeferredToolsForPrompt(); - expect(hasTool(runtime, "BrowserPreview")).toBe(true); + const restored = createRuntime({ + history: [ + assistantRow, + { + id: "tool-preview-ok", + role: "tool", + content: "", + createdAt: now(), + status: "complete", + toolName: "BrowserPreview", + toolCallId: "call-preview-ok", + toolStatus: "success", + toolArgs: {}, + toolResult: { content: [{ type: "text", text: "opened" }] }, + }, + ], + }); + expect(hasTool(restored, "BrowserPreview")).toBe(true); + await restored.dispose(); await runtime.dispose(); }); - it("restores the activation again after a mode round trip", async () => { + it("requires activation again after a fixed-catalog mode round trip", async () => { const runtime = createRuntime({ history: [assistantRow, searchRow()] }); (runtime as any).resetDeferredToolsForPrompt(); expect(hasTool(runtime, "BrowserPreview")).toBe(true); @@ -9197,7 +9193,7 @@ describe("DesktopAgentRuntime deferred tool restore (#225)", () => { runtime.setMode("plan"); runtime.setMode("agent"); - expect(hasTool(runtime, "BrowserPreview")).toBe(true); + expect(hasTool(runtime, "BrowserPreview")).toBe(false); await runtime.dispose(); }); }); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index a3f1cdc622..e1d3cf82c2 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -3830,7 +3830,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } return [ "# On-demand tools", - `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using an on-demand tool that has not been activated. A visible schema is not activation; the tool_activation section, when present, records active names.`, + `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using an on-demand tool that has not been activated. A visible schema is not activation. Successful ToolSearch results identify activated tools; if unsure, search again.`, ...lines, ].join("\n"); } diff --git a/packages/agent-runtime/src/system-transcript-runtime.test.ts b/packages/agent-runtime/src/system-transcript-runtime.test.ts index 15ced3662a..0e039d584a 100644 --- a/packages/agent-runtime/src/system-transcript-runtime.test.ts +++ b/packages/agent-runtime/src/system-transcript-runtime.test.ts @@ -2,10 +2,14 @@ import { describe, expect, it } from "vitest"; import type { Agent } from "@earendil-works/pi-agent-core"; import { createAssistantMessageEventStream, getCurrentTools, getCurrentSystemPrompt, type AssistantMessage, type Message } from "@earendil-works/pi-ai"; import type { UiMessage } from "@pi-desktop/shared"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; import { DesktopAgentRuntime, type RuntimeProviderConfig } from "./runtime.js"; const provider: RuntimeProviderConfig = { id: "fixture", name: "Fixture", modelId: "fixture", apiKey: "", authKind: "none", + modelConfig: modelConfigFromPi({ ...Object.values(DEEPSEEK_MODELS)[0], id: "fixture", provider: "fixture", baseUrl: "https://fixture.invalid/v1", + compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: true } }), baseUrl: "https://fixture.invalid/v1", supportsReasoning: false, supportedThinkingLevels: ["off"], }; const skill = { id: "fixture/notes", name: "First catalog", description: "Summarize notes" }; diff --git a/packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts b/packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts new file mode 100644 index 0000000000..c48f4622db --- /dev/null +++ b/packages/agent-runtime/src/test-helpers/fixed-tool-fixture.ts @@ -0,0 +1,77 @@ +import { randomUUID } from "node:crypto"; +import { createServer } from "node:http"; +import { vi } from "vitest"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import type { UiMessage } from "@pi-desktop/shared"; +import { modelConfigFromPi } from "../model-capabilities.js"; +import { DesktopAgentRuntime, type PluginToolDef, type RuntimeProviderConfig } from "../runtime.js"; + +export function flashProvider(): RuntimeProviderConfig { + const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; + return { id: "flash-fixture", name: "Flash", modelId: model.id, baseUrl: model.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: modelConfigFromPi(model) }; +} +export const pluginTools: PluginToolDef[] = ["plugin_alpha", "plugin_beta"].map((name) => ({ + name, description: `${name} synthetic probe`, parameters: { type: "object", properties: {}, required: [] }, risk: "low", +})); +export type Payload = { tools: { function: { name: string } }[]; messages: { role: string; content?: unknown }[] }; +type Call = { name: string; args?: Record }; +export async function wireFixture(calls: Call[]) { + const requests: Payload[] = []; + const fetch = globalThis.fetch; + const server = createServer(async (req, res) => { + const chunks: Buffer[] = []; + for await (const chunk of req) chunks.push(Buffer.from(chunk)); + requests.push(JSON.parse(Buffer.concat(chunks).toString())); + const call = calls.shift(); + const delta = call ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), + type: "function", function: { name: call.name, arguments: JSON.stringify(call.args ?? {}) } }] } + : { role: "assistant", content: "Done." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + res.end(`data: ${JSON.stringify({ choices: [{ index: 0, delta, finish_reason: call ? "tool_calls" : "stop" }] })}\n\ndata: [DONE]\n\n`); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const address = server.address(); + if (!address || typeof address === "string") throw new Error("Missing HTTP address"); + vi.stubGlobal("fetch", ((_url, init) => fetch(`http://127.0.0.1:${address.port}`, init)) satisfies typeof fetch); + return { requests, close: async () => { + vi.unstubAllGlobals(); + await new Promise((resolve, reject) => server.close((error) => error ? reject(error) : resolve())); + } }; +} +export function runtimeFixture(history: UiMessage[] = [], tools = pluginTools, provider = flashProvider(), denied = false) { + const rows = structuredClone(history); + const executed: string[] = []; + const errors: unknown[] = []; + const runtime = new DesktopAgentRuntime({ + sessionId: "fixed-tools", mode: "agent", provider, thinkingLevel: "off", history: rows, pluginTools: tools, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") rows.push((params as { message: UiMessage }).message); + else if (method === "tools.execute") { + executed.push((params as { toolName: string }).toolName); + return denied ? { ok: false, denied: true, content: "Permission denied" } as T + : { ok: true, content: "Synthetic success" } as T; + } else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ id: event.toolCallId, role: "tool", content: "", + toolCallId: event.toolCallId, toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString() }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = event.isError ? "error" : "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + return { runtime, rows, executed, errors, prompt: async (id = "user-1") => { + rows.push({ id, role: "user", content: "Run the synthetic probes", createdAt: new Date().toISOString() }); + await runtime.prompt("Run the synthetic probes", id, `turn-${id}`); + } }; +} diff --git a/scripts/e2e-fixed-tool-declarations.mjs b/scripts/e2e-fixed-tool-declarations.mjs index c44d32ef58..9264a20bad 100644 --- a/scripts/e2e-fixed-tool-declarations.mjs +++ b/scripts/e2e-fixed-tool-declarations.mjs @@ -36,12 +36,15 @@ const server = createServer(async (req, res) => { }); await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; -const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); -const provider = { id: "fixture", name: "Flash fixture", modelId: model.id, baseUrl: model.baseUrl, +const flash = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); +const model = process.env.PI_FIXED_TOOL_FIXTURE_ROUTE === "compatible" + ? { ...flash, id: "compatible-fixture", baseUrl: "https://fixture.invalid/v1", compat: undefined } + : flash; +const provider = { id: "fixture", name: "Fixed tools fixture", modelId: model.id, baseUrl: model.baseUrl, modelConfig: modelConfigFromPi(model), apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; const fetchHook = join(root, "fixture-fetch.mjs"); await writeFile(fetchHook, `const fetch = globalThis.fetch; globalThis.fetch = (url, init) => { - if (String(url) !== "https://api.deepseek.com/chat/completions") throw new Error("Unexpected fixture endpoint"); + if (String(url) !== ${JSON.stringify(`${model.baseUrl}/chat/completions`)}) throw new Error("Unexpected fixture endpoint"); return fetch(${JSON.stringify(baseUrl)}, init); };`); let pluginTools = ["plugin_alpha", "plugin_beta"].map((name) => ({ name, From c590924c4805aabb4208cac57c3f83a706315383 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 23:12:24 +0800 Subject: [PATCH 7/8] test(runtime): verify activation separately in system E2E The unflagged fixture now declares its catalog upfront. Assert persisted activation and stable request schemas instead of expecting a later declaration delta. --- scripts/e2e-system-transcript.mjs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/scripts/e2e-system-transcript.mjs b/scripts/e2e-system-transcript.mjs index 6d4139152a..a30350f0ef 100644 --- a/scripts/e2e-system-transcript.mjs +++ b/scripts/e2e-system-transcript.mjs @@ -121,7 +121,8 @@ try { const saved = await detail(); const states = saved.messages.filter((row) => row.modelSystem); assert.equal(states.length, 2); - assert(JSON.parse(states[1].modelSystem.messageJson).toolsAdded.some((tool) => tool.name === "BrowserPreview")); + assert(JSON.parse(JSON.parse(states[1].modelSystem.messageJson).sections.tool_activation).active.includes("BrowserPreview")); + assert.deepEqual(requests[1].tools, requests[0].tools); assert(requests[1].tools.some((tool) => tool.function.name === "BrowserPreview")); console.log("PASS: sidecar ToolSearch and acknowledged Host persistence"); await sidecar.dispose(); From 14a5401001d9c84223fe6f15dd4f3f33f5bc8cf6 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 23:24:28 +0800 Subject: [PATCH 8/8] test(runtime): publish process readiness atomically CI observed the readiness file between creation and its JSON write. Rename a completed sibling file into place so process cancellation assertions begin only after the PID list is readable. --- packages/agent-runtime/src/extensions/managed-exec.test.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/packages/agent-runtime/src/extensions/managed-exec.test.ts b/packages/agent-runtime/src/extensions/managed-exec.test.ts index b41a6beeb9..5b869d663a 100644 --- a/packages/agent-runtime/src/extensions/managed-exec.test.ts +++ b/packages/agent-runtime/src/extensions/managed-exec.test.ts @@ -8,7 +8,8 @@ it("cancels an owned process tree after an explicit readiness signal", async () const root = mkdtempSync(join(tmpdir(), "pi-owned-exec-")); const ready = join(root, "ready.json"); const owner = new AbortController(); - const childSource = `require('node:fs').writeFileSync(${JSON.stringify(ready)}, JSON.stringify([process.ppid,process.pid])); setInterval(()=>{},1000);`; + // Publish readiness only after the complete PID list is visible to the reader. + const childSource = `const fs=require('node:fs'); fs.writeFileSync(${JSON.stringify(`${ready}.tmp`)}, JSON.stringify([process.ppid,process.pid])); fs.renameSync(${JSON.stringify(`${ready}.tmp`)}, ${JSON.stringify(ready)}); setInterval(()=>{},1000);`; const parentSource = `require('node:child_process').spawn(process.execPath,['-e',${JSON.stringify(childSource)}],{stdio:'inherit'}); setInterval(()=>{},1000);`; const pending = managedExec(process.execPath, ["-e", parentSource], root, owner.signal); try {