From d8fd08dcd38d1a6ced2dffb464004f3776ecd7f6 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:51:29 +0800 Subject: [PATCH 1/5] fix(runtime): preserve chronological model system state Persist provider-neutral instruction and tool updates before dispatch so skill refresh, session restoration, and compaction preserve their order. Keep execution permissions authoritative and gate native updates on the original model, API, and endpoint binding. Related to #1285 --- apps/desktop/src/lib/assistant-turns.ts | 1 + apps/desktop/src/lib/transcript-projection.ts | 5 +- .../test/transcript-projection.test.mjs | 13 ++ crates/host-core/src/plugin_sessions.rs | 1 + crates/host-core/src/sessions.rs | 28 +++ crates/host-core/src/sessions/model_system.rs | 108 +++++++++++ ...039-plugin-skills-activation-and-devkit.md | 11 +- ...ed-tools-from-effective-session-context.md | 6 +- docs/adr/chronological-system-transcript.md | 63 +++++++ docs/spec/03-runtime/02-agent-runtime.md | 55 +++++- docs/spec/03-runtime/04-data-storage.md | 18 ++ docs/spec/06-delivery/04-e2e-test-plan.md | 39 ++++ .../src/extensions/prompt-chain.test.ts | 5 +- .../src/flash-transcript.test.ts | 144 +++++++++++++++ .../src/mode-tool-access.test.ts | 5 +- .../agent-runtime/src/model-capabilities.ts | 1 + packages/agent-runtime/src/output-cap.test.ts | 8 + packages/agent-runtime/src/output-cap.ts | 11 ++ .../src/plugin-skills-prompt.test.ts | 16 ++ .../agent-runtime/src/plugin-skills-prompt.ts | 34 +++- .../agent-runtime/src/plugin-skills.test.ts | 6 + packages/agent-runtime/src/plugin-skills.ts | 12 +- .../agent-runtime/src/provider-binding.ts | 5 +- .../src/provider-certificate-flow.test.ts | 5 +- .../src/provider-recovery-flow.test.ts | 1 + packages/agent-runtime/src/runtime.test.ts | 52 +++--- packages/agent-runtime/src/runtime.ts | 144 +++++++++------ packages/agent-runtime/src/session-context.ts | 6 +- packages/agent-runtime/src/sidecar.ts | 8 +- .../src/system-transcript-journal.test.ts | 86 +++++++++ .../src/system-transcript-journal.ts | 117 ++++++++++++ .../src/system-transcript-order.test.ts | 55 ++++++ .../src/system-transcript-runtime.test.ts | 144 +++++++++++++++ .../src/system-transcript.test.ts | 112 ++++++------ .../agent-runtime/src/system-transcript.ts | 122 ++++++++----- packages/agent-runtime/src/thinking-level.ts | 2 + .../src/transcript-compat.test.ts | 31 ++++ .../agent-runtime/src/transcript-compat.ts | 28 +++ .../src/transcript-payload.test.ts | 87 +++++++++ packages/shared/src/types/messages.ts | 9 + patches/@earendil-works__pi-ai@0.99.1.patch | 7 + pnpm-lock.yaml | 14 +- scripts/e2e-system-transcript.mjs | 172 ++++++++++++++++++ 43 files changed, 1580 insertions(+), 217 deletions(-) create mode 100644 crates/host-core/src/sessions/model_system.rs create mode 100644 docs/adr/chronological-system-transcript.md create mode 100644 packages/agent-runtime/src/flash-transcript.test.ts create mode 100644 packages/agent-runtime/src/system-transcript-journal.test.ts create mode 100644 packages/agent-runtime/src/system-transcript-journal.ts create mode 100644 packages/agent-runtime/src/system-transcript-order.test.ts create mode 100644 packages/agent-runtime/src/system-transcript-runtime.test.ts create mode 100644 packages/agent-runtime/src/transcript-compat.test.ts create mode 100644 packages/agent-runtime/src/transcript-compat.ts create mode 100644 packages/agent-runtime/src/transcript-payload.test.ts create mode 100644 scripts/e2e-system-transcript.mjs diff --git a/apps/desktop/src/lib/assistant-turns.ts b/apps/desktop/src/lib/assistant-turns.ts index ec80ddcda..e47a06489 100644 --- a/apps/desktop/src/lib/assistant-turns.ts +++ b/apps/desktop/src/lib/assistant-turns.ts @@ -68,6 +68,7 @@ export function messageThinking(message: UiMessage): string { } function isVisibleMessage(message: UiMessage): boolean { + if (message.role === "system" && message.modelSystem) return false; return !( message.role === "assistant" && !(message.content || "").trim() && diff --git a/apps/desktop/src/lib/transcript-projection.ts b/apps/desktop/src/lib/transcript-projection.ts index 093bdb668..e50ffefc2 100644 --- a/apps/desktop/src/lib/transcript-projection.ts +++ b/apps/desktop/src/lib/transcript-projection.ts @@ -57,7 +57,8 @@ function shape(message: UiMessage): MessageShape { const value = { content, thinking, - visible: message.role !== "assistant" || content || thinking || Boolean(message.hostedSearch || message.error), + visible: !(message.role === "system" && message.modelSystem) && + (message.role !== "assistant" || content || thinking || Boolean(message.hostedSearch || message.error)), answer: message.parentToolCallId ? content || Boolean(message.error) : content || !thinking || Boolean(message.error), @@ -72,7 +73,7 @@ function canReplace(previous: UiMessage, next: UiMessage): boolean { previous.parentToolCallId !== next.parentToolCallId || previous.agentName !== next.agentName || previous.createdAt !== next.createdAt || previous.toolCallId !== next.toolCallId || previous.toolName !== next.toolName || - previous.hostedSearch !== next.hostedSearch + previous.hostedSearch !== next.hostedSearch || Boolean(previous.modelSystem) !== Boolean(next.modelSystem) ) return false; if (isDelegationStartTool(previous.toolName) && ( previous.toolArgs !== next.toolArgs || previous.toolResult !== next.toolResult diff --git a/apps/desktop/test/transcript-projection.test.mjs b/apps/desktop/test/transcript-projection.test.mjs index 01a3de4ad..15b30035f 100644 --- a/apps/desktop/test/transcript-projection.test.mjs +++ b/apps/desktop/test/transcript-projection.test.mjs @@ -14,6 +14,19 @@ const message = (id, role, content, extra = {}) => ({ id, role, content, createdAt: "2026-09-30T00:00:00.000Z", ...extra, }); +test("model system records remain hidden after loading and live projection updates", () => { + const state = message("state", "system", "", { modelSystem: { + version: 1, beforeMessageId: "user", messageJson: JSON.stringify({ role: "system", content: "Instructions", timestamp: 1 }), + } }); + const user = message("user", "user", "Hello"); + const notice = message("notice", "system", "Visible notice"); + let rows = [state, user, notice]; + assert.deepEqual(assertProjection(rows).visible.map((row) => row.id), ["user", "notice"]); + rows = upsertLiveSessionMessage(rows, message("answer", "assistant", "Hello back")); + assert.deepEqual(assertProjection(rows).visible.map((row) => row.id), ["user", "notice", "answer"]); + assert.equal(rows[0], state); +}); + function assertProjection(messages, compactions) { const actual = getTranscriptProjection(messages, compactions); const expected = buildTranscriptEntries(messages, compactions); diff --git a/crates/host-core/src/plugin_sessions.rs b/crates/host-core/src/plugin_sessions.rs index 3dee8f6a3..bbc982407 100644 --- a/crates/host-core/src/plugin_sessions.rs +++ b/crates/host-core/src/plugin_sessions.rs @@ -260,6 +260,7 @@ fn parse_message( nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }) } diff --git a/crates/host-core/src/sessions.rs b/crates/host-core/src/sessions.rs index 0b2478225..84c19f019 100644 --- a/crates/host-core/src/sessions.rs +++ b/crates/host-core/src/sessions.rs @@ -12,6 +12,7 @@ use std::sync::{Mutex, OnceLock}; use crate::transcripts::{self, CompactionRecord, MessageRecord, RevisionRecord}; mod fork_files; +mod model_system; mod usage; pub use usage::record_usage; @@ -237,6 +238,9 @@ pub struct UiMessage { /// as an additive `hostedSearch` transcript block; no SQL migration. #[serde(default, skip_serializing_if = "Option::is_none")] pub hosted_search: Option, + /// Internal model-context state, preserved outside visible message text. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_system: Option, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -316,6 +320,9 @@ fn is_default_title(title: &str) -> bool { /// the search index row (None for tool rows, matching the FTS triggers). pub(crate) fn ui_to_record(message: &UiMessage) -> (MessageRecord, Option) { let mut meta_obj = serde_json::Map::new(); + if let Some(system) = &message.model_system { + meta_obj.insert("modelSystem".into(), system.clone()); + } if let Some(command) = &message.command { meta_obj.insert("command".into(), json!(command)); } @@ -465,6 +472,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { _ => Vec::new(), }; let meta = record.meta.unwrap_or(Value::Null); + let model_system = meta.get("modelSystem").cloned(); let command = meta .get("command") .and_then(Value::as_str) @@ -634,6 +642,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { nested_parent_tool_call_id, agent_name, hosted_search: hosted_search.clone(), + model_system: None, } } else { let content = blocks @@ -678,6 +687,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { nested_parent_tool_call_id, agent_name, hosted_search, + model_system, } } } @@ -897,6 +907,19 @@ fn clone_records_for_fork( .cloned() .unwrap_or_else(|| Uuid::new_v4().to_string()); if let Some(meta) = record.meta.as_mut().and_then(Value::as_object_mut) { + if let Some(system) = meta.get_mut("modelSystem").and_then(Value::as_object_mut) { + for key in ["beforeMessageId", "afterMessageId"] { + if let Some(new_id) = system + .get(key) + .and_then(Value::as_str) + .and_then(|id| message_ids.get(id)) + { + system.insert(key.into(), json!(new_id)); + } else { + system.remove(key); + } + } + } meta.remove("revisionRootId"); meta.remove("revisionCount"); meta.remove("activeRevision"); @@ -1903,6 +1926,7 @@ pub fn append_message( message: &UiMessage, turn_id: Option<&str>, ) -> Result<()> { + model_system::validate(message)?; let message = crate::session_collaboration::prepare_append(db, session_id, message, turn_id)?; let session_created = ensure_session_for_append(db, session_id)?; let (mut record, text) = ui_to_record(&message); @@ -3958,6 +3982,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, } } @@ -4610,6 +4635,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }; append_message(&db, &session.id, &tool, None).unwrap(); @@ -5124,6 +5150,7 @@ mod tests { nested_parent_tool_call_id: None, agent_name: None, hosted_search: None, + model_system: None, session_message: None, }; append_message(&db, &session.id, &assistant, None).unwrap(); @@ -5203,6 +5230,7 @@ mod tests { parent_tool_call_id: None, nested_parent_tool_call_id: None, agent_name: None, + model_system: None, hosted_search: Some(json!({ "status": "completed", "rounds": [ diff --git a/crates/host-core/src/sessions/model_system.rs b/crates/host-core/src/sessions/model_system.rs new file mode 100644 index 000000000..88dd3140f --- /dev/null +++ b/crates/host-core/src/sessions/model_system.rs @@ -0,0 +1,108 @@ +use super::UiMessage; +use anyhow::{anyhow, Result}; +use serde_json::Value; + +/// Internal model state uses existing transcript metadata, never visible chat text. +pub(super) fn validate(row: &UiMessage) -> Result<()> { + let Some(record) = &row.model_system else { + return Ok(()); + }; + let invalid = || anyhow!("Invalid model system record"); + if row.role != "system" || !row.content.is_empty() || record["version"] != 1 { + return Err(invalid()); + } + for key in ["beforeMessageId", "afterMessageId"] { + if record + .get(key) + .is_some_and(|value| value.as_str().is_none_or(str::is_empty)) + { + return Err(invalid()); + } + } + let message: Value = serde_json::from_str(record["messageJson"].as_str().ok_or_else(invalid)?) + .map_err(|_| invalid())?; + if message["role"] != "system" + || !message["content"].is_string() + || !message["timestamp"] + .as_f64() + .is_some_and(|value| (0.0..=8.64e15).contains(&value)) + { + return Err(invalid()); + } + if let Some(sections) = message.get("sections") { + let sections = sections.as_object().ok_or_else(invalid)?; + if sections + .values() + .any(|value| !value.is_null() && !value.is_string()) + { + return Err(invalid()); + } + } + for key in ["toolsAdded", "toolsRemoved"] { + if let Some(tools) = message.get(key) { + for tool in tools.as_array().ok_or_else(invalid)? { + if tool + .get("name") + .and_then(Value::as_str) + .is_none_or(str::is_empty) + || (key == "toolsAdded" + && (!tool["description"].is_string() || !tool["parameters"].is_object())) + { + return Err(invalid()); + } + } + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + fn row() -> UiMessage { + serde_json::from_value(json!({ + "id":"state", "role":"system", "content":"", "createdAt":"2026-10-01T00:00:00Z", + "modelSystem":{"version":1,"beforeMessageId":"user","messageJson":json!({ + "role":"system","content":"","timestamp":1, + "sections":{"skills":"catalog"},"toolsAdded":[{"name":"Read","description":"read","parameters":{}}] + }).to_string()} + })).unwrap() + } + + #[test] + fn state_roundtrips_in_canonical_metadata_and_remaps_fork_anchors() { + let row = row(); + validate(&row).unwrap(); + let (record, text) = super::super::ui_to_record(&row); + assert!(text.as_deref().unwrap_or("").is_empty()); + assert_eq!( + super::super::record_to_ui(record.clone()).model_system, + row.model_system + ); + let mut user = row.clone(); + user.id = "user".into(); + user.role = "user".into(); + user.model_system = None; + let (records, ids, _) = + super::super::clone_records_for_fork(vec![record, super::super::ui_to_record(&user).0]); + assert_eq!( + records[0].meta.as_ref().unwrap()["modelSystem"]["beforeMessageId"], + ids["user"] + ); + } + + #[test] + fn rejects_invalid_state_or_hiding_a_user_message() { + let mut row = row(); + row.role = "user".into(); + assert!(validate(&row).is_err()); + row.role = "system".into(); + row.model_system.as_mut().unwrap()["version"] = json!(2); + assert!(validate(&row).is_err()); + row.model_system.as_mut().unwrap()["version"] = json!(1); + row.model_system.as_mut().unwrap()["messageJson"] = json!("invalid JSON"); + assert!(validate(&row).is_err()); + } +} diff --git a/docs/adr/0039-plugin-skills-activation-and-devkit.md b/docs/adr/0039-plugin-skills-activation-and-devkit.md index 9a619975f..ce80c5d36 100644 --- a/docs/adr/0039-plugin-skills-activation-and-devkit.md +++ b/docs/adr/0039-plugin-skills-activation-and-devkit.md @@ -39,11 +39,12 @@ giving them the loop — scaffold, run, inspect, package. instructions.** Later text carries more weight, so a user's own instruction files keep the last word and an installed plugin can refine the built-in guidance but never the reverse. -3. **Runtime reuse keys on the catalog digest, not on bodies.** Enabling a - plugin, revoking `agent.prompt.inject` or renaming a skill changes the text - the model reads, so it retires the idle runtime rather than reusing a stale - prompt. An edit to a body needs no retirement: the `Skill` tool reads the file - at call time. +3. **Catalog changes update an idle runtime's skills section.** The chronological + system-transcript decision supersedes the original digest-based retirement: + enabling a plugin, revoking `agent.prompt.inject`, or renaming a skill appends + a section update before the next request. The digest excludes bodies; the + `Skill` tool reads the file and rechecks permissions at call time. Removing + the last catalog entry also removes the executable `Skill` declaration. 4. **Plugin authoring ships as a first-party devkit, not as a plugin.** `@pi-desktop/plugin-devkit` owns scaffold, check and pack; three surfaces share that one implementation — the `pi-plugin` CLI, the `PluginScaffold` / diff --git a/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md b/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md index 464fc3e16..617c59f03 100644 --- a/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md +++ b/docs/adr/0225-restore-deferred-tools-from-effective-session-context.md @@ -17,7 +17,11 @@ request omitted its schema. The same mismatch occurred after a mode switch. Before each new prompt and after a mode switch, the sidecar clears its in-memory deferred activation set and restores it from the effective `buildSessionContext` -projection. A successful `ToolSearch` result contributes its `addedToolNames`; +projection. When system declarations exist, their replayed active set is the +baseline; only successful results after the last declaration can add a newly +activated tool. This prevents an earlier success from resurrecting a subsequent +removal and preserves the set across a compaction checkpoint. For older histories +without declarations, a successful `ToolSearch` result contributes its `addedToolNames`; a successful result from a deferred tool contributes that tool's name. A name is restored only when it remains in the current mode's deferred catalog. Failed, interrupted, or missing-result placeholder rows are ignored, and assistant/user diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md new file mode 100644 index 000000000..74884ee9f --- /dev/null +++ b/docs/adr/chronological-system-transcript.md @@ -0,0 +1,63 @@ +# ADR: Preserve chronological model system state + +- Status: Accepted +- Date: 2026-10-01 +- Issue: #1285 +- References: [Pi PR #9548](https://github.com/earendil-works/pi/pull/9548), + [Pi model authority](pi-ai-core-0991-authority.md) +- Updates: ADR 0039 catalog refresh, ADR 0225 deferred-tool restoration + +## Context + +Moving new tool declarations to the beginning of a conversation changes its +cached prefix. Rebuilding from visible chat alone also loses the chronology of +instruction and tool changes. Pi 0.99.1 already owns tool-state diffing and +provider request projections; Desktop owns session persistence and permissions. + +## Decision + +Keep Pi's provider-neutral system messages in chronological order. Use the +existing Host-owned transcript writer and optional versioned `modelSystem` +metadata on hidden system rows, acknowledging writes before dispatch. Anchors +preserve model order when the user row was pre-persisted. Save the effective +state once at explicit compaction; restore it before the summary and retained +tail. Keep current mode/catalog/Host permissions authoritative for execution. + +Separate Desktop's runtime instructions, skill catalog and contextual guidance +into replaceable sections. The skill-loading header uses `skills`, and each +catalog entry uses `skill:`; a change or removal appends only that +entry's replacement or null tombstone. The fixed prefix prevents collisions with +other Desktop section names. Restoring a legacy aggregate `skills` section +replaces it with the header and individual entries once at continuation. +Catalog-only skill changes refresh an idle runtime; +body loading and permission checks stay at the existing Host bridge. Tool schema +changes cannot reuse an idle runtime merely because tool names are identical. + +Retain the published model/API/endpoint identity before account overrides. +Accept transcript capability opt-ins only when that identity matches the actual +request route. Let Pi adapt the request for partial or absent support; never +persist an adapter's folded projection. + +The existing Pi 0.99.1 dependency patch adds the missing mid-conversation system +capability to its `deepseek-flash` catalog entry. Authorized official-endpoint +experiments confirmed both preserved cache reuse and effective updated +instructions. Keep this correction in the single Pi catalog, not a parallel +Desktop allowlist; remove the hunk when an upgraded Pi catalog carries it. +The model/API/endpoint binding check still excludes aliases and relays. Native +tool-addition and tool-change flags are not enabled by this correction. + +## Alternatives and consequences + +Prepending a reconstructed snapshot was rejected because it changes history on +ordinary continuation. A new persistence store or adopting Pi AgentSession +would create another session owner; optional canonical metadata preserves the +current process and data boundaries without a database migration. Unknown old +histories establish their first baseline at continuation rather than inventing +past state. Downgrades remain readable but do not retain these cache guarantees. + +The added records consume transcript storage. A failed write stops dispatch as +a local preparation failure; retry uses the same row ID. Providers may still +fold unsupported updates and lose cache reuse. Even native mid-conversation +system support does not guarantee provider cache reuse. Entry-level updates +reduce new input tokens without changing instruction roles or priority. Offline tests prove ordering, +restoration and payload compatibility, not real-provider cache-hit ratios. diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index 0262e800a..db1a2ee59 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -519,8 +519,8 @@ anchors on the last assistant usage and estimates everything after it as `chars / 4`: that constant under-counts CJK text, and with no anchor left it omits system/tool overhead. The budget also computes the output-cap estimator over non-system conversation messages plus the current system prompt and active -tool schemas. System-transcript rows are metadata snapshots of that same prompt -and are excluded from this component, so the request estimate counts the prompt +tool schemas. System-transcript rows are chronological updates, already covered by that +folded prompt/tool floor, and are excluded from this component, so the request estimate counts the prompt and schemas exactly once. The budget uses the larger of this full-request estimate and the calibrated message estimate. This is a hard floor before the first calibration sample and prevents counting overhead twice after an @@ -1278,6 +1278,51 @@ payload hook keeps its own object and its return value still wins. + [optional user custom instructions] ``` +### 7.0.0 Chronological system state (issue #1285) + +The Pi agent loop owns `toolsAdded` / `toolsRemoved` declarations, including +same-name schema replacement. Desktop never moves an update ahead of the user +or ToolSearch result that preceded it. Instruction composition uses independent +`runtime`, `skills` (loading instructions), `skill:` (one catalog +entry each), and `context` sections. Refreshing an idle skill catalog appends only +changed entries, uses null to revoke removed entries, and updates the executable +catalog. Unchanged entries and loading instructions are not repeated. Restoring +an older aggregate `skills` section upgrades it once at the continuation +boundary; checkpoints retain the effective per-entry state. Bodies remain +on-demand and permission-checked. Different plugin tool schemas/permissions +invalidate idle reuse even when their names are unchanged. + +Before model dispatch, Desktop acknowledges new system records through Host's +existing transcript writer. Restart replays these provider-neutral records in +order. An old history without records receives the current declaration at its +continuation boundary; Desktop does not invent historical instructions. Explicit +compaction saves the effective system state once before summary and retained +tail; it does not replay old tool deltas or preserve removed tools. + +Mid-conversation instructions, tool additions and full tool changes are separate +capabilities. Desktop accepts catalog opt-ins only for the same published model, +wire API and effective endpoint (including path and port). Unknown models, +changed bindings and unverified relays use Pi's conservative request projection. +Partial support keeps instruction updates but folds tools as required; removals +and redefinitions use the adapter's supported fallback. Model switching never +rewrites the canonical journal. Cache savings depend on the actual provider; +unsupported routes may still rebuild the request prefix. + +The pinned Pi patch declares mid-conversation system support for +`deepseek-flash` on its published `openai-completions` binding at +`https://api.deepseek.com`. Skill changes on this binding append system updates +without rewriting the previous request prefix. This does not enable native tool +additions or changes, and does not apply to unverified gateways, aliases or APIs. +The declaration comes from the patched Pi catalog, not provider settings or a +second Desktop model catalog. + +When restoring assistant history, map the current local account ID to the Pi +model's provider identity and retain recorded model IDs. A different recorded +account or model remains distinct. Legacy rows without identity retain the +current-model fallback. Same-model Completions reasoning stays in its native +reasoning field, never appended to visible answer text because of an account +UUID/vendor-name mismatch. + ### 7.0.1 User custom system prompt files (issue #542) The `[optional user custom instructions]` layer is the pi-compatible file pair @@ -1378,7 +1423,11 @@ schemas. Providers with native deferred-tool search receive the definitions at that load point; other providers receive the active definitions normally. At the start of each new user prompt, the sidecar clears the in-memory deferred -activation set and rebuilds it from the effective context. Successful +activation set and rebuilds it from the effective context. Recorded system +messages (including a compaction checkpoint) define the active baseline. Only +successful tool results after the latest declaration can add a new activation; +older results must not resurrect a removed tool. Histories without system records +continue to use successful results throughout their effective context. `ToolSearch` results contribute their canonical `details.addedToolNames`. For compatibility, historical `details.activated` and top-level `addedToolNames` markers are also accepted. Successful results from deferred diff --git a/docs/spec/03-runtime/04-data-storage.md b/docs/spec/03-runtime/04-data-storage.md index d43eee94c..e1722bc58 100644 --- a/docs/spec/03-runtime/04-data-storage.md +++ b/docs/spec/03-runtime/04-data-storage.md @@ -179,6 +179,24 @@ per message; `seq` is implied by line order: {"type":"compaction","id":"cp1","summary":"…","firstKeptMessageId":"m2","throughMessageId":"m3","tokensBefore":917000,"retainedTail":[…],"providerId":"…","modelId":"…","createdAt":"…"} ``` +Internal system-state rows use role `system`, empty visible content, and optional +`meta.modelSystem = { version: 1, messageJson, beforeMessageId?, afterMessageId? }`. +`messageJson` is validated JSON text of a Pi system message with sections and tool +schema deltas; executable functions are excluded. JSON text preserves section and schema key +order across Rust storage; parsing for validation never reserializes it. Stable row IDs make retries +idempotent. The anchors restore logical model order when a user row was already +persisted before its preceding declaration; a surviving following anchor takes +precedence, then a preceding anchor, then the record's continuation position. +Forks remap surviving anchor IDs. Normal system notices remain visible; internal +model-state rows do not produce transcript bubbles or search text. + +Compaction details may include one `systemMessageJson` checkpoint. It replaces old +system updates in the retained tail and is restored before the summary. These +optional metadata fields use the existing JSONL/SQLite index and require no +schema migration. Old sessions remain readable; their first continuation records +a new baseline. Older app versions ignore the metadata and reconstruct their +usual current prompt, so downgrade does not promise the same cache prefix. + `sessions/.inflight.json` — the assistant reply currently streaming in the session, as one `{ schema, sessionId, turnId, savedAt, message }` object that host-core replaces atomically (temp + rename) on every diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 00272e30d..35e7005ec 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -16258,3 +16258,42 @@ renderer's durable transcript reads. No real model or provider is contacted. - Installed Electron, real account/paid API and cross-version rollback are separate release qualification. No MCP, Codemode or virtual-router migration is included. See `docs/project/pi-0991-adoption.md` for candidate evidence. + +### E2E-SYSTEM-TRANSCRIPT: Ordered model state survives restart and compaction + +- Fixture: isolated Host data directory, production AgentSidecar transport and + separate Node runtime/Pi process, local SSE provider; no real provider + credentials or user's Desktop process. The harness persists completed + messages; it does not exercise Electron's UI or persistence outbox. +- Send a prompt, activate BrowserPreview through ToolSearch, and verify the next + request includes the schema. Confirm the baseline and delta are acknowledged + by Host and survive stopping/restarting both processes without duplicate rows. +- Resume the session, update a skill catalog on the same runtime, and confirm the + next request uses the new catalog while keeping a second unchanged entry. + Runtime/adapter tests verify the native delta contains only the modified entry, + and fallback models receive the complete current catalog with removals applied. + Invoke Skill through the actual sidecar + bridge with a fixture body supplied by the embedding host. Compact, restart + the sidecar, and verify current skills and active tools survive without + replaying old tool results. Remove the final skill and verify its catalog and + executable schema disappear without recreating the runtime. +- Automated: `node scripts/e2e-system-transcript.mjs` with + `PI_DESKTOP_HOST_BIN` pointing to the candidate Host binary. Requires built + shared/host-runtime/agent-runtime packages. Renderer projection tests ensure + model-state records stay hidden while ordinary system notices remain visible. +- Adapter contract tests cover exact model/API/endpoint capability binding, + partial support, unverified relay fallback, removals/redefinitions and model + switching without canonical-history mutation. Real cache-hit improvement is + a separately authorized provider experiment, not an offline-test claim. + +- Flash regression: load the pinned Pi `deepseek-flash` model, serialize its + Desktop projection, and pass skill updates/removals through the real adapter. + The official binding preserves the earlier wire prefix; changed bindings + retain the folded fallback. Covered by `flash-transcript.test.ts`. +- Authorized live Flash acceptance: use isolated development sessions with + short and long contexts. Alternate unchanged turns and single-skill updates, + confirm current Skill bodies execute, and compare actual outgoing prefixes + plus reported usage. After restart, verify no duplicate system updates and + same-model assistant reasoning remains separate from visible content. Adapter + regressions also cover legacy identities and genuine account/model changes. Cache + percentages are observations, not deterministic pass thresholds. diff --git a/packages/agent-runtime/src/extensions/prompt-chain.test.ts b/packages/agent-runtime/src/extensions/prompt-chain.test.ts index 1669a0b55..fa212b0c4 100644 --- a/packages/agent-runtime/src/extensions/prompt-chain.test.ts +++ b/packages/agent-runtime/src/extensions/prompt-chain.test.ts @@ -39,7 +39,10 @@ it.each([ }); runtime = new DesktopAgentRuntime({ host: { - call: async (method) => { throw new Error(`Unexpected host call: ${method}`); }, + call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + throw new Error(`Unexpected host call: ${method}`); + }, onNotification: () => () => {}, }, sessionId: "prompt-chain-fixture", diff --git a/packages/agent-runtime/src/flash-transcript.test.ts b/packages/agent-runtime/src/flash-transcript.test.ts new file mode 100644 index 000000000..a423319cf --- /dev/null +++ b/packages/agent-runtime/src/flash-transcript.test.ts @@ -0,0 +1,144 @@ +import type { Agent } from "@earendil-works/pi-agent-core"; +import { DesktopAgentRuntime } from "./runtime.js"; +import { describe, expect, it } from "vitest"; +import { normalizeContext, type Message, type Model } from "@earendil-works/pi-ai"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { stream } from "@earendil-works/pi-ai/api/openai-completions"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { buildProviderModel, type RuntimeProviderConfig } from "./provider-binding.js"; + +function flashProvider(): RuntimeProviderConfig { + const flash = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); + if (!flash) throw new Error("Missing published Flash model"); + return { + id: "flash-fixture", name: "Flash fixture", modelId: flash.id, baseUrl: flash.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: true, supportedThinkingLevels: ["high"], + // Main sends this projection across the sidecar boundary. + modelConfig: JSON.parse(JSON.stringify(modelConfigFromPi(flash))), + }; +} + +type Payload = { messages: Array<{ role: string; content?: unknown }> }; +async function request(messages: Message[], baseUrl?: string): Promise { + const provider = flashProvider(); + const model = buildProviderModel({ ...provider, ...(baseUrl ? { baseUrl } : {}) }); + let captured: Payload | undefined; + const response = stream(model as Model<"openai-completions">, normalizeContext({ messages }), { + apiKey: "fixture", + fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + expect((await response.result()).stopReason).toBe("stop"); + if (!captured) throw new Error("No Flash request captured"); + return captured; +} + +describe("published Flash transcript compatibility", () => { + it("enables system updates independently of native tool-state capabilities", () => { + const provider = flashProvider(); + expect(provider.modelConfig?.compat?.supportsMidConvoSystemMessages).toBe(true); + expect(buildProviderModel(provider).compat).toMatchObject({ + supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false, + supportsMidConvoToolChanges: false, supportsAdditionalTools: false, supportsToolSearch: false, + }); + expect(buildProviderModel({ ...provider, baseUrl: "https://api.deepseek.com/" }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + + it.each([ + { modelId: "deepseek-flash-alias" }, + { baseUrl: "https://relay.example/v1" }, { baseUrl: "http://api.deepseek.com" }, + { baseUrl: "https://api.deepseek.com/v1" }, { baseUrl: "https://api.deepseek.com:8443" }, + { baseUrl: "https://api.deepseek.com?route=other" }, { baseUrl: "https://api.deepseek.com#fragment" }, + { baseUrl: "https://user@api.deepseek.com" }, + ])("retains conservative fallback on a changed binding: %j", (overrides) => { + expect(buildProviderModel({ ...flashProvider(), ...overrides }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: false }); + }); + + it("checks the selected wire API rather than the provider-wide style", () => { + const provider = flashProvider(); + expect(buildProviderModel({ ...provider, apiStyle: "responses" })).toMatchObject({ + api: "openai-completions", compat: { supportsMidConvoSystemMessages: true }, + }); + expect(buildProviderModel({ ...provider, modelConfig: { ...provider.modelConfig!, api: "openai-responses" } })) + .toMatchObject({ api: "openai-responses", compat: { supportsMidConvoSystemMessages: false } }); + }); + + it("requires the original Pi binding even when the endpoint and model name look official", () => { + const provider = flashProvider(); + for (const modelConfig of [undefined, + { ...provider.modelConfig!, source: "generic" as const }, + { ...provider.modelConfig!, transcriptBinding: undefined }, + ]) { + expect(buildProviderModel({ ...provider, modelConfig }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: false }); + } + }); + + it("appends skill changes and revocations without rewriting the previous wire prefix", async () => { + const baseline: Message[] = [ + { role: "system", content: "", timestamp: 1, sections: { + runtime: "Base rules", skills: "Load skills on demand", "skill:a": "Old Alpha", "skill:b": "Bravo", + } }, + { role: "user", content: "First request", timestamp: 2 }, + ]; + const changed: Message[] = [...baseline, + { role: "system", content: "", timestamp: 3, sections: { "skill:a": "New Alpha", "skill:b": null } }, + { role: "user", content: "Continue", timestamp: 4 }, + ]; + const before = await request(baseline); + const after = await request(changed); + expect(after.messages.slice(0, before.messages.length)).toEqual(before.messages); + expect(after.messages.map((message) => message.role)).toEqual(["system", "user", "system", "user"]); + expect(JSON.stringify(after.messages[2])).toContain("New Alpha"); + expect(JSON.stringify(after.messages[2])).toContain("skill:b"); + const relay = await request(changed, "https://relay.example/v1"); + expect(relay.messages.map((message) => message.role)).toEqual(["system", "user", "user"]); + expect(JSON.stringify(relay.messages)).toContain("New Alpha"); + expect(JSON.stringify(relay.messages)).not.toContain("Old Alpha"); + expect(JSON.stringify(relay.messages)).not.toContain("Bravo"); + }); +}); + + +describe("Flash reasoning after Desktop session restoration", () => { + it.each([ + { providerId: "flash-fixture", modelId: "deepseek-flash", sameModel: true }, + { providerId: undefined, modelId: undefined, sameModel: true }, + { providerId: "another-account", modelId: "deepseek-flash", sameModel: false }, + { providerId: "flash-fixture", modelId: "another-model", sameModel: false }, + ])("preserves source identity when restoring %j", async ({ providerId, modelId, sameModel }) => { + const provider = { ...flashProvider(), vendorKey: "deepseek" }; + let captured: Payload | undefined; + const runtime = new DesktopAgentRuntime({ + sessionId: "restore-flash", mode: "agent", provider, thinkingLevel: "high", + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + history: [{ + id: "old-answer", role: "assistant", content: "answer", thinking: "private plan", + providerId, modelId, status: "complete", createdAt: "2026-10-01T00:00:00.000Z", + }], + host: { call: async (): Promise => undefined as T }, onEvent: () => {}, + }); + const agent = (runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = (model, context) => stream(model as Model<"openai-completions">, context, { + apiKey: "fixture", fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + try { + await runtime.prompt("Continue", "new-user", "new-turn"); + expect(captured).toBeDefined(); + const assistant = captured!.messages.find((message) => message.role === "assistant"); + if (sameModel) { + expect(assistant).toMatchObject({ content: "answer", reasoning_content: "private plan" }); + } else { + expect(assistant?.content).toContain("private plan"); + expect(assistant).not.toMatchObject({ reasoning_content: "private plan" }); + } + } finally { await runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/mode-tool-access.test.ts b/packages/agent-runtime/src/mode-tool-access.test.ts index 64f96723e..042010e7e 100644 --- a/packages/agent-runtime/src/mode-tool-access.test.ts +++ b/packages/agent-runtime/src/mode-tool-access.test.ts @@ -80,7 +80,10 @@ it.each([ compactionSettings: { enabled: false, reserveTokens: 0, keepRecentTokens: 0 }, commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, subagents: [{ name: "worker", description: "Inspect", tools: ["Read"], prompt: "Inspect only.", source: "user" }], - host: { call: async (method) => { hostCalls.push(method); throw new Error("blocked tools must never reach the host"); }, onNotification: () => () => {} }, + host: { call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + hostCalls.push(method); throw new Error("blocked tools must never reach the host"); + }, onNotification: () => () => {} }, onEvent: (event) => { events.push(event); }, }); runtime.setMode(mode); diff --git a/packages/agent-runtime/src/model-capabilities.ts b/packages/agent-runtime/src/model-capabilities.ts index dda3a9bce..2c7f3cd58 100644 --- a/packages/agent-runtime/src/model-capabilities.ts +++ b/packages/agent-runtime/src/model-capabilities.ts @@ -24,6 +24,7 @@ export function modelConfigFromPi(model: Model): ModelConfig { ...metadata, ...(compat ? { compat: { ...compat } } : {}), source: "pi", + transcriptBinding: { modelId: model.id, api: model.api, baseUrl: model.baseUrl }, nativeCost: cost, // Desktop's historical tier schema differs; do not invent a translation. cost: { input: cost.input, output: cost.output, cacheRead: cost.cacheRead, cacheWrite: cost.cacheWrite }, diff --git a/packages/agent-runtime/src/output-cap.test.ts b/packages/agent-runtime/src/output-cap.test.ts index 0faaf69d8..234a118a5 100644 --- a/packages/agent-runtime/src/output-cap.test.ts +++ b/packages/agent-runtime/src/output-cap.test.ts @@ -352,3 +352,11 @@ describe("clampOutputToContext", () => { } }); }); + +// Transcript-only Pi requests have no legacy systemPrompt/tools fields. +it("counts instruction sections and tool declarations in transcript-only requests", () => { + const base = { messages: [{ role: "system", content: "" }] }; + const state = { messages: [{ role: "system", content: "", sections: { skills: "中文说明" }, toolsAdded: [{ name: "Read", parameters: { type: "object" } }] }] }; + expect(estimateOutputCapInputTokens(state)).toBeGreaterThan(estimateOutputCapInputTokens(base)); + expect(estimateOutputCapInputTokens(state)).toBeGreaterThan(4); +}); diff --git a/packages/agent-runtime/src/output-cap.ts b/packages/agent-runtime/src/output-cap.ts index ccc46f05f..728d05858 100644 --- a/packages/agent-runtime/src/output-cap.ts +++ b/packages/agent-runtime/src/output-cap.ts @@ -46,6 +46,9 @@ export type OutputCapContext = { export type OutputCapMessage = { role: string; content: unknown; + sections?: Record; + toolsAdded?: readonly unknown[]; + toolsRemoved?: readonly unknown[]; api?: Api; provider?: string; model?: string; @@ -224,6 +227,14 @@ export function estimateOutputCapInputTokens( message.content, replayOptionsFor(message, model), ); + if (message.role === "system") { + // Canonical transcript requests carry instruction sections and schema + // updates on the message, without a separate systemPrompt/tools field. + const state = [message.sections, message.toolsAdded, message.toolsRemoved] + .filter((value) => value !== undefined).map(stringifyForEstimate).join("\n"); + baseline += Math.ceil(state.length / CHARS_PER_TOKEN); + cjkChars += countCjkChars(state); + } baseline += estimated.baseline; cjkChars += estimated.cjkChars; } diff --git a/packages/agent-runtime/src/plugin-skills-prompt.test.ts b/packages/agent-runtime/src/plugin-skills-prompt.test.ts index 9060c0f77..c2a63d4f1 100644 --- a/packages/agent-runtime/src/plugin-skills-prompt.test.ts +++ b/packages/agent-runtime/src/plugin-skills-prompt.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "vitest"; import { pluginSkillsPrompt, + pluginSkillsPromptSections, SKILL_TOOL_NAME, type PluginSkillDef, } from "./plugin-skills-prompt.js"; @@ -38,3 +39,18 @@ describe("pluginSkillsPrompt", () => { expect(prompt.split("\n").filter((line) => line.startsWith("- "))).toHaveLength(1); }); }); + +it("gives skill IDs collision-free section names and stable catalog ordering", () => { + const entries = [ + { id: "__proto__", name: "Prototype" }, { id: "runtime", name: "Reserved" }, + { id: "skill:a", name: "Colon" }, { id: "a", name: "A" }, + ]; + const sections = pluginSkillsPromptSections(entries); + expect(sections).toEqual(pluginSkillsPromptSections([...entries].reverse())); + expect(Object.keys(sections)).toEqual(Object.keys(pluginSkillsPromptSections([...entries].reverse()))); + expect(sections["skill:__proto__"]).toContain("Prototype"); + expect(sections["skill:runtime"]).toContain("Reserved"); + expect(sections["skill:skill:a"]).toContain("Colon"); + expect(sections["skill:a"]).toContain("A"); + expect(pluginSkillsPromptSections([])).toEqual({}); +}); diff --git a/packages/agent-runtime/src/plugin-skills-prompt.ts b/packages/agent-runtime/src/plugin-skills-prompt.ts index 9416ca37a..73eb161f8 100644 --- a/packages/agent-runtime/src/plugin-skills-prompt.ts +++ b/packages/agent-runtime/src/plugin-skills-prompt.ts @@ -11,16 +11,30 @@ export const SKILL_TOOL_NAME = "Skill"; * through the `Skill` tool when the model decides a skill applies, so a long * document costs nothing until it is needed. */ +const SKILLS_HEADER = [ + "# Skills", + "", + `Plugins have taught you the following skills. Each entry is a set of instructions you can load with the \`${SKILL_TOOL_NAME}\` tool by passing its exact id. When a task matches a skill's description, load the skill first and follow it; do not guess at its content. Load each skill at most once per task.`, +].join("\n"); + +export const SKILL_SECTION_PREFIX = "skill:"; + +function skillEntry(skill: PluginSkillDef): string { + const description = skill.description?.trim(); + return `- \`${skill.id}\` — ${skill.name}${description ? `: ${description}` : ""}`; +} + +/** Stable entry identities let Pi append only the changed or revoked skill. */ +export function pluginSkillsPromptSections(skills: readonly PluginSkillDef[]): Record { + if (!skills.length) return {}; + return Object.fromEntries([ + ["skills", SKILLS_HEADER], + ...[...skills].sort((a, b) => a.id.localeCompare(b.id)) + .map((skill) => [`${SKILL_SECTION_PREFIX}${skill.id}`, skillEntry(skill)]), + ]); +} + export function pluginSkillsPrompt(skills: PluginSkillDef[]): string | undefined { if (!skills.length) return undefined; - return [ - "# Skills", - "", - `Plugins have taught you the following skills. Each entry is a set of instructions you can load with the \`${SKILL_TOOL_NAME}\` tool by passing its exact id. When a task matches a skill's description, load the skill first and follow it; do not guess at its content. Load each skill at most once per task.`, - "", - ...skills.map((skill) => { - const description = skill.description?.trim(); - return `- \`${skill.id}\` — ${skill.name}${description ? `: ${description}` : ""}`; - }), - ].join("\n"); + return `${SKILLS_HEADER}\n\n${skills.map(skillEntry).join("\n")}`; } diff --git a/packages/agent-runtime/src/plugin-skills.test.ts b/packages/agent-runtime/src/plugin-skills.test.ts index d9df1a82e..d87fd7b55 100644 --- a/packages/agent-runtime/src/plugin-skills.test.ts +++ b/packages/agent-runtime/src/plugin-skills.test.ts @@ -40,3 +40,9 @@ describe("pluginSkillsDigest", () => { expect(pluginSkillsDigest([{ id: "a", name: "A", description: "" }])).toBe(without); }); }); + +it("does not confuse delimiter-containing catalog fields", () => { + expect(pluginSkillsDigest([{ id: "a", name: "b:c", description: "d" }])).not.toBe( + pluginSkillsDigest([{ id: "a", name: "b", description: "c:d" }]), + ); +}); diff --git a/packages/agent-runtime/src/plugin-skills.ts b/packages/agent-runtime/src/plugin-skills.ts index 42baab695..4126b89f6 100644 --- a/packages/agent-runtime/src/plugin-skills.ts +++ b/packages/agent-runtime/src/plugin-skills.ts @@ -20,15 +20,11 @@ export type PluginSkillDef = { /** * Stable fingerprint of a skill catalog. * - * `AgentRuntime.matches()` compares this so enabling a plugin, revoking its - * prompt permission, or editing a skill's front matter starts a fresh runtime - * instead of reusing a session whose catalog is already stale. Bodies are not - * part of it: they never enter the prompt, and the `Skill` tool reads them - * fresh from disk on every call. + * The runtime compares this before appending a skills section update. Bodies + * are excluded: the Skill tool reads them through the permission-checked host + * bridge on each invocation. */ export function pluginSkillsDigest(skills?: PluginSkillDef[]): string { if (!skills?.length) return ""; - return skills - .map((skill) => `${skill.id}:${skill.name}:${skill.description ?? ""}`) - .join("|"); + return JSON.stringify(skills.map((skill) => [skill.id, skill.name, skill.description ?? ""])); } diff --git a/packages/agent-runtime/src/provider-binding.ts b/packages/agent-runtime/src/provider-binding.ts index 5733dd2ee..c42d42b5f 100644 --- a/packages/agent-runtime/src/provider-binding.ts +++ b/packages/agent-runtime/src/provider-binding.ts @@ -1,3 +1,4 @@ +import { transcriptCompat } from "./transcript-compat.js"; /** * Provider/model wiring shared by the session runtime and its subagents. * @@ -312,7 +313,7 @@ export function buildProviderModel( const binding = apiBindingForProviderModel(provider); const catalog = provider.modelConfig; const catalogModel = catalog - ? (({ source: _source, nativeCost, ...model }) => ({ + ? (({ source: _source, transcriptBinding: _transcriptBinding, nativeCost, ...model }) => ({ ...model, ...(nativeCost ? { cost: nativeCost } : {}), }))(catalog) @@ -386,7 +387,7 @@ export function buildProviderModel( }) === "on" ? true : undefined, - ...(compat ? { compat } : {}), + compat: { ...compat, ...transcriptCompat(catalog, provider.modelId, binding.api, baseUrl) }, ...(Object.keys(modelHeaders).length > 0 ? { headers: modelHeaders } : {}), } as Model; } diff --git a/packages/agent-runtime/src/provider-certificate-flow.test.ts b/packages/agent-runtime/src/provider-certificate-flow.test.ts index b533cb1d1..c3f0d8911 100644 --- a/packages/agent-runtime/src/provider-certificate-flow.test.ts +++ b/packages/agent-runtime/src/provider-certificate-flow.test.ts @@ -29,7 +29,10 @@ it.each(["session", "delegate"])( const runtime = kind === "session" ? new DesktopAgentRuntime({ sessionId: "certificate-session", mode: "agent", provider, thinkingLevel: "off", onEvent, - host: { call: async () => { throw new Error("Unexpected host request"); } }, + host: { call: async (method: string): Promise => { + if (method === "session.appendMessage") return undefined as T; + throw new Error("Unexpected host request"); + } }, commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, }) : undefined; try { diff --git a/packages/agent-runtime/src/provider-recovery-flow.test.ts b/packages/agent-runtime/src/provider-recovery-flow.test.ts index 81dc0ec83..23d529818 100644 --- a/packages/agent-runtime/src/provider-recovery-flow.test.ts +++ b/packages/agent-runtime/src/provider-recovery-flow.test.ts @@ -85,6 +85,7 @@ function fixture(steps: Step[]) { }, host: { async call(method: string): Promise { + if (method === "session.appendMessage") return undefined as T; if (method !== "tools.execute") throw new Error(`Unexpected host method: ${method}`); reads++; diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index a511a7060..9318de1cb 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -293,10 +293,10 @@ describe("system transcript reconstruction", () => { { id: "old-assistant", role: "assistant", content: "Earlier answer", createdAt: new Date(before - 1_000).toISOString(), status: "complete" }, ] }); const internals = runtime as any; - const prefix = internals.agent.state.messages[0]; + const prefix = internals.agent.state.messages.at(-1); expect(prefix.timestamp).toBeGreaterThanOrEqual(before); const rebuilt = internals.rebuiltAgentContext(); - expect(rebuilt.messages[0]).toBe(prefix); + expect(rebuilt.messages.at(-1)).toBe(prefix); expect(getCurrentTools(rebuilt.messages)).toEqual(rebuilt.tools.map(toToolDeclaration)); await runtime.dispose(); }); @@ -318,7 +318,7 @@ describe("system transcript reconstruction", () => { vi.spyOn(agent, "continue").mockImplementation(async () => { expect(agent.state.systemPrompt).toContain(marker); expect(agent.state.systemPrompt.split("SECTION_MARKER")).toHaveLength(2); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "SECTION_MARKER" }); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "SECTION_MARKER" }); expect(getCurrentTools(agent.state.messages)).toEqual(tools); response.timestamp = agent.state.messages[0]!.timestamp + 10; agent.state.messages.push(response as any); @@ -329,9 +329,9 @@ describe("system transcript reconstruction", () => { expect(agent.continue).toHaveBeenCalledOnce(); expect(agent.state.systemPrompt).toBe(before); - expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "SECTION_MARKER" }); + expect(getCurrentSystemMessage(agent.state.messages)?.sections).toMatchObject({ rules: "SECTION_MARKER" }); expect(getCurrentTools(agent.state.messages)).toEqual(tools); - expect(estimateTranscriptTokens(agent.state.messages as any).usageTokens).toBe(0); + expect(agent.state.messages.at(-1)?.timestamp).toBeGreaterThan(response.timestamp); await runtime.dispose(); }); }); @@ -2183,10 +2183,9 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(next.context.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Tool deltas append new declarations; catalog order is not semantic. - expect([...getCurrentTools(next.context.messages)].sort((a, b) => a.name.localeCompare(b.name))).toEqual( - next.context.tools.map(toToolDeclaration).sort((a: { name: string }, b: { name: string }) => a.name.localeCompare(b.name)), - ); + // The Pi loop declares changes immediately before conversion; preparation + // only changes the executable tool catalog. + expect(getCurrentTools(next.context.messages).some((tool) => tool.name === "BrowserPreview")).toBe(false); await runtime.dispose(); }); @@ -2856,7 +2855,7 @@ describe("DesktopAgentRuntime plan transitions", () => { // The progress assistant is visible in the reused bubble but must be // removed before continue() rebuilds the model context. expect(agent.state.messages.filter((message: any) => message.role !== "system")).toHaveLength(1); - expect(agent.state.messages.at(-1)?.role).toBe("user"); + expect(agent.state.messages.filter((message: AgentMessage) => message.role !== "system").at(-1)?.role).toBe("user"); expect(agent.state.systemPrompt).toContain(""); } await handleAgentEvent({ type: "agent_start" }); @@ -2945,7 +2944,7 @@ describe("DesktopAgentRuntime plan transitions", () => { ]; } else { expect(agent.state.messages.filter((message: any) => message.role !== "system")).toHaveLength(1); - expect(agent.state.messages.at(-1)?.role).toBe("user"); + expect(agent.state.messages.filter((message: AgentMessage) => message.role !== "system").at(-1)?.role).toBe("user"); expect(agent.state.systemPrompt).toContain( attempts === 2 ? "" : "", ); @@ -5178,7 +5177,7 @@ describe("DesktopAgentRuntime compaction restore", () => { const estimate = estimateAgentContextTokens((runtime as any).agent.state.messages); expect(estimate.usageTokens).toBe(0); expect(estimate.lastUsageIndex).toBeNull(); - expect(budget.tokens).toBeGreaterThan(estimate.tokens); + expect(budget.tokens).toBeGreaterThanOrEqual(estimate.tokens); expect(budget.tokens).toBeGreaterThan(0); expect(budget.tokens).toBeLessThan(250_000); await runtime.dispose(); @@ -6555,17 +6554,18 @@ describe("DesktopAgentRuntime inline context compaction", () => { ); const generateCompaction = vi.spyOn(runtime as any, "generateCompaction"); const agent = (runtime as any).agent as Agent; - const prefix = { ...agent.state.messages[0], sections: { rules: "Keep checkpoint rules" } }; - agent.state.messages[0] = prefix as any; + const index = agent.state.messages.findIndex((message) => message.role === "system"); + const prefix = { ...agent.state.messages[index], sections: { rules: "Keep checkpoint rules" } }; + agent.state.messages[index] = prefix as any; await (runtime as any).prepareNextTurn(nextTurn); // The point of this family: the window is bought back without paying for a // summary, so no provider request is made at all. - expect(agent.state.messages[0]).toBe(prefix); + expect(agent.state.messages[0]).toEqual(prefix); expect(getCurrentTools(agent.state.messages)).toEqual(agent.state.tools.map(toToolDeclaration)); expect(getCurrentSystemMessage(agent.state.messages)?.sections).toEqual({ rules: "Keep checkpoint rules" }); - expect((runtime as any).rebuiltAgentContext().messages[0]).toBe(prefix); + expect((runtime as any).rebuiltAgentContext().messages[0]).toEqual(prefix); expect(generateCompaction).not.toHaveBeenCalled(); const compaction = host.call.mock.calls.find( ([method]) => method === "session.appendCompaction", @@ -6582,7 +6582,7 @@ describe("DesktopAgentRuntime inline context compaction", () => { buildSessionContext((runtime as any).entriesWithCompaction()).messages.map( (message: any) => message.role, ), - ).toEqual(["compactionSummary"]); + ).toEqual(["system", "compactionSummary"]); const events = onEvent.mock.calls.map(([envelope]) => (envelope as any).event); expect(events).toContainEqual( expect.objectContaining({ @@ -6674,23 +6674,23 @@ describe("DesktopAgentRuntime plugin skills (D174)", () => { await runtime.dispose(); }); - it("does not reuse a runtime whose skill catalog changed", async () => { + it("reuses an idle runtime when the skill catalog changes", async () => { const runtime = createRuntime({ pluginSkills }); expect(runtimeMatches(runtime, { pluginSkills })).toBe(true); // Revoking agent.prompt.inject empties the catalog. - expect(runtimeMatches(runtime, { pluginSkills: [] })).toBe(false); + expect(runtimeMatches(runtime, { pluginSkills: [] })).toBe(true); expect( runtimeMatches(runtime, { pluginSkills: [...pluginSkills, { id: "demo.hello/other", name: "Other" }], }), - ).toBe(false); + ).toBe(true); // A renamed skill rewrites the catalog line the model reads. expect( runtimeMatches(runtime, { pluginSkills: [{ ...pluginSkills[0], name: "Renamed" }], }), - ).toBe(false); + ).toBe(true); await runtime.dispose(); }); @@ -10447,3 +10447,13 @@ describe("toolResultFromUi image restoration (issue #1073)", () => { expect(restored.content).toEqual([{ type: "text", text: expect.stringContaining("broken.png") }]); }); }); + +it("does not reuse stale plugin declarations when schema or permission metadata changes", async () => { + const plugin: PluginToolDef = { name: "plugin_fixture", description: "Inspect", parameters: { type: "object", properties: { path: { type: "string" } } } }; + const runtime = createRuntime({ pluginTools: [plugin] }); + try { + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin }] })).toBe(true); + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin, parameters: { type: "object", properties: { file: { type: "string" } } } }] })).toBe(false); + expect(runtimeMatches(runtime, { pluginTools: [{ ...plugin, planSafeActions: ["inspect"] }] })).toBe(false); + } finally { await runtime.dispose(); } +}); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 4b3a47556..f7084b85d 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -1,3 +1,4 @@ +import { orderSystemRows, SystemTranscriptJournal } from "./system-transcript-journal.js"; import { accountModelStream } from "./request-usage.js"; import { modeToolDenial, retainModeToolDeclaration, withModeExecutionGuard } from "./mode-tool-access.js"; import { restoreHostedSearchReplay } from "./hosted-search-replay.js"; @@ -40,6 +41,7 @@ import { } from "@earendil-works/pi-agent-core"; import { isContextOverflow, + getCurrentTools, Type, type Api, type AssistantMessage, @@ -136,9 +138,12 @@ import { withExplicitRequired } from "./tool-schema.js"; import { buildSessionContext } from "./session-context.js"; import { initialSystemTranscript, + CONTEXT_BUDGET_SECTION, + syncSystemSections, + systemTranscriptCheckpoint, + removeTrailingAssistantMessages, rebuildSystemTranscript, replaceSystemPrompt, - syncSystemTools, systemPromptContent, } from "./system-transcript.js"; import { @@ -205,6 +210,7 @@ import type { CustomSystemPrompt } from "./custom-system-prompt.js"; import { projectMemoryPrompt } from "./project-memory-prompt.js"; import { pluginSkillsPrompt, + pluginSkillsPromptSections, SKILL_TOOL_NAME, type PluginSkillDef, } from "./plugin-skills-prompt.js"; @@ -1794,6 +1800,8 @@ export class DesktopAgentRuntime { }; private terminatingToolCalls = new Set(); private fullEntries: MessageEntry[]; + private readonly systemJournal = new SystemTranscriptJournal(); + private composedSections: Record = {}; private activeCompaction?: ContextCompactionRecord; private compactionEnabled: boolean; private readonly compactionStrategy: CompactionStrategy; @@ -1846,7 +1854,7 @@ export class DesktopAgentRuntime { this.streamSink = createStreamCoalescer(opts.onEvent); this.onEvent = (envelope) => this.streamSink.push(envelope); this.pluginTools = opts.pluginTools ?? []; - this.pluginSkills = opts.pluginSkills ?? []; + this.pluginSkills = [...(opts.pluginSkills ?? [])].sort((left, right) => left.id.localeCompare(right.id)); this.trustedExtensionSpecs = opts.trustedExtensions ?? []; this.subagents = opts.subagents ?? []; this.subagentProviders = opts.subagentProviders ?? {}; @@ -1881,7 +1889,6 @@ export class DesktopAgentRuntime { this.delegationChains.hydrate( rebuildChainsFromTranscript(this.transcriptHistory), ); - const skillsPrompt = pluginSkillsPrompt(this.pluginSkills); // Parts: [0] is the product persona; [1:] are operational rules a custom // SYSTEM.md must not remove (tool guidance, delegation, scratch, skills). const defaultSystemPromptParts = [ @@ -1913,8 +1920,6 @@ Do not invent objections or turn speculative risks into blockers. Stop when the `Your scratch directory for this session is \`${formatScratchDirForShell(this.commandShell, this.scratchDir)}\` (in Bash: ${shellScratchVariable(this.commandShell)}). Store ad-hoc temporary and intermediate files there using absolute paths. Workspace writes must be task-related project files or required toolchain outputs. Scratch persists across turns and is deleted with the session.`, ] : []), - // Plugin skills (D174). - ...(skillsPrompt ? [skillsPrompt] : []), ]; // A custom SYSTEM.md replaces only the product persona line, never the // operational rules in the default parts: tool guidance, delegation @@ -2058,19 +2063,24 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // The provider's rule that a tool-call id is unique is enforced here, on // the last view before the wire: the request is the only place it can be // guaranteed for both a rebuilt context and one that grew in this process. - convertToLlm: (messages) => - alignRetainedReasoningIdentity( - convertToLlm(this.dropDuplicateToolCalls(messages)), - this.reasoningReplayIdentity(), - ), + convertToLlm: async (messages) => { + await this.systemJournal.persist(messages, this.fullEntries, async (message) => { + await this.host.call("session.appendMessage", { + sessionId: this.sessionId, turnId: this.turnId, message, + }); + }); + return alignRetainedReasoningIdentity( + convertToLlm(this.dropDuplicateToolCalls(messages)), this.reasoningReplayIdentity(), + ); + }, prepareNextTurnWithContext: (context, signal) => this.prepareNextTurn(context, signal), afterToolCall: async (context) => this.afterToolCall(context), initialState: { model, - tools, + tools: [], thinkingLevel: agentThinkingLevel(this.thinkingLevel), - messages: initialSystemTranscript(this.composeSystemPrompt(), tools, this.liveSessionContext().messages), + messages: initialSystemTranscript(this.composeSystemPrompt(), tools, this.liveSessionContext().messages, this.composedSections), }, // Plan transitions must be the only tool call in an assistant batch. // Sequential execution also makes the host-confirmed mode change visible @@ -2094,6 +2104,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the }, }); + // Avoid Pi inserting a tool-only baseline ahead of restored legacy rows. + this.setAgentTools(tools); + // pi awaits every listener, so a throw here would reject the run in // progress and, with nothing awaiting that rejection, could take the whole // sidecar down. Contain it: log with the session attached and let the @@ -2121,6 +2134,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.infiniteProviderRetry = enabled; } + /** Refresh catalog instructions without changing the running session owner. */ + setPluginSkills(skills: PluginSkillDef[]): void { + if (this.disposed) throw new Error("runtime disposed"); + if (this.getStatus().isRunning) throw new Error("cannot update skills during an active turn"); + const sorted = [...skills].sort((left, right) => left.id.localeCompare(right.id)); + if (pluginSkillsDigest(this.pluginSkills) === pluginSkillsDigest(sorted)) return; + this.pluginSkills = sorted; + this.rebuildToolCatalog(); + this.setAgentTools(this.activeTools()); + this.setAgentSystemPrompt(this.composeSystemPrompt()); + } + /** Switch the planning state on this Agent without creating another Agent. */ setMode(mode: Mode): void { if (this.disposed) throw new Error("runtime disposed"); @@ -2161,7 +2186,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the (this.agent.state as unknown as { systemPrompt: string }).systemPrompt = prompt; return; } - this.agent.state.messages = replaceSystemPrompt(this.agent.state.messages, prompt); + this.agent.state.messages = prompt === this.composedSystemPrompt + ? syncSystemSections(this.agent.state.messages, this.composedSections) + : replaceSystemPrompt(this.agent.state.messages, prompt); } private setAgentMessages(messages: AgentMessage[]): void { @@ -2169,17 +2196,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.agent.state.messages = messages; return; } - this.agent.state.messages = syncSystemTools( - rebuildSystemTranscript(this.agent.state.messages, messages), - this.agent.state.tools, + this.agent.state.messages = rebuildSystemTranscript( + this.agent.state.messages.filter((message) => message.role !== "system" || !this.systemJournal.isPersisted(message)), messages, ); } private setAgentTools(tools: AgentTool[]): void { this.agent.state.tools = tools; - if (this.agentUsesTranscriptSystemMessages()) { - this.agent.state.messages = syncSystemTools(this.agent.state.messages, tools); - } } private agentSystemPromptContent(): string { @@ -2198,17 +2221,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the runningDelegationIds: this.runningDelegationIds(), }) : ""; - const composed = composeModeSystemPrompt( - this.mode, - [ - this.baseSystemPrompt, + this.composedSections = { + runtime: this.baseSystemPrompt, + ...pluginSkillsPromptSections(this.pluginSkills), + context: composeModeSystemPrompt(this.mode, [ ...(this.customSystemPrompt?.append ? [this.customSystemPrompt.append] : []), ...(optionalToolsPrompt ? [optionalToolsPrompt] : []), ...(projectPrompt ? [projectPrompt] : []), ...(memoryPrompt ? [memoryPrompt] : []), ...(resumablePrompt ? [resumablePrompt] : []), - ].join("\n\n"), - ); + ].join("\n\n")), + }; + const composed = Object.values(this.composedSections).filter(Boolean).join("\n\n"); this.composedSystemPrompt = composed; return composed; } @@ -2421,9 +2445,8 @@ Do not invent objections or turn speculative risks into blockers. Stop when the /** True when this runtime can be reused for a prompt with the given config. */ matches(config: RuntimeMatchConfig): boolean { const requestedPluginTools = config.pluginTools ?? []; - const requestedPluginSkills = config.pluginSkills ?? []; - const current = this.pluginTools.map((t) => t.name).sort().join(","); - const next = requestedPluginTools.map((t) => t.name).sort().join(","); + const current = safeJson([...this.pluginTools].sort((a, b) => a.name.localeCompare(b.name))); + const next = safeJson([...requestedPluginTools].sort((a, b) => a.name.localeCompare(b.name))); const currentThinkingLevels = [ ...(this.provider.supportedThinkingLevels ?? ["off"]), ] @@ -2466,10 +2489,6 @@ Do not invent objections or turn speculative risks into blockers. Stop when the safeJson(config.customSystemPrompt ?? null) && (this.projectMemory ?? "") === (config.projectMemory?.trim() ?? "") && (this.projectPath ?? "") === (config.projectPath?.trim() ?? "") && - // Enabling a plugin, revoking agent.prompt.inject or renaming a skill - // changes the catalog digest, which retires the runtime and its stale - // prompt. Bodies are excluded: the Skill tool always reads them fresh. - pluginSkillsDigest(this.pluginSkills) === pluginSkillsDigest(requestedPluginSkills) && // Editing `~/.agents/subagents/*.md` must reach the next prompt. Definition // bodies are part of the `Task` tool's behavior, so unlike skills they // are compared in full. @@ -2770,14 +2789,17 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // (the nearest one above), keeping each call adjacent to its result as // the provider APIs require. let toolCarrier: AssistantMessage | undefined; - for (const m of history) { + for (const m of orderSystemRows(history)) { // Subagent rows belong to the transcript and to review, never to the // parent's model context (ADR 0062): the parent only ever saw the `Task` // report, and replaying a delegate's messages would both contradict that // and reintroduce the context cost delegation exists to avoid. if (m.parentToolCallId) continue; const timestamp = Date.parse(m.createdAt) || Date.now(); - if (m.role === "user") { + if (m.role === "system" && m.modelSystem) { + toolCarrier = undefined; + append(m.id, this.systemJournal.restore(m)); + } else if (m.role === "user") { toolCarrier = undefined; const attachments = (m.attachments ?? []).map((attachment) => runtimeAttachmentFromMessage( @@ -2824,8 +2846,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the role: "assistant", content, api, - provider: this.provider.id, - model: this.provider.modelId, + // Persisted providerId identifies a local account; Pi compares its + // vendor identity to decide whether native thinking can be replayed. + // Keep known account/model switches distinct; legacy rows have no + // source identity and retain the existing current-model fallback. + provider: !m.providerId || m.providerId === this.provider.id + ? this.model.provider : m.providerId, + model: m.modelId || this.model.id, usage: usageToPi(m.usage), stopReason: "stop", timestamp, @@ -2841,8 +2868,8 @@ Do not invent objections or turn speculative risks into blockers. Stop when the role: "assistant", content: [], api, - provider: this.provider.id, - model: this.provider.modelId, + provider: this.model.provider, + model: this.model.id, usage: usageToPi(undefined), stopReason: "toolUse", timestamp, @@ -2893,6 +2920,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ): Entry[] { const entries: Entry[] = [...this.fullEntries]; if (!checkpoint) return entries; + if (isRecord(checkpoint.details) && checkpoint.details.systemMessageJson) { + this.systemJournal.rememberCheckpoint(checkpoint.details.systemMessageJson); + } const throughIndex = entries.findIndex( (entry) => entry.id === checkpoint.throughMessageId, ); @@ -5306,7 +5336,15 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private restoreDeferredToolsFromContext(): void { if (this.deferredToolNames.size === 0) return; const { messages } = this.liveSessionContext(); - for (const message of messages) { + const lastSystem = messages.map((message) => message.role).lastIndexOf("system"); + if (lastSystem >= 0) { + for (const tool of getCurrentTools(messages)) { + if (this.deferredToolNames.has(tool.name)) this.activeDeferredToolNames.add(tool.name); + } + } + // Legacy histories lack declarations. For current histories, only results + // after the last declaration can represent an activation not yet declared. + for (const message of messages.slice(lastSystem + 1)) { if (message.role !== "toolResult" || message.isError) continue; if (isMissingToolResultPlaceholder(message.content)) continue; const names = @@ -5923,8 +5961,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // agentLoopContinue refuses a transcript ending in an assistant message, // and this one carries nothing worth resending anyway. - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const promptBefore = this.agentSystemPromptContent(); @@ -5974,8 +6011,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the this.pendingOverflow = false; this.suppressOverflowRunEnd = false; this.overflowRecoveryAttempted = true; - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const compacted = await this.runCompaction( "overflow", @@ -6031,8 +6067,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // pi-agent-core refuses `continue()` when the transcript ends in an // assistant message. The progress text is already visible in the reused // bubble, so it must not be sent back as model context. - const messages = [...this.agent.state.messages]; - while (messages.at(-1)?.role === "assistant") messages.pop(); + const messages = removeTrailingAssistantMessages(this.agent.state.messages); this.setAgentMessages(messages); const promptBefore = this.agentSystemPromptContent(); @@ -6358,12 +6393,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the return { context }; } - /** - * Codex's two-tier `maybe_record`, as a system-prompt append for this turn - * only. Codex writes its reminders into conversation history; we have no - * channel for a synthetic message that stays out of the transcript, and the - * append is equivalent without persisting anything. - */ + /** A replaceable reminder section expires when compaction opens a new window. */ private withContextBudgetReminder( context: AgentContext, budget: ContextBudget, @@ -6381,7 +6411,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ...(systemPrompt ? { systemPrompt } : {}), messages: [ ...context.messages, - { role: "system", content: reminder, timestamp: Date.now() }, + { role: "system", content: "", sections: { [CONTEXT_BUDGET_SECTION]: reminder }, timestamp: Date.now() }, ], } as AgentContext; } @@ -6828,6 +6858,11 @@ Do not invent objections or turn speculative risks into blockers. Stop when the mustFitSafeBudget: boolean, fallback?: ContextCompactionFallback, ): Promise { + const systemMessage = systemTranscriptCheckpoint(this.agent.state.messages); + checkpoint = { + ...checkpoint, + details: { ...(isRecord(checkpoint.details) ? checkpoint.details : {}), ...(systemMessage ? { systemMessageJson: JSON.stringify(systemMessage) } : {}) }, + }; const compactedBudget = this.contextBudget( this.liveSessionContext(checkpoint).messages, ); @@ -6853,7 +6888,10 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // resetting its `claim_*` flags when the context window turns over. this.contextReminderClaimed = false; this.contextFallbackReminderClaimed = false; - this.setAgentMessages(this.liveSessionContext().messages); + this.agent.state.messages = this.liveSessionContext().messages; + for (const message of this.agent.state.messages) { + if (message.role === "system") this.systemJournal.remember(message, checkpoint.id); + } this.emit({ type: "compaction_end", reason, diff --git a/packages/agent-runtime/src/session-context.ts b/packages/agent-runtime/src/session-context.ts index 5aae7c8de..22327fd35 100644 --- a/packages/agent-runtime/src/session-context.ts +++ b/packages/agent-runtime/src/session-context.ts @@ -1,3 +1,4 @@ +import { readSystemMessage } from "./system-transcript-journal.js"; /** * Project pi session entries into the model context. * @@ -59,6 +60,8 @@ export function sessionEntryToContextMessages( return isContextMessage(entry.message) ? [entry.message] : []; case "compaction": return [ + ...(entry.details && typeof entry.details === "object" && "systemMessageJson" in entry.details + ? [readSystemMessage(entry.details.systemMessageJson)] : []), createCompactionSummaryMessage( entry.summary, entry.tokensBefore, @@ -71,7 +74,8 @@ export function sessionEntryToContextMessages( entry.timestamp, identity, )), - ...entry.retainedTail.filter(isContextMessage), + ...entry.retainedTail.filter((message) => isContextMessage(message) && + !(message.role === "system" && entry.details && typeof entry.details === "object" && "systemMessageJson" in entry.details)), ]; case "branch_summary": return entry.summary diff --git a/packages/agent-runtime/src/sidecar.ts b/packages/agent-runtime/src/sidecar.ts index ce7db0b7b..7bc2126dc 100644 --- a/packages/agent-runtime/src/sidecar.ts +++ b/packages/agent-runtime/src/sidecar.ts @@ -241,6 +241,7 @@ async function runtimeFor( runtimes.delete(sessionId); } if (reusable) { + reusable.setPluginSkills(pluginSkills); reusable.setCompactionSettings(params.compactionSettings); reusable.setInfiniteProviderRetry(params.infiniteProviderRetry === true); reusable.setMode(mode); @@ -260,10 +261,9 @@ async function runtimeFor( // The current prompt is sent separately below. Exclude its persisted row // before attachment hydration so it cannot consume the history byte budget. if (currentPrompt !== undefined && params.userMessageId) { - const last = restoredMessages.at(-1); - if (last?.role === "user" && last.id === params.userMessageId) { - restoredMessages = restoredMessages.slice(0, -1); - } + restoredMessages = restoredMessages.filter((message) => + message.role !== "user" || message.id !== params.userMessageId, + ); } const supportsVision = visionFromModelConfig(params.provider.modelConfig); history = await hydrateAttachmentHistory(restoredMessages, { diff --git a/packages/agent-runtime/src/system-transcript-journal.test.ts b/packages/agent-runtime/src/system-transcript-journal.test.ts new file mode 100644 index 000000000..22b9816a3 --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-journal.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from "vitest"; +import type { AgentMessage, MessageEntry, CompactionEntry } from "@earendil-works/pi-agent-core"; +import type { UiMessage } from "@pi-desktop/shared"; +import { getCurrentTools, Type, type SystemMessage } from "@earendil-works/pi-ai"; +import { orderSystemRows, readSystemMessage, SystemTranscriptJournal } from "./system-transcript-journal.js"; +import { buildSessionContext } from "./session-context.js"; +import { CONTEXT_BUDGET_SECTION, syncSystemSections, systemTranscriptCheckpoint } from "./system-transcript.js"; + +const tool = { name: "Read", description: "Read", parameters: Type.Object({ path: Type.String() }) }; +const initial: SystemMessage = { role: "system", content: "", sections: { runtime: "Instructions", skills: "First skill" }, toolsAdded: [tool], timestamp: 1 }; +const user: AgentMessage = { role: "user", content: "Find files", timestamp: 2 }; +function entry(message: AgentMessage, id = "user"): MessageEntry { + return { type: "message", id, seq: 0, parentId: null, timestamp: message.timestamp, message }; +} +const userRow: UiMessage = { id: "user", role: "user", content: "Find files", createdAt: new Date(2).toISOString() }; + +describe("model system journal", () => { + it("restores the original order even when a user row was pre-persisted", async () => { + const journal = new SystemTranscriptJournal(); + const rows = [userRow]; + const entries = [entry(user)]; + const delta: SystemMessage = { role: "system", content: "", toolsRemoved: [{ name: "Read" }], timestamp: 3 }; + await journal.persist([initial, user, delta], entries, async (row) => { rows.push(row); }); + expect(rows.map((row) => row.role)).toEqual(["user", "system", "system"]); + const restored = new SystemTranscriptJournal(); + const messages = orderSystemRows(rows).map((row) => row.modelSystem ? restored.restore(row) : user); + expect(messages).toEqual([initial, user, delta]); + expect(getCurrentTools(messages)).toEqual([]); + const append = vi.fn(); + await restored.persist(messages, entries, append); + expect(append).not.toHaveBeenCalled(); + }); + + it("retries failed persistence with the same id and never installs it before acknowledgement", async () => { + const journal = new SystemTranscriptJournal(); + const entries = [entry(user)]; + const rows: UiMessage[] = []; + await expect(journal.persist([initial, user], entries, async (row) => { + rows.push(row); throw new Error("disk full"); + })).rejects.toMatchObject({ code: "LOCAL_REQUEST_ERROR", cause: new Error("disk full") }); + expect(entries).toHaveLength(1); + await journal.persist([initial, user], entries, async (row) => { rows.push(row); }); + expect(rows[0].id).toBe(rows[1].id); + expect(entries.map((item) => item.message.role)).toEqual(["system", "user"]); + }); + + it("replaces skill sections without rewriting the old prefix", () => { + const original = [initial, user]; + const changed = syncSystemSections(original, { runtime: "Instructions", skills: "Second skill" }); + expect(changed.slice(0, 2)).toEqual(original); + expect(changed.at(-1)).toMatchObject({ sections: { skills: "Second skill" } }); + expect(syncSystemSections(changed, { runtime: "Instructions", skills: "Second skill" })).toBe(changed); + expect(syncSystemSections(changed, { runtime: "Instructions", skills: "" }).at(-1)).toMatchObject({ sections: { skills: "" } }); + }); + + it("restores a compaction checkpoint once before the retained tail", async () => { + const delta: SystemMessage = { role: "system", content: "", toolsRemoved: [{ name: "Read" }], timestamp: 3 }; + const systemMessage = systemTranscriptCheckpoint([initial, user, delta])!; + const compaction: CompactionEntry = { + type: "compaction", id: "cp", seq: 1, parentId: "user", timestamp: 4, + summary: "Past work", tokensBefore: 100, retainedTail: [initial, user, delta], + details: { systemMessageJson: JSON.stringify(systemMessage) }, fromHook: false, + }; + const messages = buildSessionContext([entry(user), compaction]).messages; + expect(messages.filter((message) => message.role === "system")).toEqual([systemMessage]); + expect(messages[0]).toEqual(systemMessage); + expect(getCurrentTools(messages)).toEqual([]); + const journal = new SystemTranscriptJournal(); + journal.rememberCheckpoint(systemMessage); + const append = vi.fn(); + await journal.persist(JSON.parse(JSON.stringify(messages)), [], append); + expect(append).not.toHaveBeenCalled(); + }); + + it("expires budget reminders at compaction without dropping active instructions or tools", () => { + const checkpoint = systemTranscriptCheckpoint([initial, user, { + role: "system", content: "", sections: { [CONTEXT_BUDGET_SECTION]: "Old window is nearly full" }, timestamp: 3, + }]); + expect(checkpoint?.sections).toEqual(initial.sections); + expect(checkpoint?.toolsAdded).toEqual(initial.toolsAdded); + }); + + it.each([null, {}, { ...initial, timestamp: -1 }, { ...initial, toolsAdded: [{ name: "Read", parameters: "bad" }] }])("rejects malformed saved state", (value) => { + expect(() => readSystemMessage(value)).toThrow("Invalid persisted model system message"); + }); +}); diff --git a/packages/agent-runtime/src/system-transcript-journal.ts b/packages/agent-runtime/src/system-transcript-journal.ts new file mode 100644 index 000000000..bff05f19e --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-journal.ts @@ -0,0 +1,117 @@ +import { randomUUID } from "node:crypto"; +import { LocalRequestError, Type, contentText, toToolDeclaration, type SystemMessage } from "@earendil-works/pi-ai"; +import type { AgentMessage, MessageEntry } from "@earendil-works/pi-agent-core"; +import type { UiMessage } from "@pi-desktop/shared"; +import * as Value from "typebox/value"; + +const systemSchema = Type.Object({ + role: Type.Literal("system"), + content: Type.String(), + timestamp: Type.Number({ minimum: 0, maximum: 8.64e15 }), + sections: Type.Optional(Type.Record(Type.String(), Type.Union([Type.String(), Type.Null()]))), + toolsAdded: Type.Optional(Type.Array(Type.Object({ + name: Type.String({ minLength: 1 }), description: Type.String(), + parameters: Type.Record(Type.String(), Type.Unknown()), + constrainedSampling: Type.Optional(Type.Union([ + Type.Literal(false), + Type.Object({ type: Type.Literal("json_schema"), strict: Type.Union([Type.Literal("prefer"), Type.Literal("require")]) }), + Type.Object({ type: Type.Literal("grammar"), variants: Type.Record(Type.String(), Type.Unknown()) }), + ])), + }))), + toolsRemoved: Type.Optional(Type.Array(Type.Object({ name: Type.String({ minLength: 1 }) }))), +}); + +export function readSystemMessage(value: unknown): SystemMessage { + if (typeof value === "string") value = JSON.parse(value); + if (!Value.Check(systemSchema, value)) throw new Error("Invalid persisted model system message"); + // Tool parameter schemas and grammar variants are JSON objects at this boundary. + return value as SystemMessage; +} + +/** Reorder only internal rows, whose following user row may have arrived first. */ +export function orderSystemRows(history: readonly UiMessage[]): UiMessage[] { + const rows = history.filter((row) => !row.modelSystem); + for (const row of history) { + if (!row.modelSystem) continue; + if (row.role !== "system" || row.modelSystem.version !== 1) { + throw new Error("Invalid model system record"); + } + const before = row.modelSystem.beforeMessageId; + let index = before ? rows.findIndex((candidate) => candidate.id === before) : -1; + if (index < 0 && row.modelSystem.afterMessageId) { + const previous = rows.findIndex((candidate) => candidate.id === row.modelSystem?.afterMessageId); + if (previous >= 0) { + index = previous + 1; + while (rows[index]?.modelSystem) index++; + } + } + rows.splice(index < 0 ? rows.length : index, 0, row); + } + return rows; +} + +/** Persist provider-neutral declarations before dispatch; never persist a folded request. */ +export class SystemTranscriptJournal { + private readonly ids = new WeakMap(); + private readonly pending = new WeakMap(); + private readonly checkpoints = new Set(); + + restore(row: UiMessage): SystemMessage { + const message = readSystemMessage(row.modelSystem?.messageJson); + this.ids.set(message, row.id); + return message; + } + + isPersisted(message: AgentMessage): boolean { + return this.ids.has(message) || this.checkpoints.has(JSON.stringify(message)); + } + + async persist( + messages: readonly AgentMessage[], + entries: MessageEntry[], + append: (row: UiMessage) => Promise, + ): Promise { + for (let index = 0; index < messages.length; index++) { + const message = messages[index]; + if (message.role !== "system" || this.isPersisted(message)) continue; + const following = messages.slice(index + 1).find((candidate) => candidate.role !== "system"); + const nextEntry = following && entries.find((entry) => entry.message === following); + const preceding = messages.slice(0, index).reverse().find((candidate) => candidate.role !== "system"); + const previousEntry = preceding && entries.find((entry) => entry.message === preceding); + const id = this.pending.get(message) ?? randomUUID(); + this.pending.set(message, id); + const serialized: SystemMessage = { + ...message, content: contentText(message.content), + ...(message.toolsAdded ? { toolsAdded: message.toolsAdded.map(toToolDeclaration) } : {}), + }; + const row: UiMessage = { + id, role: "system", content: "", createdAt: new Date(message.timestamp).toISOString(), + modelSystem: { version: 1, messageJson: JSON.stringify(serialized), ...(nextEntry ? { beforeMessageId: nextEntry.id } : {}), ...(previousEntry ? { afterMessageId: previousEntry.id } : {}) }, + }; + try { + await append(row); + } catch (cause) { + // Stop before dispatch. Retrying the provider cannot repair a failed + // durable write, and the original host error may contain private paths. + throw new LocalRequestError("request-preparation", { cause }); + } + this.ids.set(message, id); + this.pending.delete(message); + const entry: MessageEntry = { + type: "message", id, seq: entries.length, parentId: null, + timestamp: message.timestamp, message, + }; + const position = nextEntry ? entries.indexOf(nextEntry) : entries.length; + entries.splice(position, 0, entry); + entries.forEach((item, seq) => { item.seq = seq; item.parentId = entries[seq - 1]?.id ?? null; }); + } + } + + rememberCheckpoint(message: unknown): void { + this.checkpoints.add(JSON.stringify(readSystemMessage(message))); + } + + remember(message: AgentMessage, id: string): void { + this.ids.set(message, id); + } +} diff --git a/packages/agent-runtime/src/system-transcript-order.test.ts b/packages/agent-runtime/src/system-transcript-order.test.ts new file mode 100644 index 000000000..65e127c11 --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-order.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from "vitest"; +import { Agent, type AgentTool, type AgentMessage } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, Type, getCurrentTools, type SystemMessage } from "@earendil-works/pi-ai"; +import { rebuildSystemTranscript } from "./system-transcript.js"; + +const read = { name: "Read", description: "Read a file", parameters: Type.Object({}) }; +const search = { name: "Search", description: "Search files", parameters: Type.Object({}) }; +const initial: SystemMessage = { role: "system", content: "Instructions", toolsAdded: [read], timestamp: 1 }; +const user: AgentMessage = { role: "user", content: "Find a file", timestamp: 2 }; +const result: AgentMessage = { + role: "toolResult", toolCallId: "search-1", toolName: "ToolSearch", + content: [{ type: "text", text: "Search activated" }], isError: false, timestamp: 3, +}; + +describe("chronological system updates", () => { + it("lets the Pi loop append schema changes and preserves the complete prefix", async () => { + const executable = (tool: typeof read): AgentTool => ({ ...tool, label: tool.name, execute: async () => ({ content: [], details: {} }) }); + const requests: AgentMessage[][] = []; + const agent = new Agent({ + initialState: { tools: [executable(read)], messages: [initial, user] }, + streamFn: async (_model, context) => { + requests.push([...context.messages]); + const stream = createAssistantMessageEventStream(); + stream.push({ type: "done", reason: "stop", message: { + role: "assistant", content: [{ type: "text", text: "Done" }], stopReason: "stop", timestamp: Date.now(), + api: "openai-completions", provider: "fixture", model: "fixture", + usage: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0, totalTokens: 2, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, + } }); + return stream; + }, + }); + await agent.continue(); + const before = [...agent.state.messages]; + const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; + agent.state.tools = [executable(replacement), executable(search)]; + await agent.prompt("Continue"); + expect(requests).toHaveLength(2); + expect(requests[1].slice(0, before.length)).toEqual(before); + expect(requests[1].slice(before.length)).toContainEqual(expect.objectContaining({ + role: "system", toolsAdded: [replacement, search], toolsRemoved: [{ name: "Read" }], + })); + expect(getCurrentTools(requests[1])).toEqual([replacement, search]); + await agent.prompt("Again"); + expect(requests[2].filter((message) => message.role === "system")).toHaveLength(2); + }); + + it("does not move a retained tool update ahead of its activation result", () => { + const update: SystemMessage = { role: "system", content: "", toolsAdded: [search], timestamp: 4 }; + const before = [initial, user, result, update]; + const after = rebuildSystemTranscript(before, [user, result]); + expect(after).toEqual(before); + expect(after[0]).toBe(initial); + expect(after.at(-1)).toBe(update); + }); +}); diff --git a/packages/agent-runtime/src/system-transcript-runtime.test.ts b/packages/agent-runtime/src/system-transcript-runtime.test.ts new file mode 100644 index 000000000..f48e132dd --- /dev/null +++ b/packages/agent-runtime/src/system-transcript-runtime.test.ts @@ -0,0 +1,144 @@ +import { describe, expect, it } from "vitest"; +import type { Agent } from "@earendil-works/pi-agent-core"; +import { createAssistantMessageEventStream, getCurrentTools, getCurrentSystemPrompt, type AssistantMessage, type Message } from "@earendil-works/pi-ai"; +import type { UiMessage } from "@pi-desktop/shared"; +import { DesktopAgentRuntime, type RuntimeProviderConfig } from "./runtime.js"; + +const provider: RuntimeProviderConfig = { + id: "fixture", name: "Fixture", modelId: "fixture", apiKey: "", authKind: "none", + baseUrl: "https://fixture.invalid/v1", supportsReasoning: false, supportedThinkingLevels: ["off"], +}; +const skill = { id: "fixture/notes", name: "First catalog", description: "Summarize notes" }; +function answer(content: AssistantMessage["content"], stopReason: "stop" | "toolUse" = "stop"): AssistantMessage { + return { + role: "assistant", api: "openai-completions", provider: "fixture", model: "fixture", + timestamp: Date.now() + 10, stopReason, content, + usage: { input: 10, output: 2, cacheRead: 0, cacheWrite: 0, totalTokens: 12, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } }, + }; +} +function fixture(history: UiMessage[] = [], skills = [skill]) { + const rows = [...history]; + const requests: Message[][] = []; + const errors: unknown[] = []; + let persistenceFailure: Error | undefined; + let respond = () => answer([{ type: "text", text: "Done." }]); + const runtime = new DesktopAgentRuntime({ + sessionId: "fixture", mode: "agent", provider, thinkingLevel: "off", history, pluginSkills: skills, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") { + if (persistenceFailure) throw persistenceFailure; + rows.push((params as { message: UiMessage }).message); + } + else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + const agent = (runtime as unknown as { agent: Agent }).agent; + agent.streamFunction = async (_model, context) => { + requests.push(structuredClone(context.messages)); + const message = respond(); + const stream = createAssistantMessageEventStream(); + stream.push({ type: "start", partial: { ...message, content: [] } }); + stream.push({ type: "done", reason: message.stopReason as "stop" | "toolUse", message }); + return stream; + }; + return { + runtime, agent, rows, requests, errors, + failPersistence: () => { persistenceFailure = new Error("disk full"); }, + respond: (next: () => AssistantMessage) => { respond = next; }, + prompt: async (id: string) => { + rows.push({ id, role: "user", content: "Continue", createdAt: new Date().toISOString() }); + await runtime.prompt("Continue", id, `turn-${id}`); + }, + }; +} + +describe("system state through the Desktop user path", () => { + it("activates ToolSearch through Pi, retains its prefix, and restores the declarations after restart", async () => { + const f = fixture(); + let requests = 0; + f.respond(() => ++requests === 1 + ? answer([{ type: "toolCall", id: "search-call", name: "ToolSearch", arguments: { query: "BrowserPreview" } }], "toolUse") + : answer([{ type: "text", text: "Ready." }])); + try { + await f.prompt("user-1"); + expect(f.requests).toHaveLength(2); + const [before, after] = f.requests; + expect(after.slice(0, before.length)).toEqual(before); + expect(getCurrentTools(before).some((tool) => tool.name === "BrowserPreview")).toBe(false); + expect(getCurrentTools(after).some((tool) => tool.name === "BrowserPreview")).toBe(true); + const activation = after.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); + expect(after.slice(activation + 1)).toContainEqual(expect.objectContaining({ role: "system", toolsAdded: expect.arrayContaining([expect.objectContaining({ name: "BrowserPreview" })]) })); + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); + const restored = fixture(structuredClone(f.rows)); + try { + await restored.prompt("user-2"); + const replay = restored.requests[0]; + expect(replay.filter((message) => message.role === "system")).toEqual(after.filter((message) => message.role === "system")); + const resultIndex = replay.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); + expect(resultIndex).toBeGreaterThan(0); + expect(replay.slice(resultIndex + 1)).toContainEqual(after.at(-1)); + expect(getCurrentTools(restored.requests[0]).some((tool) => tool.name === "BrowserPreview")).toBe(true); + expect(restored.rows.filter((row) => row.modelSystem)).toHaveLength(2); + } finally { await restored.runtime.dispose(); } + } finally { await f.runtime.dispose(); } + }); + + it("stops before provider dispatch when the Host cannot save system state", async () => { + const f = fixture(); + f.failPersistence(); + try { + await f.prompt("user-1"); + expect(f.requests).toHaveLength(0); + expect(f.errors).toContainEqual(expect.objectContaining({ + code: "INTERNAL", retriable: false, details: expect.objectContaining({ origin: "local", phase: "request-preparation" }), + })); + } finally { await f.runtime.dispose(); } + }); + + it("updates and revokes a skill catalog on the same runtime without rewriting previous requests", async () => { + const otherSkill = { id: "fixture/other", name: "Unchanged entry", description: "Keep this instruction" }; + const f = fixture([], [skill, otherSkill]); + try { + await f.prompt("user-1"); + const before = f.requests[0]; + f.runtime.setPluginSkills([{ ...skill, name: "Updated catalog" }, otherSkill]); + await f.prompt("user-2"); + const after = f.requests[1]; + expect(after.slice(0, before.length)).toEqual(before); + expect(getCurrentSystemPrompt(after)).toContain("Updated catalog"); + expect(getCurrentSystemPrompt(after)).not.toContain("First catalog"); + expect(after.filter((message) => message.role === "system")).toHaveLength(2); + const updates = after.filter((message) => message.role === "system"); + expect(updates.at(-1)?.sections).toEqual({ "skill:fixture/notes": "- `fixture/notes` — Updated catalog: Summarize notes" }); + expect(getCurrentSystemPrompt(after)).toContain("Unchanged entry"); + const restored = fixture(structuredClone(f.rows), [{ ...skill, name: "Updated catalog" }, otherSkill]); + try { + await restored.prompt("restored-user"); + expect(restored.requests[0].filter((message) => message.role === "system")).toEqual(updates); + } finally { await restored.runtime.dispose(); } + f.runtime.setPluginSkills([]); + await f.prompt("user-3"); + expect(getCurrentSystemPrompt(f.requests[2])).not.toContain("Updated catalog"); + // Revocation includes a section delta and Pi's executable-tool removal. + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(4); + expect(getCurrentTools(f.requests[2]).some((tool) => tool.name === "Skill")).toBe(false); + } finally { await f.runtime.dispose(); } + }); +}); diff --git a/packages/agent-runtime/src/system-transcript.test.ts b/packages/agent-runtime/src/system-transcript.test.ts index 0b5bf732e..8e994a499 100644 --- a/packages/agent-runtime/src/system-transcript.test.ts +++ b/packages/agent-runtime/src/system-transcript.test.ts @@ -13,22 +13,23 @@ import { } from "@earendil-works/pi-ai"; import { estimateContextTokens as estimateTranscriptTokens } from "@earendil-works/pi-ai/utils/estimate"; import { + syncSystemSections, initialSystemTranscript, rebuildSystemTranscript, replaceSystemPrompt, - syncSystemTools, + systemTranscriptCheckpoint, systemPromptContent, } from "./system-transcript.js"; const read: Tool = { name: "Read", description: "Read text", parameters: Type.Object({ path: Type.String() }) }; const edit: Tool = { ...read, name: "Edit", description: "Edit text" }; const initial: SystemMessage = { - role: "system", content: "Base instructions", timestamp: 1_000, - sections: { rules: "Keep these rules", obsolete: "Remove these rules" }, + role: "system", content: "", timestamp: 1_000, + sections: { runtime: "Base instructions", rules: "Keep these rules", obsolete: "Remove these rules" }, toolsAdded: [read, edit], }; const delta: SystemMessage = { - role: "system", content: "Additional instructions", timestamp: 1_500, + role: "system", content: "", timestamp: 1_500, sections: { rules: "Updated rules", obsolete: null }, toolsRemoved: [{ name: "Edit" }], }; @@ -49,13 +50,46 @@ function estimateContextTokens(messages: AgentMessage[]) { afterEach(() => vi.restoreAllMocks()); describe("system transcript helpers", () => { + it("updates only the changed skill entry and preserves unrelated sections", () => { + const previous: AgentMessage[] = [{ + role: "system", content: "", timestamp: 1, + sections: { runtime: "Base", skills: "Load skills", "skill:a": "A", "skill:b": "B", context: "Context", extension: "Keep" }, + }, assistant]; + const desired = { runtime: "Base", skills: "Load skills", "skill:a": "A updated", "skill:b": "B", context: "Context" }; + const updated = syncSystemSections(previous, desired); + expect(updated.slice(0, previous.length)).toEqual(previous); + expect(updated.at(-1)).toMatchObject({ sections: { "skill:a": "A updated" } }); + expect(getCurrentSystemMessage(updated)?.sections).toHaveProperty("extension", "Keep"); + expect(syncSystemSections(updated, desired)).toBe(updated); + const revoked = syncSystemSections(updated, { runtime: "Base", context: "Context" }); + expect(revoked.at(-1)).toMatchObject({ sections: { skills: null, "skill:a": null, "skill:b": null } }); + expect(getCurrentSystemPrompt(revoked)).not.toContain("A updated"); + expect(getCurrentSystemPrompt(revoked)).toContain("Keep"); + }); + + it("upgrades aggregate catalogs once and restores per-skill checkpoints without duplicate deltas", () => { + const legacy: AgentMessage[] = [{ role: "system", content: "", timestamp: 1, + sections: { runtime: "Base", skills: "Old full catalog", context: "Context" } }]; + const desired = { runtime: "Base", skills: "Load skills", "skill:z": "Z", context: "Context" }; + const upgraded = syncSystemSections(legacy, desired); + expect(getCurrentSystemPrompt(upgraded)).not.toContain("Old full catalog"); + const added = syncSystemSections(upgraded, { ...desired, "skill:a": "A" }); + expect(systemPromptContent(added)).toBe("Base\n\nLoad skills\n\nA\n\nZ\n\nContext"); + const restored = [systemTranscriptCheckpoint(added)!]; + expect(syncSystemSections(restored, { ...desired, "skill:a": "A" })).toBe(restored); + expect(systemPromptContent(restored)).toBe(systemPromptContent(added)); + const replaced = replaceSystemPrompt(restored, "Temporary prompt"); + expect(getCurrentSystemPrompt(replaced)).toBe("Temporary prompt"); + }); + it("replays sections and tool removals without giving an unchanged prefix a new timestamp", () => { const previous = [initial, delta, assistant, user]; const rebuilt = rebuildSystemTranscript(previous, [assistant, user]); expect(getCurrentSystemPrompt(rebuilt)).toBe(getCurrentSystemPrompt(previous)); - expect(rebuilt[0]).toMatchObject({ - content: "Base instructions\n\nAdditional instructions", timestamp: 1_500, - sections: { rules: "Updated rules" }, toolsAdded: [read], + expect(rebuilt).toEqual(previous); + expect(systemTranscriptCheckpoint(rebuilt)).toMatchObject({ + content: "", timestamp: 1_500, + sections: { runtime: "Base instructions", rules: "Updated rules" }, toolsAdded: [read], }); expect(getCurrentSystemMessage(rebuilt)?.sections).not.toHaveProperty("obsolete"); expect(getCurrentTools(rebuilt)).toEqual([read]); @@ -63,18 +97,17 @@ describe("system transcript helpers", () => { expect(rebuildSystemTranscript(rebuilt, [assistant, user])[0]).toBe(rebuilt[0]); }); - it("does not resurrect usage predating a folded system delta", () => { + it("keeps an earlier usage anchor and estimates an appended delta", () => { const newer = { ...delta, timestamp: 3_000 }; const rebuilt = rebuildSystemTranscript([initial, assistant, newer], [assistant]); - expect(rebuilt[0]?.timestamp).toBe(3_000); - expect(estimateContextTokens(rebuilt).lastUsageIndex).toBeNull(); + expect(rebuilt.at(-1)?.timestamp).toBe(3_000); + expect(estimateContextTokens(rebuilt).trailingTokens).toBeGreaterThan(0); }); it("keeps a complete recovery transcript and its deltas in place", () => { const previous = [initial, assistant, delta, user, { ...assistant, stopReason: "error" as const }]; const retained = previous.slice(0, -1); expect(rebuildSystemTranscript(previous, retained)).toBe(retained); - expect(syncSystemTools(retained, [read])).toBe(retained); }); it("preserves object identity when setting either the same content or rendered prompt", () => { @@ -83,14 +116,14 @@ describe("system transcript helpers", () => { expect(replaceSystemPrompt(messages, systemPromptContent(messages))).toBe(messages); }); - it("invalidates usage only on a real content change and preserves structured sections and tools", () => { + it("accounts for a real content change and preserves structured sections and tools", () => { vi.spyOn(Date, "now").mockReturnValue(5_000); const changed = replaceSystemPrompt([initial, delta, assistant], "Replacement instructions"); - expect(changed[0]).toMatchObject({ content: "Replacement instructions", timestamp: 5_000, sections: { rules: "Updated rules" } }); + expect(changed.at(-1)).toMatchObject({ content: "", timestamp: 5_000, sections: { runtime: "Replacement instructions" } }); expect(getCurrentTools(changed)).toEqual([read]); - expect(estimateContextTokens(changed).lastUsageIndex).toBeNull(); + expect(estimateContextTokens(changed).trailingTokens).toBeGreaterThan(0); expect(replaceSystemPrompt(changed, "Replacement instructions")).toBe(changed); - expect(initial.content).toBe("Base instructions"); + expect(initial.sections?.runtime).toBe("Base instructions"); const response = { ...assistant, timestamp: 6_000 }; expect(estimateContextTokens(rebuildSystemTranscript(changed, [assistant, response])).usageTokens).toBe(1_000); }); @@ -98,7 +131,7 @@ describe("system transcript helpers", () => { it("can clear content without clearing sections or tool declarations", () => { const changed = replaceSystemPrompt([initial, delta, assistant], ""); expect(systemPromptContent(changed)).toBe(""); - expect(getCurrentSystemMessage(changed)?.sections).toEqual({ rules: "Updated rules" }); + expect(getCurrentSystemMessage(changed)?.sections).toEqual({ runtime: "", rules: "Updated rules" }); expect(getCurrentTools(changed)).toEqual([read]); }); @@ -110,55 +143,32 @@ describe("system transcript helpers", () => { const response = { ...assistant, timestamp: 6_000 }; const restored = replaceSystemPrompt([...nudged, response], before); expect(getCurrentSystemPrompt(restored)).toBe(getCurrentSystemPrompt(messages)); - expect(getCurrentSystemMessage(restored)?.sections).toEqual({ rules: "Updated rules" }); - expect(restored[0]?.timestamp).toBeGreaterThan(response.timestamp); - expect(estimateContextTokens(restored).usageTokens).toBe(0); - }); - - it("records tool additions, removals, and same-name schema replacements with fresh semantic time", () => { - vi.spyOn(Date, "now").mockReturnValue(5_000); - const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; - const messages = [initial, assistant]; - const changed = syncSystemTools(messages, [replacement]); - expect(changed[1]).toMatchObject({ timestamp: 5_000, toolsAdded: [replacement], toolsRemoved: [{ name: "Read" }, { name: "Edit" }] }); - expect(getCurrentTools(changed)).toEqual([replacement]); - expect(estimateContextTokens(changed).usageTokens).toBe(0); - expect(syncSystemTools(changed, [replacement])).toBe(changed); - const response = { ...assistant, timestamp: 6_000 }; - expect(estimateContextTokens(rebuildSystemTranscript(changed, [assistant, response])).usageTokens).toBe(1_000); - const cleared = syncSystemTools(changed, []); - expect(getCurrentTools(rebuildSystemTranscript(cleared, [assistant]))).toEqual([]); - }); - - it("ignores executable-only tool changes instead of invalidating usage", () => { - const messages = [initial, assistant]; - const executable = { ...read, label: "Read", execute: vi.fn() }; - expect(syncSystemTools(messages, [executable, edit])).toBe(messages); - expect(estimateContextTokens(messages).usageTokens).toBe(1_000); + expect(getCurrentSystemMessage(restored)?.sections).toEqual({ runtime: "Base instructions", rules: "Updated rules" }); + expect(restored.at(-1)?.timestamp).toBeGreaterThan(response.timestamp); + expect(estimateContextTokens(restored).trailingTokens).toBeGreaterThan(0); }); it("advances semantic time even if the clock shares a millisecond or moves backwards", () => { vi.spyOn(Date, "now").mockReturnValue(2_000); - expect(replaceSystemPrompt([initial, assistant], "changed")[0]?.timestamp).toBe(2_001); + expect(replaceSystemPrompt([initial, assistant], "changed").at(-1)?.timestamp).toBe(2_001); vi.spyOn(Date, "now").mockReturnValue(500); - expect(syncSystemTools([initial, assistant], [read])[1]?.timestamp).toBe(2_001); + expect(replaceSystemPrompt([initial, assistant], "another change").at(-1)?.timestamp).toBe(2_001); }); it("initializes restored history conservatively without inventing an old timestamp", () => { vi.spyOn(Date, "now").mockReturnValue(5_000); const messages = initialSystemTranscript("Current config", [read], [assistant]); - expect(messages[0]).toMatchObject({ content: "Current config", toolsAdded: [read], timestamp: 5_000 }); - expect(estimateContextTokens(messages).usageTokens).toBe(0); + expect(messages.at(-1)).toMatchObject({ sections: { runtime: "Current config" }, toolsAdded: [read], timestamp: 5_000 }); + expect(estimateContextTokens(messages).trailingTokens).toBeGreaterThan(0); const rebuilt = rebuildSystemTranscript(messages, [assistant]); expect(rebuilt[0]).toBe(messages[0]); - expect(estimateContextTokens(rebuilt).usageTokens).toBe(0); + expect(estimateContextTokens(rebuilt).trailingTokens).toBeGreaterThan(0); }); it("preserves the empty transcript when there is no system state", () => { const messages: AgentMessage[] = []; expect(initialSystemTranscript("", [], messages)).toBe(messages); expect(rebuildSystemTranscript([], messages)).toBe(messages); - expect(syncSystemTools(messages, [])).toBe(messages); expect(replaceSystemPrompt(messages, "")).toBe(messages); }); @@ -176,13 +186,13 @@ describe("system transcript helpers", () => { return stream; }, }); - agent.state.messages = syncSystemTools(rebuildSystemTranscript(agent.state.messages, [user]), [tool]); - const prefix = agent.state.messages[0]; + agent.state.messages = rebuildSystemTranscript(agent.state.messages, [user]); + const prefix = agent.state.messages.filter((message) => message.role === "system"); await agent.continue(); expect(agent.state.errorMessage).toBeUndefined(); expect(requests).toHaveLength(1); - expect(requests[0]?.filter((message) => message.role === "system")).toEqual([prefix]); + expect(requests[0]?.filter((message) => message.role === "system")).toEqual(prefix); expect(getCurrentTools(requests[0]!)).toEqual([toToolDeclaration(tool)]); - expect(agent.state.messages.filter((message) => message.role === "system")).toEqual([prefix]); + expect(agent.state.messages.filter((message) => message.role === "system")).toEqual(prefix); }); }); diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index 28bd5028b..a5bac9363 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -1,22 +1,21 @@ import type { AgentMessage } from "@earendil-works/pi-agent-core"; import { contentText, - createInitialSystemMessage, getCurrentSystemMessage, getCurrentSystemPrompt, - getCurrentTools, - getToolStateChanges, type SystemMessage, type Tool, toToolDeclaration, } from "@earendil-works/pi-ai"; +import { SKILL_SECTION_PREFIX } from "./plugin-skills-prompt.js"; + function systemMessages(messages: readonly AgentMessage[]): SystemMessage[] { return messages.filter((message): message is SystemMessage => message.role === "system"); } function nextSystemTimestamp(messages: readonly AgentMessage[]): number { - // 同一毫秒内也必须晚于旧响应;时钟回拨时只前移,不伪造早于 usage 的时间。 + // Never backdate a state change relative to the response it invalidates. return messages.reduce((timestamp, message) => Math.max(timestamp, message.timestamp + 1), Date.now()); } @@ -24,71 +23,102 @@ function currentSystemMessage(messages: readonly AgentMessage[]): SystemMessage const systems = systemMessages(messages); if (systems.length < 2) return systems[0]; const current = getCurrentSystemMessage(systems)!; - // 上游折叠器保留第一条的时间,但 sections/工具删除可能来自较晚的 delta。 - // 快照的语义时间必须覆盖所有贡献者,否则旧 usage 会被错误地重新激活。 + // The upstream fold retains the first timestamp. A checkpoint must cover + // every contributing update so old usage is not accidentally reactivated. return { ...current, timestamp: systems.reduce((timestamp, message) => Math.max(timestamp, message.timestamp), systems[0]!.timestamp), }; } -/** The unrendered content, without flattening named sections into the prompt. */ +/** Desktop-owned sections retain their order ahead of extension sections. */ +export const CONTEXT_BUDGET_SECTION = "context_budget"; + +function desktopSectionNames(sections: Record): string[] { + return ["runtime", "skills", ...Object.keys(sections) + .filter((name) => name.startsWith(SKILL_SECTION_PREFIX)) + .sort((a, b) => a.localeCompare(b)), "context"]; +} + export function systemPromptContent(messages: readonly AgentMessage[]): string { - return contentText(getCurrentSystemMessage(messages)?.content ?? ""); + const current = getCurrentSystemMessage(messages); + const sections = current?.sections; + if (sections && desktopSectionNames(sections).some((name) => name in sections)) { + return desktopSectionNames(sections).map((name) => sections[name]).filter(Boolean).join("\n\n"); + } + return contentText(current?.content ?? ""); +} + +export function syncSystemSections( + messages: AgentMessage[], + desired: Record, +): AgentMessage[] { + const current = getCurrentSystemMessage(messages)?.sections ?? {}; + const sections: Record = {}; + for (const name of desktopSectionNames({ ...current, ...desired })) { + const next = desired[name] ?? null; + if ((current[name] ?? null) !== next) sections[name] = next; + } + if (Object.keys(sections).length === 0) return messages; + return [...messages, { role: "system", content: "", sections, timestamp: nextSystemTimestamp(messages) }]; } export function initialSystemTranscript( prompt: string, tools: readonly Tool[], messages: AgentMessage[], + sections: Record = { runtime: prompt }, ): AgentMessage[] { - const system = createInitialSystemMessage(prompt, tools.map(toToolDeclaration)); - // 持久化历史不记录 system 状态,重启后无法证明新配置与旧请求相同。 - // 保守使用实际初始化时间;不能沿用上游初始值 0 来让历史 usage 假装有效。 - return system - ? [{ ...system, timestamp: nextSystemTimestamp(messages) }, ...messages] - : messages; + if (messages.some((message) => message.role === "system")) { + return syncSystemSections(messages, sections); + } + if (!prompt && tools.length === 0) return messages; + // Old sessions have no recorded baseline. Declare the current state at the + // continuation boundary rather than inventing past instructions or tools. + return [...messages, { + role: "system", content: "", sections, + toolsAdded: tools.map(toToolDeclaration), timestamp: nextSystemTimestamp(messages), + }]; } export function replaceSystemPrompt(messages: AgentMessage[], prompt: string): AgentMessage[] { - const current = currentSystemMessage(messages); - if (prompt === getCurrentSystemPrompt(messages) || prompt === contentText(current?.content ?? "")) { - return messages; - } - // prompt 只替换内容;命名 sections 与最终工具状态仍由上游 replay 负责。 - // recovery 读写未渲染内容,避免把 sections 再嵌入 content 造成重复。 - return [ - { ...current, role: "system", content: prompt, timestamp: nextSystemTimestamp(messages) }, - ...messages.filter((message) => message.role !== "system"), - ]; + if (prompt === getCurrentSystemPrompt(messages) || prompt === systemPromptContent(messages)) return messages; + return syncSystemSections(messages, { runtime: prompt }); } export function rebuildSystemTranscript( previous: readonly AgentMessage[], messages: AgentMessage[], ): AgentMessage[] { - // recovery 传入完整 live transcript 的切片,保留其位置、对象及 delta,不重复加前缀。 - if (messages.some((message) => message.role === "system")) return messages; - // durable projection 不含 system;用上游 replay 还原有效 sections 和工具集合, - // 而不是将渲染后的 systemPrompt 当成新消息。折叠不产生新的语义时间。 - const system = currentSystemMessage(previous); - return system ? [system, ...messages] : messages; + let result = messages; + for (let i = 0; i < previous.length; i++) { + const message = previous[i]; + if (message.role !== "system" || result.includes(message)) continue; + const following = previous.slice(i + 1).find((item) => result.includes(item)); + const preceding = previous.slice(0, i).reverse().find((item) => result.includes(item)); + let position = following ? result.indexOf(following) : preceding ? result.indexOf(preceding) + 1 : result.length; + if (!following) while (result[position]?.role === "system") position++; + if (result === messages) result = [...messages]; + result.splice(position, 0, message); + } + return result; +} + +/** Fold state only at an explicit compaction boundary, with semantic time. */ +export function systemTranscriptCheckpoint(messages: readonly AgentMessage[]): SystemMessage | undefined { + const checkpoint = currentSystemMessage(messages); + if (!checkpoint?.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; + const { [CONTEXT_BUDGET_SECTION]: _expired, ...sections } = checkpoint.sections; + return { ...checkpoint, sections }; } -export function syncSystemTools(messages: AgentMessage[], tools: readonly Tool[]): AgentMessage[] { - const changes = getToolStateChanges(getCurrentTools(messages), tools); - if (changes.toolsAdded.length === 0 && changes.toolsRemoved.length === 0) return messages; - // 真正的工具变化成为较新的前缀,旧 usage 因此失效。保留删除/同名替换的 delta, - // 并让声明集合与可执行 catalog 一致,避免 agent-loop 下一轮再次自动追加同一变化。 - return [ - ...systemMessages(messages), - { - role: "system", - content: "", - ...(changes.toolsAdded.length ? { toolsAdded: changes.toolsAdded } : {}), - ...(changes.toolsRemoved.length ? { toolsRemoved: changes.toolsRemoved } : {}), - timestamp: nextSystemTimestamp(messages), - }, - ...messages.filter((message) => message.role !== "system"), - ]; +/** Recovery discards failed trailing responses even after a prompt cleanup delta. */ +export function removeTrailingAssistantMessages(messages: readonly AgentMessage[]): AgentMessage[] { + const result = [...messages]; + for (let index = result.length - 1; index >= 0; index--) { + if (result[index].role === "system") continue; + if (result[index].role !== "assistant") break; + result.splice(index, 1); + } + return result; } diff --git a/packages/agent-runtime/src/thinking-level.ts b/packages/agent-runtime/src/thinking-level.ts index 509bbd316..f386bbe80 100644 --- a/packages/agent-runtime/src/thinking-level.ts +++ b/packages/agent-runtime/src/thinking-level.ts @@ -23,6 +23,8 @@ export type ThinkingCapabilitySet = { */ export type ModelConfig = { source: "pi" | "models.dev" | "generic"; + /** Published identity before account endpoint or wire-model overrides. */ + transcriptBinding?: { modelId: string; api: string; baseUrl: string }; inputLimits?: Model["inputLimits"]; promptCache?: Model["promptCache"]; samplingParams?: Model["samplingParams"]; diff --git a/packages/agent-runtime/src/transcript-compat.test.ts b/packages/agent-runtime/src/transcript-compat.test.ts new file mode 100644 index 000000000..86fb4c2c6 --- /dev/null +++ b/packages/agent-runtime/src/transcript-compat.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from "vitest"; +import { genericModelConfig } from "./model-capabilities.js"; +import { buildProviderModel, type RuntimeProviderConfig } from "./provider-binding.js"; + +const baseUrl = "https://native.test/v1"; +const provider: RuntimeProviderConfig = { + id: "account", name: "Account", modelId: "exact-model", baseUrl, apiStyle: "openai_completions", + apiKey: "", authKind: "none", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: { + ...genericModelConfig("exact-model", baseUrl), source: "pi", + transcriptBinding: { modelId: "exact-model", api: "openai-completions", baseUrl }, + compat: { supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }, + }, +}; + +describe("transcript compatibility binding", () => { + it("keeps instruction and tool capabilities independent on the exact binding", () => { + expect(buildProviderModel(provider).compat).toMatchObject({ supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }); + expect(buildProviderModel({ ...provider, baseUrl: `${baseUrl}/` }).compat).toMatchObject({ supportsMidConvoSystemMessages: true }); + }); + it.each([ + { baseUrl: "https://relay.test/v1" }, { baseUrl: "https://native.test/other" }, + { baseUrl: "https://native.test:8443/v1" }, { modelId: "alias" }, { apiStyle: "responses" }, + { modelConfig: { ...provider.modelConfig!, transcriptBinding: undefined } }, + ])("falls back for an unverified route or model: %j", (overrides) => { + expect(buildProviderModel({ ...provider, ...overrides }).compat).toMatchObject({ + supportsMidConvoSystemMessages: false, supportsMidConvoToolAdditions: false, + supportsMidConvoToolChanges: false, supportsAdditionalTools: false, + }); + }); +}); diff --git a/packages/agent-runtime/src/transcript-compat.ts b/packages/agent-runtime/src/transcript-compat.ts new file mode 100644 index 000000000..365e2f6da --- /dev/null +++ b/packages/agent-runtime/src/transcript-compat.ts @@ -0,0 +1,28 @@ +import type { ModelConfig } from "./thinking-level.js"; + +const capabilities = [ + "supportsMidConvoSystemMessages", "supportsMidConvoToolAdditions", + "supportsMidConvoToolChanges", "supportsAdditionalTools", "supportsToolSearch", +] as const; + +function endpoint(value: string): string | undefined { + try { + const url = new URL(value); + if (url.username || url.password || url.search || url.hash) return undefined; + return `${url.origin}${url.pathname.replace(/\/+$/, "")}`; + } catch { + return undefined; + } +} + +/** Catalog capabilities describe one model/API/endpoint, never a vendor label. */ +export function transcriptCompat( + catalog: ModelConfig | undefined, modelId: string, api: string, baseUrl: string, +): Record<(typeof capabilities)[number], boolean> { + const binding = catalog?.transcriptBinding; + const address = endpoint(baseUrl); + const verified = catalog?.source === "pi" && binding?.modelId === modelId && + binding.api === api && address !== undefined && endpoint(binding.baseUrl) === address; + return Object.fromEntries(capabilities.map((key) => [key, verified && catalog?.compat?.[key] === true])) as + Record<(typeof capabilities)[number], boolean>; +} diff --git a/packages/agent-runtime/src/transcript-payload.test.ts b/packages/agent-runtime/src/transcript-payload.test.ts new file mode 100644 index 000000000..0cd77f8b2 --- /dev/null +++ b/packages/agent-runtime/src/transcript-payload.test.ts @@ -0,0 +1,87 @@ +import { expect, it } from "vitest"; +import { normalizeContext, Type, type Message, type Model } from "@earendil-works/pi-ai"; +import { stream } from "@earendil-works/pi-ai/api/openai-completions"; +import { genericModelConfig } from "./model-capabilities.js"; +import { buildProviderModel } from "./provider-binding.js"; + +const baseUrl = "https://native.invalid/v1"; +const read = { name: "Read", description: "Read", parameters: Type.Object({ path: Type.String() }) }; +const search = { ...read, name: "Search" }; +const messages: Message[] = [ + { role: "system", content: "", sections: { runtime: "Base", skills: "Old catalog" }, toolsAdded: [read], timestamp: 1 }, + { role: "user", content: "Find files", timestamp: 2 }, + { role: "system", content: "", sections: { skills: "New catalog" }, toolsAdded: [search], timestamp: 3 }, +]; +type Payload = { messages: Array<{ role: string; content?: unknown; tools?: unknown[] }>; tools?: Array<{ function: { name: string; parameters: unknown } }> }; +async function payload(instructions: boolean, additions: boolean, route = baseUrl, transcript = messages): Promise { + const model = buildProviderModel({ + id: "fixture", name: "Fixture", modelId: "exact", baseUrl: route, apiKey: "", authKind: "none", + supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: { + ...genericModelConfig("exact", baseUrl), source: "pi", + transcriptBinding: { modelId: "exact", api: "openai-completions", baseUrl }, + compat: { supportsMidConvoSystemMessages: instructions, supportsMidConvoToolAdditions: additions }, + }, + }); + let captured: Payload | undefined; + const response = stream(model as Model<"openai-completions">, normalizeContext({ messages: transcript }), { + apiKey: "fixture", + fetch: async (_url, init) => { + captured = JSON.parse(String(init?.body)); + return new Response('data: {"choices":[{"index":0,"delta":{"role":"assistant","content":"Done"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n', { headers: { "content-type": "text/event-stream" } }); + }, + }); + expect((await response.result()).stopReason).toBe("stop"); + if (!captured) throw new Error("No payload captured"); + return captured; +} + +it("uses native additions only when both instruction and tool capabilities are verified", async () => { + const original = structuredClone(messages); + const native = await payload(true, true); + expect(native.tools?.map((tool) => tool.function.name)).toEqual(["Read"]); + expect(native.messages.slice(0, 2).map((message) => message.role)).toEqual(["system", "user"]); + expect(native.messages.slice(2).some((message) => message.tools?.length === 1)).toBe(true); + const partial = await payload(true, false); + expect(partial.tools?.map((tool) => tool.function.name)).toEqual(["Read", "Search"]); + expect(partial.messages.filter((message) => message.role === "system")).toHaveLength(2); + const relay = await payload(true, true, "https://relay.invalid/v1"); + expect(relay.messages.map((message) => message.role)).toEqual(["system", "user"]); + expect(relay.tools?.map((tool) => tool.function.name)).toEqual(["Read", "Search"]); + expect(JSON.stringify(relay.messages)).toContain("New catalog"); + expect(JSON.stringify(relay.messages)).not.toContain("Old catalog"); + expect(await payload(true, true)).toEqual(native); + expect(messages).toEqual(original); +}); + +it("folds removals and same-name replacements into the active request tool set", async () => { + const replacement = { ...read, parameters: Type.Object({ file: Type.String() }) }; + const request = await payload(true, true, baseUrl, [...messages, { + role: "system", content: "", timestamp: 4, + toolsRemoved: [{ name: "Read" }, { name: "Search" }], toolsAdded: [replacement], + }]); + expect(request.tools).toEqual([expect.objectContaining({ function: expect.objectContaining({ name: "Read", parameters: replacement.parameters }) })]); + expect(request.messages.some((message) => message.tools)).toBe(false); +}); + +it("projects per-skill changes and revocations with native and fallback model support", async () => { + const catalog: Message[] = [ + { role: "system", content: "", timestamp: 1, sections: { + runtime: "Base", skills: "Load skills", "skill:a": "Alpha", "skill:b": "Bravo", "skill:c": "Charlie", + } }, + { role: "user", content: "Continue", timestamp: 2 }, + { role: "system", content: "", timestamp: 3, sections: { "skill:a": "Alpha updated", "skill:b": null } }, + ]; + const before = await payload(true, false, baseUrl, catalog.slice(0, 2)); + const native = await payload(true, false, baseUrl, catalog); + expect(native.messages.slice(0, before.messages.length)).toEqual(before.messages); + expect(JSON.stringify(native.messages.at(-1))).toContain("Alpha updated"); + expect(JSON.stringify(native.messages.at(-1))).not.toContain("Charlie"); + for (const route of [baseUrl, "https://relay.invalid/v1"]) { + const fallback = await payload(false, false, route, catalog); + expect(fallback.messages.map((message) => message.role)).toEqual(["system", "user"]); + expect(JSON.stringify(fallback.messages)).toContain("Alpha updated"); + expect(JSON.stringify(fallback.messages)).toContain("Charlie"); + expect(JSON.stringify(fallback.messages)).not.toContain("Bravo"); + } +}); diff --git a/packages/shared/src/types/messages.ts b/packages/shared/src/types/messages.ts index 361a476c9..f34738a16 100644 --- a/packages/shared/src/types/messages.ts +++ b/packages/shared/src/types/messages.ts @@ -70,6 +70,15 @@ export type UiMessage = { id: string; role: UiMessageRole; content: string; + /** Internal model instructions/tool declarations; never a visible chat row. */ + modelSystem?: { + version: 1; + /** The following user input can have been persisted before runtime admission. */ + beforeMessageId?: string; + afterMessageId?: string; + /** Opaque JSON preserves section and schema key order across Host storage. */ + messageJson: string; + }; /** Authenticated agent-to-agent provenance; never inferred from message text. */ sessionMessage?: SessionMessageOrigin; /** Present only on the durable user row created by a Live Voice operation. */ diff --git a/patches/@earendil-works__pi-ai@0.99.1.patch b/patches/@earendil-works__pi-ai@0.99.1.patch index df5da3d0d..a6db0ceff 100644 --- a/patches/@earendil-works__pi-ai@0.99.1.patch +++ b/patches/@earendil-works__pi-ai@0.99.1.patch @@ -1613,3 +1613,10 @@ index 000000000..5c63b882a + return stream; + } +} +diff --git a/dist/providers/data/deepseek.json b/dist/providers/data/deepseek.json +--- a/dist/providers/data/deepseek.json ++++ b/dist/providers/data/deepseek.json +@@ -1 +1 @@ +-{"openai-completions":{"chat:deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"thinkingLevelMap":{"minimal":null,"low":"low","medium":null,"high":"high","max":"max"},"input":["text","image"],"cost":{"input":0.3,"output":1.2,"cacheRead":0.006,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true},"inputLimits":{"images":{"resize":{"maxWidth":2000,"maxHeight":2000,"maxBytes":4718592,"jpegQuality":80}}},"type":"chat"},"chat:deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"input":["text"],"cost":{"input":1.32,"output":3.96,"cacheRead":0.044,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"thinkingLevelMap":{"minimal":null,"low":null,"medium":null,"high":"high","max":"max"},"type":"chat"}}} ++{"openai-completions":{"chat:deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"thinkingLevelMap":{"minimal":null,"low":"low","medium":null,"high":"high","max":"max"},"input":["text","image"],"cost":{"input":0.3,"output":1.2,"cacheRead":0.006,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"inputLimits":{"images":{"resize":{"maxWidth":2000,"maxHeight":2000,"maxBytes":4718592,"jpegQuality":80}}},"type":"chat"},"chat:deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","api":"openai-completions","baseUrl":"https://api.deepseek.com","provider":"deepseek","reasoning":true,"input":["text"],"cost":{"input":1.32,"output":3.96,"cacheRead":0.044,"cacheWrite":0},"contextWindow":1000000,"maxTokens":384000,"compat":{"supportsStore":false,"supportsDeveloperRole":false,"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek","supportsStrictMode":true,"supportsMidConvoSystemMessages":true},"thinkingLevelMap":{"minimal":null,"low":null,"medium":null,"high":"high","max":"max"},"type":"chat"}}} +\ No newline at end of file diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 4010b26e9..b984698f6 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -178,7 +178,7 @@ patchedDependencies: hash: b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7 path: patches/@earendil-works__pi-agent-core@0.99.1.patch '@earendil-works/pi-ai@0.99.1': - hash: c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4 + hash: d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4 path: patches/@earendil-works__pi-ai@0.99.1.patch '@earendil-works/pi-coding-agent@0.99.1': hash: 0e05051778ef4ffca9bac7ec68cae64f1d659d4b8640eeeb1a121e0f620a6abe @@ -209,7 +209,7 @@ importers: devDependencies: '@earendil-works/pi-ai': specifier: 0.99.1 - version: 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3) + version: 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-mcp': specifier: 0.99.1 version: 0.99.1 @@ -404,7 +404,7 @@ importers: version: 0.99.1(patch_hash=b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-ai': specifier: 0.99.1 - version: 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + version: 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-coding-agent': specifier: 0.99.1 version: 0.99.1(patch_hash=0e05051778ef4ffca9bac7ec68cae64f1d659d4b8640eeeb1a121e0f620a6abe)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(ws@8.21.1)(zod@4.4.3) @@ -5310,7 +5310,7 @@ snapshots: '@earendil-works/pi-agent-core@0.99.1(patch_hash=b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': dependencies: '@earendil-works/chord': 0.99.1 - '@earendil-works/pi-ai': 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + '@earendil-works/pi-ai': 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-telemetry': 0.99.1 diff: 8.0.4 ignore: 7.0.8 @@ -5329,7 +5329,7 @@ snapshots: - ws - zod - '@earendil-works/pi-ai@0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3)': + '@earendil-works/pi-ai@0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@6.28.0)(ws@8.21.1)(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.124.0(zod@4.4.3) '@aws-sdk/client-bedrock-runtime': 3.1127.0 @@ -5354,7 +5354,7 @@ snapshots: - ws - zod - '@earendil-works/pi-ai@0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': + '@earendil-works/pi-ai@0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3)': dependencies: '@anthropic-ai/sdk': 0.124.0(zod@4.4.3) '@aws-sdk/client-bedrock-runtime': 3.1127.0 @@ -5387,7 +5387,7 @@ snapshots: dependencies: '@earendil-works/chord': 0.99.1 '@earendil-works/pi-agent-core': 0.99.1(patch_hash=b98bc9f5653147d34650f4be892612c7b34f7c553a58cb70457cf900310347d7)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) - '@earendil-works/pi-ai': 0.99.1(patch_hash=c9e66480f4eefd4a5d3ac998d3b65bf761fb22a17a238cb44c76d31448454ec4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) + '@earendil-works/pi-ai': 0.99.1(patch_hash=d146cb1c4386252dd01f522fab0c388868fed0c86d230b4c34e5fd28a8bfc6e4)(@aws-sdk/credential-provider-node@3.972.83)(@smithy/signature-v4@5.7.3)(undici@8.10.2)(ws@8.21.1)(zod@4.4.3) '@earendil-works/pi-codemode': 0.99.1 '@earendil-works/pi-mcp': 0.99.1 '@earendil-works/pi-tui': 0.99.1 diff --git a/scripts/e2e-system-transcript.mjs b/scripts/e2e-system-transcript.mjs new file mode 100644 index 000000000..6d4139152 --- /dev/null +++ b/scripts/e2e-system-transcript.mjs @@ -0,0 +1,172 @@ +#!/usr/bin/env node +/** Production sidecar transport + isolated Host + local SSE provider. + * Message persistence is harness-owned; Electron's UI/outbox is not exercised. + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, rm } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +import { PROTOCOL_VERSION } from "../packages/shared/dist/protocol.js"; +import { withScenario } from "./e2e/fixture.mjs"; +import { resolveHostBinary } from "./e2e/host.mjs"; +import { createSession } from "./e2e/session.mjs"; + +const root = await mkdtemp(join(tmpdir(), "pi-system-state-e2e-")); +const requests = []; +const responses = []; +const server = createServer(async (req, res) => { + let body = ""; + for await (const chunk of req) body += chunk; + requests.push(JSON.parse(body)); + const tool = responses.shift(); + const delta = tool ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), type: "function", + function: { name: tool.name, arguments: JSON.stringify(tool.args) } }] } + : { role: "assistant", content: "The fixture completed." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + for (const [value, finish] of [[delta, null], [{}, tool ? "tool_calls" : "stop"]]) { + res.write(`data: ${JSON.stringify({ id: "fixture", model: "fixture", choices: [{ index: 0, delta: value, finish_reason: finish }] })}\n\n`); + } + res.end("data: [DONE]\n\n"); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; +const provider = { id: "fixture", name: "Fixture", modelId: "fixture", baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; +let skills = [ + { id: "fixture/notes", name: "First skill", description: "Summarize notes" }, + { id: "fixture/unchanged", name: "Unchanged skill", description: "Preserve this catalog entry" }, +]; +try { + await withScenario("E2E-system-transcript", async ({ host, workspace }) => { + const session = await createSession(host, workspace, "System state fixture"); + let writes = Promise.resolve(); + let writeError; + let stderr = ""; + const turns = new Map(); + const executedTools = []; + const toolRows = new Map(); + const persist = (message, turnId) => { + writes = writes.then(() => host.call("session.appendMessage", { sessionId: session.id, turnId, message })) + .catch((error) => { writeError ??= error; }); + }; + const startSidecar = () => { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, + args: [fileURLToPath(new URL("../packages/agent-runtime/dist/sidecar.js", import.meta.url))], + cwd: workspace, + env: { ...process.env, PI_DESKTOP_PLAN_UI_PROBE: "1", PI_DESKTOP_COMPACTION_STRATEGY: "fresh_window" }, + }, + onStderr: (chunk) => { stderr += chunk; }, + }); + sidecar.setHost({ call: host.call.bind(host), onNotification: () => () => {}, onExit: () => () => {} }); + sidecar.setLocalTool("Skill", async ({ args }) => { + executedTools.push(args); + return { ok: true, content: "Fixture skill body: summarize the notes." }; + }); + sidecar.onNotification((method, envelope) => { + if (method !== "agent.event") return; + const { event, turnId } = envelope; + const turn = turns.get(turnId); + if (event.type === "error") turn?.errors.push(event.error); + if (event.type === "message_end") persist(event.message, turnId); + if (event.type === "tool_start") toolRows.set(event.toolCallId, { + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + turn?.toolResults.push(event); + persist({ ...toolRows.get(event.toolCallId), toolResult: event.result, isError: event.isError, + toolStatus: event.isError ? "error" : "success" }, turnId); + } + if (event.type === "agent_end") turn?.resolve(); + }); + return sidecar; + }; + let sidecar = startSidecar(); + const params = () => ({ sessionId: session.id, mode: "agent", provider, thinkingLevel: "off", pluginSkills: skills, + projectPath: workspace, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + }); + const identity = async () => (await sidecar.call("agent.testRuntimeIdentity", { sessionId: session.id })).runtimeId; + const prompt = async (id) => { + const { turnId } = await host.call("session.beginTurn", { sessionId: session.id }); + await host.call("session.appendMessage", { sessionId: session.id, turnId, + message: { id, role: "user", content: "Continue", createdAt: new Date().toISOString() } }); + let timer; + const turn = { errors: [], toolResults: [] }; + const ended = new Promise((resolve, reject) => { + turn.resolve = resolve; + timer = setTimeout(() => reject(new Error(`Turn timed out: ${JSON.stringify(turn.errors)}\n${stderr}`)), 20_000); + }); + turns.set(turnId, turn); + try { + await sidecar.call("agent.prompt", { ...params(), turnId, userMessageId: id, content: "Continue" }); + await ended; + await writes; + if (writeError) throw writeError; + assert.deepEqual(turn.errors, [], JSON.stringify(turn.errors)); + assert(turn.toolResults.every((event) => !event.isError), JSON.stringify(turn.toolResults)); + await host.call("session.endTurn", { turnId, status: "completed", createNotification: false }); + return turn; + } finally { clearTimeout(timer); turns.delete(turnId); } + }; + const detail = async () => (await host.call("session.get", { id: session.id })).session; + try { + responses.push({ name: "ToolSearch", args: { query: "BrowserPreview" } }); + await prompt("user-1"); + const saved = await detail(); + const states = saved.messages.filter((row) => row.modelSystem); + assert.equal(states.length, 2); + assert(JSON.parse(states[1].modelSystem.messageJson).toolsAdded.some((tool) => tool.name === "BrowserPreview")); + assert(requests[1].tools.some((tool) => tool.function.name === "BrowserPreview")); + console.log("PASS: sidecar ToolSearch and acknowledged Host persistence"); + await sidecar.dispose(); + await host.stop(); await host.start(PROTOCOL_VERSION); + const recovered = await detail(); + assert.deepEqual(recovered.messages.filter((row) => row.modelSystem), states); + sidecar = startSidecar(); + await prompt("user-2"); + const resumedStates = (await detail()).messages.filter((row) => row.modelSystem); + assert.equal(resumedStates.length, 2, JSON.stringify(resumedStates.slice(2).map((row) => row.modelSystem.messageJson))); + assert(requests.at(-1).tools.some((tool) => tool.function.name === "BrowserPreview")); + console.log("PASS: Host and sidecar process restart restores tools without duplicate state"); + const priorRuntime = await identity(); + skills = [{ ...skills[0], name: "Updated skill" }, skills[1]]; + responses.push({ name: "Skill", args: { id: skills[0].id } }); + await prompt("user-3"); + assert.equal(await identity(), priorRuntime, "skill catalog update must reuse the runtime"); + assert.equal(executedTools.length, 1, "Skill execution reaches the embedding host"); + assert(JSON.stringify(requests.at(-1).messages).includes("Fixture skill body")); + assert(JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("First skill")); + assert(JSON.stringify(requests.at(-1).messages).includes("Unchanged skill")); + console.log("PASS: skill catalog update reuses runtime and Skill executes across process boundary"); + await sidecar.call("agent.compact", params()); + const compacted = await detail(); + assert(JSON.parse(compacted.compaction.details.systemMessageJson).toolsAdded.some((tool) => tool.name === "BrowserPreview")); + await sidecar.dispose(); + sidecar = startSidecar(); + await prompt("user-4"); + assert(requests.at(-1).tools.some((tool) => tool.function.name === "BrowserPreview")); + assert(!requests.at(-1).messages.some((message) => message.role === "tool")); + assert(JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("First skill")); + console.log("PASS: compaction and sidecar restart preserve current skills and active tools"); + const beforeRemoval = await identity(); + skills = []; + await prompt("user-5"); + assert.equal(await identity(), beforeRemoval); + assert(!requests.at(-1).tools.some((tool) => tool.function.name === "Skill")); + assert(!JSON.stringify(requests.at(-1).messages).includes("Updated skill")); + console.log("PASS: removing the final skill removes its catalog and executable schema"); + } finally { await sidecar.dispose(); await writes; } + }, resolveHostBinary(), root, PROTOCOL_VERSION); +} finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + await rm(root, { recursive: true, force: true }); +} From 022cb0a8ffecbbe1ad5d06db5f5e5ed08c3545ee Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:02:12 +0800 Subject: [PATCH 2/5] fix(runtime): preserve transport bindings through catalog enrichment Keep Pi's exact transport capabilities independent of models.dev-owned limits and prices so catalog enrichment cannot silently disable updates. Normalize compaction checkpoints to the durable journal shape, including extension-provided text blocks, and cover recovery cleanup boundaries. Related to #1285 --- .../electron/main/models-dev-catalog.ts | 12 +++++- apps/desktop/test/models-dev-catalog.test.mjs | 41 +++++++++++++++++++ docs/adr/chronological-system-transcript.md | 7 ++++ docs/spec/03-runtime/02-agent-runtime.md | 5 +++ .../agent-runtime/src/model-capabilities.ts | 1 + .../src/provider-binding.test.ts | 9 ++-- packages/agent-runtime/src/runtime.test.ts | 9 ++-- .../src/system-transcript-journal.test.ts | 10 +++++ .../src/system-transcript-runtime.test.ts | 3 ++ .../src/system-transcript.test.ts | 10 +++++ .../agent-runtime/src/system-transcript.ts | 8 +++- .../src/transcript-compat.test.ts | 4 ++ .../agent-runtime/src/transcript-compat.ts | 13 +++++- 13 files changed, 117 insertions(+), 15 deletions(-) diff --git a/apps/desktop/electron/main/models-dev-catalog.ts b/apps/desktop/electron/main/models-dev-catalog.ts index fbf42f28c..a8a6b310f 100644 --- a/apps/desktop/electron/main/models-dev-catalog.ts +++ b/apps/desktop/electron/main/models-dev-catalog.ts @@ -32,7 +32,7 @@ import type { ThinkingProtocol, ModelBinding, } from "@pi-desktop/shared"; -import { genericModelConfig, modelConfigWithBinding, type ModelConfig } from "@pi-desktop/agent-runtime"; +import { genericModelConfig, modelConfigWithBinding, transcriptConfigFromPi, type ModelConfig } from "@pi-desktop/agent-runtime"; import { resolveBindingLimits } from "@pi-desktop/shared"; export const MODELS_DEV_API_URL = "https://models.dev/api.json"; @@ -1424,6 +1424,16 @@ export class ModelsDevCatalog { const thinking = this.anthropicThinkingFor(input.modelId); if (thinking) baseline = { ...baseline, ...thinking }; } + // Pi owns wire capabilities, independently of models.dev's limits/prices. + // Use the original published record, never an account/relay projection. + const key = input.vendorKey?.trim().toLowerCase(); + const vendor = (key ? PI_VENDOR_ALIASES[key] ?? key : undefined) ?? this.providerKeyForRow(input); + const transport = vendor ? this.operationModels.getModel(vendor, input.modelId) : undefined; + if (transport) { + const transcript = transcriptConfigFromPi(transport); + baseline = { ...baseline, transcriptBinding: transcript.transcriptBinding, + compat: { ...baseline.compat, ...transcript.compat } }; + } if (!binding) return baseline; const limits = resolveBindingLimits(baseline, binding); return modelConfigWithBinding(limits.catalogConfig, limits.binding); diff --git a/apps/desktop/test/models-dev-catalog.test.mjs b/apps/desktop/test/models-dev-catalog.test.mjs index cdfc969a1..7d79bf3e8 100644 --- a/apps/desktop/test/models-dev-catalog.test.mjs +++ b/apps/desktop/test/models-dev-catalog.test.mjs @@ -4,11 +4,15 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import test from "node:test"; import { fileURLToPath } from "node:url"; +import { createModels, InMemoryModelsStore } from "@earendil-works/pi-ai"; +import { builtinProviders } from "@earendil-works/pi-ai/providers/all"; +import { buildProviderModel } from "../../../packages/agent-runtime/dist/provider-binding.js"; import { apiStyleForAdapter, bindingForCustomModelInfo, catalogModelIdsMatch, modelIdsMatch, + resolveApiStyle, } from "@pi-desktop/shared"; import { MODELS_DEV_API_URL, @@ -103,6 +107,43 @@ async function loadFixtureCatalog(t, fixture = catalogFixture) { return catalog; } +test("published metadata retains exact Pi transcript transport bindings through runtime launch", async (t) => { + const pi = createModels({ modelsStore: new InMemoryModelsStore(), + authContext: { env: async () => undefined, fileExists: async () => false } }); + for (const provider of builtinProviders()) pi.setProvider(provider); + for (const [vendorKey, modelId] of [ + ["deepseek", "deepseek-flash"], ["anthropic", "claude-opus-5-5"], + ["openai", "gpt-6.1-sol"], ["openai-codex", "gpt-6.1-sol"], + ]) { + const original = pi.getModel(vendorKey, modelId); + assert.ok(original, `${vendorKey}/${modelId}`); + const catalog = await loadFixtureCatalog(t, { [vendorKey]: { + name: vendorKey, api: original.baseUrl, models: { [modelId]: { + id: modelId, limit: { context: 543_210, output: 40_000 }, cost: { input: 1.25, output: 2.5 }, + } }, + } }); + const target = { providerId: "account", vendorKey, modelId, baseUrl: original.baseUrl, + apiStyle: resolveApiStyle(original.api) }; + const config = catalog.modelConfigFor(target); + assert.equal(config.source, "models.dev"); + assert.equal(config.contextWindow, 543_210); + assert.equal(config.maxTokens, 40_000); + assert.equal(config.cost.input, 1.25); + assert.deepEqual(config.transcriptBinding, { modelId, api: original.api, baseUrl: original.baseUrl }); + const provider = { ...target, id: "account", name: "Fixture", apiKey: "", authKind: "none", + supportsReasoning: false, supportedThinkingLevels: ["off"], modelConfig: config }; + const wire = buildProviderModel(provider); + assert.equal(wire.compat.supportsMidConvoSystemMessages, true, vendorKey); + assert.equal(wire.compat.supportsMidConvoToolChanges, original.compat?.supportsMidConvoToolChanges === true); + assert.equal(wire.compat.supportsAdditionalTools, original.compat?.supportsAdditionalTools === true); + const relay = { ...target, baseUrl: "https://relay.invalid/v1" }; + assert.equal(buildProviderModel({ ...provider, ...relay, modelConfig: catalog.modelConfigFor(relay) }) + .compat.supportsMidConvoSystemMessages, false); + assert.equal(buildProviderModel({ ...provider, modelId: `${modelId}-alias` }) + .compat.supportsMidConvoSystemMessages, false); + } +}); + test("the bundled models.dev OpenAI record supplies the selected model limits", async () => { const catalogPath = fileURLToPath(new URL("../resources/models.dev/api.json", import.meta.url)); const catalog = new ModelsDevCatalog({ catalogPath }); diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md index 74884ee9f..c1a484003 100644 --- a/docs/adr/chronological-system-transcript.md +++ b/docs/adr/chronological-system-transcript.md @@ -38,6 +38,13 @@ Accept transcript capability opt-ins only when that identity matches the actual request route. Let Pi adapt the request for partial or absent support; never persist an adapter's folded projection. +The current models.dev catalog remains authoritative for published limits, +prices and modalities. Independently copy only the five transcript transport +flags and their original binding from Pi's exact published model. Do not derive +transport support from a models.dev metadata match or an account endpoint +override. Both Pi and models.dev metadata projections can carry that binding; +unverified routes and generic records retain the conservative fallback. + The existing Pi 0.99.1 dependency patch adds the missing mid-conversation system capability to its `deepseek-flash` catalog entry. Authorized official-endpoint experiments confirmed both preserved cache reuse and effective updated diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index db1a2ee59..44506236b 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -1308,6 +1308,11 @@ and redefinitions use the adapter's supported fallback. Model switching never rewrites the canonical journal. Cache savings depend on the actual provider; unsupported routes may still rebuild the request prefix. +Published model limits, prices and modalities remain owned by models.dev. Its +runtime projection separately carries Pi's exact published transcript capability +flags and original binding. An enriched metadata source does not grant native +support by itself, and account overrides never replace that original binding. + The pinned Pi patch declares mid-conversation system support for `deepseek-flash` on its published `openai-completions` binding at `https://api.deepseek.com`. Skill changes on this binding append system updates diff --git a/packages/agent-runtime/src/model-capabilities.ts b/packages/agent-runtime/src/model-capabilities.ts index 2c7f3cd58..43e54643c 100644 --- a/packages/agent-runtime/src/model-capabilities.ts +++ b/packages/agent-runtime/src/model-capabilities.ts @@ -7,6 +7,7 @@ import { type ModelModality, } from "@pi-desktop/shared"; import type { ModelConfig, ThinkingCapabilitySet } from "./thinking-level.js"; +export { transcriptConfigFromPi } from "./transcript-compat.js"; export { agentThinkingLevel, diff --git a/packages/agent-runtime/src/provider-binding.test.ts b/packages/agent-runtime/src/provider-binding.test.ts index bc4036443..ad755e70a 100644 --- a/packages/agent-runtime/src/provider-binding.test.ts +++ b/packages/agent-runtime/src/provider-binding.test.ts @@ -254,12 +254,9 @@ describe("Anthropic runtime endpoint", () => { .result(); expect(result.stopReason).toBe("error"); - expect( - Object.hasOwn(model.compat ?? {}, "supportsMidConvoSystemMessages"), - ).toBe(false); - expect( - Object.hasOwn(model.compat ?? {}, "supportsMidConvoToolChanges"), - ).toBe(false); + expect(model.compat).toMatchObject({ + supportsMidConvoSystemMessages: false, supportsMidConvoToolChanges: false, + }); expect(request?.headers.get("anthropic-beta") ?? "").not.toMatch( /mid-conversation-tool-changes|inline-tools/, ); diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index 9318de1cb..689bab439 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -2213,11 +2213,10 @@ describe("DesktopAgentRuntime deferred tool catalog", () => { expect(agent.state.tools.some((tool: any) => tool.name === "BrowserPreview")).toBe( true, ); - // Tool deltas append new declarations; catalog order is not semantic. - const byName = (a: { name: string }, b: { name: string }) => a.name.localeCompare(b.name); - expect([...getCurrentTools(agent.state.messages)].sort(byName)).toEqual( - agent.state.tools.map(toToolDeclaration).sort(byName), - ); + // Reset retains executability; Pi declares it at the next dispatch, not + // while this test calls preparation helpers outside the agent loop. + expect(getCurrentTools(agent.state.messages).some((tool) => tool.name === "BrowserPreview")).toBe(false); + await runtime.dispose(); }); }); diff --git a/packages/agent-runtime/src/system-transcript-journal.test.ts b/packages/agent-runtime/src/system-transcript-journal.test.ts index 22b9816a3..3fc43d2ee 100644 --- a/packages/agent-runtime/src/system-transcript-journal.test.ts +++ b/packages/agent-runtime/src/system-transcript-journal.test.ts @@ -80,6 +80,16 @@ describe("model system journal", () => { expect(checkpoint?.toolsAdded).toEqual(initial.toolsAdded); }); + it("recognizes a restored checkpoint with text blocks without adding durable rows", async () => { + const blocks: SystemMessage = { ...initial, content: [{ type: "text", text: "Extension instructions" }] }; + const checkpoint = systemTranscriptCheckpoint([blocks, user])!; + const journal = new SystemTranscriptJournal(); + journal.rememberCheckpoint(JSON.stringify(checkpoint)); + const append = vi.fn(); + await journal.persist(JSON.parse(JSON.stringify([checkpoint, user])), [], append); + expect(append).not.toHaveBeenCalled(); + }); + it.each([null, {}, { ...initial, timestamp: -1 }, { ...initial, toolsAdded: [{ name: "Read", parameters: "bad" }] }])("rejects malformed saved state", (value) => { expect(() => readSystemMessage(value)).toThrow("Invalid persisted model system message"); }); diff --git a/packages/agent-runtime/src/system-transcript-runtime.test.ts b/packages/agent-runtime/src/system-transcript-runtime.test.ts index f48e132dd..15ced3662 100644 --- a/packages/agent-runtime/src/system-transcript-runtime.test.ts +++ b/packages/agent-runtime/src/system-transcript-runtime.test.ts @@ -86,6 +86,9 @@ describe("system state through the Desktop user path", () => { const activation = after.findIndex((message) => message.role === "toolResult" && message.toolName === "ToolSearch"); expect(after.slice(activation + 1)).toContainEqual(expect.objectContaining({ role: "system", toolsAdded: expect.arrayContaining([expect.objectContaining({ name: "BrowserPreview" })]) })); expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); + await f.prompt("same-runtime-user"); + expect(getCurrentTools(f.requests[2]).some((tool) => tool.name === "BrowserPreview")).toBe(true); + expect(f.rows.filter((row) => row.modelSystem)).toHaveLength(2); const restored = fixture(structuredClone(f.rows)); try { await restored.prompt("user-2"); diff --git a/packages/agent-runtime/src/system-transcript.test.ts b/packages/agent-runtime/src/system-transcript.test.ts index 8e994a499..3d4ec5c1d 100644 --- a/packages/agent-runtime/src/system-transcript.test.ts +++ b/packages/agent-runtime/src/system-transcript.test.ts @@ -19,6 +19,7 @@ import { replaceSystemPrompt, systemTranscriptCheckpoint, systemPromptContent, + removeTrailingAssistantMessages, } from "./system-transcript.js"; const read: Tool = { name: "Read", description: "Read text", parameters: Type.Object({ path: Type.String() }) }; @@ -50,6 +51,15 @@ function estimateContextTokens(messages: AgentMessage[]) { afterEach(() => vi.restoreAllMocks()); describe("system transcript helpers", () => { + it("removes failed trailing responses across cleanup deltas without changing system state", () => { + const messages = [initial, user, assistant, delta, { ...assistant, timestamp: 3_000 }]; + const cleaned = removeTrailingAssistantMessages(messages); + expect(cleaned).toEqual([initial, user, delta]); + expect(getCurrentTools(cleaned)).toEqual(getCurrentTools(messages)); + expect(messages).toHaveLength(5); + expect(removeTrailingAssistantMessages([user, assistant])).toEqual([user]); + expect(removeTrailingAssistantMessages([assistant, user, delta])).toEqual([assistant, user, delta]); + }); it("updates only the changed skill entry and preserves unrelated sections", () => { const previous: AgentMessage[] = [{ role: "system", content: "", timestamp: 1, diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index a5bac9363..f3e347033 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -106,8 +106,12 @@ export function rebuildSystemTranscript( /** Fold state only at an explicit compaction boundary, with semantic time. */ export function systemTranscriptCheckpoint(messages: readonly AgentMessage[]): SystemMessage | undefined { - const checkpoint = currentSystemMessage(messages); - if (!checkpoint?.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; + const current = currentSystemMessage(messages); + if (!current) return undefined; + // Use the journal's durable shape, including extension-provided text blocks. + const checkpoint: SystemMessage = { ...current, content: contentText(current.content), + ...(current.toolsAdded ? { toolsAdded: current.toolsAdded.map(toToolDeclaration) } : {}) }; + if (!checkpoint.sections || !(CONTEXT_BUDGET_SECTION in checkpoint.sections)) return checkpoint; const { [CONTEXT_BUDGET_SECTION]: _expired, ...sections } = checkpoint.sections; return { ...checkpoint, sections }; } diff --git a/packages/agent-runtime/src/transcript-compat.test.ts b/packages/agent-runtime/src/transcript-compat.test.ts index 86fb4c2c6..6e3df7ced 100644 --- a/packages/agent-runtime/src/transcript-compat.test.ts +++ b/packages/agent-runtime/src/transcript-compat.test.ts @@ -14,6 +14,10 @@ const provider: RuntimeProviderConfig = { }; describe("transcript compatibility binding", () => { + it("keeps Pi transport capabilities when models.dev owns published metadata", () => { + expect(buildProviderModel({ ...provider, modelConfig: { ...provider.modelConfig!, source: "models.dev" } }).compat) + .toMatchObject({ supportsMidConvoSystemMessages: true }); + }); it("keeps instruction and tool capabilities independent on the exact binding", () => { expect(buildProviderModel(provider).compat).toMatchObject({ supportsMidConvoSystemMessages: true, supportsMidConvoToolAdditions: false }); expect(buildProviderModel({ ...provider, baseUrl: `${baseUrl}/` }).compat).toMatchObject({ supportsMidConvoSystemMessages: true }); diff --git a/packages/agent-runtime/src/transcript-compat.ts b/packages/agent-runtime/src/transcript-compat.ts index 365e2f6da..887e40687 100644 --- a/packages/agent-runtime/src/transcript-compat.ts +++ b/packages/agent-runtime/src/transcript-compat.ts @@ -1,10 +1,21 @@ import type { ModelConfig } from "./thinking-level.js"; +import type { Api, Model } from "@earendil-works/pi-ai"; const capabilities = [ "supportsMidConvoSystemMessages", "supportsMidConvoToolAdditions", "supportsMidConvoToolChanges", "supportsAdditionalTools", "supportsToolSearch", ] as const; +/** Transport capabilities only; published limits and prices keep their owner. */ +export function transcriptConfigFromPi(model: Model): Pick { + return { + transcriptBinding: { modelId: model.id, api: model.api, baseUrl: model.baseUrl }, + compat: Object.fromEntries(capabilities.map((key) => [key, + model.compat !== undefined && key in model.compat && Reflect.get(model.compat, key) === true, + ])), + }; +} + function endpoint(value: string): string | undefined { try { const url = new URL(value); @@ -21,7 +32,7 @@ export function transcriptCompat( ): Record<(typeof capabilities)[number], boolean> { const binding = catalog?.transcriptBinding; const address = endpoint(baseUrl); - const verified = catalog?.source === "pi" && binding?.modelId === modelId && + const verified = (catalog?.source === "pi" || catalog?.source === "models.dev") && binding?.modelId === modelId && binding.api === api && address !== undefined && endpoint(binding.baseUrl) === address; return Object.fromEntries(capabilities.map((key) => [key, verified && catalog?.compat?.[key] === true])) as Record<(typeof capabilities)[number], boolean>; From 09f2cdbb7e0c624cce236e761544926e704ccabb Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:31:23 +0800 Subject: [PATCH 3/5] docs(adr): index chronological system transcript decision Register the system transcript decision in the ADR index so the documentation integrity check no longer aborts the workspace test run. --- docs/adr/README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/adr/README.md b/docs/adr/README.md index 4bac9199a..753634d3a 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -20,6 +20,7 @@ Each ADR includes: | ID | Title | Status | |---|---|---| +| chronological-system-transcript | [Preserve chronological model system state](chronological-system-transcript.md) | Accepted | | mcp-tool-approval-risk | [User MCP tools keep the normal approval path](mcp-tool-approval-risk.md) | Accepted | | models-dev-catalog-authority | [models.dev owns published model metadata](models-dev-catalog-authority.md) | Accepted for implementation | | pi-ai-core-0991-authority | [Pi 0.99.1 account model authority](pi-ai-core-0991-authority.md) | Superseded for chat model metadata | From b622cb975bc1b6c8eed343a74ce531e462e78ee3 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 02:35:22 +0800 Subject: [PATCH 4/5] test(runtime): align compaction reminder checks with system sections The reminder now travels in a replaceable transcript section so it can expire after compaction. Update the old source guard and assert that Pi folds the actual reminder content into the current system state. --- apps/desktop/test/context-compaction.test.mjs | 7 +++---- packages/agent-runtime/src/runtime.test.ts | 9 ++++++++- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/apps/desktop/test/context-compaction.test.mjs b/apps/desktop/test/context-compaction.test.mjs index d57835f90..9080321c5 100644 --- a/apps/desktop/test/context-compaction.test.mjs +++ b/apps/desktop/test/context-compaction.test.mjs @@ -114,12 +114,11 @@ test("the hard boundary is enforced by the host, with a model-side escape hatch" assert.match(runtime, /function contextFallbackReminder\(/); assert.match(runtime, /contextReminderClaimed/); assert.match(runtime, /contextFallbackReminderClaimed/); - // pi 0.86 carries the canonical system prompt in the transcript. The - // reminder is appended as a per-turn system message, not only as the legacy - // AgentContext.systemPrompt field. + // The reminder is appended as a replaceable system section so compaction + // can expire it without rewriting earlier instructions. assert.match( runtime, - /messages: \[\s*\.\.\.context\.messages,\s*\{\s*role: "system",\s*content: reminder,/, + /messages: \[\s*\.\.\.context\.messages,\s*\{\s*role: "system",\s*content: "",\s*sections: \{ \[CONTEXT_BUDGET_SECTION\]: reminder \}/, ); assert.match(hostPermissions, /"new_context"/); }); diff --git a/packages/agent-runtime/src/runtime.test.ts b/packages/agent-runtime/src/runtime.test.ts index 689bab439..61e2d83e1 100644 --- a/packages/agent-runtime/src/runtime.test.ts +++ b/packages/agent-runtime/src/runtime.test.ts @@ -5285,8 +5285,15 @@ describe("DesktopAgentRuntime per-turn context protection", () => { // compaction trigger, so the model can close out before host compaction. expect(below.context.systemPrompt).toContain(""); expect(below.context.systemPrompt).toContain("new_context"); + expect(below.context.messages.at(-1)).toMatchObject({ + role: "system", + content: "", + sections: { context_budget: expect.stringContaining("") }, + }); + expect(getCurrentSystemMessage(below.context.messages)?.sections?.context_budget).toContain("new_context"); expect(stillBelow.context.systemPrompt).not.toContain(""); - // The reminder rides on the turn's context only; nothing is persisted. + // The legacy prompt baseline stays unchanged; the returned transcript + // carries the reminder section through the normal persistence path. expect((runtime as any).agent.state.systemPrompt).not.toContain( "", ); From 907594acb86096458cd149d4f7123ad5e5bf9ff7 Mon Sep 17 00:00:00 2001 From: zszz3 <91608029+zszz3@users.noreply.github.com> Date: Sat, 3 Oct 2026 15:12:22 +0800 Subject: [PATCH 5/5] fix(runtime): keep Flash tool declarations stable across searches Declare the verified Flash catalog once per account and schema epoch so ToolSearch activation does not invalidate the earlier request prefix. Persist activation independently and reject inactive tools before Host execution, preserving existing mode and approval restrictions. Cover HTTP payloads, recovery, compaction, revoked tools and fallback limits, with paired live Flash measurements documenting the larger cold request and the short-conversation cost tradeoff. refs #1285 --- docs/adr/chronological-system-transcript.md | 39 +++- docs/spec/03-runtime/02-agent-runtime.md | 62 +++-- docs/spec/06-delivery/04-e2e-test-plan.md | 27 +++ .../zh-CN/spec/03-runtime/02-agent-runtime.md | 24 +- .../spec/06-delivery/04-e2e-test-plan.md | 27 +++ .../src/fixed-tool-declarations.test.ts | 52 +++++ .../src/fixed-tool-declarations.ts | 69 ++++++ .../src/fixed-tool-runtime.test.ts | 216 ++++++++++++++++++ packages/agent-runtime/src/runtime.ts | 71 +++++- .../agent-runtime/src/system-transcript.ts | 6 +- scripts/e2e-fixed-tool-declarations.mjs | 170 ++++++++++++++ 11 files changed, 717 insertions(+), 46 deletions(-) create mode 100644 packages/agent-runtime/src/fixed-tool-declarations.test.ts create mode 100644 packages/agent-runtime/src/fixed-tool-declarations.ts create mode 100644 packages/agent-runtime/src/fixed-tool-runtime.test.ts create mode 100644 scripts/e2e-fixed-tool-declarations.mjs diff --git a/docs/adr/chronological-system-transcript.md b/docs/adr/chronological-system-transcript.md index c1a484003..659ec00a2 100644 --- a/docs/adr/chronological-system-transcript.md +++ b/docs/adr/chronological-system-transcript.md @@ -45,7 +45,7 @@ transport support from a models.dev metadata match or an account endpoint override. Both Pi and models.dev metadata projections can carry that binding; unverified routes and generic records retain the conservative fallback. -The existing Pi 0.99.1 dependency patch adds the missing mid-conversation system +The Pi 1.0.0 dependency patch adds the missing mid-conversation system capability to its `deepseek-flash` catalog entry. Authorized official-endpoint experiments confirmed both preserved cache reuse and effective updated instructions. Keep this correction in the single Pi catalog, not a parallel @@ -53,6 +53,43 @@ Desktop allowlist; remove the hunk when an upgraded Pi catalog carries it. The model/API/endpoint binding check still excludes aliases and relays. Native tool-addition and tool-change flags are not enabled by this correction. +## Fixed declarations for the verified Flash route + +For the exact official `deepseek-flash` / `openai-completions` binding with +verified chronological system support, declare the complete current tool catalog +in deterministic name order from the first request. ToolSearch changes execution +activation only. This is a Desktop declaration policy, not an additional Pi +transport capability or an endpoint switch. Other bindings keep on-demand +schema publication; native tool-state flags alone do not prove cache stability. + +Keep activation separate from declarations in a versioned `tool_activation` +system section. The section records active deferred names and a SHA-256 identity +of the account, model, API, endpoint, declarations and deferred-name set. Updates +append after tool results and persist through the existing Host journal and +compaction checkpoint. Restoration never interprets the complete declaration +snapshot as permission to execute every tool. Successful ToolSearch results +newer than the saved activation section recover an interrupted activation. +Invalid, unknown-version or mismatched state grants no activation. Legacy +histories without this section retain their existing activation evidence rules. + +Check activation before extension hooks or Host execution, then retain all +existing mode, approval and Host restrictions. A catalog/schema/mode/account or +route change starts a new declaration epoch and invalidates prior activation; +removed tools cannot be invoked. Temporary prompt replacement must preserve the +activation metadata. Current runtime activation remains authoritative between +prompts; declarations do not re-grant revoked activation. + +DeepSeek limits a request to 128 functions. If the full catalog exceeds that +limit, or its estimated prompt/schema cost leaves less than the ordinary +retained-tail budget below the automatic compaction threshold, use the existing +on-demand path and log the fallback reason. Never truncate a catalog. Context +estimation charges the full declared catalog while fixed declarations are active. + +The first request is larger. Short conversations may cost more overall even +when later cache-hit ratios improve. Acceptance compares cold and subsequent +uncached tokens and cumulative input cost across both short and longer synthetic +conversations; no universal savings or hit-rate guarantee is made. + ## Alternatives and consequences Prepending a reconstructed snapshot was rejected because it changes history on diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index 08346ab68..15bcd08d4 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -1393,10 +1393,10 @@ grammar and validated against another fails every call. ### 7.1 Active tool context and on-demand loading (D185, ADR 0048) -The sidecar builds one complete tool registry, but it does not serialize every -registered schema into every provider request. Each new user prompt starts with -the mode's core set plus any deferred tools that can be restored from successful -activation evidence still present in the effective session context: +The sidecar builds one complete tool registry. By default, each provider request +declares the mode's core set plus activated deferred tools. The verified Flash +binding uses the fixed-declaration policy below, while preserving the same +execution activation rules: - Agent: `Read`, `Bash`, `Edit`, and `Write` (matching pi's coding-agent core) - Agent: `Skill` whenever the skill catalog is non-empty (D404, ADR 0230) — the @@ -1424,24 +1424,42 @@ The sidecar activates up to four matches, records their names in the canonical schemas. Providers with native deferred-tool search receive the definitions at that load point; other providers receive the active definitions normally. -At the start of each new user prompt, the sidecar clears the in-memory deferred -activation set and rebuilds it from the effective context. Recorded system -messages (including a compaction checkpoint) define the active baseline. Only -successful tool results after the latest declaration can add a new activation; -older results must not resurrect a removed tool. Histories without system records -continue to use successful results throughout their effective context. -`ToolSearch` results contribute their canonical `details.addedToolNames`. -For compatibility, historical `details.activated` and top-level -`addedToolNames` markers are also accepted. Successful results from deferred -tools contribute that tool's name. Only names still present in the current -mode's deferred catalog are restored. Failed rows, interrupted or -missing-result placeholders, and assistant/user prose never activate a tool. -The tool registry, host permission path, tool timeout, and workspace containment -rules remain unchanged. `ToolSearch` is local to the sidecar and does not cross -the host RPC boundary. Its activation marker is retained in the persisted tool -result, so a runtime restart or a new prompt can reuse an eligible capability -while that evidence remains in the effective context; a fresh search is still -required after the evidence is compacted away or otherwise absent. +Deferred activation is sticky for the live runtime. At restoration, recorded +system messages (including a compaction checkpoint) define the active baseline +for ordinary on-demand histories. Only successful results after the latest +system record can add activation; older results must not resurrect removed +tools. Legacy histories without system records use successful results throughout +their effective context. ToolSearch accepts canonical `details.addedToolNames` +and historical `details.activated` / top-level `addedToolNames`; successful +results from deferred tools also restore their names. Failed results, +missing-result placeholders and assistant/user prose never activate tools. +Only names in the current mode's deferred catalog are eligible. + +For the exact official `deepseek-flash` Chat Completions binding with verified +mid-conversation system support, the runtime instead declares the complete +catalog in deterministic name order on the first request. ToolSearch changes +activation without changing the declared schemas. A visible schema does not +permit execution: inactive deferred calls are rejected before extension hooks +and the Host; activated calls still require the existing mode and Host checks. +ToolSearch remains local and never grants approval or bypasses permissions. + +Fixed declarations persist separately from activation. A version-1 +`tool_activation` section records active names and a fingerprint of the account, +model, API, endpoint, schema catalog and deferred set. Activation changes append +at the continuation boundary, and the existing system journal/checkpoint saves +both declarations and activation. Restore only validated activation for a +matching fingerprint, plus successful ToolSearch results newer than that state; +never activate tools merely because the full snapshot declared them. Malformed, +unknown-version and mismatched activation state fail closed. A catalog/schema, +mode, account, model or route change creates a new epoch and requires new +activation. Removal immediately removes the tool from executable registration. + +If the full catalog exceeds 128 functions or its prompt/schema estimate cannot +leave the normal retained-tail budget below the compaction threshold, retain +on-demand declarations and emit a diagnostic explaining that ToolSearch cache +stability is not guaranteed. Do not truncate tools. Other models and unverified +routes retain the existing Pi projection. First-request schema overhead increases; +cache stability does not imply that short conversations become cheaper. For user-visible HTML deliverables, the default system prompt asks the agent to activate `BrowserPreview` once after creating the page or making its first diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 708281ee4..1407badc2 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -16318,3 +16318,30 @@ renderer's durable transcript reads. No real model or provider is contacted. same-model assistant reasoning remains separate from visible content. Adapter regressions also cover legacy identities and genuine account/model changes. Cache percentages are observations, not deterministic pass thresholds. + +### E2E-FIXED-TOOL-DECLARATIONS: Stable Flash schemas with independent activation + +- Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, + and a child-process fetch boundary redirected to local HTTP/SSE. Credentials + are dummy values and synthetic tools have no external side effects. +- Prompt, search Alpha, execute Alpha, search Beta and execute Beta. HTTP payload + tests assert identical ordered `tools` and unchanged prior message prefixes. + Directly calling a declared but inactive tool must fail before the Host. +- Restart both processes, then compact and restart again: Alpha stays activated, + Beta remains inactive despite its declaration. Activate Beta, then remove it + from the catalog and verify it cannot execute. New schema/route epochs must + not restore old grants. Host rejection and Plan guards remain effective. +- Runtime contracts additionally cover legacy migration, interrupted activation, + malformed metadata, temporary prompt replacement, deterministic catalog order, + the 128-tool boundary and insufficient-context fallback. +- Automated: `node scripts/e2e-fixed-tool-declarations.mjs` with built + shared/host-runtime/agent-runtime and `PI_DESKTOP_HOST_BIN` set to the candidate + Host binary; `fixed-tool-runtime.test.ts` and `fixed-tool-declarations.test.ts` + cover HTTP payload and policy contracts. Electron UI/outbox is outside this + fixture's scope. +- Separately authorized official Flash experiments compare several independent + on-demand/fixed sessions with the same synthetic catalog and call sequence. + Record cold requests, both activations, follow-up cache hits/misses and + cumulative input cost. Report the larger first request and possible short-chat + cost increase, alongside any longer-conversation benefit. Offline test success + alone is not evidence of provider cache behavior. diff --git a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md index 820a0d5b3..97ec5d4ab 100644 --- a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md +++ b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md @@ -978,15 +978,21 @@ sidecar 最多激活四个匹配项,并将名称写入 canonical 模式。具有本机延迟工具搜索的提供商可在该负载点接收定义;其他 提供商通常会收到活动定义。 -每个新用户提示前都会清除延迟激活集,再从有效上下文重建。成功的 -`ToolSearch` 结果读取 canonical `details.addedToolNames`;为兼容历史 -数据,也接受 `details.activated` 和顶层 `addedToolNames`。成功的延迟 -工具结果会贡献其工具名。仅恢复当前模式延迟目录中仍存在的名称;失败、 -中断、缺少结果的占位行以及助手/用户文本不会激活工具。工具注册表、主机 -权限路径、工具超时和工作区包含规则保持不变。`ToolSearch` 是 sidecar 的 -本地工具,不跨越主机 RPC 边界。激活标记保留在持久化工具结果中,因此 -只要证据仍在有效上下文,运行时重启或新提示都可以复用能力;证据被压缩 -或消失后仍需重新搜索。 +Deferred activation remains sticky within a live runtime. Restoration uses +successful activation evidence and the current catalog; old declarations do not +re-grant tools revoked from the live activation set. For official bound Flash, +full declarations and execution activation are independent: versioned +`tool_activation` sections carry the account/model/API/endpoint/catalog identity +and active names through restart and compaction. Only matching, valid state and +newer successful ToolSearch results restore activation; malformed or changed +epochs fail closed. Inactive declared tools are blocked before extension/Host +execution, and activation never bypasses mode or approval checks. The full +catalog is deterministic from the first request. More than 128 tools or an +insufficient context budget falls back to on-demand declarations with a +diagnostic, without truncation. Other bindings retain their existing projection. +Fixed declarations may increase total cost for short conversations. See the +English section 7.1 and the chronological-system-transcript ADR for the complete +contract. 对于用户可见的 HTML 可交付成果,默认系统提示要求代理 创建页面或创建第一个页面后激活 `BrowserPreview` 一次 diff --git a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md index 4e8caff87..c48a69498 100644 --- a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md @@ -9219,3 +9219,30 @@ the latest destination. These assertions measure work counts, not device FPS. - **验收:** 缓存路径、迁移和清理单测通过;Windows task-candidate 验证应覆盖更新源传输、安装器交接和文件系统行为,且不连接真实发布源。 - **里程碑:** M6+ - **状态:** 单测和源码契约覆盖(`update-cache.test.mjs`、`auto-update.test.mjs`);仍需 Windows 安装器/E2E 验证。 + +### E2E-FIXED-TOOL-DECLARATIONS: Stable Flash schemas with independent activation + +- Fixture: production AgentSidecar and isolated Host, official Pi Flash binding, + and a child-process fetch boundary redirected to local HTTP/SSE. Credentials + are dummy values and synthetic tools have no external side effects. +- Prompt, search Alpha, execute Alpha, search Beta and execute Beta. HTTP payload + tests assert identical ordered `tools` and unchanged prior message prefixes. + Directly calling a declared but inactive tool must fail before the Host. +- Restart both processes, then compact and restart again: Alpha stays activated, + Beta remains inactive despite its declaration. Activate Beta, then remove it + from the catalog and verify it cannot execute. New schema/route epochs must + not restore old grants. Host rejection and Plan guards remain effective. +- Runtime contracts additionally cover legacy migration, interrupted activation, + malformed metadata, temporary prompt replacement, deterministic catalog order, + the 128-tool boundary and insufficient-context fallback. +- Automated: `node scripts/e2e-fixed-tool-declarations.mjs` with built + shared/host-runtime/agent-runtime and `PI_DESKTOP_HOST_BIN` set to the candidate + Host binary; `fixed-tool-runtime.test.ts` and `fixed-tool-declarations.test.ts` + cover HTTP payload and policy contracts. Electron UI/outbox is outside this + fixture's scope. +- Separately authorized official Flash experiments compare several independent + on-demand/fixed sessions with the same synthetic catalog and call sequence. + Record cold requests, both activations, follow-up cache hits/misses and + cumulative input cost. Report the larger first request and possible short-chat + cost increase, alongside any longer-conversation benefit. Offline test success + alone is not evidence of provider cache behavior. diff --git a/packages/agent-runtime/src/fixed-tool-declarations.test.ts b/packages/agent-runtime/src/fixed-tool-declarations.test.ts new file mode 100644 index 000000000..8b54969b9 --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-declarations.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "vitest"; +import type { AgentTool } from "@earendil-works/pi-agent-core"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { Type, type Api, type Model } from "@earendil-works/pi-ai"; +import { toolDeclarationPolicy, toolActivationSection, restoredToolActivation, TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; +import { replaceSystemPrompt, systemTranscriptCheckpoint } from "./system-transcript.js"; + +const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; +const tool = (name: string): AgentTool => ({ name, label: name, description: "Fixture tool", parameters: Type.Object({}), + execute: async () => ({ content: [{ type: "text", text: "Done" }], details: {} }) }); +const deferred = new Set(["Alpha", "Beta"]); +const policy = (tools = [tool("Alpha"), tool("Beta")], overrides: Partial> = {}, prompt = "") => + toolDeclarationPolicy({ ...model, ...overrides } as Model, tools, deferred, prompt, "fixture-account"); + +describe("fixed tool declaration policy", () => { + it("has a deterministic declaration order and snapshot identity", () => { + const first = policy(); + const second = policy([tool("Beta"), tool("Alpha")]); + expect(second.key).toBe(first.key); + expect(second.tools?.map((tool) => tool.name)).toEqual(["Alpha", "Beta"]); + expect(policy([{ ...tool("Alpha"), parameters: Type.Object({ id: Type.String() }) }]).key).not.toBe(first.key); + }); + it.each([ + { id: "deepseek-pro" }, { api: "openai-responses" }, { baseUrl: "https://relay.invalid" }, + { baseUrl: "https://api.deepseek.com/v1" }, { compat: { supportsMidConvoSystemMessages: false } }, + ] as Partial>[])("does not enable unverified bindings: %j", (overrides) => { + expect(policy(undefined, overrides).tools).toBeUndefined(); + expect(policy(undefined, overrides).fallback).toBeUndefined(); + }); + it("falls back without truncating catalogs beyond the provider's function limit", () => { + expect(policy(Array.from({ length: 128 }, (_, index) => tool(`Tool${index}`))).tools).toHaveLength(128); + expect(policy(Array.from({ length: 129 }, (_, index) => tool(`Tool${index}`)))).toMatchObject({ fallback: "tool-count" }); + }); + it("leaves room for retained conversation and output", () => { + expect(policy([tool("Alpha")], { contextWindow: 32000, maxTokens: 4000 }, "x".repeat(80000))) + .toMatchObject({ fallback: "context-budget" }); + }); + it("restores activation independently of full declarations, including compacted and transient prompts", () => { + const current = policy(); + const messages = [{ role: "system" as const, content: "", timestamp: 1, + toolsAdded: current.tools, sections: { runtime: "Rules", [TOOL_ACTIVATION_SECTION]: toolActivationSection(current.key, new Set(["Alpha"])) } }]; + const changed = replaceSystemPrompt(messages, "Temporary nudge"); + expect(restoredToolActivation(changed, current.key)?.active).toEqual(["Alpha"]); + expect(restoredToolActivation([systemTranscriptCheckpoint(changed)!], current.key)?.active).toEqual(["Alpha"]); + expect(restoredToolActivation(messages, "different-snapshot")?.active).toEqual([]); + expect(restoredToolActivation([], current.key)).toBeUndefined(); + }); + it.each(["invalid JSON", '{"version":2}', null])("fails closed on invalid activation metadata: %s", (value) => { + expect(restoredToolActivation([{ role: "system", content: "", timestamp: 1, + sections: { [TOOL_ACTIVATION_SECTION]: value } }], policy().key)?.active).toEqual([]); + }); +}); diff --git a/packages/agent-runtime/src/fixed-tool-declarations.ts b/packages/agent-runtime/src/fixed-tool-declarations.ts new file mode 100644 index 000000000..721f39663 --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-declarations.ts @@ -0,0 +1,69 @@ +import { createHash } from "node:crypto"; +import type { AgentMessage, AgentTool } from "@earendil-works/pi-agent-core"; +import { getCurrentSystemMessage, toToolDeclaration, Type, type Api, type Model } from "@earendil-works/pi-ai"; +import * as Value from "typebox/value"; +import { contextBudgetLimitsFor, automaticCompactionThresholdFor } from "./context-budget.js"; +import { estimateOutputCapInputTokens } from "./output-cap.js"; + +export const TOOL_ACTIVATION_SECTION = "tool_activation"; +const activationSchema = Type.Object({ + version: Type.Literal(1), snapshot: Type.String({ pattern: "^[a-f0-9]{64}$" }), + active: Type.Array(Type.String({ minLength: 1 }), { uniqueItems: true }), +}, { additionalProperties: false }); + +export type ToolDeclarationPolicy = { + key: string; + tools?: AgentTool[]; + fallback?: "tool-count" | "context-budget"; +}; + +/** The verified Flash route has chronological system updates, but no tool deltas. */ +export function toolDeclarationPolicy( + model: Model, tools: readonly AgentTool[], deferred: ReadonlySet, prompt: string, accountId: string, +): ToolDeclarationPolicy { + const ordered = [...tools].sort((a, b) => a.name < b.name ? -1 : a.name > b.name ? 1 : 0); + const key = createHash("sha256").update(JSON.stringify({ + account: accountId, model: model.id, api: model.api, endpoint: model.baseUrl.replace(/\/+$/, ""), + tools: ordered.map(toToolDeclaration), deferred: [...deferred].sort(), + })).digest("hex"); + if (model.id !== "deepseek-flash" || model.api !== "openai-completions" + || model.baseUrl.replace(/\/+$/, "") !== "https://api.deepseek.com" + || !model.compat || !("supportsMidConvoSystemMessages" in model.compat) + || model.compat.supportsMidConvoSystemMessages !== true) return { key }; + // DeepSeek Chat Completions permits at most 128 functions. Never truncate. + if (ordered.length > 128) return { key, fallback: "tool-count" }; + const budget = contextBudgetLimitsFor(model); + const tokens = estimateOutputCapInputTokens({ messages: [], systemPrompt: prompt, tools: ordered }, model); + // Leave the normal retained-tail budget available to the conversation. A + // fixed catalog must not make every fresh/compacted request overflow again. + if (tokens >= automaticCompactionThresholdFor(budget) - budget.keepRecentTokens) { + return { key, fallback: "context-budget" }; + } + return { key, tools: ordered }; +} + +export function toolActivationSection(key: string, active: ReadonlySet): string { + return JSON.stringify({ version: 1, snapshot: key, active: [...active].sort() }); +} + +/** A declaration is not activation. Malformed/new-version state fails closed. */ +export function restoredToolActivation(messages: readonly AgentMessage[], key: string): { active: string[]; replayFrom: number } | undefined { + const section = getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION]; + const closed = { active: [], replayFrom: messages.length }; + if (section == null) return messages.some((message) => message.role === "system" + && TOOL_ACTIVATION_SECTION in (message.sections ?? {})) ? closed : undefined; + try { + const parsed: unknown = JSON.parse(section); + if (!Value.Check(activationSchema, parsed) || parsed.snapshot !== key) return closed; + const lastState = messages.map((message) => message.role === "system" + && TOOL_ACTIVATION_SECTION in (message.sections ?? {})).lastIndexOf(true); + return { active: parsed.active, replayFrom: lastState + 1 }; + } catch { return closed; } +} + +/** Activation is appended after the result, never inserted into old instructions. */ +export function syncToolActivation(messages: AgentMessage[], section: string): AgentMessage[] { + if (getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION] === section) return messages; + const timestamp = messages.reduce((latest, message) => Math.max(latest, message.timestamp + 1), Date.now()); + return [...messages, { role: "system", content: "", timestamp, sections: { [TOOL_ACTIVATION_SECTION]: section } }]; +} diff --git a/packages/agent-runtime/src/fixed-tool-runtime.test.ts b/packages/agent-runtime/src/fixed-tool-runtime.test.ts new file mode 100644 index 000000000..2c86736ce --- /dev/null +++ b/packages/agent-runtime/src/fixed-tool-runtime.test.ts @@ -0,0 +1,216 @@ +import { randomUUID } from "node:crypto"; +import { createServer } from "node:http"; +import { afterEach, describe, expect, it, vi } from "vitest"; +import { DEEPSEEK_MODELS } from "@earendil-works/pi-ai/providers/deepseek.models"; +import { systemTranscriptCheckpoint } from "./system-transcript.js"; +import { readSystemMessage } from "./system-transcript-journal.js"; +import type { UiMessage } from "@pi-desktop/shared"; +import { modelConfigFromPi } from "./model-capabilities.js"; +import { DesktopAgentRuntime, type PluginToolDef, type RuntimeProviderConfig } from "./runtime.js"; + +function flashProvider(): RuntimeProviderConfig { + const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash")!; + return { id: "flash-fixture", name: "Flash", modelId: model.id, baseUrl: model.baseUrl, + apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"], + modelConfig: modelConfigFromPi(model) }; +} +const pluginTools: PluginToolDef[] = ["plugin_alpha", "plugin_beta"].map((name) => ({ + name, description: `${name} synthetic probe`, parameters: { type: "object", properties: {}, required: [] }, risk: "low", +})); +type Payload = { tools: { function: { name: string } }[]; messages: { role: string; content?: unknown }[] }; +type Call = { name: string; args?: Record }; +async function wireFixture(calls: Call[]) { + const requests: Payload[] = []; + const fetch = globalThis.fetch; + const server = createServer(async (req, res) => { + const chunks: Buffer[] = []; + for await (const chunk of req) chunks.push(Buffer.from(chunk)); + requests.push(JSON.parse(Buffer.concat(chunks).toString())); + const call = calls.shift(); + const delta = call ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), + type: "function", function: { name: call.name, arguments: JSON.stringify(call.args ?? {}) } }] } + : { role: "assistant", content: "Done." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + res.end(`data: ${JSON.stringify({ choices: [{ index: 0, delta, finish_reason: call ? "tool_calls" : "stop" }] })}\n\ndata: [DONE]\n\n`); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + const address = server.address(); + if (!address || typeof address === "string") throw new Error("Missing HTTP address"); + vi.stubGlobal("fetch", ((_url, init) => fetch(`http://127.0.0.1:${address.port}`, init)) satisfies typeof fetch); + return { requests, close: async () => { + vi.unstubAllGlobals(); + await new Promise((resolve, reject) => server.close((error) => error ? reject(error) : resolve())); + } }; +} +function runtimeFixture(history: UiMessage[] = [], tools = pluginTools, provider = flashProvider(), denied = false) { + const rows = structuredClone(history); + const executed: string[] = []; + const errors: unknown[] = []; + const runtime = new DesktopAgentRuntime({ + sessionId: "fixed-tools", mode: "agent", provider, thinkingLevel: "off", history: rows, pluginTools: tools, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + host: { call: async (method: string, params?: unknown): Promise => { + if (method === "session.appendMessage") rows.push((params as { message: UiMessage }).message); + else if (method === "tools.execute") { + executed.push((params as { toolName: string }).toolName); + return denied ? { ok: false, denied: true, content: "Permission denied" } as T + : { ok: true, content: "Synthetic success" } as T; + } else throw new Error(`Unexpected host method: ${method}`); + return undefined as T; + } }, + onEvent: ({ event }) => { + if (event.type === "error") errors.push(event.error); + if (event.type === "tool_start") rows.push({ id: event.toolCallId, role: "tool", content: "", + toolCallId: event.toolCallId, toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString() }); + if (event.type === "tool_end") { + const row = rows.find((row) => row.id === event.toolCallId)!; + row.toolResult = event.result; row.isError = event.isError; row.toolStatus = event.isError ? "error" : "success"; + } + if (event.type === "message_end") { + const index = rows.findIndex((row) => row.id === event.message.id); + if (index < 0) rows.push(event.message); else rows[index] = event.message; + } + }, + }); + return { runtime, rows, executed, errors, prompt: async (id = "user-1") => { + rows.push({ id, role: "user", content: "Run the synthetic probes", createdAt: new Date().toISOString() }); + await runtime.prompt("Run the synthetic probes", id, `turn-${id}`); + } }; +} +afterEach(() => vi.unstubAllGlobals()); +describe("fixed Flash declarations through runtime and HTTP/SSE", () => { + it("rejects a declared but inactive tool without invoking Host", async () => { + const wire = await wireFixture([{ name: "plugin_beta" }]); + const f = runtimeFixture(); + try { + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual([]); + expect(f.rows.find((row) => row.toolName === "plugin_beta")).toMatchObject({ isError: true }); + expect(JSON.stringify(wire.requests[1].messages)).toContain("Call ToolSearch to activate plugin_beta"); + expect(wire.requests[1].tools).toEqual(wire.requests[0].tools); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it.each([false, true])("restores only activated tools after restart (checkpoint=%s)", async (checkpoint) => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + let declared: Payload["tools"]; + try { + await f.prompt(); + expect(f.errors).toEqual([]); + declared = wire.requests.at(-1)!.tools; + history = structuredClone(f.rows); + if (checkpoint) { + const system = systemTranscriptCheckpoint(history.filter((row) => row.modelSystem) + .map((row) => readSystemMessage(row.modelSystem!.messageJson)))!; + history = [{ id: "checkpoint", role: "system", content: "", createdAt: new Date().toISOString(), + modelSystem: { version: 1, messageJson: JSON.stringify(system) } }]; + } + } finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("restored-user"); + expect(restored.errors).toEqual([]); + expect(restored.executed).toEqual(["plugin_alpha"]); + expect(resumed.requests[0].tools).toEqual(declared!); + expect([...restored.rows].reverse().find((row) => row.toolName === "plugin_beta")).toMatchObject({ isError: true }); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it.each(["schema", "removed", "route"])("does not revive activation after a %s change", async (change) => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + try { await f.prompt(); history = structuredClone(f.rows); } + finally { await f.runtime.dispose(); await wire.close(); } + const tools = change === "removed" ? pluginTools.slice(1) : change === "schema" + ? [{ ...pluginTools[0], description: "Changed schema epoch" }, pluginTools[1]] : pluginTools; + const provider = change === "route" ? { ...flashProvider(), baseUrl: "https://relay.invalid/v1" } : flashProvider(); + const resumed = await wireFixture([{ name: "plugin_alpha" }]); + const restored = runtimeFixture(history!, tools, provider); + try { + await restored.prompt("changed-user"); + expect(restored.errors).toEqual([]); + expect(restored.executed).toEqual([]); + if (change === "removed") expect(resumed.requests[0].tools.map((tool) => tool.function.name)).not.toContain("plugin_alpha"); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("still requires Host approval after ToolSearch activation", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }]); + const f = runtimeFixture([], pluginTools, flashProvider(), true); + try { + await f.prompt(); + expect(f.executed).toEqual(["plugin_alpha"]); + expect(f.rows.find((row) => row.toolName === "plugin_alpha")).toMatchObject({ isError: true }); + expect(JSON.stringify(wire.requests.at(-1)?.messages)).toContain("Permission denied"); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it("replays a successful activation if stopped before its next declaration checkpoint", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture(); + let history: UiMessage[]; + try { + await f.prompt(); + const first = f.rows.find((row) => row.modelSystem)!; + history = structuredClone(f.rows.filter((row) => !row.modelSystem || row === first)); + } finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("resumed-user"); + expect(restored.executed).toEqual(["plugin_alpha"]); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("migrates legacy on-demand history without granting the rest of the catalog", async () => { + const wire = await wireFixture([{ name: "ToolSearch", args: { query: "plugin_alpha" } }]); + const f = runtimeFixture([], pluginTools, { ...flashProvider(), baseUrl: "https://relay.invalid" }); + let history: UiMessage[]; + try { await f.prompt(); history = structuredClone(f.rows); } + finally { await f.runtime.dispose(); await wire.close(); } + const resumed = await wireFixture([{ name: "plugin_alpha" }, { name: "plugin_beta" }]); + const restored = runtimeFixture(history!); + try { + await restored.prompt("migrated-user"); + expect(restored.executed).toEqual(["plugin_alpha"]); + expect(resumed.requests[0].tools.map((tool) => tool.function.name)).toEqual(expect.arrayContaining(["plugin_alpha", "plugin_beta"])); + } finally { await restored.runtime.dispose(); await resumed.close(); } + }); + + it("retains the Plan execution guard for predeclared tools", async () => { + const wire = await wireFixture([{ name: "Write", args: { path: "fixture", content: "denied" } }]); + const f = runtimeFixture(); + try { + f.runtime.setMode("plan"); + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual([]); + expect(f.rows.find((row) => row.toolName === "Write")).toMatchObject({ isError: true }); + } finally { await f.runtime.dispose(); await wire.close(); } + }); + + it("keeps the complete request prefix while searching and executing two different tools", async () => { + const wire = await wireFixture([ + { name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha" }, + { name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta" }, + ]); + const f = runtimeFixture(); + try { + await f.prompt(); + expect(f.errors).toEqual([]); + expect(f.executed).toEqual(["plugin_alpha", "plugin_beta"]); + expect(wire.requests).toHaveLength(5); + expect(wire.requests[0].tools.map((tool) => tool.function.name)).toEqual(expect.arrayContaining(["plugin_alpha", "plugin_beta"])); + for (let index = 1; index < wire.requests.length; index++) { + expect(wire.requests[index].tools).toEqual(wire.requests[0].tools); + const previous = wire.requests[index - 1].messages; + expect(wire.requests[index].messages.slice(0, previous.length)).toEqual(previous); + } + } finally { await f.runtime.dispose(); await wire.close(); } + }); +}); diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index aa1379404..fbb689fd4 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -1,3 +1,4 @@ +import { TOOL_ACTIVATION_SECTION, toolDeclarationPolicy, toolActivationSection, restoredToolActivation, syncToolActivation, type ToolDeclarationPolicy } from "./fixed-tool-declarations.js"; import { orderSystemRows, SystemTranscriptJournal } from "./system-transcript-journal.js"; import { accountModelStream } from "./request-usage.js"; import { modeToolDenial, retainModeToolDeclaration, withModeExecutionGuard } from "./mode-tool-access.js"; @@ -1691,8 +1692,11 @@ export class DesktopAgentRuntime { private toolCatalog = new Map(); /** Tools intentionally omitted from the initial provider request. */ private deferredToolNames = new Set(); - /** Deferred tools loaded for the current user prompt. */ + /** Deferred tools activated for this runtime and declaration epoch. */ private activeDeferredToolNames = new Set(); + private declarationPolicy?: ToolDeclarationPolicy; + private trackToolActivation = false; + private activationHydrated = false; private scratchDir?: string; private projectPath?: string; private commandShell: CommandShellOption; @@ -1880,7 +1884,6 @@ export class DesktopAgentRuntime { this.rebuildToolCatalog(); const model = buildProviderModel(this.provider); this.model = model; - const tools = this.activeTools(); const models = createProviderModels(this.provider, model); this.models = models; const runtimeApiKey = providerRequestKey(this.provider); @@ -1934,6 +1937,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ).trim(), ...defaultSystemPromptParts.slice(1), ].join("\n\n"); + this.refreshToolDeclarationPolicy(); + this.restoreDeferredToolsFromContext(); + const tools = this.declaredTools(); this.agent = new Agent({ streamFn: (m, context, options) => { this.setAgentActivity({ phase: "waiting-model", since: Date.now() }); @@ -2204,7 +2210,13 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } private setAgentTools(tools: AgentTool[]): void { - this.agent.state.tools = tools; + this.agent.state.tools = this.declarationPolicy?.tools ?? tools; + if (this.trackToolActivation && this.declarationPolicy) { + const section = toolActivationSection(this.declarationPolicy.key, this.activeDeferredToolNames); + this.composedSections = { ...this.composedSections, [TOOL_ACTIVATION_SECTION]: section }; + this.composedSystemPrompt = Object.values(this.composedSections).filter(Boolean).join("\n\n"); + this.agent.state.messages = syncToolActivation(this.agent.state.messages, section); + } } private agentSystemPromptContent(): string { @@ -2225,6 +2237,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the : ""; this.composedSections = { runtime: this.baseSystemPrompt, + ...(this.trackToolActivation && this.declarationPolicy ? { + [TOOL_ACTIVATION_SECTION]: toolActivationSection(this.declarationPolicy.key, this.activeDeferredToolNames), + } : {}), ...pluginSkillsPromptSections(this.pluginSkills), context: composeModeSystemPrompt(this.mode, [ ...(this.customSystemPrompt?.append ? [this.customSystemPrompt.append] : []), @@ -2385,6 +2400,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private async beforeToolCall( context: BeforeToolCallContext, ): Promise { + if (this.deferredToolNames.has(context.toolCall.name) && !this.activeDeferredToolNames.has(context.toolCall.name)) { + return { block: true, reason: `Call ${TOOL_SEARCH_NAME} to activate ${context.toolCall.name} before using it. A tool declaration does not grant execution permission.` }; + } const toolCalls = (context.assistantMessage.content as Array<{ type?: string }>).filter( (block) => block.type === "toolCall", ); @@ -3612,9 +3630,9 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } /** - * Build the complete registry once, then expose only the core subset to the - * first provider request. This mirrors pi's active-tool model while keeping - * the host tool implementation and permission path unchanged. + * Build the complete registry before selecting fixed or on-demand + * declarations. Execution activation and Host permissions remain separate + * from the provider-visible schemas. */ private rebuildToolCatalog(): void { const catalog = new Map(); @@ -3659,6 +3677,25 @@ Do not invent objections or turn speculative risks into blockers. Stop when the }), ); } + if (this.model) this.refreshToolDeclarationPolicy(); + } + + private refreshToolDeclarationPolicy(): void { + const previous = this.declarationPolicy; + const policy = toolDeclarationPolicy(this.model, [...this.toolCatalog.values()], this.deferredToolNames, this.composeSystemPrompt(), this.provider.id); + if (previous && previous.key !== policy.key && this.trackToolActivation) { + this.activeDeferredToolNames.clear(); + this.activationHydrated = true; + } + this.declarationPolicy = policy; + this.trackToolActivation ||= Boolean(policy.tools || policy.fallback); + if (policy.fallback && (previous?.key !== policy.key || previous.fallback !== policy.fallback)) { + process.stderr.write(`[agent-runtime] fixed tool declarations unavailable (${policy.fallback}); using on-demand declarations; ToolSearch cache stability is not guaranteed.\n`); + } + } + + private declaredTools(): AgentTool[] { + return this.declarationPolicy?.tools ?? this.activeTools(); } private isPlanSafePluginTool(name: string): boolean { @@ -3747,7 +3784,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the } return [ "# On-demand tools", - `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using one that is not in the current tool list.`, + `The following capabilities are available on demand. Call ${TOOL_SEARCH_NAME} with an exact tool name or a short capability description before using an on-demand tool that has not been activated. A visible schema is not activation; the tool_activation section, when present, records active names.`, ...lines, ].join("\n"); } @@ -3777,7 +3814,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the name: TOOL_SEARCH_NAME, label: "Tool Search", description: - "Find and activate an on-demand tool by exact name or capability. Use this before calling any tool listed under On-demand tools that is not already in the current tool list.", + "Find and activate an on-demand tool by exact name or capability. Use this before calling an inactive tool listed under On-demand tools, even when its schema is already visible. Activation does not bypass approval or mode restrictions.", parameters: Type.Object({ query: Type.String({ description: @@ -5340,9 +5377,18 @@ Do not invent objections or turn speculative risks into blockers. Stop when the */ private restoreDeferredToolsFromContext(): void { if (this.deferredToolNames.size === 0) return; + if (this.trackToolActivation && this.activationHydrated) return; const { messages } = this.liveSessionContext(); - const lastSystem = messages.map((message) => message.role).lastIndexOf("system"); - if (lastSystem >= 0) { + const restored = this.declarationPolicy && restoredToolActivation(messages, this.declarationPolicy.key); + if (restored !== undefined) { + this.trackToolActivation = true; + for (const name of restored.active) { + if (this.deferredToolNames.has(name)) this.activeDeferredToolNames.add(name); + } + } + this.activationHydrated = true; + const lastSystem = restored ? restored.replayFrom - 1 : messages.map((message) => message.role).lastIndexOf("system"); + if (!restored && lastSystem >= 0) { for (const tool of getCurrentTools(messages)) { if (this.deferredToolNames.has(tool.name)) this.activeDeferredToolNames.add(tool.name); } @@ -5350,6 +5396,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the // Legacy histories lack declarations. For current histories, only results // after the last declaration can represent an activation not yet declared. for (const message of messages.slice(lastSystem + 1)) { + if (restored && (message.role !== "toolResult" || message.toolName !== TOOL_SEARCH_NAME)) continue; if (message.role !== "toolResult" || message.isError) continue; if (isMissingToolResultPlaceholder(message.content)) continue; const names = @@ -6127,7 +6174,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the ...(typeof this.agent.state.systemPrompt === "string" ? { systemPrompt: this.agent.state.systemPrompt } : {}), - tools: this.activeTools(), + tools: this.declaredTools(), }, this.model, ); @@ -6300,7 +6347,7 @@ Do not invent objections or turn speculative risks into blockers. Stop when the private rebuiltAgentContext(): AgentContext { const messages = this.liveSessionContext().messages; - const tools = this.activeTools(); + const tools = this.declaredTools(); this.setAgentMessages(messages); this.setAgentTools(tools); return { diff --git a/packages/agent-runtime/src/system-transcript.ts b/packages/agent-runtime/src/system-transcript.ts index f3e347033..9ba1167b2 100644 --- a/packages/agent-runtime/src/system-transcript.ts +++ b/packages/agent-runtime/src/system-transcript.ts @@ -1,3 +1,4 @@ +import { TOOL_ACTIVATION_SECTION } from "./fixed-tool-declarations.js"; import type { AgentMessage } from "@earendil-works/pi-agent-core"; import { contentText, @@ -35,7 +36,7 @@ function currentSystemMessage(messages: readonly AgentMessage[]): SystemMessage export const CONTEXT_BUDGET_SECTION = "context_budget"; function desktopSectionNames(sections: Record): string[] { - return ["runtime", "skills", ...Object.keys(sections) + return ["runtime", TOOL_ACTIVATION_SECTION, "skills", ...Object.keys(sections) .filter((name) => name.startsWith(SKILL_SECTION_PREFIX)) .sort((a, b) => a.localeCompare(b)), "context"]; } @@ -83,7 +84,8 @@ export function initialSystemTranscript( export function replaceSystemPrompt(messages: AgentMessage[], prompt: string): AgentMessage[] { if (prompt === getCurrentSystemPrompt(messages) || prompt === systemPromptContent(messages)) return messages; - return syncSystemSections(messages, { runtime: prompt }); + const activation = getCurrentSystemMessage(messages)?.sections?.[TOOL_ACTIVATION_SECTION]; + return syncSystemSections(messages, { runtime: prompt, ...(activation ? { [TOOL_ACTIVATION_SECTION]: activation } : {}) }); } export function rebuildSystemTranscript( diff --git a/scripts/e2e-fixed-tool-declarations.mjs b/scripts/e2e-fixed-tool-declarations.mjs new file mode 100644 index 000000000..c44d32ef5 --- /dev/null +++ b/scripts/e2e-fixed-tool-declarations.mjs @@ -0,0 +1,170 @@ +#!/usr/bin/env node +/** Production sidecar transport + isolated Host + local SSE provider. + * Message persistence is harness-owned; Electron's UI/outbox is not exercised. + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +import { DEEPSEEK_MODELS } from "../packages/agent-runtime/node_modules/@earendil-works/pi-ai/dist/providers/deepseek.models.js"; +import { modelConfigFromPi } from "../packages/agent-runtime/dist/model-capabilities.js"; +import { PROTOCOL_VERSION } from "../packages/shared/dist/protocol.js"; +import { withScenario } from "./e2e/fixture.mjs"; +import { resolveHostBinary } from "./e2e/host.mjs"; +import { createSession } from "./e2e/session.mjs"; + +const root = await mkdtemp(join(tmpdir(), "pi-system-state-e2e-")); +const requests = []; +const responses = []; +const server = createServer(async (req, res) => { + let body = ""; + for await (const chunk of req) body += chunk; + requests.push(JSON.parse(body)); + const tool = responses.shift(); + const delta = tool ? { role: "assistant", tool_calls: [{ index: 0, id: randomUUID(), type: "function", + function: { name: tool.name, arguments: JSON.stringify(tool.args) } }] } + : { role: "assistant", content: "The fixture completed." }; + res.writeHead(200, { "content-type": "text/event-stream" }); + for (const [value, finish] of [[delta, null], [{}, tool ? "tool_calls" : "stop"]]) { + res.write(`data: ${JSON.stringify({ id: "fixture", model: "fixture", choices: [{ index: 0, delta: value, finish_reason: finish }] })}\n\n`); + } + res.end("data: [DONE]\n\n"); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +const baseUrl = `http://127.0.0.1:${server.address().port}/v1`; +const model = Object.values(DEEPSEEK_MODELS).find((model) => model.id === "deepseek-flash"); +const provider = { id: "fixture", name: "Flash fixture", modelId: model.id, baseUrl: model.baseUrl, + modelConfig: modelConfigFromPi(model), apiKey: "fixture", authKind: "api_key", supportsReasoning: false, supportedThinkingLevels: ["off"] }; +const fetchHook = join(root, "fixture-fetch.mjs"); +await writeFile(fetchHook, `const fetch = globalThis.fetch; globalThis.fetch = (url, init) => { + if (String(url) !== "https://api.deepseek.com/chat/completions") throw new Error("Unexpected fixture endpoint"); + return fetch(${JSON.stringify(baseUrl)}, init); +};`); +let pluginTools = ["plugin_alpha", "plugin_beta"].map((name) => ({ name, + description: "Synthetic read-only marker", parameters: { type: "object", properties: {}, required: [] }, risk: "low" })); +try { + await withScenario("E2E-fixed-tool-declarations", async ({ host, workspace }) => { + const session = await createSession(host, workspace, "System state fixture"); + let writes = Promise.resolve(); + let writeError; + let stderr = ""; + const turns = new Map(); + const executedTools = []; + const toolRows = new Map(); + const persist = (message, turnId) => { + writes = writes.then(() => host.call("session.appendMessage", { sessionId: session.id, turnId, message })) + .catch((error) => { writeError ??= error; }); + }; + const startSidecar = () => { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, + args: [fileURLToPath(new URL("../packages/agent-runtime/dist/sidecar.js", import.meta.url))], + cwd: workspace, + env: { ...process.env, NODE_OPTIONS: `--import=${fetchHook}`, PI_DESKTOP_PLAN_UI_PROBE: "1", PI_DESKTOP_COMPACTION_STRATEGY: "fresh_window" }, + }, + onStderr: (chunk) => { stderr += chunk; }, + }); + sidecar.setHost({ call: host.call.bind(host), onNotification: () => () => {}, onExit: () => () => {} }); + for (const name of ["plugin_alpha", "plugin_beta"]) sidecar.setLocalTool(name, async () => { + executedTools.push(name); + return { ok: true, content: `${name} synthetic result` }; + }); + sidecar.onNotification((method, envelope) => { + if (method !== "agent.event") return; + const { event, turnId } = envelope; + const turn = turns.get(turnId); + if (event.type === "error") turn?.errors.push(event.error); + if (event.type === "message_end") persist(event.message, turnId); + if (event.type === "tool_start") toolRows.set(event.toolCallId, { + id: event.toolCallId, role: "tool", content: "", toolCallId: event.toolCallId, + toolName: event.toolName, toolArgs: event.args, createdAt: new Date().toISOString(), + }); + if (event.type === "tool_end") { + turn?.toolResults.push(event); + persist({ ...toolRows.get(event.toolCallId), toolResult: event.result, isError: event.isError, + toolStatus: event.isError ? "error" : "success" }, turnId); + } + if (event.type === "agent_end") turn?.resolve(); + }); + return sidecar; + }; + let sidecar = startSidecar(); + const params = () => ({ sessionId: session.id, mode: "agent", provider, thinkingLevel: "off", pluginTools, + projectPath: workspace, + commandShell: { id: "bash", label: "Bash", dialect: "posix", available: true, isDefault: true }, + }); + const prompt = async (id, expectedErrors = 0) => { + const { turnId } = await host.call("session.beginTurn", { sessionId: session.id }); + await host.call("session.appendMessage", { sessionId: session.id, turnId, + message: { id, role: "user", content: "Continue", createdAt: new Date().toISOString() } }); + let timer; + const turn = { errors: [], toolResults: [] }; + const ended = new Promise((resolve, reject) => { + turn.resolve = resolve; + timer = setTimeout(() => reject(new Error(`Turn timed out: ${JSON.stringify(turn.errors)}\n${stderr}`)), 20_000); + }); + turns.set(turnId, turn); + try { + await sidecar.call("agent.prompt", { ...params(), turnId, userMessageId: id, content: "Continue" }); + await ended; + await writes; + if (writeError) throw writeError; + assert.deepEqual(turn.errors, [], JSON.stringify(turn.errors)); + assert.equal(turn.toolResults.filter((event) => event.isError).length, expectedErrors, JSON.stringify(turn.toolResults)); + await host.call("session.endTurn", { turnId, status: "completed", createNotification: false }); + return turn; + } finally { clearTimeout(timer); turns.delete(turnId); } + }; + const detail = async () => (await host.call("session.get", { id: session.id })).session; + try { + responses.push({ name: "ToolSearch", args: { query: "plugin_alpha" } }, { name: "plugin_alpha", args: {} }); + await prompt("user-1"); + const declared = requests[0].tools; + assert(declared.some((tool) => tool.function.name === "plugin_beta")); + for (let index = 1; index < requests.length; index++) { + assert.deepEqual(requests[index].tools, declared); + assert.deepEqual(requests[index].messages.slice(0, requests[index - 1].messages.length), requests[index - 1].messages); + } + assert.deepEqual(executedTools, ["plugin_alpha"]); + const states = (await detail()).messages.filter((row) => row.modelSystem); + assert(states.length >= 2); + console.log("PASS: fixed declarations survive ToolSearch through production sidecar and HTTP/SSE"); + await sidecar.dispose(); + await host.stop(); await host.start(PROTOCOL_VERSION); + assert.deepEqual((await detail()).messages.filter((row) => row.modelSystem), states); + sidecar = startSidecar(); + responses.push({ name: "plugin_alpha", args: {} }, { name: "plugin_beta", args: {} }); + await prompt("user-2", 1); + assert.deepEqual(executedTools, ["plugin_alpha", "plugin_alpha"]); + assert.deepEqual(requests.at(-1).tools, declared); + console.log("PASS: Host and sidecar restart restore Alpha activation without activating declared Beta"); + await sidecar.call("agent.compact", params()); + const checkpoint = JSON.parse((await detail()).compaction.details.systemMessageJson); + assert.deepEqual(JSON.parse(checkpoint.sections.tool_activation).active, ["plugin_alpha"]); + await sidecar.dispose(); sidecar = startSidecar(); + responses.push({ name: "plugin_alpha", args: {} }, { name: "plugin_beta", args: {} }); + await prompt("user-3", 1); + assert.deepEqual(executedTools, ["plugin_alpha", "plugin_alpha", "plugin_alpha"]); + assert.deepEqual(requests.at(-1).tools, declared); + console.log("PASS: compaction and process restart preserve declarations and activation separately"); + responses.push({ name: "ToolSearch", args: { query: "plugin_beta" } }, { name: "plugin_beta", args: {} }); + await prompt("user-4"); + assert.equal(executedTools.at(-1), "plugin_beta"); + assert.deepEqual(requests.at(-1).tools, declared); + pluginTools = pluginTools.slice(0, 1); + responses.push({ name: "plugin_beta", args: {} }); + await prompt("user-5", 1); + assert.equal(executedTools.length, 4); + assert(!requests.at(-1).tools.some((tool) => tool.function.name === "plugin_beta")); + console.log("PASS: changed catalog revokes Beta execution and creates a new declaration epoch"); + } finally { await sidecar.dispose(); await writes; } + }, resolveHostBinary(), root, PROTOCOL_VERSION); +} finally { + server.closeAllConnections(); + await new Promise((resolve) => server.close(resolve)); + await rm(root, { recursive: true, force: true }); +}