From b965cd7ddd1d9716e72b0fa6c76a8b3bd40c7dde Mon Sep 17 00:00:00 2001 From: Hunter Bown Date: Mon, 28 Sep 2026 20:22:27 -0700 Subject: [PATCH 1/5] feat(hooks): export a post-admission execution receipt to tool_call_after (#6689) tool_call_after hooks could not tell what a shell tool actually ran. They got the result text, the success flag, the exit code and the status, but not the command or its directory. A tool_call_before rewrite can change the before-hook input they would otherwise re-derive those from. DEEPSEEK_TOOL_EXECUTION_RECEIPT now carries schema-1 JSON with these fields: command, cwd, state (completed/interrupted), scope, exit_code (observed or null), bounded stdout/stderr previews, truncation flags and output_kind. - Identity reuses the spawn record. execute_foreground_via_background reads command and working_dir from the BackgroundShell record the process manager created, under the same lock that marks the run as foreground. - HookContext::with_tool_outcome reads metadata.execution_receipt from Ok and Err results. That one seam covers the TUI and Runtime API completion hooks, and on_error for a failed shell call. Lowercase bash failures carry the receipt too. - Exact or absent. An identity over 8 KiB, containing NUL, or with a relative or non-UTF-8 directory gets no receipt. Previews keep both ends and halve until the serialized JSON is at most 32 KiB. The hook boundary drops a non-schema-1 or oversized receipt instead of truncating it. - Scope is settled, pipe-backed, unsandboxed, local foreground runs only. No receipt for background, moved-to-/jobs, PTY/interactive, OS sandbox, external backend, read-only argv hardening, Windows, or pre-exec refusals. - Documented in docs/HOOKS.md and docs/zh_hans/HOOKS.md. The contract and test design follow Isabel Wu's reference branch wuisabel-gif/CodeWhale@feat/tool-call-after-execution-receipt. This version is re-implemented on the current with_tool_outcome seam and on main's process-manager record. Evidence (macOS, CARGO_TARGET_DIR outside the repo): cargo test -p codewhale-tui --lib -- execution_receipt -> test result: ok. 3 passed; 0 failed; 0 ignored; 13816 filtered out same run with the metadata read and identity capture disabled -> test result: FAILED. 1 passed; 2 failed (both new behavior tests fail) cargo test -p codewhale-tui --lib -- hooks:: tools::shell:: tool_routing -> test result: ok. 340 passed; 0 failed; 1 ignored cargo test -p codewhale-tui --lib -- runtime_tool_completion_fires_after_and_error_hooks runtime_shell_completion_delivers_exit_code_and_status_to_hooks -> test result: ok. 2 passed; 0 failed cargo clippy -p codewhale-tui --lib --tests --locked -- -D warnings (CI allow-list) -> Finished, 0 warnings cargo fmt --all -- --check -> clean Co-authored-by: Isabel Wu <231155141+wuisabel-gif@users.noreply.github.com> Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014ZwqatxgVFxHvovngywnks --- crates/tui/src/hooks.rs | 6 +- crates/tui/src/hooks/executor.rs | 118 +++++++++++++++++++++ crates/tui/src/tools/shell.rs | 154 ++++++++++++++++++++++++++- crates/tui/src/tools/shell/tests.rs | 157 +++++++++++++++++++++++++++- docs/HOOKS.md | 39 +++++++ docs/zh_hans/HOOKS.md | 27 +++++ 6 files changed, 492 insertions(+), 9 deletions(-) diff --git a/crates/tui/src/hooks.rs b/crates/tui/src/hooks.rs index c3a6c11eb4..d12d3af22e 100644 --- a/crates/tui/src/hooks.rs +++ b/crates/tui/src/hooks.rs @@ -31,9 +31,9 @@ pub use config::{ PROJECT_HOOKS_TEMPLATE, workspace_allows_project_hooks, }; pub(crate) use executor::{ - HOOK_CONTEXT_AGGREGATE_MAX_CHARS, HOOK_LABEL_MAX_CHARS, generic_unavailable_detail, - parse_tool_call_before_stdout, sanitize_hook_denial_reason, sanitize_hook_label, - sanitize_hook_line, sanitize_hook_text, + HOOK_CONTEXT_AGGREGATE_MAX_CHARS, HOOK_EXECUTION_RECEIPT_MAX_BYTES, HOOK_LABEL_MAX_CHARS, + generic_unavailable_detail, parse_tool_call_before_stdout, sanitize_hook_denial_reason, + sanitize_hook_label, sanitize_hook_line, sanitize_hook_text, }; #[cfg(test)] pub(crate) use executor::{ diff --git a/crates/tui/src/hooks/executor.rs b/crates/tui/src/hooks/executor.rs index 7930311fe6..f5d589946e 100644 --- a/crates/tui/src/hooks/executor.rs +++ b/crates/tui/src/hooks/executor.rs @@ -47,6 +47,10 @@ pub struct HookContext { /// command usually has no exit code, so this is how a hook tells it apart /// from a tool that reported nothing. pub tool_status: Option, + /// Serialized post-admission shell execution receipt (#6689), exported as + /// `DEEPSEEK_TOOL_EXECUTION_RECEIPT`. Complete JSON or absent — never a + /// truncated document — and at most [`HOOK_EXECUTION_RECEIPT_MAX_BYTES`]. + pub tool_execution_receipt: Option, /// Whether tool succeeded pub tool_success: Option, /// Current mode @@ -115,6 +119,7 @@ impl HookContext { }; let mut context = self.with_tool_result(&text, success, reported_tool_exit_code(result)); context.tool_status = reported_tool_status(result).map(str::to_string); + context.tool_execution_receipt = reported_tool_execution_receipt(result); context } @@ -179,6 +184,15 @@ impl HookContext { bound(&mut self.message, HOOK_MESSAGE_CONTEXT_MAX_BYTES); bound(&mut self.error_message, HOOK_ERROR_CONTEXT_MAX_BYTES); bound(&mut self.model, HOOK_OBSERVER_METADATA_MAX_BYTES); + // A receipt is complete JSON or nothing: truncating it would export a + // broken document, so an oversized one is dropped instead. + if self + .tool_execution_receipt + .as_ref() + .is_some_and(|receipt| receipt.len() > HOOK_EXECUTION_RECEIPT_MAX_BYTES) + { + self.tool_execution_receipt = None; + } if let Some(workspace) = self.workspace.take() { self.workspace = Some(PathBuf::from(truncate_env_value( &workspace.to_string_lossy(), @@ -225,6 +239,14 @@ impl HookContext { if let Some(ref status) = self.tool_status { env.insert("DEEPSEEK_TOOL_STATUS".to_string(), status.clone()); } + if let Some(ref receipt) = self.tool_execution_receipt + && receipt.len() <= HOOK_EXECUTION_RECEIPT_MAX_BYTES + { + env.insert( + "DEEPSEEK_TOOL_EXECUTION_RECEIPT".to_string(), + receipt.clone(), + ); + } if let Some(ref mode) = self.mode { env.insert("DEEPSEEK_MODE".to_string(), mode.clone()); } @@ -405,6 +427,12 @@ const HOOK_TOOL_ARGS_ENV_MAX_BYTES: usize = 10_000; /// Largest raw tool result retained in an observer job before enqueue. const HOOK_TOOL_RESULT_CONTEXT_MAX_BYTES: usize = 10_000; +/// Largest serialized shell execution receipt exported through +/// `DEEPSEEK_TOOL_EXECUTION_RECEIPT`. The shell tool fits its output previews +/// beneath this bound; the hook boundary drops anything larger rather than +/// truncate a JSON document. +pub(crate) const HOOK_EXECUTION_RECEIPT_MAX_BYTES: usize = 32 * 1024; + /// Largest error retained in an observer job before enqueue. const HOOK_ERROR_CONTEXT_MAX_BYTES: usize = 5_000; @@ -3011,6 +3039,24 @@ fn reported_tool_status( } } +/// Read the post-admission execution receipt a shell tool recorded (#6689), +/// serialized for `DEEPSEEK_TOOL_EXECUTION_RECEIPT`. +/// +/// Only a schema-1 object within the size bound counts. The receipt is built +/// by the shell tool from what its process manager recorded at spawn; it is +/// never reconstructed here from the before-hook input, which can differ from +/// what actually ran. +fn reported_tool_execution_receipt( + result: &Result, +) -> Option { + let receipt = reported_tool_metadata(result)?.get("execution_receipt")?; + if receipt.get("schema_version")?.as_u64()? != 1 { + return None; + } + let encoded = serde_json::to_string(receipt).ok()?; + (encoded.len() <= HOOK_EXECUTION_RECEIPT_MAX_BYTES).then_some(encoded) +} + /// Read the process exit code a tool reported, when it reported one. /// /// The one source for `DEEPSEEK_TOOL_EXIT_CODE`: the TUI and the Runtime API @@ -5617,6 +5663,78 @@ command = "echo project" ); } + /// #6689: `DEEPSEEK_TOOL_EXECUTION_RECEIPT` is read from the metadata a + /// shell tool recorded — on a failed call as well as a successful one — + /// and is complete JSON or absent. The existing variables do not change. + #[test] + fn execution_receipt_env_is_complete_json_or_absent() { + use crate::tools::spec::{ToolError, ToolResult}; + + let receipt = json!({"schema_version": 1, "command": "printf effective", + "cwd": "/tmp", "state": "completed", "scope": "local", "exit_code": 7, + "stdout": "\u{1f40b}", "stderr": "", "stdout_truncated": false, + "stderr_truncated": false, "output_kind": "separate"}); + let plain = HookContext::new() + .with_tool_name("Bash") + .with_tool_outcome(&Ok(ToolResult::success("out"))); + let legacy = plain.to_env_vars(); + assert!(!legacy.contains_key("DEEPSEEK_TOOL_EXECUTION_RECEIPT")); + + let with_receipt = HookContext::new() + .with_tool_name("Bash") + .with_tool_outcome(&Ok( + ToolResult::success("out").with_metadata(json!({"execution_receipt": receipt})) + )); + let mut env = with_receipt.to_env_vars(); + let encoded = env + .remove("DEEPSEEK_TOOL_EXECUTION_RECEIPT") + .expect("receipt exported"); + assert_eq!( + serde_json::from_str::(&encoded).unwrap(), + receipt + ); + assert_eq!(env, legacy, "existing variables are unchanged"); + + let failed = + HookContext::new().with_tool_outcome(&Err(ToolError::execution_failed_with_metadata( + "boom", + json!({"exit_code": 7, "execution_receipt": receipt}), + ))); + assert!( + failed + .to_env_vars() + .contains_key("DEEPSEEK_TOOL_EXECUTION_RECEIPT") + ); + + // An unknown schema or an oversized document is dropped, not cut. + for bad in [ + json!({"schema_version": 2, "command": "x"}), + json!({"command": "x"}), + json!({"schema_version": 1, + "stdout": "x".repeat(super::HOOK_EXECUTION_RECEIPT_MAX_BYTES)}), + ] { + let context = HookContext::new().with_tool_outcome(&Ok( + ToolResult::success("out").with_metadata(json!({"execution_receipt": bad})) + )); + assert!(context.tool_execution_receipt.is_none()); + } + let oversized = HookContext { + tool_execution_receipt: Some("x".repeat(super::HOOK_EXECUTION_RECEIPT_MAX_BYTES + 1)), + ..HookContext::new() + }; + assert!( + !oversized + .to_env_vars() + .contains_key("DEEPSEEK_TOOL_EXECUTION_RECEIPT") + ); + assert!( + oversized + .bounded_for_observer() + .tool_execution_receipt + .is_none() + ); + } + #[test] fn observer_context_is_bounded_before_enqueue() { let huge = "用户".repeat(20_000); diff --git a/crates/tui/src/tools/shell.rs b/crates/tui/src/tools/shell.rs index 68ded0793b..04109f8af2 100644 --- a/crates/tui/src/tools/shell.rs +++ b/crates/tui/src/tools/shell.rs @@ -4763,6 +4763,7 @@ async fn execute_foreground_via_background( extra_env: HashMap, direct_argv: bool, timeout_bounds_ms: (u64, u64), + receipt_identity: &mut Option, ) -> Result { let timeout_ms = timeout_ms.map(|timeout| timeout.clamp(timeout_bounds_ms.0, timeout_bounds_ms.1)); @@ -4798,11 +4799,23 @@ async fn execute_foreground_via_background( .ok_or_else(|| anyhow!("foreground shell did not return a process id"))?; // Classify before releasing the manager lock: even an immediately // completed foreground command must never look like background work. - manager + let process = manager .processes .get_mut(&task_id) - .ok_or_else(|| anyhow!("foreground shell {task_id} is not tracked"))? - .background = false; + .ok_or_else(|| anyhow!("foreground shell {task_id} is not tracked"))?; + process.background = false; + // #6689: the receipt identity is what the manager recorded for this + // spawn — the admitted command and the directory handed to the OS — + // never the before-hook request. Only pipe-backed, unsandboxed local + // runs qualify: a PTY, the hardened read-only argv rewrite, an OS + // sandbox wrapper, and Windows shell prefixes all change what the + // process actually executes relative to this string. + if !cfg!(windows) && !tty && !direct_argv && !spawned.sandboxed && !spawned.sandbox_denied { + *receipt_identity = Some(ShellExecutionIdentity { + command: process.command.clone(), + cwd: process.working_dir.clone(), + }); + } task_id }; let mut foreground = ForegroundShellGuard { @@ -4933,6 +4946,110 @@ impl Drop for ForegroundShellGuard { } } +/// The admitted command and working directory a foreground spawn recorded, +/// for the `tool_call_after` execution receipt (#6689). +struct ShellExecutionIdentity { + command: String, + cwd: PathBuf, +} + +/// Largest command or working directory a receipt carries. Identities are +/// exact or absent, never truncated. +const EXECUTION_RECEIPT_IDENTITY_MAX_BYTES: usize = 8 * 1024; + +/// Starting per-stream output preview in an execution receipt. Halved until +/// the serialized receipt fits `HOOK_EXECUTION_RECEIPT_MAX_BYTES`. +const EXECUTION_RECEIPT_PREVIEW_MAX_BYTES: usize = 8 * 1024; + +/// Keep both ends of an output stream — the first lines and the final +/// diagnostic — cut on UTF-8 boundaries with a visible marker. +fn execution_receipt_preview(text: &str, budget: usize) -> (String, bool) { + if text.len() <= budget { + return (text.to_owned(), false); + } + let mut head = budget / 2; + while !text.is_char_boundary(head) { + head -= 1; + } + let mut tail = text.len() - budget / 2; + while !text.is_char_boundary(tail) { + tail += 1; + } + ( + format!( + "{}\n[receipt preview truncated]\n{}", + &text[..head], + &text[tail..] + ), + true, + ) +} + +/// Build the schema-1 execution receipt for a settled foreground shell run +/// (#6689), exported to `tool_call_after` hooks as +/// `DEEPSEEK_TOOL_EXECUTION_RECEIPT`. +/// +/// Returns `None` — absence, which implies neither success nor failure — for +/// a run still in flight, a sandboxed run, or an identity that is not exact: +/// empty, over the identity bound, containing NUL, or a relative or non-UTF-8 +/// directory. `output_kind` is `"combined"` when stdout and stderr shared one +/// pipe, so `stdout` holds the combined preview. +fn shell_execution_receipt( + identity: &ShellExecutionIdentity, + result: &ShellResult, + output_kind: &'static str, +) -> Option { + let command = identity.command.as_str(); + let cwd = identity.cwd.to_str()?; + if command.is_empty() + || command.len() > EXECUTION_RECEIPT_IDENTITY_MAX_BYTES + || command.contains('\0') + || cwd.len() > EXECUTION_RECEIPT_IDENTITY_MAX_BYTES + || cwd.contains('\0') + || !identity.cwd.is_absolute() + || result.sandboxed + || result.sandbox_denied + { + return None; + } + // A nonzero exit is still a completed run; `interrupted` is a signal, + // kill, cancel, or timeout. The exit code is only ever the observed one. + let state = match result.status { + ShellStatus::Completed => "completed", + ShellStatus::Failed if result.exit_code.is_some() => "completed", + ShellStatus::Failed | ShellStatus::Killed | ShellStatus::TimedOut => "interrupted", + ShellStatus::Running => return None, + }; + let mut budget = EXECUTION_RECEIPT_PREVIEW_MAX_BYTES; + loop { + let (stdout, stdout_clipped) = execution_receipt_preview(&result.stdout, budget); + let (stderr, stderr_clipped) = execution_receipt_preview(&result.stderr, budget); + let receipt = json!({ + "schema_version": 1, + "command": command, + "cwd": cwd, + "state": state, + "scope": "local", + "exit_code": result.exit_code, + "stdout": stdout, + "stderr": stderr, + "stdout_truncated": result.stdout_truncated || stdout_clipped, + "stderr_truncated": result.stderr_truncated || stderr_clipped, + "output_kind": output_kind, + }); + // The bound is on the serialized form, so JSON escaping counts. + if serde_json::to_vec(&receipt).ok()?.len() + <= crate::hooks::HOOK_EXECUTION_RECEIPT_MAX_BYTES + { + return Some(receipt); + } + if budget == 0 { + return None; + } + budget /= 2; + } +} + const BASH_MAX_TIMEOUT_MS: u64 = i32::MAX as u64; /// Initial cadence for foreground-via-background completion polling. Fast @@ -5001,6 +5118,7 @@ fn finish_contract_bash_result( result: ShellResult, timeout_ms: Option, context: &ToolContext, + execution_receipt: Option, ) -> Result { let sandbox_denied_hint = shell_sandbox_denied_hint(context, &result); let mut output = result.stdout.clone(); @@ -5012,12 +5130,15 @@ fn finish_contract_bash_result( format!("{hint}\n\n{output}") }; } - let metadata = json!({ + let mut metadata = json!({ "evidence_routing": "inline", "exit_code": result.exit_code, "status": format!("{:?}", result.status), "duration_ms": result.duration_ms, "sandboxed": result.sandboxed, "sandbox_type": result.sandbox_type, "task_id": result.task_id, "backgrounded": result.status == ShellStatus::Running, }); + if let Some(receipt) = execution_receipt { + metadata["execution_receipt"] = receipt; + } if result.status == ShellStatus::Running { let task_id = result.task_id.as_deref().unwrap_or("unknown"); let partial = (!output.is_empty()).then(|| format!("\n\nOutput so far:\n{output}")); @@ -5928,6 +6049,7 @@ impl ToolSpec for BashTool { } let mut lifecycle_warning = None; + let mut receipt_identity = None; let result = if interactive { let mut manager = context .shell_manager @@ -6011,6 +6133,7 @@ impl ToolSpec for BashTool { } else { (1_000, 600_000) }, + &mut receipt_identity, ) .await }; @@ -6033,8 +6156,26 @@ impl ToolSpec for BashTool { .cancel_token .as_ref() .is_some_and(|token| token.is_cancelled()); + // Lowercase `bash` joins stdout and stderr on one pipe, so + // its preview is combined output, not stdout. + let execution_receipt = receipt_identity.as_ref().and_then(|identity| { + shell_execution_receipt( + identity, + &result, + if self.optional_timeout { + "combined" + } else { + "separate" + }, + ) + }); if self.optional_timeout { - return finish_contract_bash_result(result, timeout_ms, context); + return finish_contract_bash_result( + result, + timeout_ms, + context, + execution_receipt, + ); } let task_id_str = result.task_id.clone().unwrap_or_default(); let stdout_summary = summarize_output(&result.stdout); @@ -6165,6 +6306,9 @@ impl ToolSpec for BashTool { }), }), }); + if let Some(receipt) = execution_receipt { + metadata["execution_receipt"] = receipt; + } metadata["backgrounded"] = json!(background || backgrounded_foreground); if persist { metadata["persist_requested"] = json!(true); diff --git a/crates/tui/src/tools/shell/tests.rs b/crates/tui/src/tools/shell/tests.rs index 72bfcd4793..0faa75d100 100644 --- a/crates/tui/src/tools/shell/tests.rs +++ b/crates/tui/src/tools/shell/tests.rs @@ -263,6 +263,7 @@ fn contract_bash_nonzero_is_an_error_with_status_after_output() { }, None, &ToolContext::new("."), + None, ) .expect_err("nonzero must be a failed tool call"); assert!( @@ -292,6 +293,160 @@ async fn lowercase_bash_returns_one_ordered_stream() { assert_eq!(result.content, "out-1err-2out-3"); } +/// #6689: the receipt is exact-or-absent, fits the hook bound after JSON +/// escaping, and reports how the run ended without inventing an exit code. +#[test] +fn execution_receipt_is_bounded_and_reports_truthful_state() { + let tmp = tempdir().unwrap(); + let identity = |command: &str, cwd: &Path| ShellExecutionIdentity { + command: command.to_string(), + cwd: cwd.to_path_buf(), + }; + let cwd = tmp.path().canonicalize().unwrap(); + let mut result = failed_network_shell_result( + &"\u{1f40b}\n\"\u{1}".repeat(12_000), + &"err\n".repeat(12_000), + ); + result.sandboxed = false; + result.sandbox_type = None; + result.stdout_truncated = true; + result.exit_code = None; + result.status = ShellStatus::Killed; + + let receipt = shell_execution_receipt(&identity("printf hi", &cwd), &result, "separate") + .expect("receipt"); + let encoded = serde_json::to_string(&receipt).unwrap(); + assert!(encoded.len() <= crate::hooks::HOOK_EXECUTION_RECEIPT_MAX_BYTES); + assert_eq!(serde_json::from_str::(&encoded).unwrap(), receipt); + assert_eq!(receipt["schema_version"], 1); + assert_eq!(receipt["command"], "printf hi"); + assert_eq!(receipt["cwd"], cwd.to_str().unwrap()); + assert_eq!(receipt["state"], "interrupted"); + assert!(receipt["exit_code"].is_null()); + assert_eq!(receipt["stdout_truncated"], true); + assert_eq!(receipt["stderr_truncated"], true); + assert!( + receipt["stderr"] + .as_str() + .unwrap() + .contains("[receipt preview truncated]") + ); + + // A nonzero exit is a completed run, and a 64-bit code survives. + result.status = ShellStatus::Failed; + result.exit_code = Some(3_221_225_477); + let receipt = shell_execution_receipt(&identity("x", &cwd), &result, "combined").unwrap(); + assert_eq!(receipt["state"], "completed"); + assert_eq!(receipt["exit_code"], 3_221_225_477_i64); + assert_eq!(receipt["output_kind"], "combined"); + + result.status = ShellStatus::TimedOut; + result.exit_code = None; + let receipt = shell_execution_receipt(&identity("x", &cwd), &result, "separate").unwrap(); + assert_eq!(receipt["state"], "interrupted"); + + // Absent, never truncated or guessed. + result.status = ShellStatus::Running; + assert!(shell_execution_receipt(&identity("x", &cwd), &result, "separate").is_none()); + result.status = ShellStatus::Completed; + for command in ["", "bad\0command"] { + assert!(shell_execution_receipt(&identity(command, &cwd), &result, "separate").is_none()); + } + let long = "x".repeat(EXECUTION_RECEIPT_IDENTITY_MAX_BYTES + 1); + assert!(shell_execution_receipt(&identity(&long, &cwd), &result, "separate").is_none()); + let long_cwd = PathBuf::from(format!("/{long}")); + assert!(shell_execution_receipt(&identity("x", &long_cwd), &result, "separate").is_none()); + assert!( + shell_execution_receipt(&identity("x", Path::new("relative")), &result, "separate") + .is_none() + ); + result.sandboxed = true; + assert!(shell_execution_receipt(&identity("x", &cwd), &result, "separate").is_none()); +} + +/// #6689: a settled foreground run records the command and directory the +/// process manager spawned, on success and on failure, and background runs +/// carry no receipt. +#[cfg(unix)] +#[tokio::test] +async fn foreground_shell_results_carry_the_spawned_execution_receipt() { + let tmp = tempdir().unwrap(); + std::fs::create_dir(tmp.path().join("child")).unwrap(); + let mut context = ToolContext::new(tmp.path().to_path_buf()) + .with_elevated_sandbox_policy(ExecutionSandboxPolicy::DangerFullAccess); + context.auto_approve = true; + let child = tmp.path().join("child").canonicalize().unwrap(); + let receipt_cwd = |receipt: &Value| { + Path::new(receipt["cwd"].as_str().expect("cwd")) + .canonicalize() + .unwrap() + }; + + let tool = BashTool::new("Bash"); + let command = "printf effective; printf diagnostic >&2; exit 7"; + let result = tool + .execute(json!({"command": command, "cwd": "child"}), &context) + .await + .unwrap(); + let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"]; + assert_eq!(receipt["command"], command); + assert!(Path::new(receipt["cwd"].as_str().unwrap()).is_absolute()); + assert_eq!(receipt_cwd(receipt), child); + assert_eq!(receipt["stdout"], "effective"); + assert_eq!(receipt["stderr"], "diagnostic"); + assert_eq!(receipt["exit_code"], 7); + assert_eq!(receipt["state"], "completed"); + assert_eq!(receipt["output_kind"], "separate"); + + let interrupted = tool + .execute(json!({"command": "kill -TERM $$"}), &context) + .await + .unwrap(); + let receipt = &interrupted.metadata.as_ref().unwrap()["execution_receipt"]; + assert_eq!(receipt["state"], "interrupted"); + assert!(receipt["exit_code"].is_null()); + + let background = tool + .execute( + json!({"command": "printf background", "background": true}), + &context, + ) + .await + .unwrap(); + assert!( + background + .metadata + .as_ref() + .unwrap() + .get("execution_receipt") + .is_none() + ); + + // Lowercase `bash` shares one pipe: the preview is combined output, and a + // failing command's error still carries the receipt for hooks. + let result = LowercaseBashTool + .execute( + json!({"command": "printf merged; printf diagnostic >&2"}), + &context, + ) + .await + .unwrap(); + let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"]; + assert_eq!(receipt["output_kind"], "combined"); + assert_eq!(receipt["stdout"], "mergeddiagnostic"); + assert_eq!(receipt["stderr"], ""); + assert_eq!(receipt["state"], "completed"); + let error = LowercaseBashTool + .execute(json!({"command": "printf partial; exit 3"}), &context) + .await + .expect_err("nonzero exit is a failed bash call"); + let receipt = &error.metadata().expect("failure metadata")["execution_receipt"]; + assert_eq!(receipt["command"], "printf partial; exit 3"); + assert_eq!(receipt["exit_code"], 3); + assert_eq!(receipt["state"], "completed"); + assert_eq!(receipt_cwd(receipt), tmp.path().canonicalize().unwrap()); +} + #[cfg(unix)] #[tokio::test] async fn lowercase_bash_keeps_raw_command_under_readonly_policy() { @@ -2538,7 +2693,7 @@ fn contract_bash_denial_surfaces_the_escalation_shape() { let mut result = failed_network_shell_result("", "Operation not permitted"); result.sandbox_denied = true; - let error = finish_contract_bash_result(result, None, &ctx) + let error = finish_contract_bash_result(result, None, &ctx, None) .expect_err("sandbox denial is a failed call"); assert!( diff --git a/docs/HOOKS.md b/docs/HOOKS.md index 023bd404a7..dd1d3ef9d1 100644 --- a/docs/HOOKS.md +++ b/docs/HOOKS.md @@ -311,8 +311,47 @@ rebrand. | `DEEPSEEK_TOOL_SUCCESS` | `tool_call_after`, `on_error` (tool failures) | `true` / `false` | | `DEEPSEEK_TOOL_EXIT_CODE` | `tool_call_after` and `on_error` **when the tool reported one** | absent otherwise — never synthesized; set for a failing command as well as a passing one; 64-bit, so Windows crash codes such as `3221225477` survive | | `DEEPSEEK_TOOL_STATUS` | `tool_call_after` and `on_error` **when a shell tool reported one** | `completed`, `failed`, `timed_out`, `killed`, or `running` (moved to the background); absent for other tools | +| `DEEPSEEK_TOOL_EXECUTION_RECEIPT` | `tool_call_after` and `on_error` **for a settled, local, foreground shell run** | complete JSON, at most 32 KiB, or absent; see [Execution receipt](#execution-receipt) | | `DEEPSEEK_SESSION_COST` | when cost is supplied | USD, six decimal places | +### Execution receipt + +`DEEPSEEK_TOOL_EXECUTION_RECEIPT` says what a shell tool (`bash`, `Bash`, +`exec_shell`) actually ran. The before-hook input is not the same thing: a +`tool_call_before` hook can rewrite it. The receipt is built from what the +process manager recorded when it spawned the process, after admission and +any rewrite. + +```json +{"schema_version":1,"command":"printf hello","cwd":"/absolute/workspace","state":"completed","scope":"local","exit_code":0,"stdout":"hello","stderr":"","stdout_truncated":false,"stderr_truncated":false,"output_kind":"separate"} +``` + +| Field | Meaning | +| --- | --- | +| `command` | the admitted shell source handed to the shell, not the shell executable or its argv wrapper | +| `cwd` | the absolute directory the process started in, as passed to the OS (symlinks are not resolved) | +| `state` | `completed` for an observed exit, including a nonzero one; `interrupted` for a signal, kill, cancel, or timeout | +| `scope` | always `local` in schema 1 | +| `exit_code` | the observed integer, or `null`; never synthesized from `state` | +| `stdout`, `stderr` | output previews; a long stream keeps its first and last bytes around a `[receipt preview truncated]` marker | +| `stdout_truncated`, `stderr_truncated` | `true` when the tool's own output capture or the preview dropped bytes | +| `output_kind` | `separate` for `Bash` / `exec_shell`; `combined` for lowercase `bash`, whose stdout and stderr share one pipe — `stdout` then holds the combined preview and `stderr` is empty | + +The rules are conservative: + +- **Exact or absent.** `command` and `cwd` are never truncated. If either is + over 8 KiB, contains NUL, or the directory is relative or not UTF-8, the + receipt is left out. Previews shrink until the serialized JSON fits 32 KiB; + if it still cannot fit, the receipt is left out rather than cut. +- **Absence means nothing.** It implies neither success nor failure. +- **Scope.** Only a settled, pipe-backed, unsandboxed, local foreground run + gets a receipt. Background launches, a foreground run moved to `/jobs`, + PTY (`tty` / `combined_output`) and interactive sessions, OS-sandboxed and + external-backend execution, the read-only shell's hardened argv, Windows, + and calls refused before execution have none. +- It is set for a failed run as well as a passing one, so `on_error` for a + failed shell call carries it too. Every other variable is unchanged. + **Mode-spelling note.** UI-fired events (`session_start`, `session_end`, `message_submit`, `tool_call_after`, `mode_change`, `on_error`, `turn_end`, `subagent_*`, `session_busy`, `session_idle`, `session_error`, `waiting_for_user`) diff --git a/docs/zh_hans/HOOKS.md b/docs/zh_hans/HOOKS.md index fac5536695..bf4ef266cc 100644 --- a/docs/zh_hans/HOOKS.md +++ b/docs/zh_hans/HOOKS.md @@ -163,8 +163,35 @@ Observer 并**不**意味着无副作用。observer hook 是以你的凭据运 | `DEEPSEEK_TOOL_SUCCESS` | `tool_call_after`、`on_error`(工具失败) | `true` / `false` | | `DEEPSEEK_TOOL_EXIT_CODE` | `tool_call_after` 和 `on_error` **当工具报告了退出码时** | 否则不存在——绝不合成;命令失败时同样设置;64 位,因此 `3221225477` 这样的 Windows 崩溃码能完好保留 | | `DEEPSEEK_TOOL_STATUS` | `tool_call_after` 和 `on_error` **当 shell 工具报告了状态时** | `completed`、`failed`、`timed_out`、`killed` 或 `running`(已转入后台);其他工具不存在 | +| `DEEPSEEK_TOOL_EXECUTION_RECEIPT` | `tool_call_after` 和 `on_error` **仅限已结束的本地前台 shell 运行** | 完整 JSON,最多 32 KiB,否则不存在;见下方“执行回执” | | `DEEPSEEK_SESSION_COST` | 提供成本时 | USD,六位小数 | +### 执行回执 + +`DEEPSEEK_TOOL_EXECUTION_RECEIPT` 说明 shell 工具(`bash`、`Bash`、`exec_shell`)实际运行了什么。before-hook 的输入不等于实际执行的内容:`tool_call_before` hook 可以改写它。回执取自进程管理器在准入与改写之后启动进程时记录的内容。 + +```json +{"schema_version":1,"command":"printf hello","cwd":"/absolute/workspace","state":"completed","scope":"local","exit_code":0,"stdout":"hello","stderr":"","stdout_truncated":false,"stderr_truncated":false,"output_kind":"separate"} +``` + +| 字段 | 含义 | +| --- | --- | +| `command` | 交给 shell 的已准入命令源码,而不是 shell 可执行文件或其 argv 包装 | +| `cwd` | 进程启动时所在的绝对目录,即传给操作系统的路径(不解析符号链接) | +| `state` | 观察到退出(包括非零退出)为 `completed`;信号、kill、取消或超时为 `interrupted` | +| `scope` | schema 1 中始终为 `local` | +| `exit_code` | 观察到的整数,或 `null`;绝不根据 `state` 合成 | +| `stdout`、`stderr` | 输出预览;过长的流保留首尾字节,中间以 `[receipt preview truncated]` 标记 | +| `stdout_truncated`、`stderr_truncated` | 工具自身的输出捕获或预览丢弃了字节时为 `true` | +| `output_kind` | `Bash` / `exec_shell` 为 `separate`;小写 `bash` 的 stdout 与 stderr 共用一个管道,为 `combined`——此时 `stdout` 是合并后的预览,`stderr` 为空 | + +规则是保守的: + +- **要么精确,要么不存在。** `command` 和 `cwd` 绝不截断。任一超过 8 KiB、包含 NUL,或目录是相对路径或非 UTF-8 时,不导出回执。预览会缩短直到序列化后的 JSON 不超过 32 KiB;仍然放不下时不导出回执,而不是截断。 +- **不存在不代表任何结果。** 既不意味着成功,也不意味着失败。 +- **范围。** 只有已结束、基于管道、未沙箱化的本地前台运行才有回执。后台启动、转入 `/jobs` 的前台运行、PTY(`tty` / `combined_output`)与交互会话、OS 沙箱与外部后端执行、只读 shell 的加固 argv、Windows,以及执行前就被拒绝的调用都没有回执。 +- 失败的运行与成功的运行同样设置,因此 shell 调用失败时的 `on_error` 也带有它。其他变量均不变。 + **模式拼写说明。** UI 触发的事件(`session_start`、`session_end`、`message_submit`、`tool_call_after`、`mode_change`、`on_error`、`turn_end`、`subagent_*`)会将 `DEEPSEEK_MODE` 设为 UI 标签——`ACT`、`PLAN`、`OPERATE`。`tool_call_before` 在引擎内部触发,并使用引擎自己的模式拼写(`Agent`、`Plan`、`Operate`)。`mode` 条件不区分大小写比较,因此 `{ type = "mode", mode = "plan" }` 两者都能匹配,但精确字符串匹配 `$DEEPSEEK_MODE` 的 hook 应同时接受两种拼写。 **`shell_env` 是受限的那个。** 它只接收 `DEEPSEEK_TOOL_NAME` 和 `DEEPSEEK_TOOL_ARGS`——没有会话 id、工作区、模型或模式。因此,`shell_env` hook 上的 `{ type = "mode", … }` 条件会在加载时被拒绝;请改用 `tool_name` 或 `tool_category` 来限定作用域。 From ab7ad0dd7ce07655d3aed8e6f7ec27f978dcb27b Mon Sep 17 00:00:00 2001 From: Hunter B Date: Mon, 28 Sep 2026 21:26:01 -0700 Subject: [PATCH 2/5] fix(hooks): keep the execution receipt exact and hook-only (#6689) Review follow-ups on the tool_call_after execution receipt: - cwd has one spelling. An explicit cwd reached the spawn already canonicalized by ToolContext::resolve_path, while the default workspace was passed as the session opened it, so a workspace opened through a symlink reported two paths for one directory. The receipt now canonicalizes the recorded directory with tokio::fs::canonicalize (off the runtime thread); a directory that no longer resolves gets no receipt. The test opens the workspace through a symlink and compares exact strings instead of re-canonicalizing them. - A wait error is not an interruption. BackgroundShell::poll marks a failed try_wait (wait_failed); such a run gets no receipt instead of state "interrupted" for a process that may have exited normally. - PowerShell is excluded on every platform. $SHELL=pwsh on Unix makes the dispatcher wrap the source or run it from a temp -File, so the receipt's command would not be what ran. - Hook-only. The receipt is built only while a tool_call_after or on_error hook is registered, and the Runtime API strips it before persisting and emitting the item, which already carries the output. - docs/HOOKS.md and docs/zh_hans/HOOKS.md describe all four. Evidence (macOS, CARGO_TARGET_DIR outside the repo): cargo test -p codewhale-tui --lib --locked -- execution_receipt -> test result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 13816 filtered out same test with the canonicalize step removed -> test result: FAILED. 0 passed; 1 failed (left: .../link, right: /private/var/.../real) cargo test -p codewhale-tui --lib --locked -- hooks:: tools::shell:: tool_routing runtime_tool_completion runtime_shell_completion -> test result: ok. 341 passed; 0 failed; 1 ignored; 0 measured; 13477 filtered out cargo clippy -p codewhale-tui --lib --tests --locked -- -D warnings (CI allow-list) -> Finished, 0 warnings cargo fmt --all -- --check -> clean Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014ZwqatxgVFxHvovngywnks --- crates/tui/src/runtime_threads.rs | 6 +++ crates/tui/src/tools/shell.rs | 53 ++++++++++++++++++++-- crates/tui/src/tools/shell/tests.rs | 70 ++++++++++++++++++++++++----- docs/HOOKS.md | 18 +++++--- docs/zh_hans/HOOKS.md | 7 +-- 5 files changed, 129 insertions(+), 25 deletions(-) diff --git a/crates/tui/src/runtime_threads.rs b/crates/tui/src/runtime_threads.rs index 6c2d96d9d7..42d67bb31e 100644 --- a/crates/tui/src/runtime_threads.rs +++ b/crates/tui/src/runtime_threads.rs @@ -14413,6 +14413,12 @@ impl RuntimeThreadManager { } obj.insert("tool_result_for".to_string(), json!(id)); obj.insert("is_error".to_string(), json!(!output.success)); + // The shell execution receipt (#6689) is + // for completion hooks, which already + // read it from the live result; it + // repeats output previews `detail` + // holds, so it is never persisted. + obj.remove("execution_receipt"); } // Failed calls count too: a large error // output spills like any other. diff --git a/crates/tui/src/tools/shell.rs b/crates/tui/src/tools/shell.rs index 04109f8af2..e44165e6b0 100644 --- a/crates/tui/src/tools/shell.rs +++ b/crates/tui/src/tools/shell.rs @@ -1127,6 +1127,9 @@ pub struct BackgroundShell { lifecycle_seq: u64, last_lifecycle_status: Option, last_lifecycle_bytes: usize, + /// The terminal status came from a failed `try_wait`, not an observed + /// exit or signal. Such a run gets no execution receipt (#6689). + wait_failed: bool, } #[derive(Clone)] @@ -1297,6 +1300,7 @@ impl BackgroundShell { Ok(None) => false, // Still running Err(_) => { self.status = ShellStatus::Failed; + self.wait_failed = true; self.heavy_permit.take(); self.collect_output(); true @@ -2023,6 +2027,7 @@ impl ShellManager { lifecycle_seq: 0, last_lifecycle_status: None, last_lifecycle_bytes: 0, + wait_failed: false, }, ); } @@ -2879,6 +2884,7 @@ impl ShellManager { lifecycle_seq: 0, last_lifecycle_status: None, last_lifecycle_bytes: 0, + wait_failed: false, }; #[cfg(unix)] @@ -4807,10 +4813,20 @@ async fn execute_foreground_via_background( // #6689: the receipt identity is what the manager recorded for this // spawn — the admitted command and the directory handed to the OS — // never the before-hook request. Only pipe-backed, unsandboxed local - // runs qualify: a PTY, the hardened read-only argv rewrite, an OS - // sandbox wrapper, and Windows shell prefixes all change what the - // process actually executes relative to this string. - if !cfg!(windows) && !tty && !direct_argv && !spawned.sandboxed && !spawned.sandbox_denied { + // runs through a POSIX-style ` ` dispatcher + // qualify: a PTY, the hardened read-only argv rewrite, an OS sandbox + // wrapper, Windows shell prefixes, and a PowerShell dispatcher (even + // `$SHELL=pwsh` on Unix, which wraps the source or runs it from a temp + // `-File`) all change what the process executes relative to this string. + if !cfg!(windows) + && !tty + && !direct_argv + && !spawned.sandboxed + && !spawned.sandbox_denied + && !crate::shell_dispatcher::global_dispatcher() + .kind() + .is_powershell() + { *receipt_identity = Some(ShellExecutionIdentity { command: process.command.clone(), cwd: process.working_dir.clone(), @@ -4880,6 +4896,16 @@ async fn execute_foreground_via_background( if manager.poll_status(&task_id)? == ShellStatus::Running { None } else { + // #6689: a status from a failed wait is neither an observed + // exit nor an interruption, so the receipt is left out rather + // than reporting a guessed state. + if manager + .processes + .get(&task_id) + .is_none_or(|shell| shell.wait_failed) + { + *receipt_identity = None; + } let snapshot = manager.get_output(&task_id, false, 0)?; // Ordering matters: the snapshot is taken before the // acknowledgement releases the retained bytes. @@ -6050,6 +6076,13 @@ impl ToolSpec for BashTool { let mut lifecycle_warning = None; let mut receipt_identity = None; + // #6689: the receipt exists for completion hooks to read. Without one + // registered it would only ride along in tool metadata, which the + // Runtime API persists and emits for every call. + let wants_receipt = context.runtime.hook_executor.as_ref().is_some_and(|hooks| { + hooks.has_hooks_for_event(crate::hooks::HookEvent::ToolCallAfter) + || hooks.has_hooks_for_event(crate::hooks::HookEvent::OnError) + }); let result = if interactive { let mut manager = context .shell_manager @@ -6137,6 +6170,18 @@ impl ToolSpec for BashTool { ) .await }; + // One spelling per directory. An explicit `cwd` reaches the spawn + // already canonicalized by `resolve_path`; the default workspace + // arrives as the session opened it, possibly through a symlink. + // Resolve both the same way, off the runtime thread. A directory that + // no longer resolves when the run settles gets no receipt. + let receipt_identity = match receipt_identity.filter(|_| wants_receipt) { + Some(identity) => tokio::fs::canonicalize(&identity.cwd) + .await + .ok() + .map(|cwd| ShellExecutionIdentity { cwd, ..identity }), + None => None, + }; match result { Ok(result) => { diff --git a/crates/tui/src/tools/shell/tests.rs b/crates/tui/src/tools/shell/tests.rs index 0faa75d100..0d3063b7e3 100644 --- a/crates/tui/src/tools/shell/tests.rs +++ b/crates/tui/src/tools/shell/tests.rs @@ -366,23 +366,69 @@ fn execution_receipt_is_bounded_and_reports_truthful_state() { /// #6689: a settled foreground run records the command and directory the /// process manager spawned, on success and on failure, and background runs -/// carry no receipt. +/// carry no receipt. The workspace is opened through a symlink so the default +/// directory and an explicit `cwd` would otherwise be spelled differently. #[cfg(unix)] #[tokio::test] async fn foreground_shell_results_carry_the_spawned_execution_receipt() { let tmp = tempdir().unwrap(); - std::fs::create_dir(tmp.path().join("child")).unwrap(); - let mut context = ToolContext::new(tmp.path().to_path_buf()) + let real = tmp.path().join("real"); + std::fs::create_dir_all(real.join("child")).unwrap(); + let workspace = tmp.path().join("link"); + std::os::unix::fs::symlink(&real, &workspace).unwrap(); + let mut context = ToolContext::new(workspace.clone()) .with_elevated_sandbox_policy(ExecutionSandboxPolicy::DangerFullAccess); context.auto_approve = true; - let child = tmp.path().join("child").canonicalize().unwrap(); - let receipt_cwd = |receipt: &Value| { - Path::new(receipt["cwd"].as_str().expect("cwd")) - .canonicalize() + let tool = BashTool::new("Bash"); + + // No completion hook registered: nothing reads a receipt, so none rides + // along in the metadata the Runtime API persists. + let unobserved = tool + .execute(json!({"command": "pwd"}), &context) + .await + .unwrap(); + assert!( + unobserved + .metadata + .as_ref() .unwrap() - }; + .get("execution_receipt") + .is_none() + ); + + let hooks = crate::hooks::HookExecutor::new( + crate::hooks::HooksConfig { + enabled: true, + hooks: vec![crate::hooks::Hook::new( + crate::hooks::HookEvent::ToolCallAfter, + "true", + )], + ..crate::hooks::HooksConfig::default() + }, + workspace.clone(), + ); + context.runtime.hook_executor = Some(std::sync::Arc::new(hooks)); + // Exact strings, not re-canonicalized: both spellings must already agree. + let real = real.canonicalize().unwrap(); + let real_str = real.to_str().unwrap(); + let child_str = real.join("child"); + let child_str = child_str.to_str().unwrap(); + + let default_dir = tool + .execute(json!({"command": "pwd"}), &context) + .await + .unwrap(); + let explicit_dir = tool + .execute(json!({"command": "pwd", "cwd": "."}), &context) + .await + .unwrap(); + for result in [&default_dir, &explicit_dir] { + assert_eq!( + result.metadata.as_ref().unwrap()["execution_receipt"]["cwd"], + real_str + ); + } - let tool = BashTool::new("Bash"); let command = "printf effective; printf diagnostic >&2; exit 7"; let result = tool .execute(json!({"command": command, "cwd": "child"}), &context) @@ -390,8 +436,7 @@ async fn foreground_shell_results_carry_the_spawned_execution_receipt() { .unwrap(); let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"]; assert_eq!(receipt["command"], command); - assert!(Path::new(receipt["cwd"].as_str().unwrap()).is_absolute()); - assert_eq!(receipt_cwd(receipt), child); + assert_eq!(receipt["cwd"], child_str); assert_eq!(receipt["stdout"], "effective"); assert_eq!(receipt["stderr"], "diagnostic"); assert_eq!(receipt["exit_code"], 7); @@ -444,7 +489,7 @@ async fn foreground_shell_results_carry_the_spawned_execution_receipt() { assert_eq!(receipt["command"], "printf partial; exit 3"); assert_eq!(receipt["exit_code"], 3); assert_eq!(receipt["state"], "completed"); - assert_eq!(receipt_cwd(receipt), tmp.path().canonicalize().unwrap()); + assert_eq!(receipt["cwd"], real_str); } #[cfg(unix)] @@ -3929,6 +3974,7 @@ fn killed_shell_does_not_wait_for_blocked_reader_threads() { lifecycle_seq: 0, last_lifecycle_status: None, last_lifecycle_bytes: 0, + wait_failed: false, }; let started = std::time::Instant::now(); diff --git a/docs/HOOKS.md b/docs/HOOKS.md index dd1d3ef9d1..5f5bf660c3 100644 --- a/docs/HOOKS.md +++ b/docs/HOOKS.md @@ -329,7 +329,7 @@ any rewrite. | Field | Meaning | | --- | --- | | `command` | the admitted shell source handed to the shell, not the shell executable or its argv wrapper | -| `cwd` | the absolute directory the process started in, as passed to the OS (symlinks are not resolved) | +| `cwd` | the canonical absolute path of the directory the process started in: symlinks are resolved, so a directory has one spelling whether or not the call passed `cwd`; it is resolved when the run settles | | `state` | `completed` for an observed exit, including a nonzero one; `interrupted` for a signal, kill, cancel, or timeout | | `scope` | always `local` in schema 1 | | `exit_code` | the observed integer, or `null`; never synthesized from `state` | @@ -340,15 +340,21 @@ any rewrite. The rules are conservative: - **Exact or absent.** `command` and `cwd` are never truncated. If either is - over 8 KiB, contains NUL, or the directory is relative or not UTF-8, the - receipt is left out. Previews shrink until the serialized JSON fits 32 KiB; + over 8 KiB, contains NUL, or the directory is relative, not UTF-8, or no + longer resolves, the receipt is left out. So is a run whose end the shell + tool could not observe (the OS wait itself failed): its state is unknown, + and the receipt does not guess it. Previews shrink until the serialized JSON fits 32 KiB; if it still cannot fit, the receipt is left out rather than cut. - **Absence means nothing.** It implies neither success nor failure. -- **Scope.** Only a settled, pipe-backed, unsandboxed, local foreground run - gets a receipt. Background launches, a foreground run moved to `/jobs`, +- **Scope.** A receipt is built only while a `tool_call_after` or `on_error` + hook is configured, and only for a settled, pipe-backed, unsandboxed, local + foreground run. Background launches, a foreground run moved to `/jobs`, PTY (`tty` / `combined_output`) and interactive sessions, OS-sandboxed and external-backend execution, the read-only shell's hardened argv, Windows, - and calls refused before execution have none. + a PowerShell shell on any platform (it wraps the source or runs it from a + temporary script), and calls refused before execution have none. +- **Hooks only.** The receipt is not kept in the durable Runtime API item + record; that record already carries the tool output. - It is set for a failed run as well as a passing one, so `on_error` for a failed shell call carries it too. Every other variable is unchanged. diff --git a/docs/zh_hans/HOOKS.md b/docs/zh_hans/HOOKS.md index bf4ef266cc..1e23cb592c 100644 --- a/docs/zh_hans/HOOKS.md +++ b/docs/zh_hans/HOOKS.md @@ -177,7 +177,7 @@ Observer 并**不**意味着无副作用。observer hook 是以你的凭据运 | 字段 | 含义 | | --- | --- | | `command` | 交给 shell 的已准入命令源码,而不是 shell 可执行文件或其 argv 包装 | -| `cwd` | 进程启动时所在的绝对目录,即传给操作系统的路径(不解析符号链接) | +| `cwd` | 进程启动时所在目录的规范绝对路径:解析符号链接,因此无论调用是否传入 `cwd`,同一目录只有一种写法;在运行结束时解析 | | `state` | 观察到退出(包括非零退出)为 `completed`;信号、kill、取消或超时为 `interrupted` | | `scope` | schema 1 中始终为 `local` | | `exit_code` | 观察到的整数,或 `null`;绝不根据 `state` 合成 | @@ -187,9 +187,10 @@ Observer 并**不**意味着无副作用。observer hook 是以你的凭据运 规则是保守的: -- **要么精确,要么不存在。** `command` 和 `cwd` 绝不截断。任一超过 8 KiB、包含 NUL,或目录是相对路径或非 UTF-8 时,不导出回执。预览会缩短直到序列化后的 JSON 不超过 32 KiB;仍然放不下时不导出回执,而不是截断。 +- **要么精确,要么不存在。** `command` 和 `cwd` 绝不截断。任一超过 8 KiB、包含 NUL,或目录是相对路径、非 UTF-8 或已无法解析时,不导出回执。shell 工具无法观察到运行如何结束(操作系统的 wait 调用本身失败)时同样不导出:其状态未知,回执不做猜测。预览会缩短直到序列化后的 JSON 不超过 32 KiB;仍然放不下时不导出回执,而不是截断。 - **不存在不代表任何结果。** 既不意味着成功,也不意味着失败。 -- **范围。** 只有已结束、基于管道、未沙箱化的本地前台运行才有回执。后台启动、转入 `/jobs` 的前台运行、PTY(`tty` / `combined_output`)与交互会话、OS 沙箱与外部后端执行、只读 shell 的加固 argv、Windows,以及执行前就被拒绝的调用都没有回执。 +- **范围。** 只有配置了 `tool_call_after` 或 `on_error` hook 时才会生成回执,且仅限已结束、基于管道、未沙箱化的本地前台运行。后台启动、转入 `/jobs` 的前台运行、PTY(`tty` / `combined_output`)与交互会话、OS 沙箱与外部后端执行、只读 shell 的加固 argv、Windows、任何平台上的 PowerShell(它会包装源码或通过临时脚本运行),以及执行前就被拒绝的调用都没有回执。 +- **仅供 hook 使用。** 回执不写入持久化的 Runtime API 条目记录;该记录已包含工具输出。 - 失败的运行与成功的运行同样设置,因此 shell 调用失败时的 `on_error` 也带有它。其他变量均不变。 **模式拼写说明。** UI 触发的事件(`session_start`、`session_end`、`message_submit`、`tool_call_after`、`mode_change`、`on_error`、`turn_end`、`subagent_*`)会将 `DEEPSEEK_MODE` 设为 UI 标签——`ACT`、`PLAN`、`OPERATE`。`tool_call_before` 在引擎内部触发,并使用引擎自己的模式拼写(`Agent`、`Plan`、`Operate`)。`mode` 条件不区分大小写比较,因此 `{ type = "mode", mode = "plan" }` 两者都能匹配,但精确字符串匹配 `$DEEPSEEK_MODE` 的 hook 应同时接受两种拼写。 From 7136c671b9d7880f84a06d6d00e6056a8ab6a038 Mon Sep 17 00:00:00 2001 From: Hunter B Date: Mon, 28 Sep 2026 22:00:15 -0700 Subject: [PATCH 3/5] docs(changelog): note the tool_call_after execution receipt (#6689) Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014ZwqatxgVFxHvovngywnks --- CHANGELOG.md | 1 + crates/tui/CHANGELOG.md | 1 + 2 files changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index a48d83cec3..661160682f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -55,6 +55,7 @@ quieter, and Fleet runs can be checked before they spend anything. ### Added +- `tool_call_after` hooks for shell tools receive `DEEPSEEK_TOOL_EXECUTION_RECEIPT`: the command that actually ran after admission, its working directory, how it ended, and bounded stdout/stderr previews, so a hook can record exactly what executed ([#6689](https://github.com/Hmbown/Codewhale/issues/6689), requested by [@wuisabel-gif](https://github.com/wuisabel-gif)). - Runtime API: turns now record what they produced. Each item and turn carries typed artifact references (path, kind, size, revision, and a restore point when file-revert would accept one) for files a tool wrote, diff --git a/crates/tui/CHANGELOG.md b/crates/tui/CHANGELOG.md index c577e4eb20..df7fd4fe93 100644 --- a/crates/tui/CHANGELOG.md +++ b/crates/tui/CHANGELOG.md @@ -55,6 +55,7 @@ quieter, and Fleet runs can be checked before they spend anything. ### Added +- `tool_call_after` hooks for shell tools receive `DEEPSEEK_TOOL_EXECUTION_RECEIPT`: the command that actually ran after admission, its working directory, how it ended, and bounded stdout/stderr previews, so a hook can record exactly what executed ([#6689](https://github.com/Hmbown/Codewhale/issues/6689), requested by [@wuisabel-gif](https://github.com/wuisabel-gif)). - Runtime API: turns now record what they produced. Each item and turn carries typed artifact references (path, kind, size, revision, and a restore point when file-revert would accept one) for files a tool wrote, From cab3f446ee138e626ba6132d0a29c0d9b555a269 Mon Sep 17 00:00:00 2001 From: Hunter B Date: Mon, 28 Sep 2026 23:06:07 -0700 Subject: [PATCH 4/5] docs(credits): credit @wuisabel-gif for the execution-receipt design (#6689) b965cd7dd carries a Co-authored-by trailer for Isabel Wu, whose reference branch set the tool_call_after execution-receipt contract and test design. check-contributor-credit.py (Version drift job) failed because the handle was not in web/lib/release-credits.ts. Add her to the v0.10.1 band on every credit surface so the parity tests stay exact: RELEASE_CONTRIBUTORS, the 0.10.1 CHANGELOG Contributors block (and the synced crates/tui copy), the docs/CONTRIBUTORS.md v0.10.1 band, and requiredCandidateCredits. Evidence (local): python3 scripts/check-contributor-credit.py -> all credited on three surfaces scripts/sync-changelog.sh --check -> up to date scripts/release/check-versions.sh --range-audit-advisory -> OK check-feature-release-notes.sh HEAD -> OK check-bundled-plugin-claims.py, check-ohos-deps.sh -> OK vitest lib/public-copy.test.ts lib/public-surface-contract.test.ts -> 18 passed Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014ZwqatxgVFxHvovngywnks --- CHANGELOG.md | 1 + crates/tui/CHANGELOG.md | 1 + docs/CONTRIBUTORS.md | 1 + docs/public-surface-facts.json | 1 + web/lib/release-credits.ts | 1 + 5 files changed, 5 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 661160682f..01a3a6a16e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -52,6 +52,7 @@ quieter, and Fleet runs can be checked before they spend anything. - **[@dajiaohuang](https://github.com/dajiaohuang)** — `codewhale config set` checks a known setting's value against its schema type before saving it ([#6568](https://github.com/Hmbown/Codewhale/pull/6568)). - **[@BX166](https://github.com/BX166)** — reported the AICraft provider row missing its key console, docs link and guidance, and supplied the values ([#6616](https://github.com/Hmbown/Codewhale/issues/6616)). - **[@Water-Run](https://github.com/Water-Run)** — ingested namespaced model-only catalog entries so models present only in the canonical `models` map reach the offering list ([#6400](https://github.com/Hmbown/Codewhale/pull/6400)), and retired the blanket dead-code allowance with its unused feature stages, tightening the budget to match ([#6402](https://github.com/Hmbown/Codewhale/pull/6402)). +- **[@wuisabel-gif](https://github.com/wuisabel-gif)** — designed the `tool_call_after` execution-receipt contract and its tests on a reference branch, which landed re-implemented on the current hook seam ([#6689](https://github.com/Hmbown/Codewhale/issues/6689), [#6713](https://github.com/Hmbown/Codewhale/pull/6713)). ### Added diff --git a/crates/tui/CHANGELOG.md b/crates/tui/CHANGELOG.md index df7fd4fe93..0cd52e2d33 100644 --- a/crates/tui/CHANGELOG.md +++ b/crates/tui/CHANGELOG.md @@ -52,6 +52,7 @@ quieter, and Fleet runs can be checked before they spend anything. - **[@dajiaohuang](https://github.com/dajiaohuang)** — `codewhale config set` checks a known setting's value against its schema type before saving it ([#6568](https://github.com/Hmbown/Codewhale/pull/6568)). - **[@BX166](https://github.com/BX166)** — reported the AICraft provider row missing its key console, docs link and guidance, and supplied the values ([#6616](https://github.com/Hmbown/Codewhale/issues/6616)). - **[@Water-Run](https://github.com/Water-Run)** — ingested namespaced model-only catalog entries so models present only in the canonical `models` map reach the offering list ([#6400](https://github.com/Hmbown/Codewhale/pull/6400)), and retired the blanket dead-code allowance with its unused feature stages, tightening the budget to match ([#6402](https://github.com/Hmbown/Codewhale/pull/6402)). +- **[@wuisabel-gif](https://github.com/wuisabel-gif)** — designed the `tool_call_after` execution-receipt contract and its tests on a reference branch, which landed re-implemented on the current hook seam ([#6689](https://github.com/Hmbown/Codewhale/issues/6689), [#6713](https://github.com/Hmbown/Codewhale/pull/6713)). ### Added diff --git a/docs/CONTRIBUTORS.md b/docs/CONTRIBUTORS.md index ce4ef86c0f..9d4d8fd6bc 100644 --- a/docs/CONTRIBUTORS.md +++ b/docs/CONTRIBUTORS.md @@ -37,6 +37,7 @@ notes, and relevant issue/PR comments. - **[aboimpinto](https://github.com/aboimpinto)** — restored a green Linux full-workspace test gate without loosening any test, twice ([#6581](https://github.com/Hmbown/Codewhale/pull/6581), [#6666](https://github.com/Hmbown/Codewhale/pull/6666)). - **[dajiaohuang](https://github.com/dajiaohuang)** — validated `config set` values against the settings schema ([#6568](https://github.com/Hmbown/Codewhale/pull/6568)). - **[Water-Run](https://github.com/Water-Run)** — ingested namespaced model-only catalog entries so models present only in the canonical `models` map reach the offering list ([#6400](https://github.com/Hmbown/Codewhale/pull/6400)), and retired the blanket dead-code allowance and its unused feature stages, tightening the budget to match ([#6402](https://github.com/Hmbown/Codewhale/pull/6402)). +- **[wuisabel-gif](https://github.com/wuisabel-gif)** — designed the `tool_call_after` execution-receipt contract and its tests on a reference branch, which landed re-implemented on the current hook seam ([#6689](https://github.com/Hmbown/Codewhale/issues/6689), [#6713](https://github.com/Hmbown/Codewhale/pull/6713)). **Reports and reproductions** diff --git a/docs/public-surface-facts.json b/docs/public-surface-facts.json index 0cce49a193..76c9a846b6 100644 --- a/docs/public-surface-facts.json +++ b/docs/public-surface-facts.json @@ -232,6 +232,7 @@ "@aboimpinto", "@dajiaohuang", "@Water-Run", + "@wuisabel-gif", "@BX166" ] }, diff --git a/web/lib/release-credits.ts b/web/lib/release-credits.ts index a860e42b2c..2223945566 100644 --- a/web/lib/release-credits.ts +++ b/web/lib/release-credits.ts @@ -25,6 +25,7 @@ export const RELEASE_CONTRIBUTORS: string[] = [ "@aboimpinto", "@dajiaohuang", "@Water-Run", + "@wuisabel-gif", ]; /** From 69eebf8ae81c096c693eab6d00caf372cf3c27d1 Mon Sep 17 00:00:00 2001 From: Hunter B Date: Tue, 29 Sep 2026 02:56:57 -0700 Subject: [PATCH 5/5] Keep shell hook execution receipts bound to the current spawn Resolve the receipt cwd asynchronously before process creation and pass that same resolved path to the process manager and OS. A command that retargets the workspace symlink cannot change the reported starting directory. Clear inherited DEEPSEEK_TOOL_EXECUTION_RECEIPT before overlaying the current hook context in both foreground and background hook launches. Clarify that bounded previews show retained tool output, which may already omit earlier process output. Preserve the original PR and contributor attribution. Validation on the host using the existing contributor target and hermetic test HOME: - cargo test --locked -p codewhale-tui --lib -- execution_receipt runtime_tool_completion runtime_shell_completion --test-threads=1: 7 passed, 0 failed (including symlink-retarget and six actual hook children covering absent, oversized and current receipts in both launch modes). - npm test: 636 passed, 0 failed. - npm run check:web: passed. - rustfmt and git diff --check: passed. The child sandbox's earlier npm and web attempts failed on loopback EPERM and api.github.com DNS; the host reruns above completed successfully. No provider, deployment or release qualification is claimed. --- crates/tui/src/hooks/executor.rs | 58 +++++++++++++++++++++++++++++ crates/tui/src/tools/shell.rs | 34 ++++++++++------- crates/tui/src/tools/shell/tests.rs | 45 ++++++++++++++++++++++ docs/HOOKS.md | 11 +++--- docs/zh_hans/HOOKS.md | 8 ++-- 5 files changed, 134 insertions(+), 22 deletions(-) diff --git a/crates/tui/src/hooks/executor.rs b/crates/tui/src/hooks/executor.rs index f5d589946e..f2de3a0ab8 100644 --- a/crates/tui/src/hooks/executor.rs +++ b/crates/tui/src/hooks/executor.rs @@ -1463,6 +1463,9 @@ impl HookExecutor { // raw_arg: cmd.exe does not parse the CRT-style \" escapes that // Command::arg would insert, so pass the command line verbatim. cmd.arg("/C").raw_arg(command); + // Only this call's context may supply a receipt. In particular, + // a Codewhale launched from another hook must not inherit one. + cmd.env_remove("DEEPSEEK_TOOL_EXECUTION_RECEIPT"); cmd } #[cfg(not(windows))] @@ -1474,6 +1477,9 @@ impl HookExecutor { use std::os::unix::process::CommandExt as _; cmd.process_group(0); } + // Only this call's context may supply a receipt. In particular, + // a Codewhale launched from another hook must not inherit one. + cmd.env_remove("DEEPSEEK_TOOL_EXECUTION_RECEIPT"); cmd } } @@ -5735,6 +5741,58 @@ command = "echo project" ); } + /// An absent receipt must be absent in the actual child environment, + /// even when a nested Codewhale inherited an outer hook's receipt. + #[cfg(unix)] + #[test] + fn execution_receipt_never_inherits_another_calls_environment() { + let _env = lock_test_env(); + let _stale = EnvVarGuard::set("DEEPSEEK_TOOL_EXECUTION_RECEIPT", "stale-outer-receipt"); + let current = r#"{"schema_version":1,"command":"current call"}"#; + let dir = tempfile::tempdir().unwrap(); + let out = dir.path().join("receipt-env.txt"); + let command = write_hook_script( + &dir, + "capture_receipt_env.sh", + &format!( + "#!/bin/sh\nprintf '%s' \"${{DEEPSEEK_TOOL_EXECUTION_RECEIPT-unset}}\" > {}\n", + out.display() + ), + ); + for background in [false, true] { + let mut hook = Hook::new(HookEvent::ToolCallAfter, &command); + hook.background = background; + let executor = HookExecutor::new( + HooksConfig { + enabled: true, + hooks: vec![hook], + ..HooksConfig::default() + }, + dir.path().to_path_buf(), + ); + for (receipt, expected) in [ + (None, "unset"), + ( + Some("x".repeat(HOOK_EXECUTION_RECEIPT_MAX_BYTES + 1)), + "unset", + ), + (Some(current.to_string()), current), + ] { + if out.exists() { + std::fs::remove_file(&out).unwrap(); + } + let context = HookContext { + tool_execution_receipt: receipt, + ..HookContext::new() + }; + let results = executor.execute(HookEvent::ToolCallAfter, &context); + assert_eq!(results.len(), 1); + assert!(results[0].success); + assert_eq!(wait_for_captured_output(&out), expected); + } + } + } + #[test] fn observer_context_is_bounded_before_enqueue() { let huge = "用户".repeat(20_000); diff --git a/crates/tui/src/tools/shell.rs b/crates/tui/src/tools/shell.rs index e44165e6b0..b6f772aeaf 100644 --- a/crates/tui/src/tools/shell.rs +++ b/crates/tui/src/tools/shell.rs @@ -4769,11 +4769,29 @@ async fn execute_foreground_via_background( extra_env: HashMap, direct_argv: bool, timeout_bounds_ms: (u64, u64), + wants_receipt: bool, receipt_identity: &mut Option, ) -> Result { let timeout_ms = timeout_ms.map(|timeout| timeout.clamp(timeout_bounds_ms.0, timeout_bounds_ms.1)); let spawn_timeout_ms = timeout_ms.unwrap_or(timeout_bounds_ms.1); + // Freeze the receipt's directory before execution and hand that same + // resolved spelling to the OS. Looking up the original symlink after + // the command runs can name a different directory than the one it used. + // Resolution is asynchronous; failure leaves the ordinary execution + // path intact but cannot produce an exact receipt. + let receipt_cwd = if wants_receipt { + match working_dir.as_deref() { + Some(cwd) => tokio::fs::canonicalize(cwd) + .await + .ok() + .and_then(|path| path.into_os_string().into_string().ok()), + None => None, + } + } else { + None + }; + let working_dir = receipt_cwd.clone().or(working_dir); let task_id = { let mut manager = context .shell_manager @@ -4818,7 +4836,8 @@ async fn execute_foreground_via_background( // wrapper, Windows shell prefixes, and a PowerShell dispatcher (even // `$SHELL=pwsh` on Unix, which wraps the source or runs it from a temp // `-File`) all change what the process executes relative to this string. - if !cfg!(windows) + if receipt_cwd.is_some() + && !cfg!(windows) && !tty && !direct_argv && !spawned.sandboxed @@ -6166,22 +6185,11 @@ impl ToolSpec for BashTool { } else { (1_000, 600_000) }, + wants_receipt, &mut receipt_identity, ) .await }; - // One spelling per directory. An explicit `cwd` reaches the spawn - // already canonicalized by `resolve_path`; the default workspace - // arrives as the session opened it, possibly through a symlink. - // Resolve both the same way, off the runtime thread. A directory that - // no longer resolves when the run settles gets no receipt. - let receipt_identity = match receipt_identity.filter(|_| wants_receipt) { - Some(identity) => tokio::fs::canonicalize(&identity.cwd) - .await - .ok() - .map(|cwd| ShellExecutionIdentity { cwd, ..identity }), - None => None, - }; match result { Ok(result) => { diff --git a/crates/tui/src/tools/shell/tests.rs b/crates/tui/src/tools/shell/tests.rs index 0d3063b7e3..102ed841de 100644 --- a/crates/tui/src/tools/shell/tests.rs +++ b/crates/tui/src/tools/shell/tests.rs @@ -492,6 +492,51 @@ async fn foreground_shell_results_carry_the_spawned_execution_receipt() { assert_eq!(receipt["cwd"], real_str); } +/// A post-run lookup of a retargeted workspace symlink would falsely name +/// the replacement directory. The receipt must keep the actual spawn path. +#[cfg(unix)] +#[tokio::test] +async fn execution_receipt_keeps_spawn_cwd_when_the_command_retargets_the_workspace() { + let tmp = tempdir().unwrap(); + let real = tmp.path().join("real"); + let other = tmp.path().join("other"); + std::fs::create_dir(&real).unwrap(); + std::fs::create_dir(&other).unwrap(); + let workspace = tmp.path().join("link"); + std::os::unix::fs::symlink(&real, &workspace).unwrap(); + let mut context = ToolContext::new(workspace.clone()) + .with_elevated_sandbox_policy(ExecutionSandboxPolicy::DangerFullAccess); + context.auto_approve = true; + context.runtime.hook_executor = Some(std::sync::Arc::new(crate::hooks::HookExecutor::new( + crate::hooks::HooksConfig { + enabled: true, + hooks: vec![crate::hooks::Hook::new( + crate::hooks::HookEvent::ToolCallAfter, + "true", + )], + ..crate::hooks::HooksConfig::default() + }, + workspace.clone(), + ))); + let command = "pwd -P; rm ../link; ln -s other ../link; pwd -P"; + let result = BashTool::new("Bash") + .execute(json!({"command": command}), &context) + .await + .unwrap(); + assert!(result.success); + let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"]; + let actual = real.canonicalize().unwrap(); + let actual = actual.to_str().unwrap(); + assert_eq!(receipt["command"], command); + assert_eq!(receipt["cwd"], actual); + assert_eq!(receipt["stdout"], format!("{actual}\n{actual}\n")); + assert_eq!(receipt["exit_code"], 0); + assert_eq!( + workspace.canonicalize().unwrap(), + other.canonicalize().unwrap() + ); +} + #[cfg(unix)] #[tokio::test] async fn lowercase_bash_keeps_raw_command_under_readonly_policy() { diff --git a/docs/HOOKS.md b/docs/HOOKS.md index 5f5bf660c3..13eb8b92f2 100644 --- a/docs/HOOKS.md +++ b/docs/HOOKS.md @@ -329,23 +329,24 @@ any rewrite. | Field | Meaning | | --- | --- | | `command` | the admitted shell source handed to the shell, not the shell executable or its argv wrapper | -| `cwd` | the canonical absolute path of the directory the process started in: symlinks are resolved, so a directory has one spelling whether or not the call passed `cwd`; it is resolved when the run settles | +| `cwd` | the canonical absolute path of the directory the process started in: symlinks are resolved, so a directory has one spelling whether or not the call passed `cwd`; it is resolved before spawn and that same path is handed to the OS | | `state` | `completed` for an observed exit, including a nonzero one; `interrupted` for a signal, kill, cancel, or timeout | | `scope` | always `local` in schema 1 | | `exit_code` | the observed integer, or `null`; never synthesized from `state` | -| `stdout`, `stderr` | output previews; a long stream keeps its first and last bytes around a `[receipt preview truncated]` marker | +| `stdout`, `stderr` | previews of the tool's retained output, which may already omit early process output; long previews keep their first and last bytes around a `[receipt preview truncated]` marker | | `stdout_truncated`, `stderr_truncated` | `true` when the tool's own output capture or the preview dropped bytes | | `output_kind` | `separate` for `Bash` / `exec_shell`; `combined` for lowercase `bash`, whose stdout and stderr share one pipe — `stdout` then holds the combined preview and `stderr` is empty | The rules are conservative: - **Exact or absent.** `command` and `cwd` are never truncated. If either is - over 8 KiB, contains NUL, or the directory is relative, not UTF-8, or no - longer resolves, the receipt is left out. So is a run whose end the shell + over 8 KiB, contains NUL, or the directory is relative, not UTF-8, or cannot + resolve before spawn, the receipt is left out. So is a run whose end the shell tool could not observe (the OS wait itself failed): its state is unknown, and the receipt does not guess it. Previews shrink until the serialized JSON fits 32 KiB; if it still cannot fit, the receipt is left out rather than cut. -- **Absence means nothing.** It implies neither success nor failure. +- **Absence means nothing.** It implies neither success nor failure. An inherited + `DEEPSEEK_TOOL_EXECUTION_RECEIPT` is cleared before applying the current call's context. - **Scope.** A receipt is built only while a `tool_call_after` or `on_error` hook is configured, and only for a settled, pipe-backed, unsandboxed, local foreground run. Background launches, a foreground run moved to `/jobs`, diff --git a/docs/zh_hans/HOOKS.md b/docs/zh_hans/HOOKS.md index 1e23cb592c..4406931b1c 100644 --- a/docs/zh_hans/HOOKS.md +++ b/docs/zh_hans/HOOKS.md @@ -177,18 +177,18 @@ Observer 并**不**意味着无副作用。observer hook 是以你的凭据运 | 字段 | 含义 | | --- | --- | | `command` | 交给 shell 的已准入命令源码,而不是 shell 可执行文件或其 argv 包装 | -| `cwd` | 进程启动时所在目录的规范绝对路径:解析符号链接,因此无论调用是否传入 `cwd`,同一目录只有一种写法;在运行结束时解析 | +| `cwd` | 进程启动时所在目录的规范绝对路径:解析符号链接,因此无论调用是否传入 `cwd`,同一目录只有一种写法;在启动前解析,并将同一路径交给操作系统 | | `state` | 观察到退出(包括非零退出)为 `completed`;信号、kill、取消或超时为 `interrupted` | | `scope` | schema 1 中始终为 `local` | | `exit_code` | 观察到的整数,或 `null`;绝不根据 `state` 合成 | -| `stdout`、`stderr` | 输出预览;过长的流保留首尾字节,中间以 `[receipt preview truncated]` 标记 | +| `stdout`、`stderr` | 工具已保留输出的预览,其中可能已经缺少进程输出的开头;过长的预览保留自身的首尾字节,中间以 `[receipt preview truncated]` 标记 | | `stdout_truncated`、`stderr_truncated` | 工具自身的输出捕获或预览丢弃了字节时为 `true` | | `output_kind` | `Bash` / `exec_shell` 为 `separate`;小写 `bash` 的 stdout 与 stderr 共用一个管道,为 `combined`——此时 `stdout` 是合并后的预览,`stderr` 为空 | 规则是保守的: -- **要么精确,要么不存在。** `command` 和 `cwd` 绝不截断。任一超过 8 KiB、包含 NUL,或目录是相对路径、非 UTF-8 或已无法解析时,不导出回执。shell 工具无法观察到运行如何结束(操作系统的 wait 调用本身失败)时同样不导出:其状态未知,回执不做猜测。预览会缩短直到序列化后的 JSON 不超过 32 KiB;仍然放不下时不导出回执,而不是截断。 -- **不存在不代表任何结果。** 既不意味着成功,也不意味着失败。 +- **要么精确,要么不存在。** `command` 和 `cwd` 绝不截断。任一超过 8 KiB、包含 NUL,或目录是相对路径、非 UTF-8 或在启动前无法解析时,不导出回执。shell 工具无法观察到运行如何结束(操作系统的 wait 调用本身失败)时同样不导出:其状态未知,回执不做猜测。预览会缩短直到序列化后的 JSON 不超过 32 KiB;仍然放不下时不导出回执,而不是截断。 +- **不存在不代表任何结果。** 既不意味着成功,也不意味着失败。应用当前调用的上下文前,会清除继承的 `DEEPSEEK_TOOL_EXECUTION_RECEIPT`。 - **范围。** 只有配置了 `tool_call_after` 或 `on_error` hook 时才会生成回执,且仅限已结束、基于管道、未沙箱化的本地前台运行。后台启动、转入 `/jobs` 的前台运行、PTY(`tty` / `combined_output`)与交互会话、OS 沙箱与外部后端执行、只读 shell 的加固 argv、Windows、任何平台上的 PowerShell(它会包装源码或通过临时脚本运行),以及执行前就被拒绝的调用都没有回执。 - **仅供 hook 使用。** 回执不写入持久化的 Runtime API 条目记录;该记录已包含工具输出。 - 失败的运行与成功的运行同样设置,因此 shell 调用失败时的 `on_error` 也带有它。其他变量均不变。