From 93516d9367f42a2ab5a651caabc40c79bee5b464 Mon Sep 17 00:00:00 2001 From: abose Date: Fri, 9 Oct 2026 13:54:19 +0530 Subject: [PATCH 1/8] feat(ai): style pending-question composer reminders Add the reminder's localized heading, explanatory text and actions. Style the amber banner with consistent padding, aligned close and Show question buttons, and wrapping for narrow panels. Escape grid spans so the LESS compiler preserves their placement. Advance the Phoenix Pro pin to d9550d3 for the matching composer guard, question lifecycle handling and registered integration tests. Verification: 14 question composer, 19 composer focus and 13 Live Preview send checks pass on each of Linux, macOS and Windows (138 total). LESS build and targeted lint pass; the real Haiku flow is verified in the app. --- src/nls/root/strings.js | 5 ++ src/styles/Extn-AIChatPanel.less | 92 ++++++++++++++++++++++++++++++++ tracking-repos.json | 2 +- 3 files changed, 98 insertions(+), 1 deletion(-) diff --git a/src/nls/root/strings.js b/src/nls/root/strings.js index 4e58756484..953d018403 100644 --- a/src/nls/root/strings.js +++ b/src/nls/root/strings.js @@ -3041,6 +3041,11 @@ define({ "AI_CHAT_PREVIEW_VIEWING": "Previewing", "AI_CHAT_QUESTION_OTHER": "Type a custom answer\u2026", "AI_CHAT_QUESTION_SUBMIT": "Submit", + "AI_CHAT_QUESTION_REMINDER": "Please answer the question above", + "AI_CHAT_QUESTIONS_REMINDER": "Please answer the questions above", + "AI_CHAT_QUESTION_REMINDER_DETAIL": "Claude is waiting for your answer before you can send this message.", + "AI_CHAT_QUESTION_SHOW": "Show question", + "AI_CHAT_QUESTION_DISMISS": "Dismiss reminder", "AI_CHAT_IMAGE_LIMIT": "Maximum {0} images allowed", "AI_CHAT_IMAGE_REMOVE": "Remove image", "AI_CHAT_ATTACHMENTS_COUNT": "{0} attachments", diff --git a/src/styles/Extn-AIChatPanel.less b/src/styles/Extn-AIChatPanel.less index 9ff7d79129..b214f2ee75 100644 --- a/src/styles/Extn-AIChatPanel.less +++ b/src/styles/Extn-AIChatPanel.less @@ -3497,6 +3497,98 @@ } } +.ai-question-reminder { + display: grid; + grid-template-columns: 18px minmax(0, 1fr) 24px; + gap: 4px 8px; + box-sizing: border-box; + margin-bottom: 8px; + padding: 12px; + background: rgba(213, 173, 91, 0.12); + border: 1px solid rgba(213, 173, 91, 0.35); + border-left: 3px solid #cba257; + border-radius: 5px; + color: @project-panel-text-1; + font-size: @ai-text-body; + line-height: 20px; + white-space: normal; + + &[hidden] { display: none; } + + .ai-question-reminder-icon { + grid-column: 1; + grid-row: 1; + align-self: center; + color: #dfb86d; + } + + .ai-question-reminder-title { + grid-column: 2; + grid-row: 1; + align-self: center; + min-width: 0; + font-weight: 600; + overflow-wrap: anywhere; + } + + .ai-question-reminder-detail { + grid-column: ~"2 / 4"; + grid-row: 2; + min-width: 0; + font-size: @ai-dialog-note; + line-height: 18px; + color: @project-panel-text-2; + overflow-wrap: anywhere; + } + + button { + display: inline-flex; + align-items: center; + justify-content: center; + box-sizing: border-box; + font: inherit; + border-radius: 3px; + cursor: pointer; + + &:focus-visible { + outline: 1px solid @bc-btn-border-focused; + outline-offset: 2px; + } + } + + .ai-question-reminder-close { + grid-column: 3; + grid-row: 1; + width: 24px; + height: 24px; + margin: 0; + padding: 0; + background: transparent; + border: 0; + color: @project-panel-text-2; + + &:hover { background: rgba(213, 173, 91, 0.19); } + } + + .ai-question-reminder-show { + grid-column: ~"2 / 4"; + grid-row: 3; + justify-self: end; + min-height: 28px; + margin-top: 6px; + padding: 3px 10px; + border: 1px solid rgba(213, 173, 91, 0.38); + background: rgba(213, 173, 91, 0.08); + color: @project-panel-text-1; + font-size: @ai-dialog-note; + + &:hover { + background: rgba(213, 173, 91, 0.19); + border-color: rgba(213, 173, 91, 0.6); + } + } +} + .ai-question-other { display: flex; align-items: stretch; diff --git a/tracking-repos.json b/tracking-repos.json index 8574181cc9..25c96703de 100644 --- a/tracking-repos.json +++ b/tracking-repos.json @@ -1,5 +1,5 @@ { "phoenixPro": { - "commitID": "3b8bab92f704fcf5d50e863abec0fa5c8088fc1a" + "commitID": "d9550d395bed6dbad1d504b74b3d65db797b9914" } } From 09cf5927c5e5540ed5e5a57d9d4fac8de3255fa9 Mon Sep 17 00:00:00 2001 From: abose Date: Fri, 9 Oct 2026 14:10:05 +0530 Subject: [PATCH 2/8] fix(ai): follow SDK question cancellation and docs requests Observe the AskUserQuestion hook's cancellation signal as well as the query signal, and notify the panel when that specific answer wait ends. Ignore explicit stale confirmation IDs instead of answering another pending card. Leave timeout policy and durations entirely to the Claude SDK. Clarify that the documentation locator must be followed by reading its source; the preference registry can supplement an explicitly requested docs lookup. Advance the Pro pin for matching question cleanup, tests and task tracking. Validation: 18 question composer integration specs pass on each of macOS, Windows and Linux; TD-3 passes with real Haiku 5.5; targeted ESLint and diff checks pass. The user approved committing with the natural SDK timeout observation still pending. --- src-node/ai-editor-tool-specs.js | 10 +++++++-- src-node/claude-code-agent.js | 36 ++++++++++++++++++++++++-------- tracking-repos.json | 2 +- 3 files changed, 36 insertions(+), 12 deletions(-) diff --git a/src-node/ai-editor-tool-specs.js b/src-node/ai-editor-tool-specs.js index 44a7feab6c..6e1d2c3ed0 100644 --- a/src-node/ai-editor-tool-specs.js +++ b/src-node/ai-editor-tool-specs.js @@ -608,6 +608,9 @@ function getEditorToolSpecs(peerCall, options = {}) { "description, allowedValues if any, and the resolved scope of the current value).\n" + "- get: Same fields for a single preference id.\n" + "- set: Write a value into a specific scope. Calls PreferencesManager.save() after.\n\n" + + "This is the running editor's preference registry, not its documentation. If the user " + + "asks you to look something up in the Phoenix docs, use editorDocs and read the relevant " + + "documentation before answering; this tool can supplement it with current values.\n\n" + "Scope hierarchy (highest precedence wins on read): session → project → user → default.\n" + "- default: built-in fallback declared by definePreference in source. READ-ONLY.\n" + "- user: the user's global settings (persisted across all projects). User-friendly name " + @@ -874,8 +877,11 @@ function getEditorToolSpecs(peerCall, options = {}) { "Fetch with WebFetch.\n" + "- sourceRepoURL: GitHub repo for source-level lookups when the API docs don't cover " + "something. Use WebFetch on raw.githubusercontent.com URLs to read individual files.\n\n" + - "Call this once near the start of any non-trivial editor-control task, then Read / " + - "Grep / WebFetch into the surfaces it returns.", + "When the user asks for a Phoenix documentation lookup, call this tool and then read " + + "the relevant returned source with Read / Grep / WebFetch before answering. Calling " + + "this locator alone does not consult the documentation. If the source cannot be read, " + + "say so and distinguish any fallback evidence. Also call this once near the start " + + "of any non-trivial editor-control task, then read the relevant reference.", {}, async function () { let apiDocsAvailable = false; diff --git a/src-node/claude-code-agent.js b/src-node/claude-code-agent.js index 5331cf9f99..333a5f6042 100644 --- a/src-node/claude-code-agent.js +++ b/src-node/claude-code-agent.js @@ -256,8 +256,11 @@ function _registerAnswer(kind, signal) { } /** - * Deliver a browser answer to the matching pending card (by confirmId, else - * the oldest card of that kind). Returns false if nothing was waiting. + * Deliver an answer by confirmId, or to the oldest card only when no ID is supplied. + * An expired ID must never answer a newer question or permission request. + * @param {string} kind Card category. + * @param {Object} params Answer payload with an optional confirmId. + * @return {boolean} Whether a matching pending card accepted the answer. */ function _resolveAnswer(kind, params) { const bucket = _pendingAnswers[kind]; @@ -267,10 +270,10 @@ function _resolveAnswer(kind, params) { let resolve; if (params && params.confirmId !== undefined) { resolve = bucket.get(params.confirmId); - } - if (!resolve) { + } else { resolve = bucket.values().next().value; } + if (!resolve) { return false; } resolve(params || {}); return true; } @@ -318,16 +321,28 @@ async function _askPlanModeWriteConfirm(requestId, toolName, filePath, signal) { /** * Show the AskUserQuestion card in the browser and wait for the answers. - * Resolves the browser's {answers} payload, or null on abort. + * Close that card when the SDK ends its wait; timeout policy belongs to the SDK. + * @param {string} requestId Owning query ID. + * @param {Array} questions Questions supplied by Claude. + * @param {AbortSignal} signal SDK question or query cancellation signal. + * @return {Promise} The browser's answers, or null on cancellation. */ async function _askUserQuestions(requestId, questions, signal) { + if (signal.aborted) { return null; } const pending = _registerAnswer("question", signal); nodeConnector.triggerPeer("aiQuestion", { requestId: requestId, confirmId: pending.id, questions: questions }); - return pending.promise; + try { + return await pending.promise; + } finally { + nodeConnector.triggerPeer("aiQuestionClosed", { + requestId: requestId, + confirmId: pending.id + }); + } } /** @@ -1838,11 +1853,14 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, { matcher: "AskUserQuestion", hooks: [ - async (input) => { + async (input, _toolUseID, options) => { console.log("[Phoenix AI] Intercepted AskUserQuestion"); const questions = input.tool_input.questions || []; - // Wait for the user's answer from the browser UI - const answer = await _askUserQuestions(requestId, questions, signal); + // The hook may expire while the query continues. Follow its + // cancellation as well as Stop, without adding a UI timer. + const questionSignal = options && options.signal + ? AbortSignal.any([signal, options.signal]) : signal; + const answer = await _askUserQuestions(requestId, questions, questionSignal); return { hookSpecificOutput: { hookEventName: "PreToolUse", diff --git a/tracking-repos.json b/tracking-repos.json index 25c96703de..a7157f86f6 100644 --- a/tracking-repos.json +++ b/tracking-repos.json @@ -1,5 +1,5 @@ { "phoenixPro": { - "commitID": "d9550d395bed6dbad1d504b74b3d65db797b9914" + "commitID": "18cbc74c899ce06665e237be14be9554f95c11e1" } } From d692eae86cc486bcd69ef769eecb950fc5c5139a Mon Sep 17 00:00:00 2001 From: abose Date: Fri, 9 Oct 2026 15:28:06 +0530 Subject: [PATCH 3/8] fix(ai): delegate file permissions and synchronize plan modes Let the SDK decide Auto Edit and Write permissions, including paths outside the project. Keep Phoenix's outside-project confirmation in AI Edit Mode, and limit explicit Plan-write approval to the approved call. Apply runtime permission changes to the active streaming SDK query. Order mode changes, wait before file operations, and cancel safely on update failure or timeout. Reassert the selected mode after leaving Plan Mode. Mark internal plan writes as handled only after saving succeeds, so Pro can remove misleading rejection cards without hiding real filesystem errors. Add the Reject label and matching plan-action styles for Pro 9a0063e. Register 35 isolated Node-backed Jasmine cases for permission decisions, mode changes, failures and structured plan results using the shared runner. Validation: all 35 cases pass on native Linux, macOS and Windows. Combined with Plan Review and Question Composer, 222 focused checks pass. Earlier full unit runs passed on all three platforms; final real Auto/Edit/Plan prompts, lint, LESS builds and whitespace checks pass. --- src-node/claude-code-agent.js | 203 +++++++++++++++------- src-node/test-connection.js | 1 + src-node/test/test-ai-file-permissions.js | 191 ++++++++++++++++++++ src/nls/root/strings.js | 1 + src/styles/Extn-AIChatPanel.less | 6 +- test/UnitTestSuite.js | 1 + test/spec/AIFilePermissions-test.js | 185 ++++++++++++++++++++ 7 files changed, 518 insertions(+), 70 deletions(-) create mode 100644 src-node/test/test-ai-file-permissions.js create mode 100644 test/spec/AIFilePermissions-test.js diff --git a/src-node/claude-code-agent.js b/src-node/claude-code-agent.js index 333a5f6042..d7e0bb293b 100644 --- a/src-node/claude-code-agent.js +++ b/src-node/claude-code-agent.js @@ -155,6 +155,7 @@ function _fetchModelListOnce(queryResult) { // Active query state let currentAbortController = null; +let activeQuery = null; // Lazily-initialized in-process MCP server for editor context let editorMcpServer = null; @@ -965,11 +966,40 @@ exports.answerPlanModeWriteConfirm = async function (params) { * after switching from Edit Mode to Allow Everything). The next sendPrompt also * passes permissionMode in params, so this peer is only strictly required * during streaming — but calling it on every cycle keeps the agent's - * tracker authoritative. + * tracker and the SDK session authoritative. + * @param {{mode: string}} params Requested SDK permission mode. + * @return {Promise<{success: boolean}>} Resolves after the active SDK session is updated. */ exports.setPermissionMode = async function (params) { if (params && typeof params.mode === "string") { _runtimePermissionMode = params.mode; + if (activeQuery && !activeQuery.signal.aborted) { + const current = activeQuery; + // Keep rapid mode changes in order; a failed earlier update must + // not prevent a subsequent explicit choice from reaching the SDK. + current.modeUpdate = current.modeUpdate.catch(() => {}).then(async () => { + if (current.signal.aborted) { throw new Error("Permission mode update cancelled"); } + let timeout; + try { + await Promise.race([ + current.query.setPermissionMode(params.mode), + new Promise((resolve, reject) => { + timeout = setTimeout(() => reject(new Error("Permission mode update timed out")), 10000); + }) + ]); + } finally { + clearTimeout(timeout); + } + }); + try { + await current.modeUpdate; + } catch (err) { + // A stale, more permissive SDK mode must not continue after + // the user chose a different mode in the panel. + current.controller.abort(); + throw err; + } + } } return { success: true }; }; @@ -1099,11 +1129,7 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, // tool_use id → SDK tool name, so a tool_result can be interpreted in the // light of which tool produced it (see the AskUserQuestion note below). const _toolUseIdToName = {}; - // Set true once the user clicks "Allow & Switch to Edit Mode" on a - // plan-mode write confirmation. Subsequent Edit/Write attempts in the same - // turn skip the prompt and use the cached "allow" decision so a multi-edit - // turn doesn't pop a dialog before every edit. - let _planExitApprovedThisTurn = false; + const _savedPlanToolIds = new Set(); // Live preview nudge bookkeeping, per request so each new user prompt // re-arms it. _lpPendingEdits counts live-preview-related edits since the // model last inspected the preview; _lpNudgeCount enforces the hard cap. @@ -1196,13 +1222,9 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, // rather than seeing a literal []. Each sendPrompt rebuilds this // list, so adding/removing in the UI takes effect on the next turn. const _cwdForValidation = projectPath || process.cwd(); - // Where Edit/Write may land without asking: the project, any extra - // directories the user attached, and scratch space. Anything else gets - // the permission card — the same "write outside the working directory?" - // check Claude Code makes on its own, which the Write/Edit entries in - // allowedTools would otherwise skip. Seen in the wild: after - // `mkdir -p notes-app` in the project, the model wrote the files to - // /home//notes-app and nobody was asked. + // AI Edit Mode permits edits within the project, attached directories + // and scratch space. Other paths need its manual permission card. + // Auto leaves file permission decisions to the SDK instead. function _isOutsideWriteRoots(filePath) { if (!filePath || !path.isAbsolute(filePath) || _isAiScratchPath(filePath)) { return false; @@ -1214,9 +1236,27 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, return target === r || target.startsWith(r + path.sep); }); } + /** + * Wait for SDK mode changes before allowing a file hook to proceed. + * @param {string} [mode] Optional new mode, for an explicitly approved plan. + * @return {Promise} A hook denial on cancellation/failure, otherwise null. + */ + async function _fileModeDenial(mode) { + try { + if (mode) { await exports.setPermissionMode({mode}); } + if (activeQuery && activeQuery.requestId === requestId) { + await activeQuery.modeUpdate; + } + if (!signal.aborted) { return null; } + } catch (err) { + _log("File permission mode update failed:", err.message); + } + return {hookSpecificOutput: {hookEventName: "PreToolUse", permissionDecision: "deny", + permissionDecisionReason: "Permission mode update failed or the query was cancelled."}}; + } async function _denyUnlessOutsideWriteAllowed(toolName, toolInput, promptSignal) { const filePath = toolInput && toolInput.file_path; - if (_runtimePermissionMode === "bypassPermissions" || !_isOutsideWriteRoots(filePath)) { + if (_runtimePermissionMode !== "acceptEdits" || !_isOutsideWriteRoots(filePath)) { return null; } _log("Write outside project roots:", filePath); @@ -1284,13 +1324,7 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, }; } if (FILE_WRITE_TOOLS.indexOf(toolName) !== -1 && _runtimePermissionMode === "plan") { - // Plan mode entered mid-turn via EnterPlanMode: the Edit/Write - // hooks saw the query-start mode and passed the call through, - // so the CLI's own plan-mode block asks us. Same card as the - // hook path, same one-shot approval for the rest of the turn. - if (_planExitApprovedThisTurn) { - return { behavior: "allow", updatedInput: input }; - } + // Fallback for file tools not intercepted by the Edit/Write hooks. const filePath = (input && input.file_path) || ""; const approved = await _askPlanModeWriteConfirm( requestId, toolName, filePath, promptSignal); @@ -1301,8 +1335,9 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, "tool to propose your changes for approval before editing." }; } - _planExitApprovedThisTurn = true; - _runtimePermissionMode = "auto"; + if (await _fileModeDenial("auto")) { + return {behavior: "deny", message: "Permission mode update failed or the query was cancelled."}; + } return { behavior: "allow", updatedInput: input }; } // Anything else the CLI wants a human decision on: Bash or a @@ -1357,13 +1392,12 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, }; } _log("Plan approved by user, continuing in this turn"); - _planExitApprovedThisTurn = true; // The browser pushes the restored UI mode via setPermissionMode // before answering; only fill in if it hasn't. Auto (classifier // approved) is the landing mode after a plan — it suits the // implementation phase better than manual Edit Mode confirms. - if (_runtimePermissionMode === "plan") { - _runtimePermissionMode = "auto"; + if (await _fileModeDenial(_runtimePermissionMode === "plan" ? "auto" : undefined)) { + return {behavior: "deny", message: "Permission mode update failed or the query was cancelled."}; } return { behavior: "allow", updatedInput: input }; } @@ -1389,8 +1423,8 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, }, // Permission allow-rules, not a tool availability list. Bash is // deliberately absent so that nothing here can pre-approve a shell - // command: every one is judged by the permission pipeline, and in - // Auto that means the SDK's classifier, whose "ask" verdicts reach + // command. Edit and Write are absent for the same reason: in Auto + // the SDK permission pipeline decides, and its "ask" verdicts reach // canUseTool below as the panel's Allow/Deny card. The CLI happens // to ignore a Bash allow rule anyway ("Ignoring dangerous permission // Bash(*) from cliArg (bypasses classifier)"), so leaving it out @@ -1398,7 +1432,7 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, // uses the manual confirm in the Bash PreToolUse hook below, and // Allow Everything (bypassPermissions) skips permission checks. allowedTools: [ - "Read", "Edit", "Write", "Glob", "Grep", + "Read", "Glob", "Grep", "AskUserQuestion", "Task", "Agent", // Background-subagent plumbing: lets the main agent relay a // user follow-up to a running subagent (SendMessage), read its @@ -1496,7 +1530,7 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, { matcher: "Edit", hooks: [ - async (input) => { + async (input, toolUseID) => { if (_isAiScratchPath(input && input.tool_input && input.tool_input.file_path)) { return { hookSpecificOutput: { hookEventName: "PreToolUse", permissionDecision: "allow" } }; } @@ -1523,10 +1557,13 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, fs.mkdirSync(dir, { recursive: true }); } fs.writeFileSync(input.tool_input.file_path, content, "utf8"); + _savedPlanToolIds.add(toolUseID || input.tool_use_id); _lastPlanContent = content; console.log("[Phoenix AI] Captured plan edit content:", content.length + "ch"); } catch (err) { console.warn("[Phoenix AI] Failed to edit plan file:", err.message); + return {hookSpecificOutput: {hookEventName: "PreToolUse", permissionDecision: "deny", + permissionDecisionReason: "Could not save the plan file: " + err.message}}; } const planReason = "Plan file updated." + _clarificationHintFor(input); return { @@ -1537,18 +1574,19 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, } }; } + const modeDenial = await _fileModeDenial(); + if (modeDenial) { return modeDenial; } + const checkedMode = _runtimePermissionMode; + let planWriteApproved = false; const outsideDenial = await _denyUnlessOutsideWriteAllowed( "Edit", input.tool_input, signal); if (outsideDenial) { return outsideDenial; } - // Plan mode + user-file Edit: ask the user whether - // to switch to Edit Mode. Mirrors the Bash confirm - // pattern (matcher: "Bash"). Once approved, the - // _planExitApprovedThisTurn flag suppresses the - // prompt for subsequent edits in the same turn. + // Read the current mode, not the query-start mode: + // approving a plan returns subsequent edits to Auto. const filePath = input.tool_input.file_path; - if (permissionMode === "plan" && !_planExitApprovedThisTurn) { + if (_runtimePermissionMode === "plan") { const approved = await _askPlanModeWriteConfirm( requestId, "Edit", filePath, signal); if (!approved) { @@ -1562,15 +1600,18 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, } }; } - _planExitApprovedThisTurn = true; + const switchDenial = await _fileModeDenial("auto"); + if (switchDenial) { return switchDenial; } + planWriteApproved = true; } // New flow: flush dirty buffer to disk so SDK reads // the latest content, capture pre-edit content for - // snapshot tracking, then return {} (or "allow" if - // we're auto-exiting plan mode) so SDK runs native + // snapshot tracking, then let the SDK run native // Edit on disk. Its mtime/read tracker stays // consistent and the next Edit won't trip the - // "modified since read" safety check. + // "modified since read" safety check. This saves the + // user's existing buffer, not the AI's proposed edit. + // The SDK can still deny the prepared edit in Auto. const oldString = input.tool_input.old_string; let captured = { content: "" }; try { @@ -1600,11 +1641,14 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, } }; } + const finalModeDenial = await _fileModeDenial(); + if (finalModeDenial) { return finalModeDenial; } editCount++; - // In plan mode, after the user approved the - // confirmation prompt, we need an explicit "allow" - // to override the SDK's default plan-mode block. - if (permissionMode === "plan") { + // A plan confirmation approves this exact call only. + // Otherwise only AI Edit Mode pre-approves after its + // guard. A mode change during prep must not skip it. + if (planWriteApproved || + (checkedMode === "acceptEdits" && _runtimePermissionMode === "acceptEdits")) { return { hookSpecificOutput: { hookEventName: "PreToolUse", @@ -1642,7 +1686,7 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, { matcher: "Write", hooks: [ - async (input) => { + async (input, toolUseID) => { console.log("[Phoenix AI] Intercepted Write tool"); if (_isAiScratchPath(input && input.tool_input && input.tool_input.file_path)) { // The daily sweep may have removed the folder; it comes back for the write. @@ -1657,18 +1701,21 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, const writePath = input.tool_input.file_path || ""; const normalizedPath = writePath.replace(/\\/g, "/"); if (normalizedPath.includes("/.claude/plans/")) { - _lastPlanContent = input.tool_input.content || ""; - console.log("[Phoenix AI] Captured plan content:", - _lastPlanContent.length + "ch"); + const planContent = input.tool_input.content || ""; // Write to disk so Claude can read it back later try { const dir = path.dirname(writePath); if (!fs.existsSync(dir)) { fs.mkdirSync(dir, { recursive: true }); } - fs.writeFileSync(writePath, input.tool_input.content || "", "utf8"); + fs.writeFileSync(writePath, planContent, "utf8"); + _savedPlanToolIds.add(toolUseID || input.tool_use_id); + _lastPlanContent = planContent; + console.log("[Phoenix AI] Captured plan content:", planContent.length + "ch"); } catch (err) { console.warn("[Phoenix AI] Failed to write plan file:", err.message); + return {hookSpecificOutput: {hookEventName: "PreToolUse", permissionDecision: "deny", + permissionDecisionReason: "Could not save the plan file: " + err.message}}; } const planReason = "Plan file saved." + _clarificationHintFor(input); return { @@ -1679,6 +1726,10 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, } }; } + const modeDenial = await _fileModeDenial(); + if (modeDenial) { return modeDenial; } + const checkedMode = _runtimePermissionMode; + let planWriteApproved = false; const outsideDenial = await _denyUnlessOutsideWriteAllowed( "Write", input.tool_input, signal); if (outsideDenial) { @@ -1687,7 +1738,7 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, // Plan mode + user-file Write: same confirmation // path as Edit. See Edit hook above for rationale. const filePath = input.tool_input.file_path; - if (permissionMode === "plan" && !_planExitApprovedThisTurn) { + if (_runtimePermissionMode === "plan") { const approved = await _askPlanModeWriteConfirm( requestId, "Write", filePath, signal); if (!approved) { @@ -1701,19 +1752,24 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, } }; } - _planExitApprovedThisTurn = true; + const switchDenial = await _fileModeDenial("auto"); + if (switchDenial) { return switchDenial; } + planWriteApproved = true; } // Mirror Edit: flush dirty buffer, capture pre-write - // content, return {} (or "allow" in plan mode) so - // SDK writes natively. + // content, and leave Auto's permission decision to + // the SDK before it writes natively. try { await nodeConnector.execPeer("saveBufferToDisk", { filePath }); await nodeConnector.execPeer("captureFileContent", { filePath }); } catch (err) { console.warn("[Phoenix AI] Write prep failed:", filePath, err.message); } + const finalModeDenial = await _fileModeDenial(); + if (finalModeDenial) { return finalModeDenial; } editCount++; - if (permissionMode === "plan") { + if (planWriteApproved || + (checkedMode === "acceptEdits" && _runtimePermissionMode === "acceptEdits")) { return { hookSpecificOutput: { hookEventName: "PreToolUse", @@ -1890,6 +1946,9 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, matcher: "ExitPlanMode", hooks: [ async () => { + // The native tool also changes modes. Re-apply the + // panel's mode after it runs, before the next tool. + if (await _fileModeDenial(_runtimePermissionMode)) { return {}; } return { hookSpecificOutput: { hookEventName: "PostToolUse", @@ -2162,10 +2221,9 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, try { _log("Query start:", JSON.stringify(prompt).slice(0, 80), "cwd=" + (projectPath || "?")); - // Build prompt: multi-modal with images, or plain string - let sdkPrompt = prompt; + // Build the text and optional image blocks for the user message. + const contentBlocks = [{ type: "text", text: prompt }]; if (images && images.length > 0) { - const contentBlocks = [{ type: "text", text: prompt }]; images.forEach(function (img, idx) { // Infer media type from base64 header if missing let mediaType = img.mediaType; @@ -2189,20 +2247,25 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, source: { type: "base64", media_type: mediaType, data: img.base64Data } }); }); - sdkPrompt = (async function* () { - yield { - type: "user", - session_id: currentSessionId || "", - message: { role: "user", content: contentBlocks }, - parent_tool_use_id: null - }; - }()); } + // Streaming input enables SDK control requests, including permission + // mode changes, for text-only turns as well as image turns. + const sdkPrompt = (async function* () { + yield { + type: "user", + session_id: currentSessionId || "", + message: { role: "user", content: contentBlocks }, + parent_tool_use_id: null + }; + }()); + queryOptions.permissionMode = _runtimePermissionMode; const result = queryFn({ prompt: sdkPrompt, options: queryOptions }); + activeQuery = {requestId, query: result, signal, controller: currentAbortController, + modeUpdate: Promise.resolve()}; let accumulatedText = ""; let lastStreamTime = 0; @@ -2688,10 +2751,14 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, // stopped the turn, which is not a failure either. const resultToolName = _toolUseIdToName[block.tool_use_id]; const isAnsweredByDeny = resultToolName === "AskUserQuestion"; + // Successful intercepted plan writes also use deny as + // transport. Do not depend on the CLI's error wording. + const planFileSaved = _savedPlanToolIds.delete(block.tool_use_id); nodeConnector.triggerPeer("aiToolResult", { requestId: requestId, toolId: counterId, - isError: !!block.is_error && !isAnsweredByDeny, + isError: !!block.is_error && !isAnsweredByDeny && !planFileSaved, + planFileSaved: planFileSaved, imageSearch: ["mcp__phoenix-editor__searchImages", "mcp__phoenix-editor__previewImages", "mcp__phoenix-editor__useImage"] .includes(resultToolName) ? @@ -2789,5 +2856,9 @@ async function _runQuery(requestId, prompt, projectPath, model, signal, locale, sessionId: currentSessionId, sessionTitle: await _getAISessionTitle(currentSessionId, projectPath) }); + } finally { + if (activeQuery && activeQuery.requestId === requestId) { + activeQuery = null; + } } } diff --git a/src-node/test-connection.js b/src-node/test-connection.js index f6accf7cc2..f07f533805 100644 --- a/src-node/test-connection.js +++ b/src-node/test-connection.js @@ -6,6 +6,7 @@ require("./test/test-npm-node-shim"); require("./test/test-media-server"); require("./test/test-builder-hub"); require("./test/test-ai-model-effort"); +require("./test/test-ai-file-permissions"); const TEST_NODE_CONNECTOR_ID = "ph_test_connector"; const nodeConnector = NodeConnector.createNodeConnector(TEST_NODE_CONNECTOR_ID, exports); diff --git a/src-node/test/test-ai-file-permissions.js b/src-node/test/test-ai-file-permissions.js new file mode 100644 index 0000000000..8830523bc3 --- /dev/null +++ b/src-node/test/test-ai-file-permissions.js @@ -0,0 +1,191 @@ +/* + * Copyright (c) 2021 - present core.ai + * SPDX-License-Identifier: AGPL-3.0-or-later + */ + +/** Offline exercises of the real agent hooks; assertions live in the Jasmine suite. */ +const fs = require("fs"); +const path = require("path"); +const vm = require("vm"); +const NodeConnector = require("../node-connector"); + +/** + * Exercise one bounded permission scenario with a fake SDK and browser connector. + * No installed CLI, network, credentials, user files or global agent state are used. + * @param {Object} params Mode, tool, target category and optional mode/answer scenario. + * @return {Promise} Hook decisions, SDK options and observed browser operations. + */ +async function exercise(params) { + const mode = params.mode || "auto"; + const tool = params.tool || "Write"; + if (!["auto", "acceptEdits", "plan", "bypassPermissions"].includes(mode) || + !["Write", "Edit"].includes(tool)) { + throw new Error("Unsupported permission fixture"); + } + const project = path.resolve("/phoenix-fixture/project"); + const attached = path.resolve("/phoenix-fixture/attached"); + const scratch = path.resolve("/phoenix-fixture/scratch"); + const temp = path.resolve("/phoenix-fixture/temp"); + const locations = { + project: path.join(project, "file.txt"), + attached: path.join(attached, "file.txt"), + scratch: path.join(scratch, "file.txt"), + temp: path.join(temp, "file.txt"), + outside: path.resolve("/phoenix-fixture/outside/file.txt"), + plan: path.resolve("/phoenix-fixture/.claude/plans/fixture.md") + }; + const filePath = locations[params.location || "outside"]; + if (!filePath) { throw new Error("Unsupported fixture location"); } + const operations = [], events = [], modeChanges = []; + const exported = {}; + let options, inputPrompt, finish, ready, complete; + let switchedDuringPrep = false; + const finished = new Promise(resolve => { finish = resolve; }); + const queryReady = new Promise(resolve => { ready = resolve; }); + const queryComplete = new Promise(resolve => { complete = resolve; }); + const connector = { + execPeer: async function (name, args) { + operations.push({name, filePath: args && args.filePath}); + if (name === "captureFileContent" && params.switchDuringPrep && !switchedDuringPrep) { + switchedDuringPrep = true; + await exported.setPermissionMode({mode: params.switchDuringPrep}); + } + return name === "captureFileContent" ? {content: params.stale ? "changed" : "before"} : {}; + }, + triggerPeer: function (name, data) { + events.push({name, data}); + if (name === "aiBashConfirm") { + exported.answerBashConfirm({confirmId: data.confirmId, allowed: params.allow === true}); + } else if (name === "aiPlanModeWriteConfirm") { + exported.answerPlanModeWriteConfirm({confirmId: data.confirmId, approved: params.allow === true}); + } else if (name === "aiPlanProposed") { + exported.answerPlan({confirmId: data.confirmId, approved: params.allow === true}); + } else if (name === "aiComplete" || name === "aiError") { + complete(); + } + } + }; + const sdk = { + query: function (request) { + options = request.options; + inputPrompt = request.prompt; + ready(); + return { + supportedModels: async () => [], + setPermissionMode: async function (nextMode) { + modeChanges.push(nextMode); + if (params.modeFailure) { throw new Error("fixture mode update failed"); } + if (params.modeTimeout) { await new Promise(() => {}); } + }, + async *[Symbol.asyncIterator]() { + yield {type: "system", subtype: "init"}; + await finished; + if (params.emitResult) { + yield {type: "stream_event", event: {type: "content_block_start", index: 0, + content_block: {type: "tool_use", id: "fixture-tool", name: tool}}}; + yield {type: "stream_event", event: {type: "content_block_delta", index: 0, + delta: {type: "input_json_delta", partial_json: JSON.stringify({file_path: filePath})}}}; + yield {type: "stream_event", event: {type: "content_block_stop", index: 0}}; + yield {type: "user", message: {content: [{type: "tool_result", tool_use_id: "fixture-tool", + is_error: true, content: "PreToolUse:" + tool + " hook error: " + + (params.planWriteFailure ? "Could not save the plan file" : "Plan file saved.")}]}}; + } + } + }; + } + }; + const dependencies = { + fs: {existsSync: () => true, readFileSync: () => "before", mkdirSync: () => {}, writeFileSync: () => { + if (params.planWriteFailure) { throw new Error("fixture write failed"); } + operations.push({name: "writeFileSync"}); + }}, + os: {tmpdir: () => temp}, + path, + "./mcp-editor-tools": {createEditorMcpServer: () => ({})}, + "./cli-locator": {locateCli: async () => ({}), getSourceEnv: () => ({})}, + "./ai-image-preview": {}, + "./ai-system-prompt": {buildSystemPrompt: () => "", buildEditorContextLine: () => ""}, + "./ai-cli-connector": {setBrowserConnector: () => {}}, + "./ai-cli-capabilities": {}, + "./ai-model-effort": {effortForQuery: () => undefined}, + "./node-connector": {isConnected: () => true} + }; + const source = fs.readFileSync(path.join(__dirname, "..", "claude-code-agent.js"), "utf8"); + vm.runInNewContext(source + "\nqueryModule = sdkFixture;", { + exports: exported, sdkFixture: sdk, + global: {createNodeConnector: () => connector}, + process: {platform: process.platform, env: {}, cwd: () => project}, + console: {log() {}, warn() {}, error() {}}, + AbortController, Buffer, clearTimeout, + setTimeout: function (callback, ms) { + if (params.modeTimeout && ms === 10000) { + Promise.resolve().then(callback); + return null; + } + return setTimeout(callback, ms); + }, + require: function (name) { + if (!(name in dependencies)) { throw new Error("Unexpected fixture dependency " + name); } + return dependencies[name]; + } + }, {filename: "claude-code-agent.js"}); + let timeout; + const deadline = new Promise((resolve, reject) => { + timeout = setTimeout(() => reject(new Error("Permission fixture timed out")), 5000); + }); + try { + await exported.sendPrompt({prompt: "fixture", projectPath: project, permissionMode: mode, + additionalDirectories: [attached], aiScratchDir: scratch, + images: params.image ? [{mediaType: "image/png", base64Data: "aGVsbG8="}] : undefined}); + await Promise.race([queryReady, deadline]); + const promptMessages = []; + if (typeof inputPrompt !== "string") { + for await (const message of inputPrompt) { promptMessages.push(message); } + } + let modeError = null; + if (params.switchTo) { + try { + await exported.setPermissionMode({mode: params.switchTo}); + } catch (err) { modeError = err.message; } + } + let planDecision; + if (params.approvePlan) { + planDecision = await options.canUseTool("ExitPlanMode", {plan: "Fixture plan"}, + {signal: options.abortController.signal}); + if (planDecision.behavior === "allow") { + const postExit = options.hooks.PostToolUse.find(entry => entry.matcher === "ExitPlanMode"); + await postExit.hooks[0](); + } + } + const toolInput = {file_path: filePath, content: "after", old_string: "before", new_string: "after"}; + const hook = options.hooks.PreToolUse.find(entry => entry.matcher === tool).hooks[0]; + const decision = await Promise.race([ + hook({tool_name: tool, tool_input: toolInput}, "fixture-tool", {}), deadline + ]); + const nextDecision = params.followup ? await Promise.race([ + hook({tool_name: tool, tool_input: toolInput}, "fixture-next-tool", {}), deadline + ]) : undefined; + let sdkAsk; + if (params.sdkAsk) { + sdkAsk = await options.canUseTool(tool, toolInput, {signal: options.abortController.signal}); + } + if (params.emitResult) { + finish(); + await Promise.race([queryComplete, deadline]); + } + return JSON.parse(JSON.stringify({decision, nextDecision, planDecision, sdkAsk, modeError, modeChanges, + aborted: options.abortController.signal.aborted, operations, + toolResults: events.filter(event => event.name === "aiToolResult").map(event => event.data), + events: events.filter(event => ["aiBashConfirm", "aiPlanModeWriteConfirm", "aiPlanProposed"].includes(event.name)), + allowedTools: options.allowedTools, permissionMode: options.permissionMode, + streamingInput: typeof inputPrompt !== "string", promptMessages})); + } finally { + await exported.cancelQuery(); + finish(); + await Promise.race([queryComplete, deadline]); + clearTimeout(timeout); + } +} + +exports.exercise = exercise; +NodeConnector.createNodeConnector("ph_test_ai_file_permissions", exports); diff --git a/src/nls/root/strings.js b/src/nls/root/strings.js index 953d018403..94bbc78ff3 100644 --- a/src/nls/root/strings.js +++ b/src/nls/root/strings.js @@ -2974,6 +2974,7 @@ define({ "AI_CHAT_PLAN_MAXIMIZE": "Open plan in full screen", "AI_CHAT_PLAN_CLOSE_FULLSCREEN": "Minimize back to the chat (Esc)", "AI_CHAT_PLAN_APPROVE": "Approve", + "AI_CHAT_PLAN_REJECT": "Reject", "AI_CHAT_PLAN_REVISE": "Revise", "AI_CHAT_PLAN_FEEDBACK_PLACEHOLDER": "What would you like changed?", "AI_CHAT_PLAN_STOP": "Stop", diff --git a/src/styles/Extn-AIChatPanel.less b/src/styles/Extn-AIChatPanel.less index b214f2ee75..bf2c4a54c0 100644 --- a/src/styles/Extn-AIChatPanel.less +++ b/src/styles/Extn-AIChatPanel.less @@ -3824,9 +3824,6 @@ justify-content: flex-end; } - .ai-plan-stop-btn { - margin-right: auto; - } } .ai-plan-fullscreen-close { @@ -3956,6 +3953,7 @@ .ai-plan-actions { display: flex; + justify-content: flex-end; gap: 8px; padding: 8px 12px; border-top: 1px solid rgba(107, 158, 255, 0.1); @@ -6178,7 +6176,7 @@ } } - .ai-plan-stop-btn { + .ai-plan-reject-btn { background: rgba(229, 115, 115, 0.08); border: 1px solid rgba(229, 115, 115, 0.3); color: #e57373; diff --git a/test/UnitTestSuite.js b/test/UnitTestSuite.js index 8a397a00db..c7ea5e6432 100644 --- a/test/UnitTestSuite.js +++ b/test/UnitTestSuite.js @@ -167,6 +167,7 @@ define(function (require, exports, module) { require("spec/AIImageTools-test"); require("spec/AICliConnector-test"); require("spec/AIModelEffort-test"); + require("spec/AIFilePermissions-test"); // pro test suite optional components require("./pro-test-suite"); // todo TEST_MODERN diff --git a/test/spec/AIFilePermissions-test.js b/test/spec/AIFilePermissions-test.js new file mode 100644 index 0000000000..7e3c488955 --- /dev/null +++ b/test/spec/AIFilePermissions-test.js @@ -0,0 +1,185 @@ +/* + * Copyright (c) 2021 - present core.ai + * SPDX-License-Identifier: AGPL-3.0-or-later + */ +/*global describe, it, expect, beforeAll, awaitsFor */ + +define(function (require, exports, module) { + const NodeConnector = require("NodeConnector"); + + // Native unit jobs run the isolated Node fixture; browsers have no SDK host. + if (!Phoenix.isNativeApp) { return; } + + describe("unit:AI File Permissions", function () { + let connector; + beforeAll(async function () { + await awaitsFor(NodeConnector.isNodeReady, "Node runtime to be ready"); + connector = NodeConnector.createNodeConnector("ph_test_ai_file_permissions", exports); + }); + + /** + * Run one isolated real-agent permission scenario. + * @param {Object} params Mode, tool and optional user response. + * @return {Promise} Observed decisions and browser operations. + */ + function exercise(params) { + return connector.execPeer("exercise", params); + } + + ["Edit", "Write"].forEach(function (tool) { + it("Auto leaves outside-project " + tool + " to the SDK without a manual prompt", async function () { + const result = await exercise({tool, mode: "auto"}); + expect(result.decision).toEqual({}); + expect(result.events).toEqual([]); + expect(result.allowedTools).not.toContain(tool); + expect(result.operations.map(op => op.name)).toEqual(["saveBufferToDisk", "captureFileContent"]); + }); + + it("AI Edit Mode denies outside-project " + tool + " before preparation", async function () { + const result = await exercise({tool, mode: "acceptEdits"}); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.events.length).toBe(1); + expect(result.events[0].name).toBe("aiBashConfirm"); + expect(result.operations).toEqual([]); + }); + + it("AI Edit Mode permits an explicitly approved outside-project " + tool, async function () { + const result = await exercise({tool, mode: "acceptEdits", allow: true}); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("allow"); + expect(result.events.length).toBe(1); + }); + + it("switching AI Edit Mode to Auto sends " + tool + " through the SDK", async function () { + const result = await exercise({tool, mode: "acceptEdits", switchTo: "auto"}); + expect(result.modeChanges).toEqual(["auto"]); + expect(result.decision).toEqual({}); + expect(result.events).toEqual([]); + }); + + it("switching Auto to AI Edit Mode restores the outside-project " + tool + " prompt", async function () { + const result = await exercise({tool, mode: "auto", switchTo: "acceptEdits"}); + expect(result.modeChanges).toEqual(["acceptEdits"]); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.events.length).toBe(1); + expect(result.operations).toEqual([]); + }); + + it("does not pre-approve " + tool + " when mode changes during preparation", async function () { + const result = await exercise({tool, mode: "auto", switchDuringPrep: "acceptEdits"}); + expect(result.modeChanges).toEqual(["acceptEdits"]); + expect(result.decision).toEqual({}); + }); + + it("denies " + tool + " if switching out of Plan Mode fails", async function () { + const result = await exercise({tool, mode: "plan", allow: true, modeFailure: true}); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.aborted).toBe(true); + expect(result.operations).toEqual([]); + }); + + it("reports a saved plan " + tool + " as handled despite the SDK denial wrapper", async function () { + const result = await exercise({tool, mode: "plan", location: "plan", emitResult: true}); + expect(result.toolResults.length).toBe(1); + expect(result.toolResults[0].planFileSaved).toBe(true); + expect(result.toolResults[0].isError).toBe(false); + expect(result.operations.map(op => op.name)).toEqual(["writeFileSync"]); + }); + + it("keeps a failed plan " + tool + " visible as an error", async function () { + const result = await exercise({tool, mode: "plan", location: "plan", + emitResult: true, planWriteFailure: true}); + expect(result.decision.hookSpecificOutput.permissionDecisionReason).toContain("fixture write failed"); + expect(result.toolResults.length).toBe(1); + expect(result.toolResults[0].planFileSaved).toBe(false); + expect(result.toolResults[0].isError).toBe(true); + }); + + it("Plan Mode rejects unapproved " + tool + " without an outside-root prompt", async function () { + const result = await exercise({tool, mode: "plan"}); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.events.map(event => event.name)).toEqual(["aiPlanModeWriteConfirm"]); + expect(result.operations).toEqual([]); + }); + + it("approving a plan-mode " + tool + " allows only that call before switching to Auto", async function () { + const result = await exercise({tool, mode: "plan", allow: true, followup: true}); + expect(result.modeChanges).toEqual(["auto"]); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("allow"); + expect(result.nextDecision).toEqual({}); + expect(result.events.map(event => event.name)).toEqual(["aiPlanModeWriteConfirm"]); + }); + }); + + ["project", "attached", "temp", "scratch"].forEach(function (location) { + it("AI Edit Mode writes in " + location + " without a manual prompt", async function () { + const result = await exercise({mode: "acceptEdits", location}); + expect(result.events).toEqual([]); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("allow"); + }); + }); + + it("Auto lets an SDK permission request reach the user and returns their denial", async function () { + const result = await exercise({mode: "auto", sdkAsk: true}); + expect(result.decision).toEqual({}); + expect(result.events.map(event => event.name)).toEqual(["aiBashConfirm"]); + expect(result.sdkAsk.behavior).toBe("deny"); + }); + + it("approved ExitPlanMode leaves subsequent writes under Auto's SDK decision", async function () { + const result = await exercise({mode: "plan", approvePlan: true, allow: true}); + expect(result.planDecision.behavior).toBe("allow"); + expect(result.modeChanges).toEqual(["auto", "auto"]); + expect(result.decision).toEqual({}); + expect(result.events.map(event => event.name)).toEqual(["aiPlanProposed"]); + }); + + it("Allow Everything keeps its existing unrestricted behavior", async function () { + const result = await exercise({mode: "bypassPermissions"}); + expect(result.permissionMode).toBe("bypassPermissions"); + expect(result.events).toEqual([]); + expect(result.decision).toEqual({}); + }); + + it("failed SDK mode changes abort rather than continuing under a stale mode", async function () { + const result = await exercise({mode: "acceptEdits", switchTo: "auto", modeFailure: true}); + expect(result.aborted).toBe(true); + expect(result.modeError).toContain("fixture mode update failed"); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.operations).toEqual([]); + }); + + it("aborts and denies when the SDK never acknowledges a mode change", async function () { + const result = await exercise({mode: "acceptEdits", switchTo: "auto", modeTimeout: true}); + expect(result.aborted).toBe(true); + expect(result.modeError).toContain("timed out"); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.operations).toEqual([]); + }); + + it("denies ExitPlanMode if the SDK mode update fails", async function () { + const result = await exercise({mode: "plan", approvePlan: true, allow: true, modeFailure: true}); + expect(result.planDecision.behavior).toBe("deny"); + expect(result.aborted).toBe(true); + expect(result.operations).toEqual([]); + }); + + it("text-only prompts use streaming input so SDK mode changes are available", async function () { + const result = await exercise({mode: "auto"}); + expect(result.streamingInput).toBe(true); + expect(result.promptMessages[0].message.content).toEqual([{type: "text", text: "fixture"}]); + }); + + it("image prompts preserve both text and image content", async function () { + const result = await exercise({mode: "auto", image: true}); + expect(result.streamingInput).toBe(true); + expect(result.promptMessages[0].message.content.map(item => item.type)).toEqual(["text", "image"]); + }); + + it("Auto still rejects an edit that no longer matches the current buffer", async function () { + const result = await exercise({mode: "auto", tool: "Edit", stale: true}); + expect(result.decision.hookSpecificOutput.permissionDecision).toBe("deny"); + expect(result.decision.hookSpecificOutput.permissionDecisionReason).toContain("modified by the user"); + expect(result.events).toEqual([]); + }); + }); +}); From d183521d16b1a7735a0e02611c63b4d7318ea413 Mon Sep 17 00:00:00 2001 From: abose Date: Fri, 9 Oct 2026 15:30:12 +0530 Subject: [PATCH 4/8] chore(deps): advance Pro pin for plan review fixes Pin Phoenix Pro to 9a0063e so builds include the matching plan review controls, unconditional Auto approval and structured plan-save handling. Validation: tracking JSON parses and the pin matches Pro HEAD. The paired core and Pro changes passed 74 focused checks on each of Linux, macOS and Windows, plus the recorded live plan and permission flows. --- tracking-repos.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tracking-repos.json b/tracking-repos.json index a7157f86f6..092d997d18 100644 --- a/tracking-repos.json +++ b/tracking-repos.json @@ -1,5 +1,5 @@ { "phoenixPro": { - "commitID": "18cbc74c899ce06665e237be14be9554f95c11e1" + "commitID": "9a0063ead251352f5f1ebffdb71ba90389910ae5" } } From 4624c5d6c4fef7771639b1e9fd3fee16d3f60a29 Mon Sep 17 00:00:00 2001 From: abose Date: Fri, 9 Oct 2026 17:30:32 +0530 Subject: [PATCH 5/8] feat(ai): price Codex usage by reported service tier Read the service tier from each completed Codex response and estimate cost from a validated catalog with explicit model, tier and long-context rates. Keep bundled prices as the fallback and reject unsupported or stale catalogs without changing active prices or recalculating saved usage. Prepare cached prices on first use with a bounded wait and retry after failure. Share concurrent preparation, preserve usage deduplication, and leave provider-reported costs and zero-token events independent of loading. Advance the Pro pin for matching lazy file caching, tier usage names and chart coverage. Validation: 57 CLI connector, 16 pricing-cache and 33 connector-peer cases pass on each native platform. Ledger and calendar suites also pass on Linux, macOS and Windows. Changed-file lint and whitespace checks pass. --- src-node/ai-cli-connector.js | 2 + src-node/ai-cli-pricing.js | 112 ++++++++--- src-node/ai-cli-usage.js | 67 ++++++- src-node/claude-code-agent.js | 4 + src-node/external-model-pricing.json | 248 +++++++++++++++++++++++++ src-node/test/test-ai-cli-connector.js | 115 +++++++++++- test/spec/AICliConnector-test.js | 48 +++++ tracking-repos.json | 2 +- 8 files changed, 562 insertions(+), 36 deletions(-) create mode 100644 src-node/external-model-pricing.json diff --git a/src-node/ai-cli-connector.js b/src-node/ai-cli-connector.js index 22e5e84079..2576b33212 100644 --- a/src-node/ai-cli-connector.js +++ b/src-node/ai-cli-connector.js @@ -50,6 +50,7 @@ class CliConnector { emit: record => { if (this.options.emitUsage) { this.options.emitUsage(record); } }, ready: () => !this.options.ready || this.options.ready(), baseUrl: () => "http://localhost:" + this.server.address().port, + preparePricing: this.options.preparePricing, drainMs: this.options.usageDrainMs }); this.upgrade = (request, socket, head) => { @@ -412,6 +413,7 @@ exports.setBrowserConnector = function (connector, ready) { browserConnector = c /** Attach exactly once to the window's existing HTTP server. */ exports.attach = function (server) { controller = new CliConnector(server, {peer: (fn, args) => browserConnector.execPeer(fn, args), + preparePricing: () => browserConnector.execPeer("prepareExternalModelPricing"), emitUsage: record => browserConnector.triggerPeer("aiCliUsage", record), ready: () => browserReady(), emit: state => browserConnector.triggerPeer("aiCliConnectorState", state)}); }; diff --git a/src-node/ai-cli-pricing.js b/src-node/ai-cli-pricing.js index ba713dfeec..fddbfc1f6f 100644 --- a/src-node/ai-cli-pricing.js +++ b/src-node/ai-cli-pricing.js @@ -1,41 +1,95 @@ /* Copyright (c) 2026 core.ai; SPDX-License-Identifier: AGPL-3.0-or-later */ -// Standard API USD per million tokens, checked 2026-10-05. These are API-equivalent -// estimates, not subscription charges, and exclude service-tier and regional premiums. +const {z} = require("zod"); +const bundledPricing = require("./external-model-pricing.json"); + +// API-equivalent USD per million tokens, not subscription charges or regional premiums. +// Bundled prices checked 2026-10-09: // https://developers.openai.com/api/docs/pricing // https://developers.openai.com/api/docs/models/gpt-6-sol -// https://developers.openai.com/api/docs/models/gpt-5.6-sol (also the gpt-5.6 alias) -// https://developers.openai.com/api/docs/models/gpt-5.6-terra -// https://developers.openai.com/api/docs/models/gpt-5.6-luna -// https://developers.openai.com/api/docs/models/gpt-5.3-codex -// Fields: uncached input, cached input, cache write, output. null means unpublished. -const RATES = new Map([ - ["gpt-6-astra", [10, 1, 12.5, 50]], - ["gpt-6.1-sol", [2, 0.10, 2.5, 10]], - ["gpt-6-sol", [2, 0.20, 2.5, 10]], - ["gpt-6-luna", [0.10, 0.01, 0.125, 0.50]], - ["gpt-5.6-sol", [4, 0.40, 5, 20]], - ["gpt-5.6", [4, 0.40, 5, 20]], - ["gpt-5.6-terra", [2, 0.20, 2.5, 12]], - ["gpt-5.6-luna", [0.20, 0.02, 0.25, 1.20]], - ["gpt-5.3-codex", [1.75, 0.175, null, 14]] -]); +// https://developers.openai.com/api/docs/guides/fast-mode +// Unknown model/tier combinations stay unpriced; no universal premium is assumed. +const MAX_CATALOG_BYTES = 256 * 1024; +const price = z.number().finite().min(0).max(1000000).nullable(); +const ratesSchema = z.object({input: price, cacheRead: price, cacheWrite: price, output: price}).strict(); +const tiers = {standard: ratesSchema, fast: ratesSchema.optional(), ultrafast: ratesSchema.optional()}; +const modelSchema = z.object(Object.assign({}, tiers, { + longContext: z.object(Object.assign({aboveInputTokens: z.number().int().positive().max(100000000)}, tiers)) + .strict().optional() +})).strict(); +const catalogSchema = z.object({ + version: z.literal(1), + updatedAt: z.string().regex(/^\d{4}-\d{2}-\d{2}$/), + currency: z.literal("USD"), + unit: z.literal("million_tokens"), + models: z.record(z.string().min(1).max(200).regex(/^[a-zA-Z0-9][a-zA-Z0-9._:/@\[\]-]*$/), modelSchema) + .refine(models => Object.keys(models).length > 0 && Object.keys(models).length <= 500) +}).strict(); +const TOKEN_KINDS = ["input", "cacheRead", "cacheWrite", "output"]; + +/** + * Validate and copy a downloaded or cached price catalog before making any of it active. + * @param {*} catalog Candidate version-1 USD catalog. + * @return {?Object} Validated catalog, or null without modifying the active prices. + */ +function validateCatalog(catalog) { + try { + if (JSON.stringify(catalog).length > MAX_CATALOG_BYTES) { return null; } + const result = catalogSchema.safeParse(catalog); + return result.success ? result.data : null; + } catch (error) { + return null; + } +} + +/** + * Create an isolated estimator with bundled prices and atomic, validated catalog updates. + * @return {{update: function(Object): boolean, estimate: function(?string, Object, string=): ?number}} + */ +function createPricing() { + let catalog = catalogSchema.parse(bundledPricing); + return { + update(candidate) { + const next = validateCatalog(candidate); + if (!next || next.updatedAt < catalog.updatedAt) { return false; } + catalog = next; + return true; + }, + estimate(model, usage, serviceTier = "default") { + if (!Object.hasOwn(catalog.models, model)) { return null; } + const entry = catalog.models[model]; + const tier = serviceTier === "default" ? "standard" : serviceTier; + if (!["standard", "fast", "ultrafast"].includes(tier)) { return null; } + const inclusiveInput = usage.input + usage.cacheRead + usage.cacheWrite; + const table = entry.longContext && inclusiveInput > entry.longContext.aboveInputTokens ? + entry.longContext : entry; + const rates = table[tier]; + if (!rates) { return null; } + let total = 0; + for (const kind of TOKEN_KINDS) { + if (usage[kind] > 0 && rates[kind] === null) { return null; } + total += usage[kind] * (rates[kind] || 0); + } + return total / 1000000; + } + }; +} + +const pricing = createPricing(); /** - * Estimate one Codex response using its exact reported model and disjoint token counts. - * Output already includes reasoning. Unknown models/rates stay unpriced, never guessed. + * Estimate one Codex response using its exact model, normalized tier and disjoint token counts. + * Output already includes reasoning. Missing tier metadata retains the standard estimate. * @param {?string} model Exported model ID. * @param {{input: number, output: number, cacheRead: number, cacheWrite: number}} usage - * @return {?number} Standard API-equivalent USD, or null if a rate is unavailable. + * @param {string} [serviceTier="default"] Normalized per-response tier. + * @return {?number} API-equivalent USD, or null if a model/tier rate is unavailable. */ -function estimateCodexCost(model, usage) { - const rates = RATES.get(model); - if (!rates || (usage.cacheWrite > 0 && rates[2] === null)) { return null; } - // These models charge long-context rates for the entire request beyond 272k input. - const longContext = model !== "gpt-5.3-codex" && - usage.input + usage.cacheRead + usage.cacheWrite > 272000; - const inputCost = usage.input * rates[0] + usage.cacheRead * rates[1] + usage.cacheWrite * (rates[2] || 0); - return (inputCost * (longContext ? 2 : 1) + usage.output * rates[3] * (longContext ? 1.5 : 1)) / 1000000; +function estimateCodexCost(model, usage, serviceTier = "default") { + return pricing.estimate(model, usage, serviceTier); } exports.estimateCodexCost = estimateCodexCost; +exports.updatePricing = pricing.update; +exports.validateCatalog = validateCatalog; +exports.createPricing = createPricing; diff --git a/src-node/ai-cli-usage.js b/src-node/ai-cli-usage.js index cf50ad5fc3..27526840b5 100644 --- a/src-node/ai-cli-usage.js +++ b/src-node/ai-cli-usage.js @@ -31,6 +31,7 @@ const SEEN_LIMIT = 5000; const BACKLOG_LIMIT = 2000; const BACKLOG_RETRY_MS = 2000; const EXPORT_INTERVAL_MS = 2000; +const PRICING_WAIT_MS = 5000; // The largest time a JS Date can hold. const MAX_TIME_MS = 8.64e15; @@ -89,6 +90,18 @@ function fingerprint(parts) { return createHash("sha256").update(JSON.stringify(parts)).digest("hex").slice(0, 32); } +/** + * Normalize Codex's per-response tier without consulting a session's mutable configuration. + * Codex 0.162 omits this field for Standard; older exporters also retain the standard estimate. + * @param {*} value Exported service_tier attribute. + * @return {string} A known pricing tier, or "unknown" for an unsupported explicit value. + */ +function codexServiceTier(value) { + if (value === undefined || value === null || value === "default") { return "default"; } + if (value === "priority" || value === "fast") { return "fast"; } + return value === "ultrafast" ? "ultrafast" : "unknown"; +} + /** * Claude Code's per-request event. Its input excludes cache reads and writes, so the four * kinds are already disjoint; the cost is the CLI's own estimate at API list price. @@ -132,6 +145,7 @@ function codexUsage(record, attrs) { // reads. Every probe so far reported 0 cache writes, so a nonzero value is unverified. const usage = { model: modelId(attrs.model), + serviceTier: codexServiceTier(attrs.service_tier), input: Math.max(0, count(attrs.input_token_count) - cacheRead - cacheWrite), output: count(attrs.output_token_count), cacheRead: cacheRead, @@ -140,12 +154,11 @@ function codexUsage(record, attrs) { at: recordTime(record, attrs), promptId: null }; - usage.costUSD = estimateCodexCost(usage.model, usage); // No request id: the nanosecond stamp tells two identical responses apart, and a retried // export repeats it, so the retry is dropped. usage.key = fingerprint([attrs["conversation.id"], attrs["event.timestamp"], record.timeUnixNano, record.observedTimeUnixNano, attrs.input_token_count, attrs.output_token_count, attrs.cached_token_count, - attrs.cache_write_token_count, attrs.reasoning_token_count, usage.model]); + attrs.cache_write_token_count, attrs.reasoning_token_count, usage.model, usage.serviceTier]); return usage; } @@ -207,7 +220,8 @@ function userOwnsTelemetry(cli, where = {}) { class CliUsage { /** * @param {{emit: function(Object), baseUrl: function(): string, ready?: function(): boolean, - * drainMs?: number}} options - ready says whether the editor can take records now + * drainMs?: number, preparePricing?: function(): Promise, pricingWaitMs?: number}} options + * ready says whether the editor can take records now; preparePricing lazily loads cached prices. */ constructor(options) { this.options = options; @@ -216,6 +230,8 @@ class CliUsage { this.bySession = new Map(); this.backlog = []; this.retryTimer = null; + this.pricingReady = null; + this.closed = false; } /** Deliver a record now, or hold it until the editor is back. */ @@ -279,6 +295,7 @@ class CliUsage { /** Forget every session at once (PhNode exit). */ closeAll() { + this.closed = true; clearTimeout(this.retryTimer); this.retryTimer = null; for (const entry of this.byToken.values()) { clearTimeout(entry.closeTimer); } @@ -293,8 +310,32 @@ class CliUsage { return true; } + /** + * Bound first-use preparation so a reloading browser cannot hold usage indefinitely. + * Failed preparation is retried by a later completion; this batch uses current prices. + * @return {Promise} Resolves when prices are ready or the bounded attempt failed. + */ + _preparePricing() { + if (!this.pricingReady) { + let timer; + const waitMs = this.options.pricingWaitMs === undefined ? PRICING_WAIT_MS : this.options.pricingWaitMs; + const work = Promise.race([ + Promise.resolve().then(() => this.options.preparePricing()), + new Promise((_resolve, reject) => { + timer = setTimeout(() => reject(new Error("Pricing cache preparation timed out")), waitMs); + timer.unref(); + }) + ]); + const ready = work.catch(() => { + if (this.pricingReady === ready) { this.pricingReady = null; } + }).finally(() => clearTimeout(timer)); + this.pricingReady = ready; + } + return this.pricingReady; + } + _emit(entry, usage, turns) { - this._deliver({ + const record = { sessionId: entry.sessionId, cli: entry.cli, model: usage.model || null, @@ -306,7 +347,23 @@ class CliUsage { cacheWrite: usage.cacheWrite, costUSD: usage.costUSD, turns: turns - }); + }; + if (entry.cli !== "codex" || !usage.serviceTier) { + this._deliver(record); + return; + } + record.serviceTier = usage.serviceTier; + const deliverPriced = () => { + if (this.closed) { return; } + record.costUSD = estimateCodexCost(usage.model, usage, usage.serviceTier); + this._deliver(record); + }; + if (this.options.preparePricing && usage.input + usage.output + usage.cacheRead + usage.cacheWrite > 0) { + // Claude and turn-only events never load prices or wait on this preparation. + this._preparePricing().then(deliverPriced); + } else { + deliverPriced(); + } } /** diff --git a/src-node/claude-code-agent.js b/src-node/claude-code-agent.js index d7e0bb293b..409f355d24 100644 --- a/src-node/claude-code-agent.js +++ b/src-node/claude-code-agent.js @@ -36,6 +36,7 @@ const {buildSystemPrompt, buildEditorContextLine} = require("./ai-system-prompt" const CliConnector = require("./ai-cli-connector"); const CliCapabilities = require("./ai-cli-capabilities"); const ModelEffort = require("./ai-model-effort"); +const {updatePricing} = require("./ai-cli-pricing"); const NodeConnector = require("./node-connector"); exports.readImagePreview = ImagePreview.readImage; @@ -206,6 +207,9 @@ exports.getCliConnectorStatus = async params => CliConnector.getStatus(params.se /** Disconnect or reconnect only this CLI's Phoenix tools; its terminal keeps running. */ exports.setCliConnectorEnabled = async params => CliConnector.setEnabled(params.sessionId, params.enabled); +/** Validate and atomically replace this window's Codex estimate prices; never reprices saved usage. */ +exports.setExternalModelPricing = async params => updatePricing(params && params.catalog); + // Tools whose permission request in Plan Mode means "the model wants to // start editing user files" — they share the plan-mode write-confirm card. const FILE_WRITE_TOOLS = ["Edit", "Write", "MultiEdit", "NotebookEdit"]; diff --git a/src-node/external-model-pricing.json b/src-node/external-model-pricing.json new file mode 100644 index 0000000000..b574557513 --- /dev/null +++ b/src-node/external-model-pricing.json @@ -0,0 +1,248 @@ +{ + "version": 1, + "updatedAt": "2026-10-09", + "currency": "USD", + "unit": "million_tokens", + "models": { + "gpt-6-astra": { + "standard": { + "input": 10, + "cacheRead": 1, + "cacheWrite": 12.5, + "output": 50 + }, + "fast": { + "input": 20, + "cacheRead": 2, + "cacheWrite": 25.0, + "output": 100 + }, + "ultrafast": { + "input": 60, + "cacheRead": 6, + "cacheWrite": 75.0, + "output": 300 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 20, + "cacheRead": 2, + "cacheWrite": 25.0, + "output": 75.0 + }, + "fast": { + "input": 40, + "cacheRead": 4, + "cacheWrite": 50.0, + "output": 150.0 + }, + "ultrafast": { + "input": 120, + "cacheRead": 12, + "cacheWrite": 150.0, + "output": 450.0 + } + } + }, + "gpt-6.1-sol": { + "standard": { + "input": 2, + "cacheRead": 0.1, + "cacheWrite": 2.5, + "output": 10 + }, + "fast": { + "input": 4, + "cacheRead": 0.2, + "cacheWrite": 5.0, + "output": 20 + }, + "ultrafast": { + "input": 12, + "cacheRead": 0.6, + "cacheWrite": 15.0, + "output": 60 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 4, + "cacheRead": 0.2, + "cacheWrite": 5.0, + "output": 15.0 + }, + "fast": { + "input": 8, + "cacheRead": 0.4, + "cacheWrite": 10.0, + "output": 30.0 + }, + "ultrafast": { + "input": 24, + "cacheRead": 1.2, + "cacheWrite": 30.0, + "output": 90.0 + } + } + }, + "gpt-6-sol": { + "standard": { + "input": 2, + "cacheRead": 0.2, + "cacheWrite": 2.5, + "output": 10 + }, + "fast": { + "input": 4, + "cacheRead": 0.4, + "cacheWrite": 5.0, + "output": 20 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 4, + "cacheRead": 0.4, + "cacheWrite": 5.0, + "output": 15.0 + }, + "fast": { + "input": 8, + "cacheRead": 0.8, + "cacheWrite": 10.0, + "output": 30.0 + } + } + }, + "gpt-6-luna": { + "standard": { + "input": 0.1, + "cacheRead": 0.01, + "cacheWrite": 0.125, + "output": 0.5 + }, + "fast": { + "input": 0.2, + "cacheRead": 0.02, + "cacheWrite": 0.25, + "output": 1.0 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 0.2, + "cacheRead": 0.02, + "cacheWrite": 0.25, + "output": 0.75 + }, + "fast": { + "input": 0.4, + "cacheRead": 0.04, + "cacheWrite": 0.5, + "output": 1.5 + } + } + }, + "gpt-5.6-sol": { + "standard": { + "input": 4, + "cacheRead": 0.4, + "cacheWrite": 5, + "output": 20 + }, + "fast": { + "input": 8, + "cacheRead": 0.8, + "cacheWrite": 10, + "output": 40 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 8, + "cacheRead": 0.8, + "cacheWrite": 10, + "output": 30.0 + }, + "fast": { + "input": 16, + "cacheRead": 1.6, + "cacheWrite": 20, + "output": 60.0 + } + } + }, + "gpt-5.6": { + "standard": { + "input": 4, + "cacheRead": 0.4, + "cacheWrite": 5, + "output": 20 + }, + "fast": { + "input": 8, + "cacheRead": 0.8, + "cacheWrite": 10, + "output": 40 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 8, + "cacheRead": 0.8, + "cacheWrite": 10, + "output": 30.0 + }, + "fast": { + "input": 16, + "cacheRead": 1.6, + "cacheWrite": 20, + "output": 60.0 + } + } + }, + "gpt-5.6-terra": { + "standard": { + "input": 2, + "cacheRead": 0.2, + "cacheWrite": 2.5, + "output": 12 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 4, + "cacheRead": 0.4, + "cacheWrite": 5.0, + "output": 18.0 + } + } + }, + "gpt-5.6-luna": { + "standard": { + "input": 0.2, + "cacheRead": 0.02, + "cacheWrite": 0.25, + "output": 1.2 + }, + "longContext": { + "aboveInputTokens": 272000, + "standard": { + "input": 0.4, + "cacheRead": 0.04, + "cacheWrite": 0.5, + "output": 1.8 + } + } + }, + "gpt-5.3-codex": { + "standard": { + "input": 1.75, + "cacheRead": 0.175, + "cacheWrite": null, + "output": 14 + } + } + } +} diff --git a/src-node/test/test-ai-cli-connector.js b/src-node/test/test-ai-cli-connector.js index 02a788b685..b5efd9c1d2 100644 --- a/src-node/test/test-ai-cli-connector.js +++ b/src-node/test/test-ai-cli-connector.js @@ -22,6 +22,7 @@ const {runHook} = require("../ai-cli-hooks"); const {hookCommand} = require("../ai-cli-launch"); const {checkCodexServers} = require("../ai-cli-capabilities"); const {CliUsage, userOwnsTelemetry, launchSettings} = require("../ai-cli-usage"); +const {createPricing} = require("../ai-cli-pricing"); const {spawnCli} = require("../cli-locator"); const NodeConnector = require("../node-connector"); @@ -96,6 +97,69 @@ function post(port, urlPath, body, method) { /** Usage collector fixtures; none of them need a connector session or the browser. */ async function exerciseUsage(scenario) { + if (["usage-pricing-lazy", "usage-pricing-load-error", "usage-pricing-load-timeout"].includes(scenario)) { + const emitted = []; + let loads = 0, release; + const gate = new Promise(resolve => { release = resolve; }); + const usage = new CliUsage({emit: record => emitted.push(record), baseUrl: () => "http://localhost", + pricingWaitMs: scenario === "usage-pricing-load-timeout" ? 1 : undefined, + preparePricing: async () => { + loads++; + if (scenario === "usage-pricing-load-timeout" && loads === 1) { return new Promise(() => {}); } + await gate; + if (scenario === "usage-pricing-load-error" && loads === 1) { throw new Error("cache unavailable"); } + }}); + try { + const codex = usage.open("lazy-codex", "codex"); + const claude = usage.open("lazy-claude", "claude"); + usage.ingest(claude.token, otlpLogs([claudeRequest("lazy-claude", "prompt")])); + usage.recordTurn("lazy-codex", "prompt"); + usage.ingest(codex.token, otlpLogs([pricedCompletion("1791142979000000000", + {model: "gpt-6-astra", input_token_count: 0, cached_token_count: 0, output_token_count: 0})])); + const before = {loads, emitted: emitted.length}; + const payload = otlpLogs([pricedCompletion("1791142980000000000", {model: "gpt-6-astra"}), + pricedCompletion("1791142980001000000", {model: "gpt-6-astra", service_tier: "priority"})]); + usage.ingest(codex.token, payload); + usage.ingest(codex.token, payload); + await Promise.resolve(); + const waiting = {loads, emitted: emitted.length}; + release(); + await usage.pricingReady; + usage.ingest(codex.token, otlpLogs([pricedCompletion("1791142980002000000", + {model: "gpt-6-astra", service_tier: "ultrafast"})])); + await usage.pricingReady; + return {before, waiting, loads, costs: emitted.filter(record => record.cli === "codex" && record.input) + .map(record => record.costUSD)}; + } finally { + release(); + usage.closeAll(); + } + } + if (scenario === "usage-pricing-catalog") { + const pricing = createPricing(); + const catalog = {version: 1, updatedAt: "2026-10-10", currency: "USD", unit: "million_tokens", + models: {"fixture-model": { + standard: {input: 1, cacheRead: 2, cacheWrite: 3, output: 4}, + fast: {input: 5, cacheRead: 6, cacheWrite: 7, output: 8}, + longContext: {aboveInputTokens: 100, + standard: {input: 10, cacheRead: 20, cacheWrite: 30, output: 40}} + }}}; + const tokens = {input: 60, cacheRead: 30, cacheWrite: 10, output: 5}; + const accepted = pricing.update(catalog); + const estimates = [pricing.estimate("fixture-model", tokens), + pricing.estimate("fixture-model", tokens, "fast"), + pricing.estimate("fixture-model", Object.assign({}, tokens, {input: 61})), + pricing.estimate("gpt-6-astra", tokens)]; + const invalid = [Object.assign({}, catalog, {currency: "EUR"}), + Object.assign({}, catalog, {version: 2}), Object.assign({}, catalog, {unit: "per_token"}), + Object.assign({}, catalog, {updatedAt: "2026-10-01"}), + Object.assign({}, catalog, {models: {}}), + Object.assign({}, catalog, {models: {bad: {standard: {input: -1, output: 3}}}})]; + const rejected = invalid.map(candidate => !pricing.update(candidate)); + // The estimator owns a validated copy, not a reference to the caller's mutable data. + catalog.models["fixture-model"].standard.input = 999; + return {accepted, estimates, rejected, retained: pricing.estimate("fixture-model", tokens)}; + } const emitted = []; let ready = true; const usage = new CliUsage({emit: record => emitted.push(record), ready: () => ready, @@ -142,6 +206,53 @@ async function exerciseUsage(scenario) { usage.closeAll(); return {emitted}; } + if (scenario === "usage-pricing-tiers" || scenario === "usage-pricing-chart") { + const {token} = usage.open("codex-session", "codex"); + const chart = scenario === "usage-pricing-chart"; + const tiers = chart ? ["priority", "ultrafast", "default"] : + ["priority", "ultrafast", "default", undefined, "fast"]; + const records = tiers.map((tier, index) => { + const values = {model: "gpt-6-astra"}; + if (tier !== undefined) { values.service_tier = tier; } + if (chart) { + Object.assign(values, {input_token_count: 100000, cached_token_count: 30000, + output_token_count: 40000}); + } + return pricedCompletion(String(1791142980000 + index) + "000000", values); + }); + usage.ingest(token, otlpLogs(records)); + usage.ingest(token, otlpLogs(records)); + usage.closeAll(); + return {emitted}; + } + if (scenario === "usage-pricing-tier-long") { + const {token} = usage.open("codex-session", "codex"); + const records = []; + for (const service_tier of ["fast", "ultrafast"]) { + for (const input_token_count of [272000, 272001]) { + records.push(pricedCompletion(String(1791142980000 + records.length) + "000000", + {model: "gpt-6.1-sol", service_tier, input_token_count, + cached_token_count: 71000, cache_write_token_count: 1000, output_token_count: 1000})); + } + } + usage.ingest(token, otlpLogs(records)); + usage.closeAll(); + return {emitted}; + } + if (scenario === "usage-pricing-unknown-tier") { + const {token} = usage.open("codex-session", "codex"); + const cases = [ + {model: "gpt-6-astra", service_tier: "future-tier"}, + {model: "gpt-6-astra", service_tier: ""}, + {model: "gpt-6-astra", service_tier: 123}, + {model: "gpt-6-luna", service_tier: "ultrafast"}, + {model: "gpt-future", service_tier: "fast"} + ]; + usage.ingest(token, otlpLogs(cases.map((values, index) => + pricedCompletion(String(1791142980000 + index) + "000000", values)))); + usage.closeAll(); + return {emitted}; + } if (scenario === "usage-pricing-unknown") { const {token} = usage.open("codex-session", "codex"); usage.ingest(token, otlpLogs(["gpt-future", "gpt-6-sol-new", "gpt-5.3-codex", " ", "x".repeat(201)] @@ -256,7 +367,9 @@ async function exerciseUsageLaunch() { /** Run one fixture with independent files, sockets and sessions, cleaning all owned resources. */ exports.exercise = async function ({scenario}) { if (["usage-claude", "usage-codex", "usage-malformed", "usage-backlog", "usage-http", "usage-pricing", - "usage-pricing-long", "usage-pricing-unknown"].includes(scenario)) { + "usage-pricing-long", "usage-pricing-unknown", "usage-pricing-tiers", "usage-pricing-tier-long", + "usage-pricing-unknown-tier", "usage-pricing-catalog", "usage-pricing-chart", + "usage-pricing-lazy", "usage-pricing-load-error", "usage-pricing-load-timeout"].includes(scenario)) { return exerciseUsage(scenario); } if (scenario === "usage-launch") { diff --git a/test/spec/AICliConnector-test.js b/test/spec/AICliConnector-test.js index 46c4bf8654..c6129ab7b6 100644 --- a/test/spec/AICliConnector-test.js +++ b/test/spec/AICliConnector-test.js @@ -312,6 +312,54 @@ define(function (require, exports, module) { expect(result.emitted.map(record => record.model)) .toEqual(["gpt-future", "gpt-6-sol-new", "gpt-5.3-codex", null, null]); }); + it("loads prices once on first Codex tokens, without loading for Claude, turns or warmups", async function () { + const result = await run("usage-pricing-lazy"); + expect(result.before).toEqual({loads: 0, emitted: 3}); + expect(result.waiting).toEqual({loads: 1, emitted: 3}); + expect(result.loads).toBe(1); + expect(result.costs).toEqual([0.00273, 0.00546, 0.01638]); + }); + it("keeps pricing and deduplicating Codex usage when loading cached prices fails", async function () { + const result = await run("usage-pricing-load-error"); + expect(result.loads).toBe(2); + expect(result.costs).toEqual([0.00273, 0.00546, 0.01638]); + }); + it("bounds a stalled cache request and retries preparation on later Codex usage", async function () { + const result = await run("usage-pricing-load-timeout"); + expect(result.loads).toBe(2); + expect(result.costs).toEqual([0.00273, 0.00546, 0.01638]); + }); + it("prices each reported tier independently across a session switch and deduplicates repeated exports", async function () { + const result = await run("usage-pricing-tiers"); + expect(result.emitted.length).toBe(5); + expect(result.emitted.map(record => record.serviceTier)) + .toEqual(["fast", "ultrafast", "default", "default", "fast"]); + expect(result.emitted.every(record => record.model === "gpt-6-astra" && record.turns === 0)).toBeTrue(); + const expected = [0.00546, 0.01638, 0.00273, 0.00273, 0.00546]; + result.emitted.forEach((record, index) => expect(record.costUSD).toBeCloseTo(expected[index], 9)); + }); + it("applies each tier's long-context rates to input, cache reads, cache writes and output", async function () { + const result = await run("usage-pricing-tier-long"); + expect(result.emitted.length).toBe(4); + const expected = [0.8392, 1.668408, 2.5176, 5.005224]; + result.emitted.forEach((record, index) => expect(record.costUSD).toBeCloseTo(expected[index], 9)); + }); + it("leaves explicit unknown tiers and unpublished model-tier combinations unpriced", async function () { + const result = await run("usage-pricing-unknown-tier"); + expect(result.emitted.map(record => record.serviceTier)) + .toEqual(["unknown", "unknown", "unknown", "ultrafast", "fast"]); + expect(result.emitted.map(record => record.costUSD)).toEqual([null, null, null, null, null]); + }); + it("applies a validated catalog atomically and retains it after malformed, older or mutated updates", async function () { + const result = await run("usage-pricing-catalog"); + expect(result.accepted).toBeTrue(); + expect(result.estimates[0]).toBeCloseTo(0.00017, 9); + expect(result.estimates[1]).toBeCloseTo(0.00059, 9); + expect(result.estimates[2]).toBeCloseTo(0.00171, 9); + expect(result.estimates[3]).toBeNull(); + expect(result.rejected).toEqual([true, true, true, true, true, true]); + expect(result.retained).toBeCloseTo(0.00017, 9); + }); it("skips misshapen OTLP exports and out-of-range times without throwing", async function () { const result = await run("usage-malformed"); expect(result.thrown).toEqual([]); diff --git a/tracking-repos.json b/tracking-repos.json index 092d997d18..bb0824a475 100644 --- a/tracking-repos.json +++ b/tracking-repos.json @@ -1,5 +1,5 @@ { "phoenixPro": { - "commitID": "9a0063ead251352f5f1ebffdb71ba90389910ae5" + "commitID": "3f6435595cc9cf0aafd2790fcbe36df3f86a9516" } } From f38f24e96e96c27f6b137f98ed50780b89fedde8 Mon Sep 17 00:00:00 2001 From: abose Date: Fri, 9 Oct 2026 21:42:30 +0530 Subject: [PATCH 6/8] feat(ai): support reusable live preview choices Extend askInLivePreview with structured choices, reusable CSS and script files, per-choice parameters and explicit reload-on-cleanup behavior. Document scoped page hooks, interactive pin-and-choose controls and the shared preview lifecycle. Update model guidance and localized question status messages for the matching Pro implementation. Fix the permission-test fixture's missing pricing dependency and give the Markdown splash test its own document instead of depending on the file left by lightbox cleanup. Advance the Pro pin to include the question renderer, shared floating controls, usage rollover and tooltip fixes. Validation: recorded Linux/macOS question suites pass 34/32/6/42 cases; Linux guidance suites pass 10/13/17. Permission cases pass 35 on macOS and Windows, with full unit runs of 3115 on Linux and 3112 on Windows. The Chromium Live Preview CI command passes 334 cases at the earlier question suite snapshot. Latest Windows/browser question coverage remains pending. Changed-file lint, whitespace checks and Pro pin consistency pass. --- src-node/ai-editor-tool-specs.js | 88 +++++++++++++++-------- src-node/ai-system-prompt.js | 11 +-- src-node/test/test-ai-file-permissions.js | 1 + src/nls/root/strings.js | 3 + test/spec/md-editor-integ-test.js | 12 ++++ tracking-repos.json | 2 +- 6 files changed, 82 insertions(+), 35 deletions(-) diff --git a/src-node/ai-editor-tool-specs.js b/src-node/ai-editor-tool-specs.js index 6e1d2c3ed0..a2eba6a729 100644 --- a/src-node/ai-editor-tool-specs.js +++ b/src-node/ai-editor-tool-specs.js @@ -727,39 +727,69 @@ function getEditorToolSpecs(peerCall, options = {}) { addTool( "askInLivePreview", - "Show a question card over the page in the live preview and wait for the user's answer. Use it whenever " + - "showing beats telling: a choice about one element or about the whole page, or presenting variants or " + - "a mockup for a reaction. The card's frame is Phoenix's: a title bar reading 'Phoenix AI asks' with " + - "minimize and close, a text field with Send for an answer in the user's own words, drag, resize and " + - "placement. You write only the body: uiFile, an HTML fragment with its own