diff --git a/src-node/ai-cli-connector.js b/src-node/ai-cli-connector.js index 22e5e84079..2576b33212 100644 --- a/src-node/ai-cli-connector.js +++ b/src-node/ai-cli-connector.js @@ -50,6 +50,7 @@ class CliConnector { emit: record => { if (this.options.emitUsage) { this.options.emitUsage(record); } }, ready: () => !this.options.ready || this.options.ready(), baseUrl: () => "http://localhost:" + this.server.address().port, + preparePricing: this.options.preparePricing, drainMs: this.options.usageDrainMs }); this.upgrade = (request, socket, head) => { @@ -412,6 +413,7 @@ exports.setBrowserConnector = function (connector, ready) { browserConnector = c /** Attach exactly once to the window's existing HTTP server. */ exports.attach = function (server) { controller = new CliConnector(server, {peer: (fn, args) => browserConnector.execPeer(fn, args), + preparePricing: () => browserConnector.execPeer("prepareExternalModelPricing"), emitUsage: record => browserConnector.triggerPeer("aiCliUsage", record), ready: () => browserReady(), emit: state => browserConnector.triggerPeer("aiCliConnectorState", state)}); }; diff --git a/src-node/ai-cli-pricing.js b/src-node/ai-cli-pricing.js index ba713dfeec..fddbfc1f6f 100644 --- a/src-node/ai-cli-pricing.js +++ b/src-node/ai-cli-pricing.js @@ -1,41 +1,95 @@ /* Copyright (c) 2026 core.ai; SPDX-License-Identifier: AGPL-3.0-or-later */ -// Standard API USD per million tokens, checked 2026-10-05. These are API-equivalent -// estimates, not subscription charges, and exclude service-tier and regional premiums. +const {z} = require("zod"); +const bundledPricing = require("./external-model-pricing.json"); + +// API-equivalent USD per million tokens, not subscription charges or regional premiums. +// Bundled prices checked 2026-10-09: // https://developers.openai.com/api/docs/pricing // https://developers.openai.com/api/docs/models/gpt-6-sol -// https://developers.openai.com/api/docs/models/gpt-5.6-sol (also the gpt-5.6 alias) -// https://developers.openai.com/api/docs/models/gpt-5.6-terra -// https://developers.openai.com/api/docs/models/gpt-5.6-luna -// https://developers.openai.com/api/docs/models/gpt-5.3-codex -// Fields: uncached input, cached input, cache write, output. null means unpublished. -const RATES = new Map([ - ["gpt-6-astra", [10, 1, 12.5, 50]], - ["gpt-6.1-sol", [2, 0.10, 2.5, 10]], - ["gpt-6-sol", [2, 0.20, 2.5, 10]], - ["gpt-6-luna", [0.10, 0.01, 0.125, 0.50]], - ["gpt-5.6-sol", [4, 0.40, 5, 20]], - ["gpt-5.6", [4, 0.40, 5, 20]], - ["gpt-5.6-terra", [2, 0.20, 2.5, 12]], - ["gpt-5.6-luna", [0.20, 0.02, 0.25, 1.20]], - ["gpt-5.3-codex", [1.75, 0.175, null, 14]] -]); +// https://developers.openai.com/api/docs/guides/fast-mode +// Unknown model/tier combinations stay unpriced; no universal premium is assumed. +const MAX_CATALOG_BYTES = 256 * 1024; +const price = z.number().finite().min(0).max(1000000).nullable(); +const ratesSchema = z.object({input: price, cacheRead: price, cacheWrite: price, output: price}).strict(); +const tiers = {standard: ratesSchema, fast: ratesSchema.optional(), ultrafast: ratesSchema.optional()}; +const modelSchema = z.object(Object.assign({}, tiers, { + longContext: z.object(Object.assign({aboveInputTokens: z.number().int().positive().max(100000000)}, tiers)) + .strict().optional() +})).strict(); +const catalogSchema = z.object({ + version: z.literal(1), + updatedAt: z.string().regex(/^\d{4}-\d{2}-\d{2}$/), + currency: z.literal("USD"), + unit: z.literal("million_tokens"), + models: z.record(z.string().min(1).max(200).regex(/^[a-zA-Z0-9][a-zA-Z0-9._:/@\[\]-]*$/), modelSchema) + .refine(models => Object.keys(models).length > 0 && Object.keys(models).length <= 500) +}).strict(); +const TOKEN_KINDS = ["input", "cacheRead", "cacheWrite", "output"]; + +/** + * Validate and copy a downloaded or cached price catalog before making any of it active. + * @param {*} catalog Candidate version-1 USD catalog. + * @return {?Object} Validated catalog, or null without modifying the active prices. + */ +function validateCatalog(catalog) { + try { + if (JSON.stringify(catalog).length > MAX_CATALOG_BYTES) { return null; } + const result = catalogSchema.safeParse(catalog); + return result.success ? result.data : null; + } catch (error) { + return null; + } +} + +/** + * Create an isolated estimator with bundled prices and atomic, validated catalog updates. + * @return {{update: function(Object): boolean, estimate: function(?string, Object, string=): ?number}} + */ +function createPricing() { + let catalog = catalogSchema.parse(bundledPricing); + return { + update(candidate) { + const next = validateCatalog(candidate); + if (!next || next.updatedAt < catalog.updatedAt) { return false; } + catalog = next; + return true; + }, + estimate(model, usage, serviceTier = "default") { + if (!Object.hasOwn(catalog.models, model)) { return null; } + const entry = catalog.models[model]; + const tier = serviceTier === "default" ? "standard" : serviceTier; + if (!["standard", "fast", "ultrafast"].includes(tier)) { return null; } + const inclusiveInput = usage.input + usage.cacheRead + usage.cacheWrite; + const table = entry.longContext && inclusiveInput > entry.longContext.aboveInputTokens ? + entry.longContext : entry; + const rates = table[tier]; + if (!rates) { return null; } + let total = 0; + for (const kind of TOKEN_KINDS) { + if (usage[kind] > 0 && rates[kind] === null) { return null; } + total += usage[kind] * (rates[kind] || 0); + } + return total / 1000000; + } + }; +} + +const pricing = createPricing(); /** - * Estimate one Codex response using its exact reported model and disjoint token counts. - * Output already includes reasoning. Unknown models/rates stay unpriced, never guessed. + * Estimate one Codex response using its exact model, normalized tier and disjoint token counts. + * Output already includes reasoning. Missing tier metadata retains the standard estimate. * @param {?string} model Exported model ID. * @param {{input: number, output: number, cacheRead: number, cacheWrite: number}} usage - * @return {?number} Standard API-equivalent USD, or null if a rate is unavailable. + * @param {string} [serviceTier="default"] Normalized per-response tier. + * @return {?number} API-equivalent USD, or null if a model/tier rate is unavailable. */ -function estimateCodexCost(model, usage) { - const rates = RATES.get(model); - if (!rates || (usage.cacheWrite > 0 && rates[2] === null)) { return null; } - // These models charge long-context rates for the entire request beyond 272k input. - const longContext = model !== "gpt-5.3-codex" && - usage.input + usage.cacheRead + usage.cacheWrite > 272000; - const inputCost = usage.input * rates[0] + usage.cacheRead * rates[1] + usage.cacheWrite * (rates[2] || 0); - return (inputCost * (longContext ? 2 : 1) + usage.output * rates[3] * (longContext ? 1.5 : 1)) / 1000000; +function estimateCodexCost(model, usage, serviceTier = "default") { + return pricing.estimate(model, usage, serviceTier); } exports.estimateCodexCost = estimateCodexCost; +exports.updatePricing = pricing.update; +exports.validateCatalog = validateCatalog; +exports.createPricing = createPricing; diff --git a/src-node/ai-cli-usage.js b/src-node/ai-cli-usage.js index cf50ad5fc3..27526840b5 100644 --- a/src-node/ai-cli-usage.js +++ b/src-node/ai-cli-usage.js @@ -31,6 +31,7 @@ const SEEN_LIMIT = 5000; const BACKLOG_LIMIT = 2000; const BACKLOG_RETRY_MS = 2000; const EXPORT_INTERVAL_MS = 2000; +const PRICING_WAIT_MS = 5000; // The largest time a JS Date can hold. const MAX_TIME_MS = 8.64e15; @@ -89,6 +90,18 @@ function fingerprint(parts) { return createHash("sha256").update(JSON.stringify(parts)).digest("hex").slice(0, 32); } +/** + * Normalize Codex's per-response tier without consulting a session's mutable configuration. + * Codex 0.162 omits this field for Standard; older exporters also retain the standard estimate. + * @param {*} value Exported service_tier attribute. + * @return {string} A known pricing tier, or "unknown" for an unsupported explicit value. + */ +function codexServiceTier(value) { + if (value === undefined || value === null || value === "default") { return "default"; } + if (value === "priority" || value === "fast") { return "fast"; } + return value === "ultrafast" ? "ultrafast" : "unknown"; +} + /** * Claude Code's per-request event. Its input excludes cache reads and writes, so the four * kinds are already disjoint; the cost is the CLI's own estimate at API list price. @@ -132,6 +145,7 @@ function codexUsage(record, attrs) { // reads. Every probe so far reported 0 cache writes, so a nonzero value is unverified. const usage = { model: modelId(attrs.model), + serviceTier: codexServiceTier(attrs.service_tier), input: Math.max(0, count(attrs.input_token_count) - cacheRead - cacheWrite), output: count(attrs.output_token_count), cacheRead: cacheRead, @@ -140,12 +154,11 @@ function codexUsage(record, attrs) { at: recordTime(record, attrs), promptId: null }; - usage.costUSD = estimateCodexCost(usage.model, usage); // No request id: the nanosecond stamp tells two identical responses apart, and a retried // export repeats it, so the retry is dropped. usage.key = fingerprint([attrs["conversation.id"], attrs["event.timestamp"], record.timeUnixNano, record.observedTimeUnixNano, attrs.input_token_count, attrs.output_token_count, attrs.cached_token_count, - attrs.cache_write_token_count, attrs.reasoning_token_count, usage.model]); + attrs.cache_write_token_count, attrs.reasoning_token_count, usage.model, usage.serviceTier]); return usage; } @@ -207,7 +220,8 @@ function userOwnsTelemetry(cli, where = {}) { class CliUsage { /** * @param {{emit: function(Object), baseUrl: function(): string, ready?: function(): boolean, - * drainMs?: number}} options - ready says whether the editor can take records now + * drainMs?: number, preparePricing?: function(): Promise, pricingWaitMs?: number}} options + * ready says whether the editor can take records now; preparePricing lazily loads cached prices. */ constructor(options) { this.options = options; @@ -216,6 +230,8 @@ class CliUsage { this.bySession = new Map(); this.backlog = []; this.retryTimer = null; + this.pricingReady = null; + this.closed = false; } /** Deliver a record now, or hold it until the editor is back. */ @@ -279,6 +295,7 @@ class CliUsage { /** Forget every session at once (PhNode exit). */ closeAll() { + this.closed = true; clearTimeout(this.retryTimer); this.retryTimer = null; for (const entry of this.byToken.values()) { clearTimeout(entry.closeTimer); } @@ -293,8 +310,32 @@ class CliUsage { return true; } + /** + * Bound first-use preparation so a reloading browser cannot hold usage indefinitely. + * Failed preparation is retried by a later completion; this batch uses current prices. + * @return {Promise} Resolves when prices are ready or the bounded attempt failed. + */ + _preparePricing() { + if (!this.pricingReady) { + let timer; + const waitMs = this.options.pricingWaitMs === undefined ? PRICING_WAIT_MS : this.options.pricingWaitMs; + const work = Promise.race([ + Promise.resolve().then(() => this.options.preparePricing()), + new Promise((_resolve, reject) => { + timer = setTimeout(() => reject(new Error("Pricing cache preparation timed out")), waitMs); + timer.unref(); + }) + ]); + const ready = work.catch(() => { + if (this.pricingReady === ready) { this.pricingReady = null; } + }).finally(() => clearTimeout(timer)); + this.pricingReady = ready; + } + return this.pricingReady; + } + _emit(entry, usage, turns) { - this._deliver({ + const record = { sessionId: entry.sessionId, cli: entry.cli, model: usage.model || null, @@ -306,7 +347,23 @@ class CliUsage { cacheWrite: usage.cacheWrite, costUSD: usage.costUSD, turns: turns - }); + }; + if (entry.cli !== "codex" || !usage.serviceTier) { + this._deliver(record); + return; + } + record.serviceTier = usage.serviceTier; + const deliverPriced = () => { + if (this.closed) { return; } + record.costUSD = estimateCodexCost(usage.model, usage, usage.serviceTier); + this._deliver(record); + }; + if (this.options.preparePricing && usage.input + usage.output + usage.cacheRead + usage.cacheWrite > 0) { + // Claude and turn-only events never load prices or wait on this preparation. + this._preparePricing().then(deliverPriced); + } else { + deliverPriced(); + } } /** diff --git a/src-node/ai-editor-tool-specs.js b/src-node/ai-editor-tool-specs.js index 44a7feab6c..a2eba6a729 100644 --- a/src-node/ai-editor-tool-specs.js +++ b/src-node/ai-editor-tool-specs.js @@ -608,6 +608,9 @@ function getEditorToolSpecs(peerCall, options = {}) { "description, allowedValues if any, and the resolved scope of the current value).\n" + "- get: Same fields for a single preference id.\n" + "- set: Write a value into a specific scope. Calls PreferencesManager.save() after.\n\n" + + "This is the running editor's preference registry, not its documentation. If the user " + + "asks you to look something up in the Phoenix docs, use editorDocs and read the relevant " + + "documentation before answering; this tool can supplement it with current values.\n\n" + "Scope hierarchy (highest precedence wins on read): session → project → user → default.\n" + "- default: built-in fallback declared by definePreference in source. READ-ONLY.\n" + "- user: the user's global settings (persisted across all projects). User-friendly name " + @@ -724,39 +727,69 @@ function getEditorToolSpecs(peerCall, options = {}) { addTool( "askInLivePreview", - "Show a question card over the page in the live preview and wait for the user's answer. Use it whenever " + - "showing beats telling: a choice about one element or about the whole page, or presenting variants or " + - "a mockup for a reaction. The card's frame is Phoenix's: a title bar reading 'Phoenix AI asks' with " + - "minimize and close, a text field with Send for an answer in the user's own words, drag, resize and " + - "placement. You write only the body: uiFile, an HTML fragment with its own