From e8b5e2cdc99472f9ba2efe34522d49800c79add2 Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Wed, 2 Sep 2026 15:36:57 +0530 Subject: [PATCH 1/9] feat(models): replace Axon catalog with seven OSS models Replace the Axon context-window model variants with an OSS catalog served through the MatterAI gateway: meta/muse-spark-1.2-contributor, deepseek/deepseek-v4-flash-0731, zai/glm-5.3, zai/glm-5.3-flash, gpt-5.6-sol, gpt-5.6-luna, and gemini-3.7-flash. - Extension catalog (kilocode-models.ts) and the webview copy (useOpenRouterModelProviders) now share a single OSS_MODEL_BASE variant with per-model published pricing (USD per token, OpenRouter format) and 1:1 API model ids; getKilocodeApiModelId passes unknown ids through. - Default model is deepseek/deepseek-v4-flash-0731: openRouterDefaultModelId in @roo-code/types, the CLI kilocodeModel defaults (config defaults, provider settings, browser auth), and web-evals MODEL_DEFAULT. - Model selector drops the 232k/400k context toggle and its sort-priority helper; stored Axon selections reset to the default on next launch via the existing isValidKilocodeModel stale-model check. - Update kilocode-models, kilocode-openrouter, and CLI provider-merge tests for the new catalog and default. --- apps/web-evals/src/lib/schemas.ts | 2 +- .../persistence-provider-merge.test.ts | 2 +- cli/src/config/defaults.ts | 2 +- cli/src/constants/providers/models.ts | 2 +- cli/src/constants/providers/settings.ts | 4 +- cli/src/utils/browserAuth.ts | 2 +- packages/types/src/providers/openrouter.ts | 17 +- .../__tests__/kilocode-models.spec.ts | 129 ++++--- .../__tests__/kilocode-openrouter.spec.ts | 34 +- src/api/providers/kilocode-models.ts | 314 +++++++----------- .../kilocode/chat/ModelSelector.tsx | 83 ----- .../ui/hooks/kilocode/usePreferredModels.ts | 2 +- .../ui/hooks/useOpenRouterModelProviders.ts | 300 ++++++----------- 13 files changed, 347 insertions(+), 546 deletions(-) diff --git a/apps/web-evals/src/lib/schemas.ts b/apps/web-evals/src/lib/schemas.ts index 1a86346af8..25faa9547f 100644 --- a/apps/web-evals/src/lib/schemas.ts +++ b/apps/web-evals/src/lib/schemas.ts @@ -6,7 +6,7 @@ import { rooCodeSettingsSchema } from "@roo-code/types" * CreateRun */ -export const MODEL_DEFAULT = "axon-auto-232k" +export const MODEL_DEFAULT = "deepseek/deepseek-v4-flash-0731" export const CONCURRENCY_MIN = 1 export const CONCURRENCY_MAX = 25 diff --git a/cli/src/config/__tests__/persistence-provider-merge.test.ts b/cli/src/config/__tests__/persistence-provider-merge.test.ts index 008bc7aedf..a422caffb7 100644 --- a/cli/src/config/__tests__/persistence-provider-merge.test.ts +++ b/cli/src/config/__tests__/persistence-provider-merge.test.ts @@ -51,7 +51,7 @@ describe("Provider Merging", () => { expect(result.config.providers[0].id).toBe("default") expect(result.config.providers[0]).toHaveProperty("kilocodeToken") expect(result.config.providers[0]).toHaveProperty("kilocodeModel") - expect(result.config.providers[0].kilocodeModel).toBe("axon-auto-232k") + expect(result.config.providers[0].kilocodeModel).toBe("deepseek/deepseek-v4-flash-0731") expect(result.validation.valid).toBe(true) }) diff --git a/cli/src/config/defaults.ts b/cli/src/config/defaults.ts index 922ce4853e..8c6981ccbc 100644 --- a/cli/src/config/defaults.ts +++ b/cli/src/config/defaults.ts @@ -55,7 +55,7 @@ export const DEFAULT_CONFIG = { id: "default", provider: "kilocode", kilocodeToken: "", - kilocodeModel: "axon-auto-232k", + kilocodeModel: "deepseek/deepseek-v4-flash-0731", }, ], autoApproval: DEFAULT_AUTO_APPROVAL, diff --git a/cli/src/constants/providers/models.ts b/cli/src/constants/providers/models.ts index ca1dec4770..b739de9c12 100644 --- a/cli/src/constants/providers/models.ts +++ b/cli/src/constants/providers/models.ts @@ -475,7 +475,7 @@ export function sortModelsByPreference(models: ModelRecord): string[] { } // Sort rest alphabetically - restModelIds.sort((a, b) => a.localeCompare(b)) + // restModelIds.sort((a, b) => a.localeCompare(b)) return [...preferredModelIds, ...restModelIds] } diff --git a/cli/src/constants/providers/settings.ts b/cli/src/constants/providers/settings.ts index 9871c5badc..45cb72fbd2 100644 --- a/cli/src/constants/providers/settings.ts +++ b/cli/src/constants/providers/settings.ts @@ -551,7 +551,7 @@ export const getProviderSettings = (provider: ProviderName, config: ProviderSett return [ createFieldConfig("kilocodeToken", config), createFieldConfig("kilocodeOrganizationId", config, "personal"), - createFieldConfig("kilocodeModel", config, "axon-auto-232k"), + createFieldConfig("kilocodeModel", config, "deepseek/deepseek-v4-flash-0731"), ] default: @@ -563,7 +563,7 @@ export const getProviderSettings = (provider: ProviderName, config: ProviderSett * Provider-specific default models */ export const PROVIDER_DEFAULT_MODELS: Record = { - kilocode: "axon-auto-232k", + kilocode: "deepseek/deepseek-v4-flash-0731", anthropic: "claude-3-5-sonnet-20241022", "openai-native": "gpt-4o", openrouter: "anthropic/claude-3-5-sonnet", diff --git a/cli/src/utils/browserAuth.ts b/cli/src/utils/browserAuth.ts index 2916b510d0..c24714fa8f 100644 --- a/cli/src/utils/browserAuth.ts +++ b/cli/src/utils/browserAuth.ts @@ -203,7 +203,7 @@ export async function performBrowserAuth(source: string = "axon-code-cli"): Prom id: "default", provider: "kilocode", kilocodeToken: token, - kilocodeModel: "axon-auto-232k", + kilocodeModel: "deepseek/deepseek-v4-flash-0731", }, ], } diff --git a/packages/types/src/providers/openrouter.ts b/packages/types/src/providers/openrouter.ts index 53c047e6c8..283f0ff4c2 100644 --- a/packages/types/src/providers/openrouter.ts +++ b/packages/types/src/providers/openrouter.ts @@ -1,20 +1,19 @@ import type { ModelInfo } from "../model.js" -// https://openrouter.ai/models?order=newest&supported_parameters=tools -export const openRouterDefaultModelId = "axon-auto-232k" +// MatterAI OSS models served through the MatterAI gateway +export const openRouterDefaultModelId = "deepseek/deepseek-v4-flash-0731" export const openRouterDefaultModelInfo: ModelInfo = { maxTokens: 64000, contextWindow: 232_000, supportsImages: true, supportsComputerUse: false, - supportsPromptCache: false, - inputPrice: 1.0, - outputPrice: 4.0, - cacheWritesPrice: 0.0, - cacheReadsPrice: 0.0, - description: - "Axon Auto starts with Eido 3 Code Flash and dynamically selects Flash, Mini, or Pro as the task evolves. Pricing is dynamic and follows the model used for each request.", + supportsPromptCache: true, + inputPrice: 0.00000015, + outputPrice: 0.0000005, + cacheWritesPrice: 0, + cacheReadsPrice: 0.00000003, + description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", } export const OPENROUTER_DEFAULT_PROVIDER_NAME = "[default]" diff --git a/src/api/providers/__tests__/kilocode-models.spec.ts b/src/api/providers/__tests__/kilocode-models.spec.ts index 1a55ea2219..dfaec86e25 100644 --- a/src/api/providers/__tests__/kilocode-models.spec.ts +++ b/src/api/providers/__tests__/kilocode-models.spec.ts @@ -1,48 +1,97 @@ import { getKilocodeApiModelId, KILO_CODE_MODELS, type KiloCodeModel } from "../kilocode-models" -const getSharedMetadata = ({ name: _name, context_length: _contextLength, ...metadata }: KiloCodeModel) => metadata - -describe("KiloCode Axon context variants", () => { - it.each([ - ["auto", "axon-auto-232k", "axon-auto-400k", 232000], - ["flash", "axon-eido-3.2-flash", "axon-eido-3.2-flash-400k", 232000], - ["eido-3.2", "axon-eido-3.2-232k", "axon-eido-3.2-400k", 232000], - ["pro", "axon-eido-3.2-code-pro-232k", "axon-eido-3.2-code-pro-400k", 232000], - ["lumen", "axon-lumen-4-code-232k", "axon-lumen-4-code-400k", 232000], - ])( - "provides lower-context and 400k %s variants with identical model metadata", - (_tier, modelLowerId, model400kId, lowerContextLength) => { - const modelLower = KILO_CODE_MODELS[modelLowerId] - const model400k = KILO_CODE_MODELS[model400kId] - - expect(modelLower).toBeDefined() - expect(model400k).toBeDefined() - expect(modelLower?.context_length).toBe(lowerContextLength) - expect(model400k?.context_length).toBe(400000) - expect(getSharedMetadata(modelLower!)).toEqual(getSharedMetadata(model400k!)) - }, - ) - - it.each([ - ["axon-auto-232k", "axon-auto"], - ["axon-auto-400k", "axon-auto"], - ["axon-eido-3.2-flash", "axon-eido-3.2-flash"], - ["axon-eido-3.2-flash-400k", "axon-eido-3.2-flash"], - ["axon-eido-3.2-232k", "axon-eido-3.2"], - ["axon-eido-3.2-400k", "axon-eido-3.2"], - ["axon-eido-3.2-code-pro-232k", "axon-eido-3.2-code-pro"], - ["axon-eido-3.2-code-pro-400k", "axon-eido-3.2-code-pro"], - ["axon-lumen-4-code-232k", "axon-lumen-4-code"], - ["axon-lumen-4-code-400k", "axon-lumen-4-code"], - ])("sends %s to its upstream model %s", (selectedId, apiModelId) => { - expect(getKilocodeApiModelId(selectedId)).toBe(apiModelId) +const OSS_MODEL_IDS = [ + "meta/muse-spark-1.2-contributor", + "deepseek/deepseek-v4-flash-0731", + "zai/glm-5.3", + "zai/glm-5.3-flash", + "gpt-5.6-sol", + "gpt-5.6-luna", + "gemini-3.7-flash", +] + +const getSharedMetadata = ({ + id: _id, + name: _name, + description: _description, + context_length: _contextLength, + owned_by: _ownedBy, + openrouter: _openrouter, + pricing: _pricing, + ...metadata +}: KiloCodeModel) => metadata + +describe("KiloCode OSS model catalog", () => { + it("exposes exactly the seven OSS models", () => { + expect(Object.keys(KILO_CODE_MODELS).sort()).toEqual([...OSS_MODEL_IDS].sort()) + }) + + it("no longer exposes axon models", () => { + expect(KILO_CODE_MODELS["axon-auto-232k"]).toBeUndefined() + expect(KILO_CODE_MODELS["axon-eido-3.2-code-pro-400k"]).toBeUndefined() }) - it.each(["axon-auto-232k", "axon-auto-400k"])("marks %s as dynamically priced", (modelId) => { + it.each([...OSS_MODEL_IDS])("sends %s to the API unchanged", (modelId) => { const model = KILO_CODE_MODELS[modelId] - expect(model?.pricing).toMatchObject({ type: "dynamic", display: "dynamic pricing" }) - expect(model?.pricing.prompt).toBeUndefined() - expect(model?.pricing.completion).toBeUndefined() + expect(model).toBeDefined() + expect(model?.id).toBe(modelId) + expect(model?.context_length).toBe(232000) + expect(getKilocodeApiModelId(modelId)).toBe(modelId) + }) + + it("prices each OSS model at its published rate", () => { + expect(KILO_CODE_MODELS["meta/muse-spark-1.2-contributor"]?.pricing).toMatchObject({ + prompt: "0.0000001", + completion: "0.0000002", + input_cache_reads: "0.000000002", + }) + expect(KILO_CODE_MODELS["deepseek/deepseek-v4-flash-0731"]?.pricing).toMatchObject({ + prompt: "0.00000014", + completion: "0.00000028", + input_cache_reads: "0.000000028", + }) + expect(KILO_CODE_MODELS["zai/glm-5.3"]?.pricing).toMatchObject({ + prompt: "0.0000014", + completion: "0.0000044", + input_cache_reads: "0.00000014", + }) + expect(KILO_CODE_MODELS["zai/glm-5.3-flash"]?.pricing).toMatchObject({ + prompt: "0.00000015", + completion: "0.0000005", + input_cache_reads: "0.00000003", + }) + expect(KILO_CODE_MODELS["gpt-5.6-sol"]?.pricing).toMatchObject({ + prompt: "0.000005", + completion: "0.00003", + input_cache_reads: "0.0000005", + }) + expect(KILO_CODE_MODELS["gpt-5.6-luna"]?.pricing).toMatchObject({ + prompt: "0.0000002", + completion: "0.0000012", + input_cache_reads: "0.00000002", + }) + expect(KILO_CODE_MODELS["gemini-3.7-flash"]?.pricing).toMatchObject({ + prompt: "0.00000075", + completion: "0.00000375", + input_cache_reads: "0.000000075", + }) + }) + + it("keeps image, request, and cache-write pricing at zero", () => { + for (const model of Object.values(KILO_CODE_MODELS)) { + expect(model.pricing.image).toBe("0") + expect(model.pricing.request).toBe("0") + expect(model.pricing.input_cache_writes).toBe("0") + } + }) + + it("shares identical base metadata across all OSS models", () => { + const models = OSS_MODEL_IDS.map((modelId) => KILO_CODE_MODELS[modelId]!) + const [first, ...rest] = models.map(getSharedMetadata) + + for (const metadata of rest) { + expect(metadata).toEqual(first) + } }) }) diff --git a/src/api/providers/__tests__/kilocode-openrouter.spec.ts b/src/api/providers/__tests__/kilocode-openrouter.spec.ts index aa366413f3..42d580fc1c 100644 --- a/src/api/providers/__tests__/kilocode-openrouter.spec.ts +++ b/src/api/providers/__tests__/kilocode-openrouter.spec.ts @@ -34,23 +34,23 @@ vitest.mock("openai") vitest.mock("delay", () => ({ default: vitest.fn(() => Promise.resolve()) })) vitest.mock("../fetchers/modelCache", () => ({ getModels: vitest.fn().mockResolvedValue({ - "axon-eido-3.2-code-pro-232k": { + "zai/glm-5.3": { maxTokens: 64000, contextWindow: 232000, supportsImages: true, supportsPromptCache: false, - inputPrice: 3, - outputPrice: 9, - description: "Axon Eido 3.2 Pro", + inputPrice: 0, + outputPrice: 0, + description: "GLM 5.3", }, - "axon-eido-3.2-code-pro-400k": { + "zai/glm-5.3-flash": { maxTokens: 64000, - contextWindow: 400000, + contextWindow: 232000, supportsImages: true, supportsPromptCache: false, - inputPrice: 3, - outputPrice: 9, - description: "Axon Eido 3.2 Pro", + inputPrice: 0, + outputPrice: 0, + description: "GLM 5.3 Flash", }, "anthropic/claude-sonnet-4": { maxTokens: 8192, @@ -69,26 +69,26 @@ vitest.mock("../fetchers/modelEndpointCache", () => ({ getModelEndpoints: vitest.fn().mockResolvedValue({}), })) vitest.mock("../kilocode/getKilocodeDefaultModel", () => ({ - getKilocodeDefaultModel: vitest.fn().mockResolvedValue("axon-auto-232k"), + getKilocodeDefaultModel: vitest.fn().mockResolvedValue("zai/glm-5.3-flash"), })) describe("KilocodeOpenrouterHandler", () => { const mockOptions: ApiHandlerOptions = { kilocodeToken: "test-token", - kilocodeModel: "axon-auto-232k", + kilocodeModel: "zai/glm-5.3-flash", } beforeEach(() => vitest.clearAllMocks()) describe("customRequestOptions", () => { - it("reports the selected 400k context window", async () => { + it("reports the selected model context window", async () => { const handler = new KilocodeOpenrouterHandler({ kilocodeToken: "test-token", - kilocodeModel: "axon-eido-3.2-code-pro-400k", + kilocodeModel: "zai/glm-5.3", }) await handler.fetchModel() - expect(handler.customRequestOptions()?.headers[X_MODEL_CONTEXT_WINDOW]).toBe("400000") + expect(handler.customRequestOptions()?.headers[X_MODEL_CONTEXT_WINDOW]).toBe("232000") }) it("includes taskId header when provided in metadata", () => { @@ -261,10 +261,10 @@ describe("KilocodeOpenrouterHandler", () => { }) describe("createMessage", () => { - it("sends the upstream model ID for a context-window variant", async () => { + it("sends the selected OSS model ID to the API", async () => { const handler = new KilocodeOpenrouterHandler({ kilocodeToken: "test-token", - kilocodeModel: "axon-eido-3.2-code-pro-232k", + kilocodeModel: "zai/glm-5.3", }) const mockStream = { @@ -285,7 +285,7 @@ describe("KilocodeOpenrouterHandler", () => { ]) await generator.next() - expect(mockCreate).toHaveBeenCalledWith(expect.objectContaining({ model: "axon-eido-3.2-code-pro" }), { + expect(mockCreate).toHaveBeenCalledWith(expect.objectContaining({ model: "zai/glm-5.3" }), { headers: clientMetadataHeaders, }) }) diff --git a/src/api/providers/kilocode-models.ts b/src/api/providers/kilocode-models.ts index 6e2daba3e3..443e3a1bbc 100644 --- a/src/api/providers/kilocode-models.ts +++ b/src/api/providers/kilocode-models.ts @@ -28,11 +28,15 @@ export type KiloCodeModel = { } } -type KiloCodeModelVariant = Omit +type KiloCodeModelVariant = Omit< + KiloCodeModel, + "id" | "name" | "description" | "context_length" | "owned_by" | "openrouter" +> -const AXON_AUTO: KiloCodeModelVariant = { - description: - "Axon Auto starts with Eido 3.2 Flash and dynamically selects Flash, 3.2, or Pro as the task evolves. Pricing is dynamic and follows the model used for each request.", +// Shared metadata for the OSS models served through the MatterAI gateway. +// Pricing strings are USD per token (OpenRouter format); per-model rates are +// set on each entry below. +const OSS_MODEL_BASE: KiloCodeModelVariant = { input_modalities: ["text", "image"], max_output_length: 64000, output_modalities: ["text"], @@ -47,213 +51,127 @@ const AXON_AUTO: KiloCodeModelVariant = { "stop", ], supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - type: "dynamic", - display: "dynamic pricing", - image: "0", - request: "0", - input_cache_reads: "0", - input_cache_writes: "0", - }, -} - -const AXON_EIDO_3_2_CODE_PRO: KiloCodeModelVariant = { - description: - "Axon Eido 3.2 Pro is the frontier Orbital model for coding tasks, long running agents and general intelligence, fine-tuned on open source models.", - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - prompt: "0.000003", - completion: "0.000009", - image: "0", - request: "0", - input_cache_reads: "0", - input_cache_writes: "0", - }, -} - -const AXON_EIDO_3_2: KiloCodeModelVariant = { - description: - "Axon Eido 3.2 is a general purpose super intelligent LLM coding model for high-effort day-to-day tasks", - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - prompt: "0.000002", - completion: "0.000006", - image: "0", - request: "0", - input_cache_reads: "0.0000005", - input_cache_writes: "0", - }, -} - -const AXON_LUMEN_4_CODE: KiloCodeModelVariant = { - description: - "Axon Lumen 4 is the ultra-intelligent frontier model for complex agentic coding tasks and general intelligence.", - input_modalities: ["text", "image"], - max_output_length: 128000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - prompt: "0.000005", - completion: "0.000025", - image: "0", - request: "0", - input_cache_reads: "0", - input_cache_writes: "0", - }, -} - -const AXON_EIDO_3_2_FLASH: KiloCodeModelVariant = { - description: "Axon Eido 3.2 Flash is a fast and low cost general purpose model for low-effort day-to-day tasks", - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", + created: 1786032000, pricing: { - prompt: "0.0", - completion: "0.0", image: "0", request: "0", - input_cache_reads: "0", input_cache_writes: "0", }, } export const KILO_CODE_MODELS: Record = { - "axon-auto-232k": { - ...AXON_AUTO, - id: "axon-auto", - name: "Axon Auto (232K context)", + "meta/muse-spark-1.2-contributor": { + ...OSS_MODEL_BASE, + id: "meta/muse-spark-1.2-contributor", + name: "Muse Spark 1.2 Contributor", + description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.", context_length: 232000, - }, - "axon-auto-400k": { - ...AXON_AUTO, - id: "axon-auto", - name: "Axon Auto (400K context)", - context_length: 400000, - }, - "axon-eido-3.2-flash": { - ...AXON_EIDO_3_2_FLASH, - id: "axon-eido-3.2-flash", - name: "Axon Eido 3.2 Flash (232K context)", + owned_by: "meta", + openrouter: { slug: "meta/muse-spark-1.2-contributor" }, + // $0.10/M input, $0.002/M cache read, $0.20/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.0000001", + completion: "0.0000002", + input_cache_reads: "0.000000002", + }, + }, + "deepseek/deepseek-v4-flash-0731": { + ...OSS_MODEL_BASE, + id: "deepseek/deepseek-v4-flash-0731", + name: "DeepSeek V4 Flash", + description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", context_length: 232000, - }, - "axon-eido-3.2-flash-400k": { - ...AXON_EIDO_3_2_FLASH, - id: "axon-eido-3.2-flash", - name: "Axon Eido 3.2 Flash (400K context)", - context_length: 400000, - }, - "axon-eido-3.2-232k": { - ...AXON_EIDO_3_2, - id: "axon-eido-3.2", - name: "Axon Eido 3.2 (232K context)", + owned_by: "deepseek", + openrouter: { slug: "deepseek/deepseek-v4-flash-0731" }, + // $0.14/M input, $0.028/M cache read, $0.28/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.00000014", + completion: "0.00000028", + input_cache_reads: "0.000000028", + }, + }, + "zai/glm-5.3-flash": { + ...OSS_MODEL_BASE, + id: "zai/glm-5.3-flash", + name: "GLM 5.3 Flash", + description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", context_length: 232000, - }, - "axon-eido-3.2-400k": { - ...AXON_EIDO_3_2, - id: "axon-eido-3.2", - name: "Axon Eido 3.2 (400K context)", - context_length: 400000, - }, - "axon-eido-3.2-code-pro-232k": { - ...AXON_EIDO_3_2_CODE_PRO, - id: "axon-eido-3.2-code-pro", - name: "Axon Eido 3.2 Pro (232K context)", + owned_by: "zai", + openrouter: { slug: "zai/glm-5.3-flash" }, + // $0.15/M input, $0.03/M cache read, $0.50/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.00000015", + completion: "0.0000005", + input_cache_reads: "0.00000003", + }, + }, + "zai/glm-5.3": { + ...OSS_MODEL_BASE, + id: "zai/glm-5.3", + name: "GLM 5.3", + description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", context_length: 232000, - }, - "axon-eido-3.2-code-pro-400k": { - ...AXON_EIDO_3_2_CODE_PRO, - id: "axon-eido-3.2-code-pro", - name: "Axon Eido 3.2 Pro (400K context)", - context_length: 400000, - }, - "axon-lumen-4-code-232k": { - ...AXON_LUMEN_4_CODE, - id: "axon-lumen-4-code", - name: "Axon Lumen 4 (232K context)", + owned_by: "zai", + openrouter: { slug: "zai/glm-5.3" }, + // $1.40/M input, $0.14/M cache read, $4.40/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.0000014", + completion: "0.0000044", + input_cache_reads: "0.00000014", + }, + }, + "gpt-5.6-luna": { + ...OSS_MODEL_BASE, + id: "gpt-5.6-luna", + name: "GPT-5.6 Luna", + description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", context_length: 232000, - }, - "axon-lumen-4-code-400k": { - ...AXON_LUMEN_4_CODE, - id: "axon-lumen-4-code", - name: "Axon Lumen 4 (400K context)", - context_length: 400000, + owned_by: "openai", + openrouter: { slug: "gpt-5.6-luna" }, + // $0.20/M input, $0.02/M cache read, $1.20/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.0000002", + completion: "0.0000012", + input_cache_reads: "0.00000002", + }, + }, + "gpt-5.6-sol": { + ...OSS_MODEL_BASE, + id: "gpt-5.6-sol", + name: "GPT-5.6 Sol", + description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", + context_length: 232000, + owned_by: "openai", + openrouter: { slug: "gpt-5.6-sol" }, + // $5/M input, $0.50/M cache read, $30/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.000005", + completion: "0.00003", + input_cache_reads: "0.0000005", + }, + }, + "gemini-3.7-flash": { + ...OSS_MODEL_BASE, + id: "gemini-3.7-flash", + name: "Gemini 3.7 Flash", + description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", + context_length: 232000, + owned_by: "google", + openrouter: { slug: "gemini-3.7-flash" }, + // $0.75/M input, $0.075/M cache read, $3.75/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.00000075", + completion: "0.00000375", + input_cache_reads: "0.000000075", + }, }, } @@ -267,7 +185,7 @@ export function isValidKilocodeModel(modelId: string): boolean { /** * Resolves an extension model option to the model ID understood by the API. - * Context-window variants are local catalog choices and share an upstream model. + * OSS catalog entries map 1:1 to their API model ID; unknown ids pass through. */ export function getKilocodeApiModelId(modelId: string): string { return KILO_CODE_MODELS[modelId]?.id ?? modelId diff --git a/webview-ui/src/components/kilocode/chat/ModelSelector.tsx b/webview-ui/src/components/kilocode/chat/ModelSelector.tsx index aef5228a35..362e9b9cc9 100644 --- a/webview-ui/src/components/kilocode/chat/ModelSelector.tsx +++ b/webview-ui/src/components/kilocode/chat/ModelSelector.tsx @@ -56,15 +56,6 @@ const MODEL_QUALIFIER_PATTERN = /\s*(\((?:232k context|400k context|free)\))$/i const isStandardContextWindow = (cw?: number): boolean => cw === 232000 const isExtendedContextWindow = (cw?: number): boolean => cw === 400000 -// Sort priority for Axon models within a context window group: Auto, Flash, Eido 3.2, Pro, Lumen -const getModelSortPriority = (modelId: string): number => { - if (modelId.startsWith("axon-auto-")) return 0 - if (modelId.includes("flash")) return 1 - if (modelId.includes("pro")) return 4 - if (modelId.includes("lumen")) return 5 - return 3 -} - const ModelLabel = ({ label }: { label: string }) => { const match = label.match(MODEL_QUALIFIER_PATTERN) @@ -246,7 +237,6 @@ export const ModelSelector = ({ !modelId.startsWith("opencode:")) ) }) - .sort((a, b) => getModelSortPriority(a) - getModelSortPriority(b)) .map((modelId) => { const baseLabel = providerModels[modelId]?.displayName ?? prettyModelName(modelId) const label = removeContextSuffix(baseLabel) @@ -345,78 +335,6 @@ export const ModelSelector = ({ [currentTaskItem, provider, currentApiConfigName, apiConfiguration, modelIdKey], ) - const onContextToggle = useCallback( - (mode: "232k" | "400k") => { - if (mode === contextMode) return - setContextMode(mode) - - // If current model is an Axon model, switch the active selection to the matching context variant - const isThirdParty = - selectedModelId?.startsWith("ollama:") || - selectedModelId?.startsWith("opencode:") || - selectedModelId?.startsWith("matterai3p:") - - if (!isThirdParty && selectedModelId) { - const nextModelId = - mode === "400k" ? get400kAxonVariant(selectedModelId) : get232kAxonFallback(selectedModelId) - if ( - providerModels[nextModelId] && - (!isPlanRestrictedAxonModel(nextModelId) || canAccessAxonModel(nextModelId, profilePlan)) - ) { - selectAxonModel(nextModelId) - } - } - }, - [contextMode, selectedModelId, providerModels, profilePlan, selectAxonModel], - ) - - const renderContextToggleHeader = () => { - return ( -
- Context Window -
- - -
-
- ) - } - const onChange = (value: string) => { if (isPlanRestrictedAxonModel(value) && !canAccessAxonModel(value, profilePlan)) return @@ -714,7 +632,6 @@ export const ModelSelector = ({ const is400k = providerModels[option.value]?.contextWindow === 400000 || is400kAxonModel(option.value) return }} - headerComponent={renderContextToggleHeader()} onRefresh={handleRefreshModels} // Always show refresh since matterai3p is always enabled /> ) diff --git a/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts b/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts index 847625aff7..de73318679 100644 --- a/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts +++ b/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts @@ -26,7 +26,7 @@ export const usePreferredModels = (models: Record | null) => restModelIds.push(key) } } - restModelIds.sort((a, b) => a.localeCompare(b)) + // restModelIds.sort((a, b) => a.localeCompare(b)) return [...preferredModelIds, ...restModelIds] }, [models]) diff --git a/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts b/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts index de75b77557..9a399ed24a 100644 --- a/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts +++ b/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts @@ -40,142 +40,15 @@ type KiloCodeModel = { } } -type KiloCodeModelVariant = Omit - -const AXON_AUTO: KiloCodeModelVariant = { - description: - "Axon Auto starts with Eido 3.2 Flash and dynamically selects Flash, 3.2, or Pro as the task evolves. Pricing is dynamic and follows the model used for each request.", - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - type: "dynamic", - display: "dynamic pricing", - image: "0", - request: "0", - input_cache_reads: "0", - input_cache_writes: "0", - }, -} - -const AXON_EIDO_3_2_CODE_PRO: KiloCodeModelVariant = { - description: - "Axon Eido 3.2 Pro is the frontier model for coding tasks, long running agents and general intelligence, fine-tuned on open source models.", - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - prompt: "0.000003", - completion: "0.000009", - image: "0", - request: "0", - input_cache_reads: "0", - input_cache_writes: "0", - }, -} - -const AXON_EIDO_3_2: KiloCodeModelVariant = { - description: - "Axon Eido 3.2 is a general purpose super intelligent LLM coding model for high-effort day-to-day tasks", - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - prompt: "0.000002", - completion: "0.000006", - image: "0", - request: "0", - input_cache_reads: "0.0000005", - input_cache_writes: "0", - }, -} - -const AXON_LUMEN_4_CODE: KiloCodeModelVariant = { - description: - "Axon Lumen 4 is the ultra-intelligent frontier model for complex agentic coding tasks and general intelligence.", - input_modalities: ["text", "image"], - max_output_length: 128000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, - datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", - pricing: { - prompt: "0.000005", - completion: "0.000025", - image: "0", - request: "0", - input_cache_reads: "0", - input_cache_writes: "0", - }, -} +type KiloCodeModelVariant = Omit< + KiloCodeModel, + "id" | "name" | "description" | "context_length" | "owned_by" | "openrouter" +> -const AXON_EIDO_3_2_FLASH: KiloCodeModelVariant = { - description: "Axon Eido 3.2 Flash is a fast and low cost general purpose model for low-effort day-to-day tasks", +// Shared metadata for the OSS models served through the MatterAI gateway. +// Pricing strings are USD per token (OpenRouter format); per-model rates are +// set on each entry below. +const OSS_MODEL_BASE: KiloCodeModelVariant = { input_modalities: ["text", "image"], max_output_length: 64000, output_modalities: ["text"], @@ -190,82 +63,127 @@ const AXON_EIDO_3_2_FLASH: KiloCodeModelVariant = { "stop", ], supported_features: ["tools", "structured_outputs", "web_search"], - openrouter: { - slug: "matterai/axon", - }, datacenters: [{ country_code: "US" }], - created: 1750426201, - owned_by: "matterai", + created: 1786032000, pricing: { - prompt: "0.0", - completion: "0.0", image: "0", request: "0", - input_cache_reads: "0", input_cache_writes: "0", }, } const KILO_CODE_MODELS: Record = { - "axon-auto-232k": { - ...AXON_AUTO, - id: "axon-auto", - name: "Axon Auto (232K context)", + "meta/muse-spark-1.2-contributor": { + ...OSS_MODEL_BASE, + id: "meta/muse-spark-1.2-contributor", + name: "Muse Spark 1.2 Contributor", + description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.", context_length: 232000, + owned_by: "meta", + openrouter: { slug: "meta/muse-spark-1.2-contributor" }, + // $0.10/M input, $0.002/M cache read, $0.20/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.0000001", + completion: "0.0000002", + input_cache_reads: "0.000000002", + }, }, - "axon-auto-400k": { - ...AXON_AUTO, - id: "axon-auto", - name: "Axon Auto (400K context)", - context_length: 400000, - }, - "axon-eido-3.2-flash": { - ...AXON_EIDO_3_2_FLASH, - id: "axon-eido-3.2-flash", - name: "Axon Eido 3.2 Flash (232K context)", + "deepseek/deepseek-v4-flash-0731": { + ...OSS_MODEL_BASE, + id: "deepseek/deepseek-v4-flash-0731", + name: "DeepSeek V4 Flash", + description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", context_length: 232000, + owned_by: "deepseek", + openrouter: { slug: "deepseek/deepseek-v4-flash-0731" }, + // $0.14/M input, $0.028/M cache read, $0.28/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.00000014", + completion: "0.00000028", + input_cache_reads: "0.000000028", + }, }, - "axon-eido-3.2-flash-400k": { - ...AXON_EIDO_3_2_FLASH, - id: "axon-eido-3.2-flash", - name: "Axon Eido 3.2 Flash (400K context)", - context_length: 400000, - }, - "axon-eido-3.2-232k": { - ...AXON_EIDO_3_2, - id: "axon-eido-3.2", - name: "Axon Eido 3.2 (232K context)", + "zai/glm-5.3-flash": { + ...OSS_MODEL_BASE, + id: "zai/glm-5.3-flash", + name: "GLM 5.3 Flash", + description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", context_length: 232000, + owned_by: "zai", + openrouter: { slug: "zai/glm-5.3-flash" }, + // $0.15/M input, $0.03/M cache read, $0.50/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.00000015", + completion: "0.0000005", + input_cache_reads: "0.00000003", + }, }, - "axon-eido-3.2-400k": { - ...AXON_EIDO_3_2, - id: "axon-eido-3.2", - name: "Axon Eido 3.2 (400K context)", - context_length: 400000, - }, - "axon-eido-3.2-code-pro-232k": { - ...AXON_EIDO_3_2_CODE_PRO, - id: "axon-eido-3.2-code-pro", - name: "Axon Eido 3.2 Pro (232K context)", + "zai/glm-5.3": { + ...OSS_MODEL_BASE, + id: "zai/glm-5.3", + name: "GLM 5.3", + description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", context_length: 232000, + owned_by: "zai", + openrouter: { slug: "zai/glm-5.3" }, + // $1.40/M input, $0.14/M cache read, $4.40/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.0000014", + completion: "0.0000044", + input_cache_reads: "0.00000014", + }, }, - "axon-eido-3.2-code-pro-400k": { - ...AXON_EIDO_3_2_CODE_PRO, - id: "axon-eido-3.2-code-pro", - name: "Axon Eido 3.2 Pro (400K context)", - context_length: 400000, + "gpt-5.6-luna": { + ...OSS_MODEL_BASE, + id: "gpt-5.6-luna", + name: "GPT-5.6 Luna", + description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", + context_length: 232000, + owned_by: "openai", + openrouter: { slug: "gpt-5.6-luna" }, + // $0.20/M input, $0.02/M cache read, $1.20/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.0000002", + completion: "0.0000012", + input_cache_reads: "0.00000002", + }, }, - "axon-lumen-4-code-232k": { - ...AXON_LUMEN_4_CODE, - id: "axon-lumen-4-code", - name: "Axon Lumen 4 (232K context)", + "gpt-5.6-sol": { + ...OSS_MODEL_BASE, + id: "gpt-5.6-sol", + name: "GPT-5.6 Sol", + description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", context_length: 232000, + owned_by: "openai", + openrouter: { slug: "gpt-5.6-sol" }, + // $5/M input, $0.50/M cache read, $30/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.000005", + completion: "0.00003", + input_cache_reads: "0.0000005", + }, }, - "axon-lumen-4-code-400k": { - ...AXON_LUMEN_4_CODE, - id: "axon-lumen-4-code", - name: "Axon Lumen 4 (400K context)", - context_length: 400000, + "gemini-3.7-flash": { + ...OSS_MODEL_BASE, + id: "gemini-3.7-flash", + name: "Gemini 3.7 Flash", + description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", + context_length: 232000, + owned_by: "google", + openrouter: { slug: "gemini-3.7-flash" }, + // $0.75/M input, $0.075/M cache read, $3.75/M output + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: "0.00000075", + completion: "0.00000375", + input_cache_reads: "0.000000075", + }, }, } From 2a4ca5a001d8b3d05cd049ed3d7996278f98dff1 Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Wed, 2 Sep 2026 15:37:30 +0530 Subject: [PATCH 2/9] fix(assistant-message): keep id-less native tool calls and run trailing blocks after a read-only batch - AssistantMessageParser: synthesize a stable `native-tool-call-` id when a delta carries an index but no id instead of dropping the call. Some OpenAI-compatible providers never send an id (or only send it in a later fragment); dropping the first delta silently discarded the whole tool call, and every later argument fragment for that index then hit "arguments for unknown tool call" and was dropped too. `name` is now optional in NativeToolCall since only the first delta carries it. - presentAssistantMessage: after a parallel read-only batch stops at the first non-parallelizable block (e.g. execute_command following a run of searches), continue the presentation chain so trailing blocks still execute; the batch only marks userMessageContentReady when it consumed every block, instead of firing the next API request early. - The tool-failure fallback only signals readiness for the LAST content block, so a mid-message failure no longer tells the task loop the message is complete while later blocks still have to run. - Regression tests for id-less tool call accumulation, the trailing execute_command after a read-only batch, and mid-message failure readiness. --- .../AssistantMessageParser.ts | 21 +- .../__tests__/AssistantMessageParser.spec.ts | 39 ++++ .../__tests__/presentAssistantMessage.spec.ts | 194 ++++++++++++++++++ .../kilocode/native-tool-call.ts | 4 +- .../presentAssistantMessage.ts | 28 ++- 5 files changed, 276 insertions(+), 10 deletions(-) diff --git a/src/core/assistant-message/AssistantMessageParser.ts b/src/core/assistant-message/AssistantMessageParser.ts index d6de1347ea..6010f406f6 100644 --- a/src/core/assistant-message/AssistantMessageParser.ts +++ b/src/core/assistant-message/AssistantMessageParser.ts @@ -195,8 +195,9 @@ export class AssistantMessageParser { * forked_change: Now yields partial tool calls immediately when tool name is known, * allowing the UI to show "Editing filename..." during streaming. * - * @param toolCalls Array of native tool call objects (may be partial during streaming). We - * currently set parallel_tool_calls to false, so in theory there should only be 1 call. + * @param toolCalls Array of native tool call objects (may be partial during streaming). + * Native tool calling sets parallel_tool_calls = true, so this may contain + * several independent calls that all must be accumulated and executed. */ public *processNativeToolCalls(toolCalls: NativeToolCall[]): Generator { for (const toolCall of toolCalls) { @@ -215,11 +216,17 @@ export class AssistantMessageParser { toolCallId = toolCall.id this.nativeToolCallIndexToId.set(toolCall.index, toolCallId) } else { - console.warn( - "[AssistantMessageParser] Skipping tool call: has index but no id in mapping:", - toolCall, - ) - continue + // Some OpenAI-compatible providers never send an id for a tool call + // (or only send it in a later fragment). Dropping the call here used + // to silently discard the whole tool call — every later fragment for + // this index then hit "arguments for unknown tool call" and was + // dropped too. Synthesize a stable id from the index instead; the + // assistant message and its tool_results are built from these ids, + // so pairing stays consistent. If the real id arrives in a later + // delta, the mapping below still wins and accumulation continues + // unchanged. + toolCallId = `native-tool-call-${toolCall.index}` + this.nativeToolCallIndexToId.set(toolCall.index, toolCallId) } } else if (toolCall.id) { toolCallId = toolCall.id diff --git a/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts b/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts index be38f110ea..4bf22eed57 100644 --- a/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts +++ b/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts @@ -574,6 +574,45 @@ describe("AssistantMessageParser (streaming)", () => { expect(pkgUse?.params.content).toBe(pkgContent) expect(tsUse?.params.content).toBe(tsConfigContent) }) + + it("should accumulate a tool call whose deltas carry an index but never an id", () => { + // Some OpenAI-compatible providers never send an id per tool call. The + // parser used to drop the first delta ("has index but no id") and then + // every argument fragment for that index ("arguments for unknown tool + // call"), so the whole tool call silently vanished. + const yielded1 = [ + ...parser.processNativeToolCalls([ + { + index: 2, + type: "function", + function: { + name: "read_file", + arguments: '{"file_path": "src/a.ts"', + }, + }, + ]), + ] + // Name known, JSON incomplete: only the partial block is emitted. + expect(yielded1).toHaveLength(1) + + const yielded2 = [ + ...parser.processNativeToolCalls([ + { + index: 2, + function: { arguments: "}" }, + }, + ]), + ] + expect(yielded2).toHaveLength(1) + expect(yielded2[0].id).toBe("native-tool-call-2") + + const toolUse = parser + .getContentBlocks() + .find((b) => b.type === "tool_use" && (b as ToolUse).toolUseId === "native-tool-call-2") as ToolUse + expect(toolUse).toBeDefined() + expect(toolUse.partial).toBe(false) + expect(toolUse.params.file_path).toBe("src/a.ts") + }) }) describe("size limit handling", () => { diff --git a/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts b/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts index 45958fc6be..20f925a0a6 100644 --- a/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts +++ b/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts @@ -59,6 +59,7 @@ vi.mock("../../task/Task", () => ({ })) import { presentAssistantMessage } from "../presentAssistantMessage" +import { executeCommandTool } from "../../tools/executeCommandTool" import { readFileTool } from "../../tools/readFileTool" import { searchFilesTool } from "../../tools/searchFilesTool" import { TelemetryService } from "@roo-code/telemetry" @@ -411,4 +412,197 @@ describe("presentAssistantMessage", () => { expect(cline.userMessageContentReady).toBe(true) expect(cline.presentAssistantMessageLocked).toBe(false) }) + + it("runs the trailing non-parallelizable tool after a read-only batch (execute_command not dropped)", async () => { + // Regression: a native parallel response of 3x search_files + read_file + + // execute_command executed only the 4 read-only calls. The batch scheduler + // set userMessageContentReady=true after committing the batch, so the task + // loop fired the next request before execute_command was ever presented. + const getState = vi.fn().mockResolvedValue({ mode: "code", customModes: [] }) + vi.mocked(executeCommandTool).mockImplementationOnce( + async (_cline: any, block: any, _ask: any, _handleError: any, pushToolResult: any) => { + pushToolResult(`command:${block.params.command}`) + }, + ) + const cline = { + abort: false, + taskId: "task-1", + instanceId: "instance-1", + presentAssistantMessageLocked: false, + presentAssistantMessageHasPendingUpdates: false, + currentStreamingContentIndex: 0, + assistantMessageContent: [ + { + type: "tool_use", + name: "search_files", + params: { path: "src", regex: "tool_calls", file_pattern: "*.ts" }, + partial: false, + toolUseId: "search_files:0", + }, + { + type: "tool_use", + name: "read_file", + params: { file_path: "src/api/providers/openai.ts" }, + partial: false, + toolUseId: "read_file:1", + }, + { + type: "tool_use", + name: "search_files", + params: { path: "src/controller", regex: "processRequest" }, + partial: false, + toolUseId: "search_files:2", + }, + { + type: "tool_use", + name: "search_files", + params: { path: "src", regex: "getAxonModel" }, + partial: false, + toolUseId: "search_files:3", + }, + { + type: "tool_use", + name: "execute_command", + params: { command: "git status --short" }, + partial: false, + toolUseId: "execute_command:4", + }, + ], + didCompleteReadingStream: true, + userMessageContentReady: false, + userMessageContent: [], + didRejectTool: false, + didAlreadyUseTool: false, + currentStreamingDidCheckpoint: false, + diffEnabled: false, + autoApproveAllCommands: false, + consecutiveMistakeCount: 0, + providerRef: { + deref: () => ({ + getState, + }), + }, + browserSession: { + closeBrowser: vi.fn().mockResolvedValue(undefined), + }, + say: vi.fn().mockResolvedValue(undefined), + ask: vi.fn().mockResolvedValue({ response: "yesButtonClicked" }), + processQueuedMessages: vi.fn(), + getToolCallSignature: vi.fn((name: string, params: unknown) => JSON.stringify({ name, params })), + checkAndRegisterToolCall: vi.fn().mockReturnValue(false), + recordToolUsage: vi.fn(), + toolRepetitionDetector: { + check: vi.fn().mockReturnValue({ allowExecution: true }), + }, + checkpointSave: vi.fn().mockResolvedValue(undefined), + removeStalePartialToolAskMessage: vi.fn().mockResolvedValue(undefined), + checkAndCondenseContext: vi.fn().mockResolvedValue(undefined), + } as any + + await presentAssistantMessage(cline) + + // All four read-only calls ran concurrently, and the trailing + // execute_command was still presented and executed afterwards. + expect(searchFilesTool).toHaveBeenCalledTimes(3) + expect(readFileTool).toHaveBeenCalledTimes(1) + expect(executeCommandTool).toHaveBeenCalledTimes(1) + + // Results are committed in model order with one tool_result per tool_use. + expect(cline.userMessageContent.map((item: any) => item.tool_use_id)).toEqual([ + "search_files:0", + "read_file:1", + "search_files:2", + "search_files:3", + "execute_command:4", + ]) + + expect(cline.currentStreamingContentIndex).toBe(5) + expect(cline.userMessageContentReady).toBe(true) + }) + + it("does not mark the message ready while a later block is still pending after a tool failure", async () => { + // The finally-block fallback pushes a synthetic tool_result when a tool + // fails without producing one. It must only flip userMessageContentReady + // for the LAST block — a mid-message failure used to signal "message + // complete" while later blocks still had to execute. + const getState = vi.fn().mockResolvedValue({ mode: "code", customModes: [] }) + vi.mocked(readFileTool).mockImplementationOnce(async () => { + throw new Error("boom") + }) + let readyWhenSecondRan: boolean | undefined + vi.mocked(readFileTool).mockImplementationOnce( + async (_cline: any, _block: any, _ask: any, _handleError: any, pushToolResult: any) => { + readyWhenSecondRan = cline.userMessageContentReady + pushToolResult("ok") + }, + ) + const cline = { + abort: false, + taskId: "task-1", + instanceId: "instance-1", + presentAssistantMessageLocked: false, + presentAssistantMessageHasPendingUpdates: false, + currentStreamingContentIndex: 0, + assistantMessageContent: [ + { + type: "tool_use", + name: "read_file", + params: { file_path: "a.ts" }, + partial: false, + toolUseId: "read_file:0", + }, + { + type: "tool_use", + name: "read_file", + params: { file_path: "b.ts" }, + partial: false, + toolUseId: "read_file:1", + }, + ], + didCompleteReadingStream: true, + userMessageContentReady: false, + userMessageContent: [], + didRejectTool: false, + didAlreadyUseTool: false, + currentStreamingDidCheckpoint: false, + diffEnabled: false, + autoApproveAllCommands: false, + consecutiveMistakeCount: 0, + providerRef: { + deref: () => ({ + getState, + }), + }, + browserSession: { + closeBrowser: vi.fn().mockResolvedValue(undefined), + }, + say: vi.fn().mockResolvedValue(undefined), + ask: vi.fn().mockResolvedValue({ response: "yesButtonClicked" }), + processQueuedMessages: vi.fn(), + getToolCallSignature: vi.fn((name: string, params: unknown) => JSON.stringify({ name, params })), + checkAndRegisterToolCall: vi.fn().mockReturnValue(false), + recordToolUsage: vi.fn(), + toolRepetitionDetector: { + check: vi.fn().mockReturnValue({ allowExecution: true }), + }, + checkpointSave: vi.fn().mockResolvedValue(undefined), + removeStalePartialToolAskMessage: vi.fn().mockResolvedValue(undefined), + checkAndCondenseContext: vi.fn().mockResolvedValue(undefined), + } as any + + await presentAssistantMessage(cline) + + // The second tool must have observed the message as NOT ready when it ran. + expect(readyWhenSecondRan).toBe(false) + + // Both results present: the fallback for the failed call plus the real one. + const toolResults = cline.userMessageContent.filter((item: any) => item.type === "tool_result") + expect(toolResults).toHaveLength(2) + expect(toolResults[0].tool_use_id).toBe("read_file:0") + expect(toolResults[0].content[0].text).toContain("did not produce a result") + expect(toolResults[1].tool_use_id).toBe("read_file:1") + + expect(cline.currentStreamingContentIndex).toBe(2) + expect(cline.userMessageContentReady).toBe(true) + }) }) diff --git a/src/core/assistant-message/kilocode/native-tool-call.ts b/src/core/assistant-message/kilocode/native-tool-call.ts index d1aa148182..8bb3431ccc 100644 --- a/src/core/assistant-message/kilocode/native-tool-call.ts +++ b/src/core/assistant-message/kilocode/native-tool-call.ts @@ -6,7 +6,9 @@ export interface NativeToolCall { id?: string // Only present in first delta type?: string function?: { - name: string + // name is only present in the first delta; subsequent deltas carry + // arguments only (standard OpenAI-compatible streaming shape). + name?: string arguments: string // JSON string (may be partial during streaming) } // forked_change: Track if this is an MCP tool and which server diff --git a/src/core/assistant-message/presentAssistantMessage.ts b/src/core/assistant-message/presentAssistantMessage.ts index 29f3a5c834..79775b8038 100644 --- a/src/core/assistant-message/presentAssistantMessage.ts +++ b/src/core/assistant-message/presentAssistantMessage.ts @@ -112,6 +112,13 @@ export async function presentAssistantMessage(cline: Task, options: PresentAssis } finally { cline.presentAssistantMessageLocked = false } + // The batch stops at the first block that cannot run concurrently (e.g. + // execute_command following a run of read-only calls). Continue the + // presentation chain so those trailing blocks still execute — the batch + // only marks the message ready when it consumed every block. + if (!cline.abort && cline.currentStreamingContentIndex < cline.assistantMessageContent.length) { + await presentAssistantMessage(cline) + } return } } @@ -967,8 +974,17 @@ export async function presentAssistantMessage(cline: Task, options: PresentAssis // Make sure the task loop can move forward even on failure — // otherwise the next iteration may hang waiting for a result. + // Only signal readiness when this is the LAST content block: for a + // mid-message failure the presentation chain still has blocks to + // execute, and flipping the flag here would let the task loop fire + // the next request before they are presented (same bug class as the + // parallel read-only batch skipping a trailing execute_command). cline.didAlreadyUseTool = true - if (cline.didCompleteReadingStream && !isParallelWorker) { + if ( + cline.didCompleteReadingStream && + !isParallelWorker && + blockIndex >= cline.assistantMessageContent.length - 1 + ) { cline.userMessageContentReady = true } } catch (e) { @@ -1104,7 +1120,15 @@ async function executeParallelReadOnlyBlocks(cline: Task, blockIndexes: number[] } cline.currentStreamingContentIndex = blockIndexes[blockIndexes.length - 1] + 1 - cline.userMessageContentReady = true + + // Only tell the task loop the assistant message is fully presented when the + // batch ran through the last content block. getParallelReadOnlyBlockIndexes + // stops at the first non-read-only block (e.g. execute_command after a run of + // searches), so setting this unconditionally made the task loop fire the next + // API request while that trailing tool_use was never presented or executed. + if (cline.currentStreamingContentIndex >= cline.assistantMessageContent.length) { + cline.userMessageContentReady = true + } } /** From bb438917226ab1ca85da389f56bf14ffeca06fc4 Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Wed, 2 Sep 2026 15:37:48 +0530 Subject: [PATCH 3/9] feat(context): add pre-compaction budget warning and compaction handoff prefix Port codex-style context window management so the model adapts before and after auto-compaction instead of redoing work: - Task: when usage enters a 10-point band below the condense threshold (CONTEXT_WARNING_BAND_PERCENT), push a one-time user message telling the model to finish in-flight edits, stop broad searches and large reads, and keep the todo list current before compaction replaces raw tool outputs with a summary. The warning re-arms when usage drops back below the band (e.g. after a successful condensation), so each context window gets at most one warning. - Condense: the summary prompt now records exploration already performed (searches, reads, investigations and their conclusions) and demands enough detail (paths, signatures, line numbers, error text) to continue without re-reading files. The inserted summary message is prefixed with a [CONTEXT COMPACTION] handoff note (SUMMARY_PREFIX) instructing the model to treat the summary as the authoritative record and continue from NEXT STEPS instead of re-searching, re-reading, or re-deriving. - Update condense tests for the prefix and new prompt wording. --- src/core/condense/__tests__/condense.spec.ts | 4 +-- src/core/condense/__tests__/index.spec.ts | 10 +++--- src/core/condense/index.ts | 14 ++++++++- src/core/task/Task.ts | 32 ++++++++++++++++++++ 4 files changed, 52 insertions(+), 8 deletions(-) diff --git a/src/core/condense/__tests__/condense.spec.ts b/src/core/condense/__tests__/condense.spec.ts index 5eb97b3e8a..1498c80402 100644 --- a/src/core/condense/__tests__/condense.spec.ts +++ b/src/core/condense/__tests__/condense.spec.ts @@ -6,7 +6,7 @@ import { TelemetryService } from "@roo-code/telemetry" import { BaseProvider } from "../../../api/providers/base-provider" import { ApiMessage } from "../../task-persistence/apiMessages" -import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP } from "../index" +import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP, SUMMARY_PREFIX } from "../index" // Create a mock ApiHandler for testing class MockApiHandler extends BaseProvider { @@ -81,7 +81,7 @@ describe("Condense", () => { // Verify we have a summary message const summaryMessage = result.messages.find((msg) => msg.isSummary) expect(summaryMessage).toBeTruthy() - expect(summaryMessage?.content).toBe("Mock summary of the conversation") + expect(summaryMessage?.content).toBe(`${SUMMARY_PREFIX}\n\nMock summary of the conversation`) // Verify we have the expected number of messages // [first message, summary, last N messages] diff --git a/src/core/condense/__tests__/index.spec.ts b/src/core/condense/__tests__/index.spec.ts index fa59b5fba2..62fb2b6231 100644 --- a/src/core/condense/__tests__/index.spec.ts +++ b/src/core/condense/__tests__/index.spec.ts @@ -7,7 +7,7 @@ import { TelemetryService } from "@roo-code/telemetry" import { ApiHandler } from "../../../api" import { ApiMessage } from "../../task-persistence/apiMessages" import { maybeRemoveImageBlocks } from "../../../api/transform/image-cleaning" -import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP } from "../index" +import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP, SUMMARY_PREFIX } from "../index" vi.mock("../../../api/transform/image-cleaning", () => ({ maybeRemoveImageBlocks: vi.fn((messages: ApiMessage[], _apiHandler: ApiHandler) => [...messages]), @@ -197,7 +197,7 @@ describe("summarizeConversation", () => { // Check that the summary message was inserted correctly const summaryMessage = result.messages[1] expect(summaryMessage.role).toBe("assistant") - expect(summaryMessage.content).toBe("This is a summary") + expect(summaryMessage.content).toBe(`${SUMMARY_PREFIX}\n\nThis is a summary`) expect(summaryMessage.isSummary).toBe(true) // Check that the last N_MESSAGES_TO_KEEP messages are preserved @@ -275,7 +275,7 @@ describe("summarizeConversation", () => { // Verify that createMessage was called with the correct prompt expect(mockApiHandler.createMessage).toHaveBeenCalledWith( - expect.stringContaining("Your task is to create a detailed summary of the conversation"), + expect.stringContaining("Summarize this conversation with maximum information density"), expect.any(Array), ) @@ -653,7 +653,7 @@ describe("summarizeConversation with custom settings", () => { // Verify the default prompt was used let createMessageCalls = (mockMainApiHandler.createMessage as Mock).mock.calls expect(createMessageCalls.length).toBe(1) - expect(createMessageCalls[0][0]).toContain("Your task is to create a detailed summary") + expect(createMessageCalls[0][0]).toContain("Summarize this conversation with maximum information density") // Reset mock and test with undefined vi.clearAllMocks() @@ -670,7 +670,7 @@ describe("summarizeConversation with custom settings", () => { // Verify the default prompt was used again createMessageCalls = (mockMainApiHandler.createMessage as Mock).mock.calls expect(createMessageCalls.length).toBe(1) - expect(createMessageCalls[0][0]).toContain("Your task is to create a detailed summary") + expect(createMessageCalls[0][0]).toContain("Summarize this conversation with maximum information density") }) /** diff --git a/src/core/condense/index.ts b/src/core/condense/index.ts index 2183d3679b..933876c1f1 100644 --- a/src/core/condense/index.ts +++ b/src/core/condense/index.ts @@ -28,6 +28,8 @@ FILES: For each file that was read, modified, or created: - Changes made (if any) - Key code: function signatures, type definitions, error messages, constants +EXPLORATION ALREADY DONE: Searches, file reads, and investigations already performed and what each concluded — so they are NOT repeated after compaction. + FAILED APPROACHES: What was tried and did not work. Why it failed. What should NOT be attempted again. KNOWLEDGE STATE: @@ -41,10 +43,18 @@ NEXT STEPS: Immediate next action — include a VERBATIM quote of the most recen RULES: - Preserve every file path, function name, variable name, error message, and identifier exactly as it appeared. Never paraphrase identifiers. +- The summary is the only surviving record: anything omitted is lost and must be re-derived by re-reading files. Include enough detail (paths, signatures, line numbers, error text) to continue without re-reading. - Prefer listing over prose. Every token counts. - Output ONLY the summary. No preamble, no "Here is the summary:", no commentary after. ` +// Prepended to the summary message inserted into the conversation after +// compaction. Ports codex's compact/summary_prefix.md: tells the model the +// history was compacted and that it must build on the recorded work instead +// of redoing it (re-searching, re-reading files, re-deriving facts). +export const SUMMARY_PREFIX = `\ +[CONTEXT COMPACTION] An earlier assistant began this task and produced the summary below as a handoff. Treat it as the authoritative record of all work so far: do NOT repeat anything it describes — no re-running searches, no re-reading files it covers, no re-deriving facts it states. Continue from NEXT STEPS.` + export type SummarizeResponse = { messages: ApiMessage[] // The messages after summarization summary: string // The summary text; empty string for no summary @@ -181,9 +191,11 @@ export async function summarizeConversation( return { ...response, cost, error } } + // The prefix tells the model this is a compaction handoff and that it must + // not redo the work recorded in the summary. const summaryMessage: ApiMessage = { role: "assistant", - content: summary, + content: `${SUMMARY_PREFIX}\n\n${summary}`, ts: keepMessages[0].ts, isSummary: true, } diff --git a/src/core/task/Task.ts b/src/core/task/Task.ts index 19aa45d750..948d6497fb 100644 --- a/src/core/task/Task.ts +++ b/src/core/task/Task.ts @@ -156,6 +156,7 @@ import { getAppUrl } from "@roo-code/types" const MAX_EXPONENTIAL_BACKOFF_SECONDS = 600 // 10 minutes const DEFAULT_USAGE_COLLECTION_TIMEOUT_MS = 5000 // 5 seconds const FORCED_CONTEXT_REDUCTION_PERCENT = 75 // Keep 75% of context (remove 25%) on context window errors +const CONTEXT_WARNING_BAND_PERCENT = 10 // Warn the model this many percentage points below the condense threshold const MAX_CONTEXT_WINDOW_RETRIES = 3 // Maximum retries for context window errors // kilocode_change: Idle timeout for stream consumption. If no chunk is received @@ -509,6 +510,9 @@ export class Task extends EventEmitter implements TaskLike { assistantMessageParser: AssistantMessageParser private lastUsedInstructions?: string private skipPrevResponseIdOnce: boolean = false + // forked_change: set once per context window when the context budget warning + // fires; re-arms automatically when usage drops back below the warning band + private contextBudgetWarningIssued: boolean = false // Token Usage Cache private tokenUsageSnapshot?: TokenUsage @@ -3946,6 +3950,34 @@ export class Task extends EventEmitter implements TaskLike { // Check if we're at or above the threshold if (contextPercent < effectiveThreshold) { + // forked_change start: codex-style context budget warning. + // When usage enters the band just below the condense threshold, warn + // the model ONCE so it adapts before compaction discards raw tool + // outputs: finish in-flight edits, stop broad exploration, keep the + // todo list current. Re-arms when usage drops back below the band + // (i.e. after a successful condensation), so each context window + // gets at most one warning. + if (contextWindow > 0) { + const warningThreshold = Math.max( + MIN_CONDENSE_THRESHOLD, + effectiveThreshold - CONTEXT_WARNING_BAND_PERCENT, + ) + if (contextPercent < warningThreshold) { + this.contextBudgetWarningIssued = false + } else if (!this.contextBudgetWarningIssued) { + this.contextBudgetWarningIssued = true + const tokensRemaining = Math.max(0, contextWindow - contextTokens) + console.log( + `[Task#${this.taskId}] Context budget warning: ${contextPercent.toFixed(1)}% used ` + + `(~${tokensRemaining} of ${contextWindow} tokens remain), condense threshold ${effectiveThreshold}%.`, + ) + this.userMessageContent.push({ + type: "text", + text: `\nContext window nearly full: ${contextPercent.toFixed(0)}% used (~${tokensRemaining.toLocaleString()} of ${contextWindow.toLocaleString()} tokens remain). Auto-compaction triggers at ${effectiveThreshold}% and replaces raw tool outputs with a summary.\nAdapt now:\n- Finish in-flight edits before starting anything new.\n- No broad searches or large file reads; use targeted reads of specific regions only.\n- Do not re-read files whose contents are already in context.\n- Keep the todo list current so pending work survives compaction.\n`, + }) + } + } + // forked_change end return false } From e33136a3f48906dced981e22413f5c045954fd25 Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Wed, 2 Sep 2026 15:38:05 +0530 Subject: [PATCH 4/9] feat(prompts): rework edit batching guidance and add multi-repo workspace rules - file_edit/multi_file_edit guidance now keys on edits that are confirmed and ready now rather than a blanket "always batch 2+ edits": make a ready edit immediately instead of holding it back, keep batches small and cohesive (the edits belonging to the current step), and never accumulate a whole task into one giant multi-file edit. - Replace "Plan before editing" with "Edit early, iterate in small steps": make the first edit as soon as one file's change is confirmed, alternate editing and checking (typecheck, test, targeted read) as the intended workflow, and track remaining work with update_todo_list instead of holding a full multi-file plan in context. Do not re-read a file just to confirm an edit succeeded. - Add a Multi-repo workspaces section: work inside the repo that owns the code being changed, cross into another repo only when the task requires it, and never interleave reads across repos. - Refresh prompt snapshots. --- .../architect-mode-prompt.snap | 27 ++++++++++++------- .../ask-mode-prompt.snap | 27 ++++++++++++------- .../mcp-server-creation-disabled.snap | 27 ++++++++++++------- .../partial-reads-enabled.snap | 27 ++++++++++++------- .../consistent-system-prompt.snap | 27 ++++++++++++------- .../with-computer-use-support.snap | 27 ++++++++++++------- .../system-prompt/with-diff-enabled-true.snap | 27 ++++++++++++------- .../with-different-viewport-size.snap | 27 ++++++++++++------- .../system-prompt/with-undefined-mcp-hub.snap | 27 ++++++++++++------- src/core/prompts/system.ts | 27 ++++++++++++------- 10 files changed, 180 insertions(+), 90 deletions(-) diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap index d67b9e076d..4a53942ed0 100644 --- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap +++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap @@ -567,8 +567,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -578,11 +578,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -608,7 +608,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap index 79bfc84eca..ed7c8c459a 100644 --- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap +++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap @@ -427,8 +427,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -438,11 +438,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -468,7 +468,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -554,10 +555,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap index 67be5f472b..f049284600 100644 --- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap +++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap @@ -566,8 +566,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -577,11 +577,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -607,7 +607,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -693,10 +694,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap index 10613f49cf..410f130279 100644 --- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap +++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap @@ -572,8 +572,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -583,11 +583,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -613,7 +613,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -699,10 +700,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap index d67b9e076d..4a53942ed0 100644 --- a/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap +++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap @@ -567,8 +567,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -578,11 +578,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -608,7 +608,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap index ffa6e30c24..2d6eedc8e2 100644 --- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap +++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap @@ -620,8 +620,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -631,11 +631,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -661,7 +661,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -747,10 +748,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap index d67b9e076d..4a53942ed0 100644 --- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap +++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap @@ -567,8 +567,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -578,11 +578,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -608,7 +608,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap index 4afe89345e..299127f4bf 100644 --- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap +++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap @@ -620,8 +620,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -631,11 +631,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -661,7 +661,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -747,10 +748,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap index d67b9e076d..4a53942ed0 100644 --- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap +++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap @@ -567,8 +567,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead. -- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -578,11 +578,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. `edits` — An array of edit objects. Each edit has: @@ -608,7 +608,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → `file_edit` -- 2+ edits → `multi_file_edit` (always) +- 2+ edits ready now → `multi_file_edit` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch. @@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency diff --git a/src/core/prompts/system.ts b/src/core/prompts/system.ts index 95a61f31c6..08c0308ada 100644 --- a/src/core/prompts/system.ts +++ b/src/core/prompts/system.ts @@ -107,8 +107,8 @@ Common tool calls and explanations - You know the exact text that should be replaced and its updated form. **When NOT to use**: -- If you have **2 or more edits** to make (even to the same file), use \`multi_file_edit\` instead. -- Never call \`file_edit\` multiple times in sequence. Batch your edits with \`multi_file_edit\`. +- If you have **2 or more independent edits** that are all confirmed and ready right now, use \`multi_file_edit\` instead. +- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards. **Parameters**: 1. \`file_path\` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts). @@ -118,11 +118,11 @@ Common tool calls and explanations ## multi_file_edit -**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make. +**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step. **When to use**: -- You have **2 or more edits** to make, whether to the same file or different files. -- You want to batch edits efficiently instead of making multiple separate tool calls. +- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files. +- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task. **Parameters**: 1. \`edits\` — An array of edit objects. Each edit has: @@ -148,7 +148,8 @@ Common tool calls and explanations **Guidance for choosing between file_edit and multi_file_edit**: - 1 edit → \`file_edit\` -- 2+ edits → \`multi_file_edit\` (always) +- 2+ edits ready now → \`multi_file_edit\` +- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task. **Editing discipline (CRITICAL)**: - ALWAYS copy \`old_string\` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed \`old_string\` will silently mismatch. @@ -234,10 +235,18 @@ Use zero context for discovery, then read the relevant file region. If results a - After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output. - If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time. -## Plan before editing +## Edit early, iterate in small steps -- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. -- Then execute the edits in one pass (batched via \`multi_file_edit\`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits. +- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure. +- Track remaining work with \`update_todo_list\` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context. +- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit. +- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.) + +## Multi-repo workspaces + +- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem." +- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos. ## Investigation efficiency From f82bfd8cb64de628572ef26f0729388d52d4a251 Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Wed, 2 Sep 2026 15:38:49 +0530 Subject: [PATCH 5/9] chore(release): v6.8.4 - Bump the extension version from 6.8.2 to 6.8.4. - Add the v6.8.3 changelog entry documenting the OSS model catalog that replaces the Axon models, including per-model pricing and the new deepseek/deepseek-v4-flash-0731 default. - Consume the applied changesets (axon default model, pasted image paths, Fireworks provider removal). --- .changeset/axon-auto-default-model.md | 10 ---------- .changeset/pasted-image-paths.md | 11 ----------- .changeset/remove-fireworks-provider.md | 13 ------------- CHANGELOG.md | 6 ++++++ src/package.json | 2 +- 5 files changed, 7 insertions(+), 35 deletions(-) delete mode 100644 .changeset/axon-auto-default-model.md delete mode 100644 .changeset/pasted-image-paths.md delete mode 100644 .changeset/remove-fireworks-provider.md diff --git a/.changeset/axon-auto-default-model.md b/.changeset/axon-auto-default-model.md deleted file mode 100644 index 7aea09693d..0000000000 --- a/.changeset/axon-auto-default-model.md +++ /dev/null @@ -1,10 +0,0 @@ ---- -"kilo-code": minor ---- - -Make Axon Auto the default model - -- Changed `openRouterDefaultModelId` in `@roo-code/types` from `axon-eido-3-code-mini-232k` to `axon-auto-232k`, so the Orbital/KiloCode provider defaults to the Axon Auto model. -- Updated the `openRouterDefaultModelInfo` fallback description to describe Axon Auto (dynamic Flash/Mini/Pro selection). -- Updated `getAxonPlanFallback` in `model-plan-access` to fall back to `axon-auto-232k` instead of `axon-eido-3-code-mini-232k` for plan-restricted users. -- Added the `axon-eido-3-flash-400k` model variant to the KiloCode model catalog (extension and webview), gated behind the same Pro Plus/Ultra 400k plan check as the other 400k variants; `is400kAxonModel` now recognizes `axon-eido-3-flash-*` ids and the 232k fallback maps it to `axon-eido-3-flash`. diff --git a/.changeset/pasted-image-paths.md b/.changeset/pasted-image-paths.md deleted file mode 100644 index 062e1d7ab4..0000000000 --- a/.changeset/pasted-image-paths.md +++ /dev/null @@ -1,11 +0,0 @@ -## "kilo-code": patch - -Attach files pasted or dropped into the chat input - -- Pasting a supported file path (absolute, relative to the workspace, or a `file://` URI) into the chat textarea now attaches the file instead of inserting the path as plain text, so the content is actually sent to the LLM. -- Copying a file directly (e.g. from Finder/Explorer) and pasting it now attaches non-image documents (PDF, DOCX, XLSX, CSV, JSON, MD, TXT, TSV) too — previously only image blobs were handled and other files vanished silently. -- Supports images (`.png`, `.jpg`, `.jpeg`, `.webp`) and documents (`.csv`, `.docx`, `.json`, `.md`, `.pdf`, `.txt`, `.text`, `.tsv`, `.xlsx`), reusing the same extraction/size limits and the existing `selectedAttachments` response channel as the attachment picker. -- Refactored `process-attachments.ts` to share the per-file processing logic between the file picker, the pasted-path flow, and the pasted-blob flow. -- Document attachment chips now show the extension-specific material icon (e.g. PDF, Excel, Word) instead of a generic file icon, reusing the same `vscode-material-icons` mapping as mention chips. -- Unified the image and document attachment chips into a single shared flex-wrap container with matching pill styling, so images and files render at the same size and flow together across rows. -- Drag-and-drop now attaches image and document files too: paths dropped from the VS Code explorer are routed to attachments (instead of mentions) when they match a supported type, and non-image file blobs dropped from the OS are processed by the extension. `handleDrop` only intercepts drags that look like file/path drops (external files or path-like text); plain text drags within the editor fall through to native behavior. Drag-and-drop still requires holding Shift because VS Code intercepts native file drags otherwise. diff --git a/.changeset/remove-fireworks-provider.md b/.changeset/remove-fireworks-provider.md deleted file mode 100644 index 54ac902544..0000000000 --- a/.changeset/remove-fireworks-provider.md +++ /dev/null @@ -1,13 +0,0 @@ ---- -"kilo-code": minor ---- - -Remove the Fireworks AI provider - -- Removed the `FireworksHandler` from `src/api/providers`, the `fireworks` entry from the third-party model fetcher, the `fireworks` provider schema (`fireworksSchema`, `fireworksModels`, `fireworksDefaultModelId`) from `@roo-code/types`, and the `Fireworks` settings UI component. -- Removed Fireworks from `MODELS_BY_PROVIDER`, `modelIdKeysByProvider`, `PROVIDER_CONFIGS`, `thirdPartyProviderRequiresApiKey`, `ProfileValidator`, `taskMetadata`, `webviewMessageHandler`, `ClineProvider`, `Task`, `useProviderModels`, `useSelectedModel`, and the `OpenAI` provider's Fireworks detection branch in `src/api/providers/openai.ts`. -- Removed the Fireworks section from the `ThirdPartyProviders` settings panel and the Fireworks `HardcodedModelRecord` for `fireworks:accounts/fireworks/routers/kimi-k2p5-turbo`. -- Removed Fireworks entries from the CLI: `cli/src/constants/providers/{settings,models,validation,labels}.ts`, `cli/src/types/messages.ts`, `cli/src/config/schema.json`, `cli/docs/PROVIDER_CONFIGURATION.md`, and `cli/src/constants/providers/__tests__/models.test.ts`. -- Removed the Fireworks provider docs page and sidebar entry from `apps/kilocode-docs`. -- Removed `fireworksApiKey` and `getFireworksApiKey` i18n strings from all 22 webview locales (ar, ca, cs, de, en, es, fr, hi, id, it, ja, ko, nl, pl, pt-BR, ru, th, tr, uk, vi, zh-CN, zh-TW). -- Removed the unused `fireworks-ic.png` icon. diff --git a/CHANGELOG.md b/CHANGELOG.md index da995ee7e6..7f20bfcae2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,11 @@ # Changelog +## [v6.8.3] - 2026-09-02 + +### Changed + +- **OSS model catalog replaces Axon models.** The KiloCode model catalog (extension `kilocode-models.ts` and the webview `useOpenRouterModelProviders` copy) now exposes seven OSS models — `meta/muse-spark-1.2-contributor` (Muse Spark 1.2 Contributor), `deepseek/deepseek-v4-flash-0731` (DeepSeek V4 Flash), `zai/glm-5.3` (GLM 5.3), `zai/glm-5.3-flash` (GLM 5.3 Flash), `gpt-5.6-sol` (GPT-5.6 Sol), `gpt-5.6-luna` (GPT-5.6 Luna), and `gemini-3.7-flash` (Gemini 3.7 Flash) — in place of the Axon context-window variants. Each OSS model carries its published per-token pricing (Muse Spark 1.2 Contributor $0.10/M input, $0.002/M cache read, $0.20/M output; DeepSeek V4 Flash $0.14/M input, $0.028/M cache read, $0.28/M output; GLM 5.3 $1.40/M input, $0.14/M cache read, $4.40/M output; GLM 5.3 Flash $0.15/M input, $0.03/M cache read, $0.50/M output; GPT-5.6 Sol $5/M input, $0.50/M cache read, $30/M output; GPT-5.6 Luna $0.20/M input, $0.02/M cache read, $1.20/M output; Gemini 3.7 Flash $0.75/M input, $0.075/M cache read, $3.75/M output). The default model is `deepseek/deepseek-v4-flash-0731` (`openRouterDefaultModelId` in `@roo-code/types`, plus the CLI `kilocodeModel` defaults and web-evals `MODEL_DEFAULT`). Stored Axon model selections are now stale and reset to the default on next launch via the existing `isValidKilocodeModel` stale-model check. + ## [v6.8.2] - 2026-08-28 ### Added diff --git a/src/package.json b/src/package.json index 5191be3023..a9efcd95f6 100644 --- a/src/package.json +++ b/src/package.json @@ -3,7 +3,7 @@ "displayName": "%extension.displayName%", "description": "%extension.description%", "publisher": "matterai", - "version": "6.8.2", + "version": "6.8.4", "icon": "assets/icons/matterai-ic.png", "galleryBanner": { "color": "#FFFFFF", From 1233955e153eb65217dc54c1db0c5174110cf4ac Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Thu, 3 Sep 2026 12:22:16 +0530 Subject: [PATCH 6/9] feat(models): dynamic backend model catalog and per-model usage visibility - Fetch the model catalog from the MatterAI backend and register it via registerDynamicKilocodeModels, replacing static hardcoded lists; refresh on window focus, manual refresh, and a 10-minute poller with forceRefresh cache bypass - Show per-model usage (weekly/monthly share of the shared plan pool, cost-multiplier badges) in the usage dialog and a new read-only Settings Model Usage section; /axoncode/profile now returns modelUsage - Fix refreshKilocodeModels wiping OpenRouter models by fetching openrouter and kilocode-openrouter in parallel and merging results - Fetch the catalog at provider startup so stale axon model selections are reset via isValidKilocodeModel against a populated catalog --- CHANGELOG.md | 13 +- .../__tests__/kilocode-models.spec.ts | 169 +++++++++-- src/api/providers/fetchers/modelCache.ts | 21 +- src/api/providers/fetchers/openrouter.ts | 236 ++++++++------- src/api/providers/kilocode-models.ts | 271 ++++++++++------- src/core/webview/ClineProvider.ts | 111 +++++++ .../webview/__tests__/ClineProvider.spec.ts | 9 + src/core/webview/webviewMessageHandler.ts | 9 +- src/shared/WebviewMessage.ts | 15 + src/shared/api.ts | 1 + .../src/components/chat/UsageDialog.tsx | 52 +++- .../kilocode/chat/ModelSelector.tsx | 47 ++- .../kilocode/hooks/useProviderModels.ts | 1 + .../settings/ModelUsageSettings.tsx | 205 +++++++++++++ .../src/components/settings/SettingsView.tsx | 9 +- .../ui/hooks/useOpenRouterModelProviders.ts | 273 ++++++++---------- .../components/ui/hooks/useRouterModels.ts | 45 ++- webview-ui/src/i18n/locales/en/settings.json | 1 + 18 files changed, 1077 insertions(+), 411 deletions(-) create mode 100644 webview-ui/src/components/settings/ModelUsageSettings.tsx diff --git a/CHANGELOG.md b/CHANGELOG.md index 7f20bfcae2..eef5475856 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,11 +1,22 @@ # Changelog -## [v6.8.3] - 2026-09-02 +## [v6.8.4] - 2026-09-03 + +### Added + +- **Per-model usage visibility.** The usage dialog and a new Settings → Model Usage section now show each tracked OSS model's share of the shared plan pool as weekly/monthly percentages (no credit amounts), alongside the model's plan-cost multiplier badge (e.g. `5x cost`). The new settings section also renders the weekly/monthly plan windows with reset times and a refresh action, and is excluded from the save-button flow since it is read-only. Backend: `/axoncode/profile` now returns a `modelUsage` array (`model`, `multiplier`, `weeklyPercentage`, `monthlyPercentage`); charges for tracked OSS models are multiplied by their plan-cost multiplier before draining the shared pool. ### Changed +- **Dynamic model catalog synchronization.** Models are now dynamically fetched from the MatterAI backend `/v1/models` (or `/v1/web/models`) and registered into the client model registry (`registerDynamicKilocodeModels`), replacing static hardcoded lists while continuing to exclude deprecated axon models from active selection. Models automatically refresh on window focus, via the refresh button in the model selector, and through a 10-minute background poller. Cache bypass (`forceRefresh`) ensures database updates reflect immediately without restarting the extension. + - **OSS model catalog replaces Axon models.** The KiloCode model catalog (extension `kilocode-models.ts` and the webview `useOpenRouterModelProviders` copy) now exposes seven OSS models — `meta/muse-spark-1.2-contributor` (Muse Spark 1.2 Contributor), `deepseek/deepseek-v4-flash-0731` (DeepSeek V4 Flash), `zai/glm-5.3` (GLM 5.3), `zai/glm-5.3-flash` (GLM 5.3 Flash), `gpt-5.6-sol` (GPT-5.6 Sol), `gpt-5.6-luna` (GPT-5.6 Luna), and `gemini-3.7-flash` (Gemini 3.7 Flash) — in place of the Axon context-window variants. Each OSS model carries its published per-token pricing (Muse Spark 1.2 Contributor $0.10/M input, $0.002/M cache read, $0.20/M output; DeepSeek V4 Flash $0.14/M input, $0.028/M cache read, $0.28/M output; GLM 5.3 $1.40/M input, $0.14/M cache read, $4.40/M output; GLM 5.3 Flash $0.15/M input, $0.03/M cache read, $0.50/M output; GPT-5.6 Sol $5/M input, $0.50/M cache read, $30/M output; GPT-5.6 Luna $0.20/M input, $0.02/M cache read, $1.20/M output; Gemini 3.7 Flash $0.75/M input, $0.075/M cache read, $3.75/M output). The default model is `deepseek/deepseek-v4-flash-0731` (`openRouterDefaultModelId` in `@roo-code/types`, plus the CLI `kilocodeModel` defaults and web-evals `MODEL_DEFAULT`). Stored Axon model selections are now stale and reset to the default on next launch via the existing `isValidKilocodeModel` stale-model check. +### Fixed + +- **OpenRouter models no longer wiped on background refresh.** `refreshKilocodeModels` (window-focus refresh, 10-minute poller, startup fetch) previously posted `openrouter: {}` to the webview, blanking the OpenRouter model list until the next manual reload. It now fetches both `openrouter` and `kilocode-openrouter` in parallel (mirroring the `requestRouterModels` handler) and posts the merged payload, logging and skipping only the provider that failed. +- **Stale model selections reset on launch.** The dynamic catalog is now fetched once at provider startup (after `ContextProxy` initialization) and the webview state re-posted, so the `isValidKilocodeModel` stale-model check runs against a populated catalog instead of the empty-catalog bypass. Previously, a stored axon model survived until the first focus- or webview-triggered fetch completed. + ## [v6.8.2] - 2026-08-28 ### Added diff --git a/src/api/providers/__tests__/kilocode-models.spec.ts b/src/api/providers/__tests__/kilocode-models.spec.ts index dfaec86e25..d1ec044a42 100644 --- a/src/api/providers/__tests__/kilocode-models.spec.ts +++ b/src/api/providers/__tests__/kilocode-models.spec.ts @@ -1,15 +1,75 @@ -import { getKilocodeApiModelId, KILO_CODE_MODELS, type KiloCodeModel } from "../kilocode-models" - -const OSS_MODEL_IDS = [ - "meta/muse-spark-1.2-contributor", - "deepseek/deepseek-v4-flash-0731", - "zai/glm-5.3", - "zai/glm-5.3-flash", - "gpt-5.6-sol", - "gpt-5.6-luna", - "gemini-3.7-flash", +import { + getKilocodeApiModelId, + isValidKilocodeModel, + KILO_CODE_MODELS, + registerDynamicKilocodeModels, + type KiloCodeModel, +} from "../kilocode-models" + +// Raw catalog entries as served by the MatterAI backend /v1/web/models +// endpoint (non-axon models only). The static catalog ships empty; models +// arrive at runtime via registerDynamicKilocodeModels. +const RAW_OSS_MODELS: Array> = [ + { + id: "meta/muse-spark-1.3-contributor", + name: "Muse Spark 1.3 Contributor", + description: "Meta Muse Spark 1.3 Contributor is an open general purpose model for everyday coding tasks.", + context_length: 232000, + owned_by: "meta", + pricing: { prompt: "0.0000001", completion: "0.0000002", input_cache_reads: "0.000000002" }, + }, + { + id: "deepseek/deepseek-v4-flash-0731", + name: "DeepSeek V4 Flash", + description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", + context_length: 232000, + owned_by: "deepseek", + pricing: { prompt: "0.00000014", completion: "0.00000028", input_cache_reads: "0.000000028" }, + }, + { + id: "zai/glm-5.3", + name: "GLM 5.3", + description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", + context_length: 232000, + owned_by: "zai", + pricing: { prompt: "0.0000014", completion: "0.0000044", input_cache_reads: "0.00000014" }, + }, + { + id: "zai/glm-5.3-flash", + name: "GLM 5.3 Flash", + description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", + context_length: 232000, + owned_by: "zai", + pricing: { prompt: "0.00000015", completion: "0.0000005", input_cache_reads: "0.00000003" }, + }, + { + id: "gpt-5.6-sol", + name: "GPT-5.6 Sol", + description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", + context_length: 232000, + owned_by: "openai", + pricing: { prompt: "0.000005", completion: "0.00003", input_cache_reads: "0.0000005" }, + }, + { + id: "gpt-5.6-luna", + name: "GPT-5.6 Luna", + description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", + context_length: 232000, + owned_by: "openai", + pricing: { prompt: "0.0000002", completion: "0.0000012", input_cache_reads: "0.00000002" }, + }, + { + id: "gemini-3.7-flash", + name: "Gemini 3.7 Flash", + description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", + context_length: 232000, + owned_by: "google", + pricing: { prompt: "0.00000075", completion: "0.00000375", input_cache_reads: "0.000000075" }, + }, ] +const OSS_MODEL_IDS = RAW_OSS_MODELS.map((raw) => raw.id as string) + const getSharedMetadata = ({ id: _id, name: _name, @@ -21,14 +81,22 @@ const getSharedMetadata = ({ ...metadata }: KiloCodeModel) => metadata -describe("KiloCode OSS model catalog", () => { - it("exposes exactly the seven OSS models", () => { +describe("KiloCode dynamic model catalog", () => { + // KILO_CODE_MODELS is module-level mutable state shared across tests in + // this file; seed the backend catalog before each test. + beforeEach(() => { + registerDynamicKilocodeModels(RAW_OSS_MODELS) + }) + + it("registers exactly the backend catalog models", () => { expect(Object.keys(KILO_CODE_MODELS).sort()).toEqual([...OSS_MODEL_IDS].sort()) }) - it("no longer exposes axon models", () => { - expect(KILO_CODE_MODELS["axon-auto-232k"]).toBeUndefined() - expect(KILO_CODE_MODELS["axon-eido-3.2-code-pro-400k"]).toBeUndefined() + it("skips axon models and entries without an id", () => { + registerDynamicKilocodeModels([{ id: "axon-eido-3.2-code-flash", name: "Axon" }, { name: "no id" }]) + + expect(KILO_CODE_MODELS["axon-eido-3.2-code-flash"]).toBeUndefined() + expect(Object.keys(KILO_CODE_MODELS).sort()).toEqual([...OSS_MODEL_IDS].sort()) }) it.each([...OSS_MODEL_IDS])("sends %s to the API unchanged", (modelId) => { @@ -37,11 +105,16 @@ describe("KiloCode OSS model catalog", () => { expect(model).toBeDefined() expect(model?.id).toBe(modelId) expect(model?.context_length).toBe(232000) + expect(model?.openrouter.slug).toBe(modelId) expect(getKilocodeApiModelId(modelId)).toBe(modelId) }) - it("prices each OSS model at its published rate", () => { - expect(KILO_CODE_MODELS["meta/muse-spark-1.2-contributor"]?.pricing).toMatchObject({ + it("passes unknown ids through unchanged", () => { + expect(getKilocodeApiModelId("acme/unknown")).toBe("acme/unknown") + }) + + it("prices each model at its published rate", () => { + expect(KILO_CODE_MODELS["meta/muse-spark-1.3-contributor"]?.pricing).toMatchObject({ prompt: "0.0000001", completion: "0.0000002", input_cache_reads: "0.000000002", @@ -86,7 +159,7 @@ describe("KiloCode OSS model catalog", () => { } }) - it("shares identical base metadata across all OSS models", () => { + it("shares identical base metadata across all models", () => { const models = OSS_MODEL_IDS.map((modelId) => KILO_CODE_MODELS[modelId]!) const [first, ...rest] = models.map(getSharedMetadata) @@ -94,4 +167,64 @@ describe("KiloCode OSS model catalog", () => { expect(metadata).toEqual(first) } }) + + it("falls back to sane defaults for sparse entries", () => { + registerDynamicKilocodeModels([{ id: "acme/prototype" }]) + + expect(KILO_CODE_MODELS["acme/prototype"]).toMatchObject({ + name: "acme/prototype", + description: "acme/prototype via MatterAI", + context_length: 232000, + max_output_length: 64000, + owned_by: "acme", + input_modalities: ["text", "image"], + output_modalities: ["text"], + }) + expect(KILO_CODE_MODELS["acme/prototype"].pricing).toMatchObject({ + prompt: "0", + completion: "0", + input_cache_reads: "0", + input_cache_writes: "0", + }) + }) + + it("coerces numeric pricing to strings", () => { + registerDynamicKilocodeModels([ + { + id: "acme/numeric", + pricing: { prompt: 0.0000001, completion: 0.0000002, input_cache_reads: 0.000000002 }, + }, + ]) + + // String() renders sub-1e-6 numbers in scientific notation; parsePrice + // consumes these via parseFloat, so the coercion stays lossless. + expect(KILO_CODE_MODELS["acme/numeric"].pricing).toMatchObject({ + prompt: "1e-7", + completion: "2e-7", + input_cache_reads: "2e-9", + }) + }) + + it("validates registered models and rejects stale non-auto axon selections", () => { + expect(isValidKilocodeModel("zai/glm-5.3")).toBe(true) + expect(isValidKilocodeModel("axon-eido-3.2-code-flash")).toBe(false) + expect(isValidKilocodeModel("unknown/model")).toBe(false) + }) + + it("always accepts axon-auto router ids", () => { + expect(isValidKilocodeModel("axon-auto")).toBe(true) + expect(isValidKilocodeModel("axon-auto-400k")).toBe(true) + }) +}) + +describe("isValidKilocodeModel with an empty catalog", () => { + it("accepts any model before the first backend fetch", () => { + for (const key of Object.keys(KILO_CODE_MODELS)) { + delete KILO_CODE_MODELS[key] + } + + expect(Object.keys(KILO_CODE_MODELS).length).toBe(0) + expect(isValidKilocodeModel("axon-eido-3.2-code-flash")).toBe(true) + expect(isValidKilocodeModel("anything")).toBe(true) + }) }) diff --git a/src/api/providers/fetchers/modelCache.ts b/src/api/providers/fetchers/modelCache.ts index 4c68f17a22..b93f102164 100644 --- a/src/api/providers/fetchers/modelCache.ts +++ b/src/api/providers/fetchers/modelCache.ts @@ -57,9 +57,9 @@ export /*kilocode_change*/ async function readModels(router: RouterName): Promis * @returns The models from the cache or the fetched models. */ export const getModels = async (options: GetModelsOptions): Promise => { - const { provider } = options + const { provider, forceRefresh } = options - let models = getModelsFromCache(provider) + let models = !forceRefresh ? getModelsFromCache(provider) : undefined if (models) { return models @@ -71,7 +71,10 @@ export const getModels = async (options: GetModelsOptions): Promise // forked_change start: base url and bearer token models = await getOpenRouterModels({ openRouterBaseUrl: options.baseUrl, - headers: options.apiKey ? { Authorization: `Bearer ${options.apiKey}` } : undefined, + headers: { + ...(options.apiKey ? { Authorization: `Bearer ${options.apiKey}` } : {}), + ...(forceRefresh ? { "Cache-Control": "no-cache" } : {}), + }, }) // forked_change end break @@ -96,9 +99,19 @@ export const getModels = async (options: GetModelsOptions): Promise ? `https://api.matterai.so/organizations/${options.kilocodeOrganizationId}` : "https://api.matterai.so/v1/web" const openRouterBaseUrl = getKiloUrlFromToken(backendUrl, options.kilocodeToken ?? "") + const headers: Record = {} + if (options.kilocodeToken) { + headers["Authorization"] = `Bearer ${options.kilocodeToken}` + } + if (options.kilocodeOrganizationId) { + headers["X-KILOCODE-ORGANIZATIONID"] = options.kilocodeOrganizationId + } + if (forceRefresh) { + headers["Cache-Control"] = "no-cache" + } models = await getOpenRouterModels({ openRouterBaseUrl, - headers: options.kilocodeToken ? { Authorization: `Bearer ${options.kilocodeToken}` } : undefined, + headers, }) break } diff --git a/src/api/providers/fetchers/openrouter.ts b/src/api/providers/fetchers/openrouter.ts index 943897cbda..576a671715 100644 --- a/src/api/providers/fetchers/openrouter.ts +++ b/src/api/providers/fetchers/openrouter.ts @@ -133,7 +133,7 @@ export async function getOpenRouterModels( const models: Record = {} // First, add static models from KILO_CODE_MODELS - const { KILO_CODE_MODELS } = await import("../kilocode-models") + const { KILO_CODE_MODELS, registerDynamicKilocodeModels } = await import("../kilocode-models") for (const [id, model] of Object.entries(KILO_CODE_MODELS)) { models[id] = parseOpenRouterModel({ @@ -159,59 +159,67 @@ export async function getOpenRouterModels( } // Then, fetch dynamic models from MatterAI API - // try { - // const headers: Record = { - // ...DEFAULT_HEADERS, - // ...options?.headers, - // } - - // const response = await axios.get("https://api.matterai.so/v1/models/openrouter", { headers }) - - // const rawModels = response.data?.data || [] - - // for (const rawModel of rawModels) { - // // Filter out image generation models (only text output) - // const outputModalities = rawModel.output_modalities || [] - // if (outputModalities.includes("image") && !outputModalities.includes("text")) { - // continue - // } - - // // Prefix the model ID with "openrouter/" - // const modelId = `openrouter/${rawModel.id}` - - // // Don't override static models - // if (models[modelId]) { - // continue - // } - - // models[modelId] = parseOpenRouterModel({ - // id: modelId, - // model: { - // name: rawModel.name || rawModel.id, - // description: rawModel.description, - // context_length: rawModel.context_length || 8192, - // max_completion_tokens: rawModel.max_output_length, - // pricing: rawModel.pricing - // ? { - // prompt: rawModel.pricing.prompt, - // completion: rawModel.pricing.completion, - // input_cache_read: rawModel.pricing.input_cache_reads, - // input_cache_write: rawModel.pricing.input_cache_writes, - // } - // : undefined, - // }, - // displayName: rawModel.name, - // inputModality: rawModel.input_modalities, - // outputModality: rawModel.output_modalities, - // maxTokens: rawModel.max_output_length, - // supportedParameters: rawModel.supported_sampling_parameters, - // }) - // } - // } catch (error) { - // console.error( - // `Error fetching OpenRouter models from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`, - // ) - // } + try { + const headers: Record = { + ...DEFAULT_HEADERS, + ...options?.headers, + } + + const base = (options?.openRouterBaseUrl || "https://api.matterai.so/v1/web").replace(/\/+$/, "") + let url = base.endsWith("/models") ? base : `${base}/models` + if (headers["Cache-Control"] === "no-cache") { + url += (url.includes("?") ? "&" : "?") + "forceRefresh=true" + } + + const response = await axios.get(url, { headers }) + + const rawModels = response.data?.data || [] + + if (Array.isArray(rawModels) && rawModels.length > 0) { + registerDynamicKilocodeModels(rawModels) + } + + for (const rawModel of rawModels) { + if (!rawModel?.id || rawModel.id.startsWith("axon-")) { + continue + } + + // Filter out image generation models (only text output) + const outputModalities = rawModel.output_modalities || [] + if (outputModalities.includes("image") && !outputModalities.includes("text")) { + continue + } + + const modelId = rawModel.id + + models[modelId] = parseOpenRouterModel({ + id: modelId, + model: { + name: rawModel.name || rawModel.id, + description: rawModel.description, + context_length: rawModel.context_length || 232000, + max_completion_tokens: rawModel.max_output_length, + pricing: rawModel.pricing + ? { + prompt: rawModel.pricing.prompt, + completion: rawModel.pricing.completion, + input_cache_read: rawModel.pricing.input_cache_reads, + input_cache_write: rawModel.pricing.input_cache_writes, + } + : undefined, + }, + displayName: rawModel.name, + inputModality: rawModel.input_modalities, + outputModality: rawModel.output_modalities, + maxTokens: rawModel.max_output_length, + supportedParameters: rawModel.supported_sampling_parameters, + }) + } + } catch (error) { + console.error( + `Error fetching OpenRouter models from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`, + ) + } return models } @@ -227,7 +235,7 @@ export async function getOpenRouterModelEndpoints( const models: Record = {} // First, check static models from KILO_CODE_MODELS - const { KILO_CODE_MODELS } = await import("../kilocode-models") + const { KILO_CODE_MODELS, registerDynamicKilocodeModels } = await import("../kilocode-models") const staticModel = KILO_CODE_MODELS[modelId] if (staticModel) { @@ -255,60 +263,70 @@ export async function getOpenRouterModelEndpoints( } // If not found in static models, fetch from MatterAI API - // try { - // const headers: Record = { - // ...DEFAULT_HEADERS, - // ...options?.headers, - // } - - // const response = await axios.get("https://api.matterai.so/v1/models/openrouter", { headers }) - - // const rawModels = response.data?.data || [] - - // // Strip "openrouter/" prefix if present for matching - // const strippedModelId = modelId.replace(/^openrouter\//, "") - - // const rawModel = (rawModels as MatterAiOpenRouterModel[]).find((m) => m.id === strippedModelId) - - // if (!rawModel) { - // return models - // } - - // // Filter out image generation models - // const outputModalities = rawModel.output_modalities || [] - // if (outputModalities.includes("image") && !outputModalities.includes("text")) { - // return models - // } - - // const prefixedModelId = `openrouter/${rawModel.id}` - - // models["MatterAI"] = parseOpenRouterModel({ - // id: prefixedModelId, - // model: { - // name: rawModel.name || rawModel.id, - // description: rawModel.description, - // context_length: rawModel.context_length || 8192, - // max_completion_tokens: rawModel.max_output_length, - // pricing: rawModel.pricing - // ? { - // prompt: rawModel.pricing.prompt, - // completion: rawModel.pricing.completion, - // input_cache_read: rawModel.pricing.input_cache_reads, - // input_cache_write: rawModel.pricing.input_cache_writes, - // } - // : undefined, - // }, - // displayName: rawModel.name, - // inputModality: rawModel.input_modalities, - // outputModality: rawModel.output_modalities, - // maxTokens: rawModel.max_output_length, - // supportedParameters: rawModel.supported_sampling_parameters, - // }) - // } catch (error) { - // console.error( - // `Error fetching OpenRouter model endpoints from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`, - // ) - // } + try { + const headers: Record = { + ...DEFAULT_HEADERS, + ...options?.headers, + } + + const base = (options?.openRouterBaseUrl || "https://api.matterai.so/v1/web").replace(/\/+$/, "") + let url = base.endsWith("/models") ? base : `${base}/models` + if (headers["Cache-Control"] === "no-cache") { + url += (url.includes("?") ? "&" : "?") + "forceRefresh=true" + } + + const response = await axios.get(url, { headers }) + + const rawModels = response.data?.data || [] + + if (Array.isArray(rawModels) && rawModels.length > 0) { + registerDynamicKilocodeModels(rawModels) + } + + // Strip "openrouter/" prefix if present for matching + const strippedModelId = modelId.replace(/^openrouter\//, "") + + const rawModel = (rawModels as MatterAiOpenRouterModel[]).find( + (m) => m.id === strippedModelId || m.id === modelId, + ) + + if (!rawModel) { + return models + } + + // Filter out image generation models + const outputModalities = rawModel.output_modalities || [] + if (outputModalities.includes("image") && !outputModalities.includes("text")) { + return models + } + + models["MatterAI"] = parseOpenRouterModel({ + id: rawModel.id, + model: { + name: rawModel.name || rawModel.id, + description: rawModel.description, + context_length: rawModel.context_length || 232000, + max_completion_tokens: rawModel.max_output_length, + pricing: rawModel.pricing + ? { + prompt: rawModel.pricing.prompt, + completion: rawModel.pricing.completion, + input_cache_read: rawModel.pricing.input_cache_reads, + input_cache_write: rawModel.pricing.input_cache_writes, + } + : undefined, + }, + displayName: rawModel.name, + inputModality: rawModel.input_modalities, + outputModality: rawModel.output_modalities, + maxTokens: rawModel.max_output_length, + supportedParameters: rawModel.supported_sampling_parameters, + }) + } catch (error) { + console.error( + `Error fetching OpenRouter model endpoints from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`, + ) + } return models } diff --git a/src/api/providers/kilocode-models.ts b/src/api/providers/kilocode-models.ts index 443e3a1bbc..878077fad3 100644 --- a/src/api/providers/kilocode-models.ts +++ b/src/api/providers/kilocode-models.ts @@ -61,118 +61,118 @@ const OSS_MODEL_BASE: KiloCodeModelVariant = { } export const KILO_CODE_MODELS: Record = { - "meta/muse-spark-1.2-contributor": { - ...OSS_MODEL_BASE, - id: "meta/muse-spark-1.2-contributor", - name: "Muse Spark 1.2 Contributor", - description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.", - context_length: 232000, - owned_by: "meta", - openrouter: { slug: "meta/muse-spark-1.2-contributor" }, - // $0.10/M input, $0.002/M cache read, $0.20/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.0000001", - completion: "0.0000002", - input_cache_reads: "0.000000002", - }, - }, - "deepseek/deepseek-v4-flash-0731": { - ...OSS_MODEL_BASE, - id: "deepseek/deepseek-v4-flash-0731", - name: "DeepSeek V4 Flash", - description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", - context_length: 232000, - owned_by: "deepseek", - openrouter: { slug: "deepseek/deepseek-v4-flash-0731" }, - // $0.14/M input, $0.028/M cache read, $0.28/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.00000014", - completion: "0.00000028", - input_cache_reads: "0.000000028", - }, - }, - "zai/glm-5.3-flash": { - ...OSS_MODEL_BASE, - id: "zai/glm-5.3-flash", - name: "GLM 5.3 Flash", - description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", - context_length: 232000, - owned_by: "zai", - openrouter: { slug: "zai/glm-5.3-flash" }, - // $0.15/M input, $0.03/M cache read, $0.50/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.00000015", - completion: "0.0000005", - input_cache_reads: "0.00000003", - }, - }, - "zai/glm-5.3": { - ...OSS_MODEL_BASE, - id: "zai/glm-5.3", - name: "GLM 5.3", - description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", - context_length: 232000, - owned_by: "zai", - openrouter: { slug: "zai/glm-5.3" }, - // $1.40/M input, $0.14/M cache read, $4.40/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.0000014", - completion: "0.0000044", - input_cache_reads: "0.00000014", - }, - }, - "gpt-5.6-luna": { - ...OSS_MODEL_BASE, - id: "gpt-5.6-luna", - name: "GPT-5.6 Luna", - description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", - context_length: 232000, - owned_by: "openai", - openrouter: { slug: "gpt-5.6-luna" }, - // $0.20/M input, $0.02/M cache read, $1.20/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.0000002", - completion: "0.0000012", - input_cache_reads: "0.00000002", - }, - }, - "gpt-5.6-sol": { - ...OSS_MODEL_BASE, - id: "gpt-5.6-sol", - name: "GPT-5.6 Sol", - description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", - context_length: 232000, - owned_by: "openai", - openrouter: { slug: "gpt-5.6-sol" }, - // $5/M input, $0.50/M cache read, $30/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.000005", - completion: "0.00003", - input_cache_reads: "0.0000005", - }, - }, - "gemini-3.7-flash": { - ...OSS_MODEL_BASE, - id: "gemini-3.7-flash", - name: "Gemini 3.7 Flash", - description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", - context_length: 232000, - owned_by: "google", - openrouter: { slug: "gemini-3.7-flash" }, - // $0.75/M input, $0.075/M cache read, $3.75/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.00000075", - completion: "0.00000375", - input_cache_reads: "0.000000075", - }, - }, + // "meta/muse-spark-1.2-contributor": { + // ...OSS_MODEL_BASE, + // id: "meta/muse-spark-1.2-contributor", + // name: "Muse Spark 1.2 Contributor", + // description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "meta", + // openrouter: { slug: "meta/muse-spark-1.2-contributor" }, + // // $0.10/M input, $0.002/M cache read, $0.20/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.0000001", + // completion: "0.0000002", + // input_cache_reads: "0.000000002", + // }, + // }, + // "deepseek/deepseek-v4-flash-0731": { + // ...OSS_MODEL_BASE, + // id: "deepseek/deepseek-v4-flash-0731", + // name: "DeepSeek V4 Flash", + // description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", + // context_length: 232000, + // owned_by: "deepseek", + // openrouter: { slug: "deepseek/deepseek-v4-flash-0731" }, + // // $0.14/M input, $0.028/M cache read, $0.28/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.00000014", + // completion: "0.00000028", + // input_cache_reads: "0.000000028", + // }, + // }, + // "zai/glm-5.3-flash": { + // ...OSS_MODEL_BASE, + // id: "zai/glm-5.3-flash", + // name: "GLM 5.3 Flash", + // description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "zai", + // openrouter: { slug: "zai/glm-5.3-flash" }, + // // $0.15/M input, $0.03/M cache read, $0.50/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.00000015", + // completion: "0.0000005", + // input_cache_reads: "0.00000003", + // }, + // }, + // "zai/glm-5.3": { + // ...OSS_MODEL_BASE, + // id: "zai/glm-5.3", + // name: "GLM 5.3", + // description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", + // context_length: 232000, + // owned_by: "zai", + // openrouter: { slug: "zai/glm-5.3" }, + // // $1.40/M input, $0.14/M cache read, $4.40/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.0000014", + // completion: "0.0000044", + // input_cache_reads: "0.00000014", + // }, + // }, + // "gpt-5.6-luna": { + // ...OSS_MODEL_BASE, + // id: "gpt-5.6-luna", + // name: "GPT-5.6 Luna", + // description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "openai", + // openrouter: { slug: "gpt-5.6-luna" }, + // // $0.20/M input, $0.02/M cache read, $1.20/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.0000002", + // completion: "0.0000012", + // input_cache_reads: "0.00000002", + // }, + // }, + // "gpt-5.6-sol": { + // ...OSS_MODEL_BASE, + // id: "gpt-5.6-sol", + // name: "GPT-5.6 Sol", + // description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", + // context_length: 232000, + // owned_by: "openai", + // openrouter: { slug: "gpt-5.6-sol" }, + // // $5/M input, $0.50/M cache read, $30/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.000005", + // completion: "0.00003", + // input_cache_reads: "0.0000005", + // }, + // }, + // "gemini-3.7-flash": { + // ...OSS_MODEL_BASE, + // id: "gemini-3.7-flash", + // name: "Gemini 3.7 Flash", + // description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "google", + // openrouter: { slug: "gemini-3.7-flash" }, + // // $0.75/M input, $0.075/M cache read, $3.75/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.00000075", + // completion: "0.00000375", + // input_cache_reads: "0.000000075", + // }, + // }, } /** @@ -180,6 +180,12 @@ export const KILO_CODE_MODELS: Record = { * Used to detect stale model selections after extension updates remove models. */ export function isValidKilocodeModel(modelId: string): boolean { + if (Object.keys(KILO_CODE_MODELS).length === 0) { + return true + } + if (modelId.startsWith("axon-auto")) { + return true + } return modelId in KILO_CODE_MODELS } @@ -190,3 +196,44 @@ export function isValidKilocodeModel(modelId: string): boolean { export function getKilocodeApiModelId(modelId: string): string { return KILO_CODE_MODELS[modelId]?.id ?? modelId } + +/** + * Registers models fetched dynamically from the MatterAI backend catalog into KILO_CODE_MODELS. + */ +export function registerDynamicKilocodeModels(rawModels: Array>): void { + for (const raw of rawModels) { + if (!raw?.id || raw.id.startsWith("axon-")) continue + const pricingObj = raw.pricing || {} + KILO_CODE_MODELS[raw.id] = { + ...OSS_MODEL_BASE, + id: raw.id, + name: raw.name || raw.id, + description: raw.description || `${raw.name || raw.id} via MatterAI`, + context_length: raw.context_length || 232000, + max_output_length: raw.max_output_length || 64000, + input_modalities: raw.input_modalities || ["text", "image"], + output_modalities: raw.output_modalities || ["text"], + supported_sampling_parameters: + raw.supported_sampling_parameters || OSS_MODEL_BASE.supported_sampling_parameters, + supported_features: raw.supported_features || OSS_MODEL_BASE.supported_features, + owned_by: raw.owned_by || raw.id.split("/")[0] || "matterai", + openrouter: { slug: raw.id }, + pricing: { + ...OSS_MODEL_BASE.pricing, + prompt: typeof pricingObj.prompt === "string" ? pricingObj.prompt : String(pricingObj.prompt ?? "0"), + completion: + typeof pricingObj.completion === "string" + ? pricingObj.completion + : String(pricingObj.completion ?? "0"), + input_cache_reads: + typeof pricingObj.input_cache_reads === "string" + ? pricingObj.input_cache_reads + : String(pricingObj.input_cache_reads ?? "0"), + input_cache_writes: + typeof pricingObj.input_cache_writes === "string" + ? pricingObj.input_cache_writes + : String(pricingObj.input_cache_writes ?? "0"), + }, + } + } +} diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts index 73ff9cf04f..c5ec1c13ca 100644 --- a/src/core/webview/ClineProvider.ts +++ b/src/core/webview/ClineProvider.ts @@ -107,6 +107,8 @@ import { stringifyError } from "../../shared/kilocode/errorUtils" import isWsl from "is-wsl" import { getKilocodeDefaultModel } from "../../api/providers/kilocode/getKilocodeDefaultModel" import { isValidKilocodeModel } from "../../api/providers/kilocode-models" +import { getModels, flushModels } from "../../api/providers/fetchers/modelCache" +import { type RouterName, type ModelRecord, type GetModelsOptions } from "../../shared/api" import { getKiloCodeWrapperProperties } from "../../core/kilocode/wrapper" import { getKiloUrlFromToken } from "@roo-code/types" // kilocode_change import { getKilocodeConfig, getWorkspaceProjectId, KilocodeConfig } from "../../utils/kilo-config-file" // kilocode_change @@ -167,6 +169,7 @@ export class ClineProvider private usageGitHead?: string private usageGitReportInFlight = false private settingsStateSyncTimer?: NodeJS.Timeout + private modelsRefreshInterval?: NodeJS.Timeout private recentTasksCache?: string[] private pendingOperations: Map = new Map() @@ -326,15 +329,46 @@ export class ClineProvider } // Multi-window synchronization: refresh secrets and global state when window gains focus + let lastModelFocusRefresh = 0 const windowStateDisposable = vscode.window.onDidChangeWindowState(async (e) => { if (e.focused && this.contextProxy.isInitialized) { await this.contextProxy.refreshSecrets() await this.contextProxy.refreshGlobalState() await this.postStateToWebview() + + const now = Date.now() + if (now - lastModelFocusRefresh > 5000) { + lastModelFocusRefresh = now + this.refreshKilocodeModels({ forceRefresh: true }).catch((err) => + this.log(`Error refreshing models on focus: ${err}`), + ) + } } }) this.disposables.push(windowStateDisposable) + // Background 10-minute periodic model check + this.modelsRefreshInterval = setInterval( + () => { + this.refreshKilocodeModels({ forceRefresh: true }).catch((err) => + this.log(`Error in 10-min background models refresh: ${err}`), + ) + }, + 10 * 60 * 1000, + ) + this.modelsRefreshInterval.unref?.() + + // kilocode_change: populate the dynamic model catalog once at startup so + // stale model selections (e.g. removed axon models) are validated and reset + // by getState() as soon as the catalog is available. Without this, + // isValidKilocodeModel() treats every model as valid until the first fetch + // completes, letting stale selections survive the launch. + if (this.contextProxy.isInitialized) { + this.refreshKilocodeModels() + .then(() => this.postStateToWebview()) + .catch((err) => this.log(`Error during startup models refresh: ${err}`)) + } + // kilocode_change: Setup git HEAD watcher for real-time branch updates this.setupGitHeadWatcher() @@ -504,6 +538,68 @@ export class ClineProvider } } + public async refreshKilocodeModels(options: { forceRefresh?: boolean } = {}): Promise { + try { + const { apiConfiguration } = await this.getState() + if (options.forceRefresh) { + await flushModels("kilocode-openrouter") + await flushModels("openrouter") + } + + // forked_change start: fetch both openrouter and kilocode-openrouter + // Mirrors the requestRouterModels handler so the webview receives the + // full routerModels payload; posting an empty openrouter map would + // wipe the OpenRouter model list in the webview on every refresh. + const modelFetchPromises: Array<{ key: RouterName; options: GetModelsOptions }> = [ + { + key: "openrouter", + options: { + provider: "openrouter", + apiKey: apiConfiguration?.openRouterApiKey, + baseUrl: apiConfiguration?.openRouterBaseUrl, + forceRefresh: options.forceRefresh, + }, + }, + { + key: "kilocode-openrouter", + options: { + provider: "kilocode-openrouter", + kilocodeToken: apiConfiguration?.kilocodeToken, + kilocodeOrganizationId: apiConfiguration?.kilocodeOrganizationId, + forceRefresh: options.forceRefresh, + }, + }, + ] + + const results = await Promise.allSettled( + modelFetchPromises.map(async ({ options }) => await getModels(options)), + ) + + const routerModels: Record = { + openrouter: {}, + "kilocode-openrouter": {}, + } + + results.forEach((result, index) => { + if (result.status === "fulfilled") { + routerModels[modelFetchPromises[index].key] = result.value + } else { + this.log( + `Failed to refresh ${modelFetchPromises[index].key} models: ${result.reason instanceof Error ? result.reason.message : String(result.reason)}`, + ) + } + }) + + this.postMessageToWebview({ + type: "routerModels", + routerModels, + }) + // forked_change end + } catch (error) { + this.log(`Failed to refresh kilocode models: ${error}`) + } + } + /** * Synchronize cloud profiles with local profiles. */ @@ -905,6 +1001,8 @@ export class ClineProvider this.usageGitRefsWatcher = undefined if (this.usageGitPoller) clearInterval(this.usageGitPoller) this.usageGitPoller = undefined + if (this.modelsRefreshInterval) clearInterval(this.modelsRefreshInterval) + this.modelsRefreshInterval = undefined await this.mcpHub?.unregisterClient() this.mcpHub = undefined this.marketplaceManager?.cleanup() @@ -2700,6 +2798,19 @@ ${prompt} providerSettings.apiProvider = apiProvider } + // Validate global kilocodeModel against available models. + if (providerSettings?.apiProvider === "kilocode" && providerSettings?.kilocodeModel) { + if (!isValidKilocodeModel(providerSettings.kilocodeModel)) { + const staleModel = providerSettings.kilocodeModel + const defaultModel = await getKilocodeDefaultModel() + providerSettings.kilocodeModel = defaultModel + await this.contextProxy.setProviderSettings(providerSettings) + this.log( + `[ModelValidation] Reset stale kilocodeModel "${staleModel}" to default "${defaultModel}" — model no longer available`, + ) + } + } + let organizationAllowList = ORGANIZATION_ALLOW_ALL try { diff --git a/src/core/webview/__tests__/ClineProvider.spec.ts b/src/core/webview/__tests__/ClineProvider.spec.ts index c6207a4af9..c8c3f7a390 100644 --- a/src/core/webview/__tests__/ClineProvider.spec.ts +++ b/src/core/webview/__tests__/ClineProvider.spec.ts @@ -2035,6 +2035,15 @@ describe("ClineProvider", () => { beforeEach(async () => { await provider.resolveWebviewView(mockWebviewView) logSpy = vi.spyOn(provider, "log").mockImplementation(() => {}) + + // Seed the dynamic catalog the way the backend /v1/web/models fetch + // does at runtime, so isValidKilocodeModel validates against real + // entries instead of hitting the empty-catalog bypass. + const { registerDynamicKilocodeModels } = await import("../../../api/providers/kilocode-models") + registerDynamicKilocodeModels([ + { id: "deepseek/deepseek-v4-flash-0731", name: "DeepSeek V4 Flash" }, + { id: "zai/glm-5.3", name: "GLM 5.3" }, + ]) }) afterEach(() => { diff --git a/src/core/webview/webviewMessageHandler.ts b/src/core/webview/webviewMessageHandler.ts index 01f733b414..721d7f63c5 100644 --- a/src/core/webview/webviewMessageHandler.ts +++ b/src/core/webview/webviewMessageHandler.ts @@ -1905,11 +1905,17 @@ ${comment.suggestion} // forked_change start: openrouter auth, kilocode provider const openRouterApiKey = apiConfiguration.openRouterApiKey || message?.values?.openRouterApiKey const openRouterBaseUrl = apiConfiguration.openRouterBaseUrl || message?.values?.openRouterBaseUrl + const forceRefresh = Boolean(message?.values?.forceRefresh) const modelFetchPromises: Array<{ key: RouterName; options: GetModelsOptions }> = [ { key: "openrouter", - options: { provider: "openrouter", apiKey: openRouterApiKey, baseUrl: openRouterBaseUrl }, + options: { + provider: "openrouter", + apiKey: openRouterApiKey, + baseUrl: openRouterBaseUrl, + forceRefresh, + }, }, { key: "kilocode-openrouter", @@ -1917,6 +1923,7 @@ ${comment.suggestion} provider: "kilocode-openrouter", kilocodeToken: apiConfiguration.kilocodeToken, kilocodeOrganizationId: apiConfiguration.kilocodeOrganizationId, + forceRefresh, }, }, ] diff --git a/src/shared/WebviewMessage.ts b/src/shared/WebviewMessage.ts index 6b6fb9331e..f19ad27e4c 100644 --- a/src/shared/WebviewMessage.ts +++ b/src/shared/WebviewMessage.ts @@ -514,6 +514,10 @@ export type ProfileData = { // Tiered usage windows (weekly / monthly). Each window is expressed // as a fraction of the user's monthly plan limit. tieredUsage?: AxonCodeTieredUsage + // Per-model usage for the tracked OSS models. Each entry attributes the + // model's share of the shared plan pool as percentages of the same + // weekly/monthly windows (no credit amounts are exposed). + modelUsage?: AxonCodeModelUsage[] weeklyReset?: AxonCodeWeeklyResetAvailability // Overage lets the plan keep running on shared org API credits once the // plan windows hit 98%. `enabled` is true only when the org has turned @@ -558,6 +562,17 @@ export interface AxonCodeTieredUsage { monthly: AxonCodeWindowUsage } +export interface AxonCodeModelUsage { + // OSS catalog model id (e.g. "zai/glm-5.3"). + model: string + // Plan-cost multiplier applied to this model's charges (e.g. 5 for + // deepseek, 1 for untracked/axon models). + multiplier: number + // Share of the shared weekly / monthly plan windows, as percentages. + weeklyPercentage: number + monthlyPercentage: number +} + export interface ProfileDataResponsePayload { success: boolean data?: ProfileData diff --git a/src/shared/api.ts b/src/shared/api.ts index 78b047edd7..846a2384b0 100644 --- a/src/shared/api.ts +++ b/src/shared/api.ts @@ -148,6 +148,7 @@ export const getModelMaxOutputTokens = ({ type CommonFetchParams = { apiKey?: string baseUrl?: string + forceRefresh?: boolean } // Exhaustive, value-level map for all dynamic providers. diff --git a/webview-ui/src/components/chat/UsageDialog.tsx b/webview-ui/src/components/chat/UsageDialog.tsx index 0c2f9cc64e..b734560eda 100644 --- a/webview-ui/src/components/chat/UsageDialog.tsx +++ b/webview-ui/src/components/chat/UsageDialog.tsx @@ -3,7 +3,7 @@ import { Dialog, DialogContent, DialogHeader, DialogTitle, DialogDescription } f import { vscode } from "@/utils/vscode" import { useExtensionState } from "@/context/ExtensionStateContext" import { ProfileData, WebviewMessage, AxonCodeTieredUsage } from "@roo/WebviewMessage" -import { Activity, Gauge, Sparkles, Wallet, ShieldCheck, Layers } from "lucide-react" +import { Activity, Cpu, Gauge, Sparkles, Wallet, ShieldCheck, Layers } from "lucide-react" interface UsageDialogProps { open: boolean @@ -306,6 +306,56 @@ export const UsageDialog: React.FC = ({ )} + + {/* 3. MODEL USAGE */} + {profileData?.modelUsage && profileData.modelUsage.length > 0 && ( +
+
+ + Model Usage +
+
+ Each model's share of your plan windows (weekly / monthly). +
+
+ {profileData.modelUsage.map((entry) => ( +
+
+ + {entry.model} + + + {entry.multiplier}x cost + +
+ {(["weekly", "monthly"] as const).map((window) => { + const raw = + window === "weekly" ? entry.weeklyPercentage : entry.monthlyPercentage + const pct = Math.max(0, Math.min(100, raw || 0)) + return ( +
+ + {window} + +
+
+
+ + {pct.toFixed(1)}% + +
+ ) + })} +
+ ))} +
+
+ )}
diff --git a/webview-ui/src/components/kilocode/chat/ModelSelector.tsx b/webview-ui/src/components/kilocode/chat/ModelSelector.tsx index 362e9b9cc9..fbd2999bb5 100644 --- a/webview-ui/src/components/kilocode/chat/ModelSelector.tsx +++ b/webview-ui/src/components/kilocode/chat/ModelSelector.tsx @@ -53,7 +53,6 @@ const sanitizeModelLabel = (modelId: string, provider: string): string => { const MODEL_QUALIFIER_PATTERN = /\s*(\((?:232k context|400k context|free)\))$/i -const isStandardContextWindow = (cw?: number): boolean => cw === 232000 const isExtendedContextWindow = (cw?: number): boolean => cw === 400000 const ModelLabel = ({ label }: { label: string }) => { @@ -86,8 +85,16 @@ export const ModelSelector = ({ }: ModelSelectorProps) => { const { t } = useAppTranslation() const { currentTaskItem } = useExtensionState() - const { provider, providerModels, providerDefaultModel, isLoading, isError, proModelIds, proModelsEnabled } = - useProviderModels(apiConfiguration) + const { + provider, + providerModels, + providerDefaultModel, + isLoading, + isError, + proModelIds, + proModelsEnabled, + refetchRouterModels, + } = useProviderModels(apiConfiguration) // Check if a third-party model is selected const thirdPartySelectedModel = apiConfiguration?.thirdPartySelectedModel @@ -131,8 +138,13 @@ export const ModelSelector = ({ const { data: ollamaModels, refetch: refetchOllamaModels } = useThirdPartyModels("ollama", ollamaEnabled) const { data: opencodeModels, refetch: refetchOpencodeModels } = useThirdPartyModels("opencode", opencodeEnabled) - // Refresh all third-party models + // Refresh all models (kilocode/openrouter + third-party) const handleRefreshModels = useCallback(() => { + vscode.postMessage({ type: "flushRouterModels", text: "kilocode-openrouter" }) + vscode.postMessage({ type: "requestRouterModels", values: { forceRefresh: true } }) + if (refetchRouterModels) { + refetchRouterModels() + } refetchMatterai3pModels() if (ollamaEnabled) { refetchOllamaModels() @@ -140,7 +152,14 @@ export const ModelSelector = ({ if (opencodeEnabled) { refetchOpencodeModels() } - }, [ollamaEnabled, opencodeEnabled, refetchMatterai3pModels, refetchOllamaModels, refetchOpencodeModels]) + }, [ + ollamaEnabled, + opencodeEnabled, + refetchMatterai3pModels, + refetchOllamaModels, + refetchOpencodeModels, + refetchRouterModels, + ]) // Separate matterai3p models (always shown after Axon models) const matterai3pOptions = useMemo(() => { @@ -202,10 +221,10 @@ export const ModelSelector = ({ // Filter Axon models to only include models corresponding to active context window const filteredAxonModelIds = modelsIds.filter((modelId) => { const cw = providerModels[modelId]?.contextWindow - if (isStandardContextWindow(cw) || isExtendedContextWindow(cw)) { - return contextMode === "400k" ? isExtendedContextWindow(cw) : isStandardContextWindow(cw) + if (contextMode === "400k") { + return isExtendedContextWindow(cw) } - return true + return !isExtendedContextWindow(cw) }) // Determine the active model ID representation for the selected context @@ -229,12 +248,14 @@ export const ModelSelector = ({ .concat(filteredAxonModelIds) .filter((modelId) => { const cw = providerModels[modelId]?.contextWindow + if (contextMode === "400k") { + return isExtendedContextWindow(cw) + } return ( - (contextMode === "400k" ? isExtendedContextWindow(cw) : isStandardContextWindow(cw)) || - (!cw && - !modelId.startsWith("matterai3p:") && - !modelId.startsWith("ollama:") && - !modelId.startsWith("opencode:")) + !isExtendedContextWindow(cw) && + !modelId.startsWith("matterai3p:") && + !modelId.startsWith("ollama:") && + !modelId.startsWith("opencode:") ) }) .map((modelId) => { diff --git a/webview-ui/src/components/kilocode/hooks/useProviderModels.ts b/webview-ui/src/components/kilocode/hooks/useProviderModels.ts index dd7a023e44..157359104a 100644 --- a/webview-ui/src/components/kilocode/hooks/useProviderModels.ts +++ b/webview-ui/src/components/kilocode/hooks/useProviderModels.ts @@ -326,6 +326,7 @@ export const useProviderModels = (apiConfiguration?: ProviderSettings) => { providerDefaultModel: defaultModel, isLoading: routerModels.isLoading, isError: routerModels.isError, + refetchRouterModels: routerModels.refetch, // kilocode_change: pro models support proModelIds, proModelsEnabled, diff --git a/webview-ui/src/components/settings/ModelUsageSettings.tsx b/webview-ui/src/components/settings/ModelUsageSettings.tsx new file mode 100644 index 0000000000..624ca41927 --- /dev/null +++ b/webview-ui/src/components/settings/ModelUsageSettings.tsx @@ -0,0 +1,205 @@ +import React, { useEffect, useState } from "react" +import { vscode } from "@/utils/vscode" +import { useExtensionState } from "@/context/ExtensionStateContext" +import { AxonCodeModelUsage, AxonCodeTieredUsage, ProfileData, WebviewMessage } from "@roo/WebviewMessage" +import { Cpu, Gauge, RefreshCw } from "lucide-react" + +function formatRelativeTime(isoStr?: string): string { + if (!isoStr) return "on session start" + const now = Date.now() + const target = new Date(isoStr).getTime() + if (Number.isNaN(target)) return "on session start" + const diff = target - now + if (diff <= 0) return "now" + const sec = Math.floor(diff / 1000) + const min = Math.floor(sec / 60) + const hrs = Math.floor(min / 60) + const days = Math.floor(hrs / 24) + if (days >= 1) return `in ${days} day${days > 1 ? "s" : ""}` + if (hrs >= 1) return `in ${hrs}h ${min % 60}m` + if (min >= 1) return `in ${min}m` + return "soon" +} + +const clampPercentage = (value: number | undefined) => Math.max(0, Math.min(100, value || 0)) + +const WindowBar: React.FC<{ label: string; percentage: number | undefined; resetsAt?: string }> = ({ + label, + percentage, + resetsAt, +}) => { + const pct = clampPercentage(percentage) + return ( +
+
+ {label} + {pct.toFixed(1)}% used +
+
+
+
+ {resetsAt && ( +
+ Resets {formatRelativeTime(resetsAt)} +
+ )} +
+ ) +} + +const ModelUsageRow: React.FC<{ entry: AxonCodeModelUsage }> = ({ entry }) => { + const weeklyPct = clampPercentage(entry.weeklyPercentage) + const monthlyPct = clampPercentage(entry.monthlyPercentage) + return ( +
+
+ {entry.model} + + {entry.multiplier}x limit + +
+ {( + [ + ["Weekly", weeklyPct], + ["Monthly", monthlyPct], + ] as const + ).map(([label, pct]) => ( +
+ + {label} + +
+
+
+ + {pct.toFixed(1)}% + +
+ ))} +
+ ) +} + +export const ModelUsageSettings = () => { + const { apiConfiguration } = useExtensionState() + const [profileData, setProfileData] = useState(null) + const [isLoading, setIsLoading] = useState(false) + + const requestUsage = React.useCallback(() => { + setIsLoading(true) + vscode.postMessage({ type: "fetchProfileDataRequest" }) + }, []) + + useEffect(() => { + requestUsage() + }, [apiConfiguration?.kilocodeToken, requestUsage]) + + useEffect(() => { + const handleMessage = (event: MessageEvent) => { + const message = event.data + if (message.type === "profileDataResponse") { + const payload = message.payload as any + if (payload?.success && payload.data) { + setProfileData(payload.data) + } + setIsLoading(false) + } + } + + window.addEventListener("message", handleMessage) + return () => { + window.removeEventListener("message", handleMessage) + } + }, []) + + const tiered = profileData?.tieredUsage as AxonCodeTieredUsage | undefined + const modelUsage = profileData?.modelUsage ?? [] + + return ( +
+ {/* Plan windows */} +
+
+
+ + Plan Usage Windows +
+ +
+ + {!apiConfiguration?.kilocodeToken ? ( +
+ Log in with your Kilocode / AxonCode account to see plan usage. +
+ ) : isLoading && !profileData ? ( +
+ Loading plan usage... +
+ ) : tiered ? ( +
+ {tiered.weekly && ( + + )} + {tiered.monthly && ( + + )} +
+ ) : profileData?.usagePercentage !== undefined ? ( + + ) : ( +
+ No usage data available. +
+ )} +
+ + {/* Per-model usage */} +
+
+ + Model Usage +
+
+ Each tracked model's share of your shared plan pool (weekly / monthly). Models with a cost + multiplier drain the pool faster per request. +
+ {!apiConfiguration?.kilocodeToken ? ( +
+ Log in to see per-model usage. +
+ ) : modelUsage.length > 0 ? ( +
+ {modelUsage.map((entry) => ( + + ))} +
+ ) : ( +
+ {isLoading ? "Loading model usage..." : "No model usage recorded in this cycle yet."} +
+ )} +
+
+ ) +} diff --git a/webview-ui/src/components/settings/SettingsView.tsx b/webview-ui/src/components/settings/SettingsView.tsx index d918b614d8..7f83dc27c8 100644 --- a/webview-ui/src/components/settings/SettingsView.tsx +++ b/webview-ui/src/components/settings/SettingsView.tsx @@ -3,6 +3,7 @@ import { Blocks, CheckCheck, CircleUserRound, + Cpu, Database, // GitPullRequest, // Info, // kilocode_change: hidden for now @@ -61,6 +62,7 @@ import { AutoApproveSettings } from "./AutoApproveSettings" // import { BrowserSettings } from "./BrowserSettings" // import { CheckpointSettings } from "./CheckpointSettings" import { CodeIndexSettings } from "./CodeIndexSettings" +import { ModelUsageSettings } from "./ModelUsageSettings" // import { CodeReviewSettings as CodeReviewSettingsComponent } from "./CodeReviewSettings" // import { ContextManagementSettings } from "./ContextManagementSettings" // import { DisplaySettings } from "./DisplaySettings" // kilocode_change @@ -105,6 +107,7 @@ const sectionNames = [ "plugins", "codeIndex", // kilocode_change // "codeReview", // kilocode_change + "modelUsage", // kilocode_change "developerTools", // kilocode_change: renamed from about ] as const @@ -609,6 +612,7 @@ const SettingsView = forwardRef((props, ref) { id: "language", icon: Languages }, { id: "mcp", icon: Server }, { id: "codeIndex", icon: Database }, // kilocode_change + { id: "modelUsage", icon: Cpu }, // kilocode_change { id: "thirdPartyProviders", icon: Plug }, { id: "developerTools", icon: Wrench }, // kilocode_change: renamed from about with wrench icon @@ -743,7 +747,7 @@ const SettingsView = forwardRef((props, ref)

- {!["mcp", "plugins"].includes(activeTab) && ( + {!["mcp", "plugins", "modelUsage"].includes(activeTab) && ( ((props, ref) {/* Code Index Section */} {activeTab === "codeIndex" && } + {/* Model Usage Section */} + {activeTab === "modelUsage" && } + {/* Code Review Section */} {/* {activeTab === "codeReview" && ( - -// Shared metadata for the OSS models served through the MatterAI gateway. -// Pricing strings are USD per token (OpenRouter format); per-model rates are -// set on each entry below. -const OSS_MODEL_BASE: KiloCodeModelVariant = { - input_modalities: ["text", "image"], - max_output_length: 64000, - output_modalities: ["text"], - supported_sampling_parameters: [ - "temperature", - "top_p", - "top_k", - "repetition_penalty", - "frequency_penalty", - "presence_penalty", - "seed", - "stop", - ], - supported_features: ["tools", "structured_outputs", "web_search"], - datacenters: [{ country_code: "US" }], - created: 1786032000, - pricing: { - image: "0", - request: "0", - input_cache_writes: "0", - }, -} - +// Static catalog kept commented as a reference for the entry shape; the live +// catalog is fetched dynamically from the MatterAI backend. Entries spread an +// OSS_MODEL_BASE constant removed as dead code; restore from git history if +// re-enabling static entries. const KILO_CODE_MODELS: Record = { - "meta/muse-spark-1.2-contributor": { - ...OSS_MODEL_BASE, - id: "meta/muse-spark-1.2-contributor", - name: "Muse Spark 1.2 Contributor", - description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.", - context_length: 232000, - owned_by: "meta", - openrouter: { slug: "meta/muse-spark-1.2-contributor" }, - // $0.10/M input, $0.002/M cache read, $0.20/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.0000001", - completion: "0.0000002", - input_cache_reads: "0.000000002", - }, - }, - "deepseek/deepseek-v4-flash-0731": { - ...OSS_MODEL_BASE, - id: "deepseek/deepseek-v4-flash-0731", - name: "DeepSeek V4 Flash", - description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", - context_length: 232000, - owned_by: "deepseek", - openrouter: { slug: "deepseek/deepseek-v4-flash-0731" }, - // $0.14/M input, $0.028/M cache read, $0.28/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.00000014", - completion: "0.00000028", - input_cache_reads: "0.000000028", - }, - }, - "zai/glm-5.3-flash": { - ...OSS_MODEL_BASE, - id: "zai/glm-5.3-flash", - name: "GLM 5.3 Flash", - description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", - context_length: 232000, - owned_by: "zai", - openrouter: { slug: "zai/glm-5.3-flash" }, - // $0.15/M input, $0.03/M cache read, $0.50/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.00000015", - completion: "0.0000005", - input_cache_reads: "0.00000003", - }, - }, - "zai/glm-5.3": { - ...OSS_MODEL_BASE, - id: "zai/glm-5.3", - name: "GLM 5.3", - description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", - context_length: 232000, - owned_by: "zai", - openrouter: { slug: "zai/glm-5.3" }, - // $1.40/M input, $0.14/M cache read, $4.40/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.0000014", - completion: "0.0000044", - input_cache_reads: "0.00000014", - }, - }, - "gpt-5.6-luna": { - ...OSS_MODEL_BASE, - id: "gpt-5.6-luna", - name: "GPT-5.6 Luna", - description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", - context_length: 232000, - owned_by: "openai", - openrouter: { slug: "gpt-5.6-luna" }, - // $0.20/M input, $0.02/M cache read, $1.20/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.0000002", - completion: "0.0000012", - input_cache_reads: "0.00000002", - }, - }, - "gpt-5.6-sol": { - ...OSS_MODEL_BASE, - id: "gpt-5.6-sol", - name: "GPT-5.6 Sol", - description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", - context_length: 232000, - owned_by: "openai", - openrouter: { slug: "gpt-5.6-sol" }, - // $5/M input, $0.50/M cache read, $30/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.000005", - completion: "0.00003", - input_cache_reads: "0.0000005", - }, - }, - "gemini-3.7-flash": { - ...OSS_MODEL_BASE, - id: "gemini-3.7-flash", - name: "Gemini 3.7 Flash", - description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", - context_length: 232000, - owned_by: "google", - openrouter: { slug: "gemini-3.7-flash" }, - // $0.75/M input, $0.075/M cache read, $3.75/M output - pricing: { - ...OSS_MODEL_BASE.pricing, - prompt: "0.00000075", - completion: "0.00000375", - input_cache_reads: "0.000000075", - }, - }, + // "meta/muse-spark-1.2-contributor": { + // ...OSS_MODEL_BASE, + // id: "meta/muse-spark-1.2-contributor", + // name: "Muse Spark 1.2 Contributor", + // description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "meta", + // openrouter: { slug: "meta/muse-spark-1.2-contributor" }, + // // $0.10/M input, $0.002/M cache read, $0.20/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.0000001", + // completion: "0.0000002", + // input_cache_reads: "0.000000002", + // }, + // }, + // "deepseek/deepseek-v4-flash-0731": { + // ...OSS_MODEL_BASE, + // id: "deepseek/deepseek-v4-flash-0731", + // name: "DeepSeek V4 Flash", + // description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.", + // context_length: 232000, + // owned_by: "deepseek", + // openrouter: { slug: "deepseek/deepseek-v4-flash-0731" }, + // // $0.14/M input, $0.028/M cache read, $0.28/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.00000014", + // completion: "0.00000028", + // input_cache_reads: "0.000000028", + // }, + // }, + // "zai/glm-5.3-flash": { + // ...OSS_MODEL_BASE, + // id: "zai/glm-5.3-flash", + // name: "GLM 5.3 Flash", + // description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "zai", + // openrouter: { slug: "zai/glm-5.3-flash" }, + // // $0.15/M input, $0.03/M cache read, $0.50/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.00000015", + // completion: "0.000005", + // input_cache_reads: "0.00000003", + // }, + // }, + // "zai/glm-5.3": { + // ...OSS_MODEL_BASE, + // id: "zai/glm-5.3", + // name: "GLM 5.3", + // description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.", + // context_length: 232000, + // owned_by: "zai", + // openrouter: { slug: "zai/glm-5.3" }, + // // $1.40/M input, $0.14/M cache read, $4.40/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.0000014", + // completion: "0.0000044", + // input_cache_reads: "0.00000014", + // }, + // }, + // "gpt-5.6-luna": { + // ...OSS_MODEL_BASE, + // id: "gpt-5.6-luna", + // name: "GPT-5.6 Luna", + // description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "openai", + // openrouter: { slug: "gpt-5.6-luna" }, + // // $0.20/M input, $0.02/M cache read, $1.20/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.0000002", + // completion: "0.0000012", + // input_cache_reads: "0.00000002", + // }, + // }, + // "gpt-5.6-sol": { + // ...OSS_MODEL_BASE, + // id: "gpt-5.6-sol", + // name: "GPT-5.6 Sol", + // description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.", + // context_length: 232000, + // owned_by: "openai", + // openrouter: { slug: "gpt-5.6-sol" }, + // // $5/M input, $0.50/M cache read, $30/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.000005", + // completion: "0.00003", + // input_cache_reads: "0.0000005", + // }, + // }, + // "gemini-3.7-flash": { + // ...OSS_MODEL_BASE, + // id: "gemini-3.7-flash", + // name: "Gemini 3.7 Flash", + // description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.", + // context_length: 232000, + // owned_by: "google", + // openrouter: { slug: "gemini-3.7-flash" }, + // // $0.75/M input, $0.075/M cache read, $3.75/M output + // pricing: { + // ...OSS_MODEL_BASE.pricing, + // prompt: "0.00000075", + // completion: "0.00000375", + // input_cache_reads: "0.000000075", + // }, + // }, } const parsePrice = (value?: string): number | undefined => { @@ -251,11 +223,20 @@ async function getOpenRouterProvidersForModel(modelId: string, _baseUrl?: string const models: Record = {} const model = KILO_CODE_MODELS[modelId] - if (!model) { + if (model) { + models["KiloCode"] = toOpenRouterModelProvider(model) return models } - models["KiloCode"] = toOpenRouterModelProvider(model) + // Dynamic fallback provider if model not in static map + models["KiloCode"] = { + maxTokens: 64000, + contextWindow: 232000, + supportsImages: true, + supportsPromptCache: true, + displayName: modelId, + label: "KiloCode", + } return models } diff --git a/webview-ui/src/components/ui/hooks/useRouterModels.ts b/webview-ui/src/components/ui/hooks/useRouterModels.ts index 37fe58751c..eefe923edf 100644 --- a/webview-ui/src/components/ui/hooks/useRouterModels.ts +++ b/webview-ui/src/components/ui/hooks/useRouterModels.ts @@ -1,11 +1,12 @@ -import { useQuery } from "@tanstack/react-query" +import { useEffect } from "react" +import { useQuery, useQueryClient } from "@tanstack/react-query" import { RouterModels } from "@roo/api" import { ExtensionMessage } from "@roo/ExtensionMessage" import { vscode } from "@src/utils/vscode" -const getRouterModels = async () => +const getRouterModels = async (forceRefresh: boolean = false) => new Promise((resolve, reject) => { const cleanup = () => { window.removeEventListener("message", handler) @@ -32,7 +33,7 @@ const getRouterModels = async () => } window.addEventListener("message", handler) - vscode.postMessage({ type: "requestRouterModels" }) + vscode.postMessage({ type: "requestRouterModels", values: { forceRefresh } }) }) // forked_change start @@ -50,6 +51,40 @@ type RouterModelsQueryKey = { // Requesty, Unbound, etc should perhaps also be here, but they already have their own hacks for reloading } -export const useRouterModels = (queryKey: RouterModelsQueryKey) => - useQuery({ queryKey: ["routerModels", queryKey], queryFn: () => getRouterModels() }) +export const useRouterModels = (queryKey: RouterModelsQueryKey) => { + const queryClient = useQueryClient() + + useEffect(() => { + let lastRefresh = 0 + const triggerRefresh = () => { + const now = Date.now() + if (now - lastRefresh > 5000) { + lastRefresh = now + queryClient.invalidateQueries({ queryKey: ["routerModels"] }) + } + } + + const onFocus = () => triggerRefresh() + window.addEventListener("focus", onFocus) + + const onMessage = (event: MessageEvent) => { + if (event.data?.type === "action" && event.data?.action === "didBecomeVisible") { + triggerRefresh() + } + } + window.addEventListener("message", onMessage) + + return () => { + window.removeEventListener("focus", onFocus) + window.removeEventListener("message", onMessage) + } + }, [queryClient]) + + return useQuery({ + queryKey: ["routerModels", queryKey], + queryFn: () => getRouterModels(), + refetchInterval: 10 * 60 * 1000, + refetchOnWindowFocus: true, + }) +} // forked_change end diff --git a/webview-ui/src/i18n/locales/en/settings.json b/webview-ui/src/i18n/locales/en/settings.json index 784fa22ddd..afe3e89964 100644 --- a/webview-ui/src/i18n/locales/en/settings.json +++ b/webview-ui/src/i18n/locales/en/settings.json @@ -37,6 +37,7 @@ "language": "Language", "codeIndex": "Code Indexing", "codeReview": "AI Code Reviews", + "modelUsage": "Model Usage", "developerTools": "Developer Tools" }, "slashCommands": { From 97044d2cd646ab82e020f5e21957270fc8fa178e Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Thu, 3 Sep 2026 12:55:29 +0530 Subject: [PATCH 7/9] UI cleanups --- webview-ui/src/components/chat/ChatRow.tsx | 2 +- webview-ui/src/components/chat/ChatTabs.tsx | 2 +- webview-ui/src/components/chat/chatLayout.ts | 2 +- webview-ui/src/components/kilocode/StickyUserMessage.tsx | 4 ++-- 4 files changed, 5 insertions(+), 5 deletions(-) diff --git a/webview-ui/src/components/chat/ChatRow.tsx b/webview-ui/src/components/chat/ChatRow.tsx index 9153c0b46a..0d28fb3353 100644 --- a/webview-ui/src/components/chat/ChatRow.tsx +++ b/webview-ui/src/components/chat/ChatRow.tsx @@ -1872,7 +1872,7 @@ export const ChatRowContent = ({
*/}
= ({ tabs, onSelect, onClose, onAddTab, onDragLeave={handleDragLeave} onDrop={(e) => handleDrop(e, tab.taskId)} onDragEnd={handleDragEnd} - className={`flex items-center gap-1.5 h-[30px] pl-2.5 pr-1.5 rounded-t-lg min-w-0 max-w-[240px] cursor-pointer group select-none transition-colors duration-150 ${ + className={`flex items-center gap-1.5 h-[30px] pl-2.5 pr-1.5 rounded-t-sm min-w-0 max-w-[240px] cursor-pointer group select-none transition-colors duration-150 ${ tab.isActive ? "bg-[var(--vscode-tab-activeBackground,var(--vscode-editor-background))] text-[var(--vscode-tab-activeForeground,var(--vscode-foreground))] cursor-default shadow-xs border-t border-x border-[var(--vscode-panel-border)]/40 -mb-[1px] pb-[1px]" : "bg-transparent hover:bg-[var(--vscode-tab-hoverBackground)] text-[var(--vscode-tab-inactiveForeground)] hover:text-[var(--vscode-tab-activeForeground,var(--vscode-foreground))]" diff --git a/webview-ui/src/components/chat/chatLayout.ts b/webview-ui/src/components/chat/chatLayout.ts index 2bcc80ca13..56c1a426f4 100644 --- a/webview-ui/src/components/chat/chatLayout.ts +++ b/webview-ui/src/components/chat/chatLayout.ts @@ -1,2 +1,2 @@ /** Shared horizontal inset for chat content and the message composer. */ -export const CHAT_CONTENT_HORIZONTAL_PADDING = "px-5" +export const CHAT_CONTENT_HORIZONTAL_PADDING = "px-3.5" diff --git a/webview-ui/src/components/kilocode/StickyUserMessage.tsx b/webview-ui/src/components/kilocode/StickyUserMessage.tsx index cface1b835..2fb541da8f 100644 --- a/webview-ui/src/components/kilocode/StickyUserMessage.tsx +++ b/webview-ui/src/components/kilocode/StickyUserMessage.tsx @@ -80,8 +80,8 @@ const StickyUserMessage = ({ task, messages, stickyIndex }: StickyUserMessagePro return (
Date: Thu, 3 Sep 2026 14:39:28 +0530 Subject: [PATCH 8/9] feat(models): provider logo icons from backend catalog - Add iconUrl to ModelInfo and forward it from the MatterAI /v1/web/models catalog through parseOpenRouterModel, registerDynamicKilocodeModels, and the static KILO_CODE_MODELS loop - New ProviderLogo component renders the SVG on a white circular background (visible on dark themes) - Show provider logos in the model selector (dropdown items + trigger), usage dialog, and settings model usage section - Add iconUrl to AxonCodeModelUsage for /axoncode/profile model usage entries --- packages/types/src/model.ts | 3 +++ src/api/providers/fetchers/openrouter.ts | 8 ++++++++ src/api/providers/kilocode-models.ts | 3 +++ src/shared/WebviewMessage.ts | 3 +++ webview-ui/src/components/chat/UsageDialog.tsx | 6 ++++-- .../components/kilocode/chat/ModelSelector.tsx | 10 ++++++++-- .../components/settings/ModelUsageSettings.tsx | 6 +++++- webview-ui/src/components/ui/index.ts | 1 + webview-ui/src/components/ui/provider-logo.tsx | 18 ++++++++++++++++++ 9 files changed, 53 insertions(+), 5 deletions(-) create mode 100644 webview-ui/src/components/ui/provider-logo.tsx diff --git a/packages/types/src/model.ts b/packages/types/src/model.ts index 45bb1c578d..bc26cd93ea 100644 --- a/packages/types/src/model.ts +++ b/packages/types/src/model.ts @@ -79,6 +79,9 @@ export const modelInfoSchema = z.object({ // forked_change start displayName: z.string().nullish(), preferredIndex: z.number().nullish(), + // Provider logo URL (SVG) from the MatterAI catalog; rendered by the UI + // on a white circular background. + iconUrl: z.string().nullish(), // forked_change end // Flag to indicate if the model is deprecated and should not be used deprecated: z.boolean().optional(), diff --git a/src/api/providers/fetchers/openrouter.ts b/src/api/providers/fetchers/openrouter.ts index 576a671715..530065c35c 100644 --- a/src/api/providers/fetchers/openrouter.ts +++ b/src/api/providers/fetchers/openrouter.ts @@ -107,6 +107,7 @@ const matterAiOpenRouterModelSchema = z.object({ input_modalities: z.array(z.string()).optional(), output_modalities: z.array(z.string()).optional(), supported_sampling_parameters: z.array(z.string()).optional(), + iconUrl: z.string().optional(), pricing: z .object({ prompt: z.string().optional(), @@ -155,6 +156,7 @@ export async function getOpenRouterModels( outputModality: model.output_modalities, maxTokens: model.max_output_length, supportedParameters: model.supported_sampling_parameters, + iconUrl: model.iconUrl, }) } @@ -213,6 +215,7 @@ export async function getOpenRouterModels( outputModality: rawModel.output_modalities, maxTokens: rawModel.max_output_length, supportedParameters: rawModel.supported_sampling_parameters, + iconUrl: rawModel.iconUrl, }) } } catch (error) { @@ -258,6 +261,7 @@ export async function getOpenRouterModelEndpoints( outputModality: staticModel.output_modalities, maxTokens: staticModel.max_output_length, supportedParameters: staticModel.supported_sampling_parameters, + iconUrl: staticModel.iconUrl, }) return models } @@ -321,6 +325,7 @@ export async function getOpenRouterModelEndpoints( outputModality: rawModel.output_modalities, maxTokens: rawModel.max_output_length, supportedParameters: rawModel.supported_sampling_parameters, + iconUrl: rawModel.iconUrl, }) } catch (error) { console.error( @@ -343,6 +348,7 @@ export const parseOpenRouterModel = ({ outputModality, maxTokens, supportedParameters, + iconUrl, // kilocode_change }: { id: string model: OpenRouterBaseModel @@ -351,6 +357,7 @@ export const parseOpenRouterModel = ({ outputModality: string[] | null | undefined maxTokens: number | null | undefined supportedParameters?: string[] + iconUrl?: string // kilocode_change }): ModelInfo => { const cacheWritesPrice = model.pricing?.input_cache_write ? parseApiPrice(model.pricing?.input_cache_write) @@ -375,6 +382,7 @@ export const parseOpenRouterModel = ({ // forked_change start displayName, preferredIndex: model.preferredIndex, + iconUrl, // forked_change end } diff --git a/src/api/providers/kilocode-models.ts b/src/api/providers/kilocode-models.ts index 878077fad3..aa18bb761e 100644 --- a/src/api/providers/kilocode-models.ts +++ b/src/api/providers/kilocode-models.ts @@ -16,6 +16,8 @@ export type KiloCodeModel = { datacenters: Array<{ country_code: string }> created: number owned_by: string + // Provider logo URL (SVG) from the backend catalog, rendered by the webview. + iconUrl?: string pricing: { type?: "dynamic" display?: string @@ -218,6 +220,7 @@ export function registerDynamicKilocodeModels(rawModels: Array = ({ key={entry.model} className="flex flex-col gap-1.5 p-2 rounded bg-[var(--vscode-sideBar-background)] border border-[var(--vscode-panel-border)]/50">
- - {entry.model} + + + {entry.model} {entry.multiplier}x cost diff --git a/webview-ui/src/components/kilocode/chat/ModelSelector.tsx b/webview-ui/src/components/kilocode/chat/ModelSelector.tsx index fbd2999bb5..98305e2056 100644 --- a/webview-ui/src/components/kilocode/chat/ModelSelector.tsx +++ b/webview-ui/src/components/kilocode/chat/ModelSelector.tsx @@ -1,4 +1,4 @@ -import { DropdownOption, DropdownOptionType, SelectDropdown, StandardTooltip } from "@/components/ui" +import { DropdownOption, DropdownOptionType, ProviderLogo, SelectDropdown, StandardTooltip } from "@/components/ui" import { usePreferredModels } from "@/components/ui/hooks/kilocode/usePreferredModels" import { useThirdPartyModels } from "@/components/ui/hooks/useOllamaModels" import { Alert02Icon } from "@/utils/customIcons" @@ -462,6 +462,7 @@ export const ModelSelector = ({ : "hover:bg-[var(--vscode-button-hoverBackground)] hover:text-[var(--vscode-button-foreground)] text-[var(--vscode-foreground)] opacity-70", )}> {isConfigureOption ? : null} +
@@ -651,7 +652,12 @@ export const ModelSelector = ({ renderItem={renderItem} renderValue={(option) => { const is400k = providerModels[option.value]?.contextWindow === 400000 || is400kAxonModel(option.value) - return + return ( + + + + + ) }} onRefresh={handleRefreshModels} // Always show refresh since matterai3p is always enabled /> diff --git a/webview-ui/src/components/settings/ModelUsageSettings.tsx b/webview-ui/src/components/settings/ModelUsageSettings.tsx index 624ca41927..2c3545ed54 100644 --- a/webview-ui/src/components/settings/ModelUsageSettings.tsx +++ b/webview-ui/src/components/settings/ModelUsageSettings.tsx @@ -1,4 +1,5 @@ import React, { useEffect, useState } from "react" +import { ProviderLogo } from "@/components/ui" import { vscode } from "@/utils/vscode" import { useExtensionState } from "@/context/ExtensionStateContext" import { AxonCodeModelUsage, AxonCodeTieredUsage, ProfileData, WebviewMessage } from "@roo/WebviewMessage" @@ -56,7 +57,10 @@ const ModelUsageRow: React.FC<{ entry: AxonCodeModelUsage }> = ({ entry }) => { return (
- {entry.model} + + + {entry.model} + {entry.multiplier}x limit diff --git a/webview-ui/src/components/ui/index.ts b/webview-ui/src/components/ui/index.ts index 1edc71d6ab..98f11fd539 100644 --- a/webview-ui/src/components/ui/index.ts +++ b/webview-ui/src/components/ui/index.ts @@ -10,6 +10,7 @@ export * from "./dropdown-menu" export * from "./input" export * from "./labeled-progress" export * from "./popover" +export * from "./provider-logo" export * from "./progress" export * from "./searchable-select" export * from "./separator" diff --git a/webview-ui/src/components/ui/provider-logo.tsx b/webview-ui/src/components/ui/provider-logo.tsx new file mode 100644 index 0000000000..d3c0aaa028 --- /dev/null +++ b/webview-ui/src/components/ui/provider-logo.tsx @@ -0,0 +1,18 @@ +import { cn } from "@src/lib/utils" + +/** + * Provider logo rendered on a white circular background. Catalog icons are + * transparent SVGs (some monochrome), so they need the white backing to stay + * visible on dark VS Code themes. + */ +export const ProviderLogo = ({ src, className }: { src?: string | null; className?: string }) => { + if (!src) { + return null + } + + return ( + + + + ) +} From 5b6144509de1a7e31aa3043f4dcf3e91121bb8f5 Mon Sep 17 00:00:00 2001 From: "matterai-app[bot]" Date: Thu, 3 Sep 2026 14:56:36 +0530 Subject: [PATCH 9/9] docs(readme): update AI Models section to the current OSS catalog Replace the retired axon-mini/axon-code/axon-code-2 table with the seven OSS models served dynamically from the MatterAI backend catalog, with providers, credit multipliers, and use cases. --- README.md | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 6d6a97cd42..b2b2265d92 100644 --- a/README.md +++ b/README.md @@ -85,13 +85,17 @@ Orbital provides a comprehensive suite of intelligent tools: ## 🤖 AI Models -Choose the right model for your workflow — switch seamlessly between efficiency and power: - -| Model | Credit Usage | Use Case | Capabilities | -| ----------------- | ------------ | --------------------- | --------------------------------------------------------------------------- | -| **`axon-mini`** | 0.5x | Quick tasks | Lightweight, fast, cost-effective for high-volume usage | -| **`axon-code`** | 0.8x | Daily coding tasks | High intelligence, balanced performance, best for regular coding | -| **`axon-code-2`** | 1x | Complex agentic tasks | Maximum intelligence, complex context harness, best for complex development | +Choose the right model for your workflow — switch seamlessly between efficiency and power. The model catalog is served dynamically from the MatterAI backend, so new models appear automatically in the model selector: + +| Model | Provider | Credit Usage | Use Case | Capabilities | +| ------------------------------ | -------- | ------------ | --------------------- | --------------------------------------------------------------- | +| **Muse Spark 1.3 Contributor** | Meta | 2x | Everyday coding | Open general-purpose model for everyday coding tasks | +| **DeepSeek V4 Flash** | DeepSeek | 5x | Fast, low-cost coding | Fast, low-cost open model for day-to-day coding tasks | +| **GLM 5.3** | Z.ai | 4x | Complex agentic tasks | Frontier open model for complex coding and long-running agents | +| **GLM 5.3 Flash** | Z.ai | 4x | Everyday coding | Fast, low-cost open model for everyday coding tasks | +| **GPT-5.6 Luna** | OpenAI | 2x | Everyday coding | Fast, low-cost open model for everyday coding tasks | +| **GPT-5.6 Sol** | OpenAI | 5x | Complex reasoning | Open reasoning model for complex coding and long-running agents | +| **Gemini 3.8 Flash** | Google | 3x | Everyday coding | Fast, low-cost model for everyday coding tasks | ## 📦 Installation