- )
- }
-
const onChange = (value: string) => {
if (isPlanRestrictedAxonModel(value) && !canAccessAxonModel(value, profilePlan)) return
@@ -714,7 +632,6 @@ export const ModelSelector = ({
const is400k = providerModels[option.value]?.contextWindow === 400000 || is400kAxonModel(option.value)
return
}}
- headerComponent={renderContextToggleHeader()}
onRefresh={handleRefreshModels} // Always show refresh since matterai3p is always enabled
/>
)
diff --git a/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts b/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts
index 847625aff7..de73318679 100644
--- a/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts
+++ b/webview-ui/src/components/ui/hooks/kilocode/usePreferredModels.ts
@@ -26,7 +26,7 @@ export const usePreferredModels = (models: Record | null) =>
restModelIds.push(key)
}
}
- restModelIds.sort((a, b) => a.localeCompare(b))
+ // restModelIds.sort((a, b) => a.localeCompare(b))
return [...preferredModelIds, ...restModelIds]
}, [models])
diff --git a/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts b/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts
index de75b77557..9a399ed24a 100644
--- a/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts
+++ b/webview-ui/src/components/ui/hooks/useOpenRouterModelProviders.ts
@@ -40,142 +40,15 @@ type KiloCodeModel = {
}
}
-type KiloCodeModelVariant = Omit
-
-const AXON_AUTO: KiloCodeModelVariant = {
- description:
- "Axon Auto starts with Eido 3.2 Flash and dynamically selects Flash, 3.2, or Pro as the task evolves. Pricing is dynamic and follows the model used for each request.",
- input_modalities: ["text", "image"],
- max_output_length: 64000,
- output_modalities: ["text"],
- supported_sampling_parameters: [
- "temperature",
- "top_p",
- "top_k",
- "repetition_penalty",
- "frequency_penalty",
- "presence_penalty",
- "seed",
- "stop",
- ],
- supported_features: ["tools", "structured_outputs", "web_search"],
- openrouter: {
- slug: "matterai/axon",
- },
- datacenters: [{ country_code: "US" }],
- created: 1750426201,
- owned_by: "matterai",
- pricing: {
- type: "dynamic",
- display: "dynamic pricing",
- image: "0",
- request: "0",
- input_cache_reads: "0",
- input_cache_writes: "0",
- },
-}
-
-const AXON_EIDO_3_2_CODE_PRO: KiloCodeModelVariant = {
- description:
- "Axon Eido 3.2 Pro is the frontier model for coding tasks, long running agents and general intelligence, fine-tuned on open source models.",
- input_modalities: ["text", "image"],
- max_output_length: 64000,
- output_modalities: ["text"],
- supported_sampling_parameters: [
- "temperature",
- "top_p",
- "top_k",
- "repetition_penalty",
- "frequency_penalty",
- "presence_penalty",
- "seed",
- "stop",
- ],
- supported_features: ["tools", "structured_outputs", "web_search"],
- openrouter: {
- slug: "matterai/axon",
- },
- datacenters: [{ country_code: "US" }],
- created: 1750426201,
- owned_by: "matterai",
- pricing: {
- prompt: "0.000003",
- completion: "0.000009",
- image: "0",
- request: "0",
- input_cache_reads: "0",
- input_cache_writes: "0",
- },
-}
-
-const AXON_EIDO_3_2: KiloCodeModelVariant = {
- description:
- "Axon Eido 3.2 is a general purpose super intelligent LLM coding model for high-effort day-to-day tasks",
- input_modalities: ["text", "image"],
- max_output_length: 64000,
- output_modalities: ["text"],
- supported_sampling_parameters: [
- "temperature",
- "top_p",
- "top_k",
- "repetition_penalty",
- "frequency_penalty",
- "presence_penalty",
- "seed",
- "stop",
- ],
- supported_features: ["tools", "structured_outputs", "web_search"],
- openrouter: {
- slug: "matterai/axon",
- },
- datacenters: [{ country_code: "US" }],
- created: 1750426201,
- owned_by: "matterai",
- pricing: {
- prompt: "0.000002",
- completion: "0.000006",
- image: "0",
- request: "0",
- input_cache_reads: "0.0000005",
- input_cache_writes: "0",
- },
-}
-
-const AXON_LUMEN_4_CODE: KiloCodeModelVariant = {
- description:
- "Axon Lumen 4 is the ultra-intelligent frontier model for complex agentic coding tasks and general intelligence.",
- input_modalities: ["text", "image"],
- max_output_length: 128000,
- output_modalities: ["text"],
- supported_sampling_parameters: [
- "temperature",
- "top_p",
- "top_k",
- "repetition_penalty",
- "frequency_penalty",
- "presence_penalty",
- "seed",
- "stop",
- ],
- supported_features: ["tools", "structured_outputs", "web_search"],
- openrouter: {
- slug: "matterai/axon",
- },
- datacenters: [{ country_code: "US" }],
- created: 1750426201,
- owned_by: "matterai",
- pricing: {
- prompt: "0.000005",
- completion: "0.000025",
- image: "0",
- request: "0",
- input_cache_reads: "0",
- input_cache_writes: "0",
- },
-}
+type KiloCodeModelVariant = Omit<
+ KiloCodeModel,
+ "id" | "name" | "description" | "context_length" | "owned_by" | "openrouter"
+>
-const AXON_EIDO_3_2_FLASH: KiloCodeModelVariant = {
- description: "Axon Eido 3.2 Flash is a fast and low cost general purpose model for low-effort day-to-day tasks",
+// Shared metadata for the OSS models served through the MatterAI gateway.
+// Pricing strings are USD per token (OpenRouter format); per-model rates are
+// set on each entry below.
+const OSS_MODEL_BASE: KiloCodeModelVariant = {
input_modalities: ["text", "image"],
max_output_length: 64000,
output_modalities: ["text"],
@@ -190,82 +63,127 @@ const AXON_EIDO_3_2_FLASH: KiloCodeModelVariant = {
"stop",
],
supported_features: ["tools", "structured_outputs", "web_search"],
- openrouter: {
- slug: "matterai/axon",
- },
datacenters: [{ country_code: "US" }],
- created: 1750426201,
- owned_by: "matterai",
+ created: 1786032000,
pricing: {
- prompt: "0.0",
- completion: "0.0",
image: "0",
request: "0",
- input_cache_reads: "0",
input_cache_writes: "0",
},
}
const KILO_CODE_MODELS: Record = {
- "axon-auto-232k": {
- ...AXON_AUTO,
- id: "axon-auto",
- name: "Axon Auto (232K context)",
+ "meta/muse-spark-1.2-contributor": {
+ ...OSS_MODEL_BASE,
+ id: "meta/muse-spark-1.2-contributor",
+ name: "Muse Spark 1.2 Contributor",
+ description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.",
context_length: 232000,
+ owned_by: "meta",
+ openrouter: { slug: "meta/muse-spark-1.2-contributor" },
+ // $0.10/M input, $0.002/M cache read, $0.20/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.0000001",
+ completion: "0.0000002",
+ input_cache_reads: "0.000000002",
+ },
},
- "axon-auto-400k": {
- ...AXON_AUTO,
- id: "axon-auto",
- name: "Axon Auto (400K context)",
- context_length: 400000,
- },
- "axon-eido-3.2-flash": {
- ...AXON_EIDO_3_2_FLASH,
- id: "axon-eido-3.2-flash",
- name: "Axon Eido 3.2 Flash (232K context)",
+ "deepseek/deepseek-v4-flash-0731": {
+ ...OSS_MODEL_BASE,
+ id: "deepseek/deepseek-v4-flash-0731",
+ name: "DeepSeek V4 Flash",
+ description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.",
context_length: 232000,
+ owned_by: "deepseek",
+ openrouter: { slug: "deepseek/deepseek-v4-flash-0731" },
+ // $0.14/M input, $0.028/M cache read, $0.28/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.00000014",
+ completion: "0.00000028",
+ input_cache_reads: "0.000000028",
+ },
},
- "axon-eido-3.2-flash-400k": {
- ...AXON_EIDO_3_2_FLASH,
- id: "axon-eido-3.2-flash",
- name: "Axon Eido 3.2 Flash (400K context)",
- context_length: 400000,
- },
- "axon-eido-3.2-232k": {
- ...AXON_EIDO_3_2,
- id: "axon-eido-3.2",
- name: "Axon Eido 3.2 (232K context)",
+ "zai/glm-5.3-flash": {
+ ...OSS_MODEL_BASE,
+ id: "zai/glm-5.3-flash",
+ name: "GLM 5.3 Flash",
+ description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.",
context_length: 232000,
+ owned_by: "zai",
+ openrouter: { slug: "zai/glm-5.3-flash" },
+ // $0.15/M input, $0.03/M cache read, $0.50/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.00000015",
+ completion: "0.0000005",
+ input_cache_reads: "0.00000003",
+ },
},
- "axon-eido-3.2-400k": {
- ...AXON_EIDO_3_2,
- id: "axon-eido-3.2",
- name: "Axon Eido 3.2 (400K context)",
- context_length: 400000,
- },
- "axon-eido-3.2-code-pro-232k": {
- ...AXON_EIDO_3_2_CODE_PRO,
- id: "axon-eido-3.2-code-pro",
- name: "Axon Eido 3.2 Pro (232K context)",
+ "zai/glm-5.3": {
+ ...OSS_MODEL_BASE,
+ id: "zai/glm-5.3",
+ name: "GLM 5.3",
+ description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.",
context_length: 232000,
+ owned_by: "zai",
+ openrouter: { slug: "zai/glm-5.3" },
+ // $1.40/M input, $0.14/M cache read, $4.40/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.0000014",
+ completion: "0.0000044",
+ input_cache_reads: "0.00000014",
+ },
},
- "axon-eido-3.2-code-pro-400k": {
- ...AXON_EIDO_3_2_CODE_PRO,
- id: "axon-eido-3.2-code-pro",
- name: "Axon Eido 3.2 Pro (400K context)",
- context_length: 400000,
+ "gpt-5.6-luna": {
+ ...OSS_MODEL_BASE,
+ id: "gpt-5.6-luna",
+ name: "GPT-5.6 Luna",
+ description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.",
+ context_length: 232000,
+ owned_by: "openai",
+ openrouter: { slug: "gpt-5.6-luna" },
+ // $0.20/M input, $0.02/M cache read, $1.20/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.0000002",
+ completion: "0.0000012",
+ input_cache_reads: "0.00000002",
+ },
},
- "axon-lumen-4-code-232k": {
- ...AXON_LUMEN_4_CODE,
- id: "axon-lumen-4-code",
- name: "Axon Lumen 4 (232K context)",
+ "gpt-5.6-sol": {
+ ...OSS_MODEL_BASE,
+ id: "gpt-5.6-sol",
+ name: "GPT-5.6 Sol",
+ description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.",
context_length: 232000,
+ owned_by: "openai",
+ openrouter: { slug: "gpt-5.6-sol" },
+ // $5/M input, $0.50/M cache read, $30/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.000005",
+ completion: "0.00003",
+ input_cache_reads: "0.0000005",
+ },
},
- "axon-lumen-4-code-400k": {
- ...AXON_LUMEN_4_CODE,
- id: "axon-lumen-4-code",
- name: "Axon Lumen 4 (400K context)",
- context_length: 400000,
+ "gemini-3.7-flash": {
+ ...OSS_MODEL_BASE,
+ id: "gemini-3.7-flash",
+ name: "Gemini 3.7 Flash",
+ description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.",
+ context_length: 232000,
+ owned_by: "google",
+ openrouter: { slug: "gemini-3.7-flash" },
+ // $0.75/M input, $0.075/M cache read, $3.75/M output
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: "0.00000075",
+ completion: "0.00000375",
+ input_cache_reads: "0.000000075",
+ },
},
}
From 2a4ca5a001d8b3d05cd049ed3d7996278f98dff1 Mon Sep 17 00:00:00 2001
From: "matterai-app[bot]"
Date: Wed, 2 Sep 2026 15:37:30 +0530
Subject: [PATCH 2/9] fix(assistant-message): keep id-less native tool calls
and run trailing blocks after a read-only batch
- AssistantMessageParser: synthesize a stable `native-tool-call-` id
when a delta carries an index but no id instead of dropping the call.
Some OpenAI-compatible providers never send an id (or only send it in a
later fragment); dropping the first delta silently discarded the whole
tool call, and every later argument fragment for that index then hit
"arguments for unknown tool call" and was dropped too. `name` is now
optional in NativeToolCall since only the first delta carries it.
- presentAssistantMessage: after a parallel read-only batch stops at the
first non-parallelizable block (e.g. execute_command following a run of
searches), continue the presentation chain so trailing blocks still
execute; the batch only marks userMessageContentReady when it consumed
every block, instead of firing the next API request early.
- The tool-failure fallback only signals readiness for the LAST content
block, so a mid-message failure no longer tells the task loop the
message is complete while later blocks still have to run.
- Regression tests for id-less tool call accumulation, the trailing
execute_command after a read-only batch, and mid-message failure
readiness.
---
.../AssistantMessageParser.ts | 21 +-
.../__tests__/AssistantMessageParser.spec.ts | 39 ++++
.../__tests__/presentAssistantMessage.spec.ts | 194 ++++++++++++++++++
.../kilocode/native-tool-call.ts | 4 +-
.../presentAssistantMessage.ts | 28 ++-
5 files changed, 276 insertions(+), 10 deletions(-)
diff --git a/src/core/assistant-message/AssistantMessageParser.ts b/src/core/assistant-message/AssistantMessageParser.ts
index d6de1347ea..6010f406f6 100644
--- a/src/core/assistant-message/AssistantMessageParser.ts
+++ b/src/core/assistant-message/AssistantMessageParser.ts
@@ -195,8 +195,9 @@ export class AssistantMessageParser {
* forked_change: Now yields partial tool calls immediately when tool name is known,
* allowing the UI to show "Editing filename..." during streaming.
*
- * @param toolCalls Array of native tool call objects (may be partial during streaming). We
- * currently set parallel_tool_calls to false, so in theory there should only be 1 call.
+ * @param toolCalls Array of native tool call objects (may be partial during streaming).
+ * Native tool calling sets parallel_tool_calls = true, so this may contain
+ * several independent calls that all must be accumulated and executed.
*/
public *processNativeToolCalls(toolCalls: NativeToolCall[]): Generator {
for (const toolCall of toolCalls) {
@@ -215,11 +216,17 @@ export class AssistantMessageParser {
toolCallId = toolCall.id
this.nativeToolCallIndexToId.set(toolCall.index, toolCallId)
} else {
- console.warn(
- "[AssistantMessageParser] Skipping tool call: has index but no id in mapping:",
- toolCall,
- )
- continue
+ // Some OpenAI-compatible providers never send an id for a tool call
+ // (or only send it in a later fragment). Dropping the call here used
+ // to silently discard the whole tool call — every later fragment for
+ // this index then hit "arguments for unknown tool call" and was
+ // dropped too. Synthesize a stable id from the index instead; the
+ // assistant message and its tool_results are built from these ids,
+ // so pairing stays consistent. If the real id arrives in a later
+ // delta, the mapping below still wins and accumulation continues
+ // unchanged.
+ toolCallId = `native-tool-call-${toolCall.index}`
+ this.nativeToolCallIndexToId.set(toolCall.index, toolCallId)
}
} else if (toolCall.id) {
toolCallId = toolCall.id
diff --git a/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts b/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts
index be38f110ea..4bf22eed57 100644
--- a/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts
+++ b/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts
@@ -574,6 +574,45 @@ describe("AssistantMessageParser (streaming)", () => {
expect(pkgUse?.params.content).toBe(pkgContent)
expect(tsUse?.params.content).toBe(tsConfigContent)
})
+
+ it("should accumulate a tool call whose deltas carry an index but never an id", () => {
+ // Some OpenAI-compatible providers never send an id per tool call. The
+ // parser used to drop the first delta ("has index but no id") and then
+ // every argument fragment for that index ("arguments for unknown tool
+ // call"), so the whole tool call silently vanished.
+ const yielded1 = [
+ ...parser.processNativeToolCalls([
+ {
+ index: 2,
+ type: "function",
+ function: {
+ name: "read_file",
+ arguments: '{"file_path": "src/a.ts"',
+ },
+ },
+ ]),
+ ]
+ // Name known, JSON incomplete: only the partial block is emitted.
+ expect(yielded1).toHaveLength(1)
+
+ const yielded2 = [
+ ...parser.processNativeToolCalls([
+ {
+ index: 2,
+ function: { arguments: "}" },
+ },
+ ]),
+ ]
+ expect(yielded2).toHaveLength(1)
+ expect(yielded2[0].id).toBe("native-tool-call-2")
+
+ const toolUse = parser
+ .getContentBlocks()
+ .find((b) => b.type === "tool_use" && (b as ToolUse).toolUseId === "native-tool-call-2") as ToolUse
+ expect(toolUse).toBeDefined()
+ expect(toolUse.partial).toBe(false)
+ expect(toolUse.params.file_path).toBe("src/a.ts")
+ })
})
describe("size limit handling", () => {
diff --git a/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts b/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts
index 45958fc6be..20f925a0a6 100644
--- a/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts
+++ b/src/core/assistant-message/__tests__/presentAssistantMessage.spec.ts
@@ -59,6 +59,7 @@ vi.mock("../../task/Task", () => ({
}))
import { presentAssistantMessage } from "../presentAssistantMessage"
+import { executeCommandTool } from "../../tools/executeCommandTool"
import { readFileTool } from "../../tools/readFileTool"
import { searchFilesTool } from "../../tools/searchFilesTool"
import { TelemetryService } from "@roo-code/telemetry"
@@ -411,4 +412,197 @@ describe("presentAssistantMessage", () => {
expect(cline.userMessageContentReady).toBe(true)
expect(cline.presentAssistantMessageLocked).toBe(false)
})
+
+ it("runs the trailing non-parallelizable tool after a read-only batch (execute_command not dropped)", async () => {
+ // Regression: a native parallel response of 3x search_files + read_file +
+ // execute_command executed only the 4 read-only calls. The batch scheduler
+ // set userMessageContentReady=true after committing the batch, so the task
+ // loop fired the next request before execute_command was ever presented.
+ const getState = vi.fn().mockResolvedValue({ mode: "code", customModes: [] })
+ vi.mocked(executeCommandTool).mockImplementationOnce(
+ async (_cline: any, block: any, _ask: any, _handleError: any, pushToolResult: any) => {
+ pushToolResult(`command:${block.params.command}`)
+ },
+ )
+ const cline = {
+ abort: false,
+ taskId: "task-1",
+ instanceId: "instance-1",
+ presentAssistantMessageLocked: false,
+ presentAssistantMessageHasPendingUpdates: false,
+ currentStreamingContentIndex: 0,
+ assistantMessageContent: [
+ {
+ type: "tool_use",
+ name: "search_files",
+ params: { path: "src", regex: "tool_calls", file_pattern: "*.ts" },
+ partial: false,
+ toolUseId: "search_files:0",
+ },
+ {
+ type: "tool_use",
+ name: "read_file",
+ params: { file_path: "src/api/providers/openai.ts" },
+ partial: false,
+ toolUseId: "read_file:1",
+ },
+ {
+ type: "tool_use",
+ name: "search_files",
+ params: { path: "src/controller", regex: "processRequest" },
+ partial: false,
+ toolUseId: "search_files:2",
+ },
+ {
+ type: "tool_use",
+ name: "search_files",
+ params: { path: "src", regex: "getAxonModel" },
+ partial: false,
+ toolUseId: "search_files:3",
+ },
+ {
+ type: "tool_use",
+ name: "execute_command",
+ params: { command: "git status --short" },
+ partial: false,
+ toolUseId: "execute_command:4",
+ },
+ ],
+ didCompleteReadingStream: true,
+ userMessageContentReady: false,
+ userMessageContent: [],
+ didRejectTool: false,
+ didAlreadyUseTool: false,
+ currentStreamingDidCheckpoint: false,
+ diffEnabled: false,
+ autoApproveAllCommands: false,
+ consecutiveMistakeCount: 0,
+ providerRef: {
+ deref: () => ({
+ getState,
+ }),
+ },
+ browserSession: {
+ closeBrowser: vi.fn().mockResolvedValue(undefined),
+ },
+ say: vi.fn().mockResolvedValue(undefined),
+ ask: vi.fn().mockResolvedValue({ response: "yesButtonClicked" }),
+ processQueuedMessages: vi.fn(),
+ getToolCallSignature: vi.fn((name: string, params: unknown) => JSON.stringify({ name, params })),
+ checkAndRegisterToolCall: vi.fn().mockReturnValue(false),
+ recordToolUsage: vi.fn(),
+ toolRepetitionDetector: {
+ check: vi.fn().mockReturnValue({ allowExecution: true }),
+ },
+ checkpointSave: vi.fn().mockResolvedValue(undefined),
+ removeStalePartialToolAskMessage: vi.fn().mockResolvedValue(undefined),
+ checkAndCondenseContext: vi.fn().mockResolvedValue(undefined),
+ } as any
+
+ await presentAssistantMessage(cline)
+
+ // All four read-only calls ran concurrently, and the trailing
+ // execute_command was still presented and executed afterwards.
+ expect(searchFilesTool).toHaveBeenCalledTimes(3)
+ expect(readFileTool).toHaveBeenCalledTimes(1)
+ expect(executeCommandTool).toHaveBeenCalledTimes(1)
+
+ // Results are committed in model order with one tool_result per tool_use.
+ expect(cline.userMessageContent.map((item: any) => item.tool_use_id)).toEqual([
+ "search_files:0",
+ "read_file:1",
+ "search_files:2",
+ "search_files:3",
+ "execute_command:4",
+ ])
+
+ expect(cline.currentStreamingContentIndex).toBe(5)
+ expect(cline.userMessageContentReady).toBe(true)
+ })
+
+ it("does not mark the message ready while a later block is still pending after a tool failure", async () => {
+ // The finally-block fallback pushes a synthetic tool_result when a tool
+ // fails without producing one. It must only flip userMessageContentReady
+ // for the LAST block — a mid-message failure used to signal "message
+ // complete" while later blocks still had to execute.
+ const getState = vi.fn().mockResolvedValue({ mode: "code", customModes: [] })
+ vi.mocked(readFileTool).mockImplementationOnce(async () => {
+ throw new Error("boom")
+ })
+ let readyWhenSecondRan: boolean | undefined
+ vi.mocked(readFileTool).mockImplementationOnce(
+ async (_cline: any, _block: any, _ask: any, _handleError: any, pushToolResult: any) => {
+ readyWhenSecondRan = cline.userMessageContentReady
+ pushToolResult("ok")
+ },
+ )
+ const cline = {
+ abort: false,
+ taskId: "task-1",
+ instanceId: "instance-1",
+ presentAssistantMessageLocked: false,
+ presentAssistantMessageHasPendingUpdates: false,
+ currentStreamingContentIndex: 0,
+ assistantMessageContent: [
+ {
+ type: "tool_use",
+ name: "read_file",
+ params: { file_path: "a.ts" },
+ partial: false,
+ toolUseId: "read_file:0",
+ },
+ {
+ type: "tool_use",
+ name: "read_file",
+ params: { file_path: "b.ts" },
+ partial: false,
+ toolUseId: "read_file:1",
+ },
+ ],
+ didCompleteReadingStream: true,
+ userMessageContentReady: false,
+ userMessageContent: [],
+ didRejectTool: false,
+ didAlreadyUseTool: false,
+ currentStreamingDidCheckpoint: false,
+ diffEnabled: false,
+ autoApproveAllCommands: false,
+ consecutiveMistakeCount: 0,
+ providerRef: {
+ deref: () => ({
+ getState,
+ }),
+ },
+ browserSession: {
+ closeBrowser: vi.fn().mockResolvedValue(undefined),
+ },
+ say: vi.fn().mockResolvedValue(undefined),
+ ask: vi.fn().mockResolvedValue({ response: "yesButtonClicked" }),
+ processQueuedMessages: vi.fn(),
+ getToolCallSignature: vi.fn((name: string, params: unknown) => JSON.stringify({ name, params })),
+ checkAndRegisterToolCall: vi.fn().mockReturnValue(false),
+ recordToolUsage: vi.fn(),
+ toolRepetitionDetector: {
+ check: vi.fn().mockReturnValue({ allowExecution: true }),
+ },
+ checkpointSave: vi.fn().mockResolvedValue(undefined),
+ removeStalePartialToolAskMessage: vi.fn().mockResolvedValue(undefined),
+ checkAndCondenseContext: vi.fn().mockResolvedValue(undefined),
+ } as any
+
+ await presentAssistantMessage(cline)
+
+ // The second tool must have observed the message as NOT ready when it ran.
+ expect(readyWhenSecondRan).toBe(false)
+
+ // Both results present: the fallback for the failed call plus the real one.
+ const toolResults = cline.userMessageContent.filter((item: any) => item.type === "tool_result")
+ expect(toolResults).toHaveLength(2)
+ expect(toolResults[0].tool_use_id).toBe("read_file:0")
+ expect(toolResults[0].content[0].text).toContain("did not produce a result")
+ expect(toolResults[1].tool_use_id).toBe("read_file:1")
+
+ expect(cline.currentStreamingContentIndex).toBe(2)
+ expect(cline.userMessageContentReady).toBe(true)
+ })
})
diff --git a/src/core/assistant-message/kilocode/native-tool-call.ts b/src/core/assistant-message/kilocode/native-tool-call.ts
index d1aa148182..8bb3431ccc 100644
--- a/src/core/assistant-message/kilocode/native-tool-call.ts
+++ b/src/core/assistant-message/kilocode/native-tool-call.ts
@@ -6,7 +6,9 @@ export interface NativeToolCall {
id?: string // Only present in first delta
type?: string
function?: {
- name: string
+ // name is only present in the first delta; subsequent deltas carry
+ // arguments only (standard OpenAI-compatible streaming shape).
+ name?: string
arguments: string // JSON string (may be partial during streaming)
}
// forked_change: Track if this is an MCP tool and which server
diff --git a/src/core/assistant-message/presentAssistantMessage.ts b/src/core/assistant-message/presentAssistantMessage.ts
index 29f3a5c834..79775b8038 100644
--- a/src/core/assistant-message/presentAssistantMessage.ts
+++ b/src/core/assistant-message/presentAssistantMessage.ts
@@ -112,6 +112,13 @@ export async function presentAssistantMessage(cline: Task, options: PresentAssis
} finally {
cline.presentAssistantMessageLocked = false
}
+ // The batch stops at the first block that cannot run concurrently (e.g.
+ // execute_command following a run of read-only calls). Continue the
+ // presentation chain so those trailing blocks still execute — the batch
+ // only marks the message ready when it consumed every block.
+ if (!cline.abort && cline.currentStreamingContentIndex < cline.assistantMessageContent.length) {
+ await presentAssistantMessage(cline)
+ }
return
}
}
@@ -967,8 +974,17 @@ export async function presentAssistantMessage(cline: Task, options: PresentAssis
// Make sure the task loop can move forward even on failure —
// otherwise the next iteration may hang waiting for a result.
+ // Only signal readiness when this is the LAST content block: for a
+ // mid-message failure the presentation chain still has blocks to
+ // execute, and flipping the flag here would let the task loop fire
+ // the next request before they are presented (same bug class as the
+ // parallel read-only batch skipping a trailing execute_command).
cline.didAlreadyUseTool = true
- if (cline.didCompleteReadingStream && !isParallelWorker) {
+ if (
+ cline.didCompleteReadingStream &&
+ !isParallelWorker &&
+ blockIndex >= cline.assistantMessageContent.length - 1
+ ) {
cline.userMessageContentReady = true
}
} catch (e) {
@@ -1104,7 +1120,15 @@ async function executeParallelReadOnlyBlocks(cline: Task, blockIndexes: number[]
}
cline.currentStreamingContentIndex = blockIndexes[blockIndexes.length - 1] + 1
- cline.userMessageContentReady = true
+
+ // Only tell the task loop the assistant message is fully presented when the
+ // batch ran through the last content block. getParallelReadOnlyBlockIndexes
+ // stops at the first non-read-only block (e.g. execute_command after a run of
+ // searches), so setting this unconditionally made the task loop fire the next
+ // API request while that trailing tool_use was never presented or executed.
+ if (cline.currentStreamingContentIndex >= cline.assistantMessageContent.length) {
+ cline.userMessageContentReady = true
+ }
}
/**
From bb438917226ab1ca85da389f56bf14ffeca06fc4 Mon Sep 17 00:00:00 2001
From: "matterai-app[bot]"
Date: Wed, 2 Sep 2026 15:37:48 +0530
Subject: [PATCH 3/9] feat(context): add pre-compaction budget warning and
compaction handoff prefix
Port codex-style context window management so the model adapts before and
after auto-compaction instead of redoing work:
- Task: when usage enters a 10-point band below the condense threshold
(CONTEXT_WARNING_BAND_PERCENT), push a one-time
user message telling the model to finish in-flight edits, stop broad
searches and large reads, and keep the todo list current before
compaction replaces raw tool outputs with a summary. The warning re-arms
when usage drops back below the band (e.g. after a successful
condensation), so each context window gets at most one warning.
- Condense: the summary prompt now records exploration already performed
(searches, reads, investigations and their conclusions) and demands
enough detail (paths, signatures, line numbers, error text) to continue
without re-reading files. The inserted summary message is prefixed with
a [CONTEXT COMPACTION] handoff note (SUMMARY_PREFIX) instructing the
model to treat the summary as the authoritative record and continue
from NEXT STEPS instead of re-searching, re-reading, or re-deriving.
- Update condense tests for the prefix and new prompt wording.
---
src/core/condense/__tests__/condense.spec.ts | 4 +--
src/core/condense/__tests__/index.spec.ts | 10 +++---
src/core/condense/index.ts | 14 ++++++++-
src/core/task/Task.ts | 32 ++++++++++++++++++++
4 files changed, 52 insertions(+), 8 deletions(-)
diff --git a/src/core/condense/__tests__/condense.spec.ts b/src/core/condense/__tests__/condense.spec.ts
index 5eb97b3e8a..1498c80402 100644
--- a/src/core/condense/__tests__/condense.spec.ts
+++ b/src/core/condense/__tests__/condense.spec.ts
@@ -6,7 +6,7 @@ import { TelemetryService } from "@roo-code/telemetry"
import { BaseProvider } from "../../../api/providers/base-provider"
import { ApiMessage } from "../../task-persistence/apiMessages"
-import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP } from "../index"
+import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP, SUMMARY_PREFIX } from "../index"
// Create a mock ApiHandler for testing
class MockApiHandler extends BaseProvider {
@@ -81,7 +81,7 @@ describe("Condense", () => {
// Verify we have a summary message
const summaryMessage = result.messages.find((msg) => msg.isSummary)
expect(summaryMessage).toBeTruthy()
- expect(summaryMessage?.content).toBe("Mock summary of the conversation")
+ expect(summaryMessage?.content).toBe(`${SUMMARY_PREFIX}\n\nMock summary of the conversation`)
// Verify we have the expected number of messages
// [first message, summary, last N messages]
diff --git a/src/core/condense/__tests__/index.spec.ts b/src/core/condense/__tests__/index.spec.ts
index fa59b5fba2..62fb2b6231 100644
--- a/src/core/condense/__tests__/index.spec.ts
+++ b/src/core/condense/__tests__/index.spec.ts
@@ -7,7 +7,7 @@ import { TelemetryService } from "@roo-code/telemetry"
import { ApiHandler } from "../../../api"
import { ApiMessage } from "../../task-persistence/apiMessages"
import { maybeRemoveImageBlocks } from "../../../api/transform/image-cleaning"
-import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP } from "../index"
+import { summarizeConversation, getMessagesSinceLastSummary, N_MESSAGES_TO_KEEP, SUMMARY_PREFIX } from "../index"
vi.mock("../../../api/transform/image-cleaning", () => ({
maybeRemoveImageBlocks: vi.fn((messages: ApiMessage[], _apiHandler: ApiHandler) => [...messages]),
@@ -197,7 +197,7 @@ describe("summarizeConversation", () => {
// Check that the summary message was inserted correctly
const summaryMessage = result.messages[1]
expect(summaryMessage.role).toBe("assistant")
- expect(summaryMessage.content).toBe("This is a summary")
+ expect(summaryMessage.content).toBe(`${SUMMARY_PREFIX}\n\nThis is a summary`)
expect(summaryMessage.isSummary).toBe(true)
// Check that the last N_MESSAGES_TO_KEEP messages are preserved
@@ -275,7 +275,7 @@ describe("summarizeConversation", () => {
// Verify that createMessage was called with the correct prompt
expect(mockApiHandler.createMessage).toHaveBeenCalledWith(
- expect.stringContaining("Your task is to create a detailed summary of the conversation"),
+ expect.stringContaining("Summarize this conversation with maximum information density"),
expect.any(Array),
)
@@ -653,7 +653,7 @@ describe("summarizeConversation with custom settings", () => {
// Verify the default prompt was used
let createMessageCalls = (mockMainApiHandler.createMessage as Mock).mock.calls
expect(createMessageCalls.length).toBe(1)
- expect(createMessageCalls[0][0]).toContain("Your task is to create a detailed summary")
+ expect(createMessageCalls[0][0]).toContain("Summarize this conversation with maximum information density")
// Reset mock and test with undefined
vi.clearAllMocks()
@@ -670,7 +670,7 @@ describe("summarizeConversation with custom settings", () => {
// Verify the default prompt was used again
createMessageCalls = (mockMainApiHandler.createMessage as Mock).mock.calls
expect(createMessageCalls.length).toBe(1)
- expect(createMessageCalls[0][0]).toContain("Your task is to create a detailed summary")
+ expect(createMessageCalls[0][0]).toContain("Summarize this conversation with maximum information density")
})
/**
diff --git a/src/core/condense/index.ts b/src/core/condense/index.ts
index 2183d3679b..933876c1f1 100644
--- a/src/core/condense/index.ts
+++ b/src/core/condense/index.ts
@@ -28,6 +28,8 @@ FILES: For each file that was read, modified, or created:
- Changes made (if any)
- Key code: function signatures, type definitions, error messages, constants
+EXPLORATION ALREADY DONE: Searches, file reads, and investigations already performed and what each concluded — so they are NOT repeated after compaction.
+
FAILED APPROACHES: What was tried and did not work. Why it failed. What should NOT be attempted again.
KNOWLEDGE STATE:
@@ -41,10 +43,18 @@ NEXT STEPS: Immediate next action — include a VERBATIM quote of the most recen
RULES:
- Preserve every file path, function name, variable name, error message, and identifier exactly as it appeared. Never paraphrase identifiers.
+- The summary is the only surviving record: anything omitted is lost and must be re-derived by re-reading files. Include enough detail (paths, signatures, line numbers, error text) to continue without re-reading.
- Prefer listing over prose. Every token counts.
- Output ONLY the summary. No preamble, no "Here is the summary:", no commentary after.
`
+// Prepended to the summary message inserted into the conversation after
+// compaction. Ports codex's compact/summary_prefix.md: tells the model the
+// history was compacted and that it must build on the recorded work instead
+// of redoing it (re-searching, re-reading files, re-deriving facts).
+export const SUMMARY_PREFIX = `\
+[CONTEXT COMPACTION] An earlier assistant began this task and produced the summary below as a handoff. Treat it as the authoritative record of all work so far: do NOT repeat anything it describes — no re-running searches, no re-reading files it covers, no re-deriving facts it states. Continue from NEXT STEPS.`
+
export type SummarizeResponse = {
messages: ApiMessage[] // The messages after summarization
summary: string // The summary text; empty string for no summary
@@ -181,9 +191,11 @@ export async function summarizeConversation(
return { ...response, cost, error }
}
+ // The prefix tells the model this is a compaction handoff and that it must
+ // not redo the work recorded in the summary.
const summaryMessage: ApiMessage = {
role: "assistant",
- content: summary,
+ content: `${SUMMARY_PREFIX}\n\n${summary}`,
ts: keepMessages[0].ts,
isSummary: true,
}
diff --git a/src/core/task/Task.ts b/src/core/task/Task.ts
index 19aa45d750..948d6497fb 100644
--- a/src/core/task/Task.ts
+++ b/src/core/task/Task.ts
@@ -156,6 +156,7 @@ import { getAppUrl } from "@roo-code/types"
const MAX_EXPONENTIAL_BACKOFF_SECONDS = 600 // 10 minutes
const DEFAULT_USAGE_COLLECTION_TIMEOUT_MS = 5000 // 5 seconds
const FORCED_CONTEXT_REDUCTION_PERCENT = 75 // Keep 75% of context (remove 25%) on context window errors
+const CONTEXT_WARNING_BAND_PERCENT = 10 // Warn the model this many percentage points below the condense threshold
const MAX_CONTEXT_WINDOW_RETRIES = 3 // Maximum retries for context window errors
// kilocode_change: Idle timeout for stream consumption. If no chunk is received
@@ -509,6 +510,9 @@ export class Task extends EventEmitter implements TaskLike {
assistantMessageParser: AssistantMessageParser
private lastUsedInstructions?: string
private skipPrevResponseIdOnce: boolean = false
+ // forked_change: set once per context window when the context budget warning
+ // fires; re-arms automatically when usage drops back below the warning band
+ private contextBudgetWarningIssued: boolean = false
// Token Usage Cache
private tokenUsageSnapshot?: TokenUsage
@@ -3946,6 +3950,34 @@ export class Task extends EventEmitter implements TaskLike {
// Check if we're at or above the threshold
if (contextPercent < effectiveThreshold) {
+ // forked_change start: codex-style context budget warning.
+ // When usage enters the band just below the condense threshold, warn
+ // the model ONCE so it adapts before compaction discards raw tool
+ // outputs: finish in-flight edits, stop broad exploration, keep the
+ // todo list current. Re-arms when usage drops back below the band
+ // (i.e. after a successful condensation), so each context window
+ // gets at most one warning.
+ if (contextWindow > 0) {
+ const warningThreshold = Math.max(
+ MIN_CONDENSE_THRESHOLD,
+ effectiveThreshold - CONTEXT_WARNING_BAND_PERCENT,
+ )
+ if (contextPercent < warningThreshold) {
+ this.contextBudgetWarningIssued = false
+ } else if (!this.contextBudgetWarningIssued) {
+ this.contextBudgetWarningIssued = true
+ const tokensRemaining = Math.max(0, contextWindow - contextTokens)
+ console.log(
+ `[Task#${this.taskId}] Context budget warning: ${contextPercent.toFixed(1)}% used ` +
+ `(~${tokensRemaining} of ${contextWindow} tokens remain), condense threshold ${effectiveThreshold}%.`,
+ )
+ this.userMessageContent.push({
+ type: "text",
+ text: `\nContext window nearly full: ${contextPercent.toFixed(0)}% used (~${tokensRemaining.toLocaleString()} of ${contextWindow.toLocaleString()} tokens remain). Auto-compaction triggers at ${effectiveThreshold}% and replaces raw tool outputs with a summary.\nAdapt now:\n- Finish in-flight edits before starting anything new.\n- No broad searches or large file reads; use targeted reads of specific regions only.\n- Do not re-read files whose contents are already in context.\n- Keep the todo list current so pending work survives compaction.\n`,
+ })
+ }
+ }
+ // forked_change end
return false
}
From e33136a3f48906dced981e22413f5c045954fd25 Mon Sep 17 00:00:00 2001
From: "matterai-app[bot]"
Date: Wed, 2 Sep 2026 15:38:05 +0530
Subject: [PATCH 4/9] feat(prompts): rework edit batching guidance and add
multi-repo workspace rules
- file_edit/multi_file_edit guidance now keys on edits that are confirmed
and ready now rather than a blanket "always batch 2+ edits": make a
ready edit immediately instead of holding it back, keep batches small
and cohesive (the edits belonging to the current step), and never
accumulate a whole task into one giant multi-file edit.
- Replace "Plan before editing" with "Edit early, iterate in small
steps": make the first edit as soon as one file's change is confirmed,
alternate editing and checking (typecheck, test, targeted read) as the
intended workflow, and track remaining work with update_todo_list
instead of holding a full multi-file plan in context. Do not re-read a
file just to confirm an edit succeeded.
- Add a Multi-repo workspaces section: work inside the repo that owns the
code being changed, cross into another repo only when the task requires
it, and never interleave reads across repos.
- Refresh prompt snapshots.
---
.../architect-mode-prompt.snap | 27 ++++++++++++-------
.../ask-mode-prompt.snap | 27 ++++++++++++-------
.../mcp-server-creation-disabled.snap | 27 ++++++++++++-------
.../partial-reads-enabled.snap | 27 ++++++++++++-------
.../consistent-system-prompt.snap | 27 ++++++++++++-------
.../with-computer-use-support.snap | 27 ++++++++++++-------
.../system-prompt/with-diff-enabled-true.snap | 27 ++++++++++++-------
.../with-different-viewport-size.snap | 27 ++++++++++++-------
.../system-prompt/with-undefined-mcp-hub.snap | 27 ++++++++++++-------
src/core/prompts/system.ts | 27 ++++++++++++-------
10 files changed, 180 insertions(+), 90 deletions(-)
diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap
index d67b9e076d..4a53942ed0 100644
--- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap
+++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/architect-mode-prompt.snap
@@ -567,8 +567,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -578,11 +578,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -608,7 +608,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap
index 79bfc84eca..ed7c8c459a 100644
--- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap
+++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/ask-mode-prompt.snap
@@ -427,8 +427,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -438,11 +438,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -468,7 +468,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -554,10 +555,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap
index 67be5f472b..f049284600 100644
--- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap
+++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/mcp-server-creation-disabled.snap
@@ -566,8 +566,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -577,11 +577,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -607,7 +607,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -693,10 +694,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap
index 10613f49cf..410f130279 100644
--- a/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap
+++ b/src/core/prompts/__tests__/__snapshots__/add-custom-instructions/partial-reads-enabled.snap
@@ -572,8 +572,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -583,11 +583,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -613,7 +613,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -699,10 +700,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap
index d67b9e076d..4a53942ed0 100644
--- a/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap
+++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/consistent-system-prompt.snap
@@ -567,8 +567,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -578,11 +578,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -608,7 +608,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap
index ffa6e30c24..2d6eedc8e2 100644
--- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap
+++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-computer-use-support.snap
@@ -620,8 +620,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -631,11 +631,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -661,7 +661,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -747,10 +748,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap
index d67b9e076d..4a53942ed0 100644
--- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap
+++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-diff-enabled-true.snap
@@ -567,8 +567,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -578,11 +578,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -608,7 +608,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap
index 4afe89345e..299127f4bf 100644
--- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap
+++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-different-viewport-size.snap
@@ -620,8 +620,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -631,11 +631,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -661,7 +661,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -747,10 +748,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap
index d67b9e076d..4a53942ed0 100644
--- a/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap
+++ b/src/core/prompts/__tests__/__snapshots__/system-prompt/with-undefined-mcp-hub.snap
@@ -567,8 +567,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use `multi_file_edit` instead.
-- Never call `file_edit` multiple times in sequence. Batch your edits with `multi_file_edit`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use `multi_file_edit` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. `file_path` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -578,11 +578,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. `edits` — An array of edit objects. Each edit has:
@@ -608,7 +608,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → `file_edit`
-- 2+ edits → `multi_file_edit` (always)
+- 2+ edits ready now → `multi_file_edit`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy `old_string` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed `old_string` will silently mismatch.
@@ -694,10 +695,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via `multi_file_edit`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with `update_todo_list` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
diff --git a/src/core/prompts/system.ts b/src/core/prompts/system.ts
index 95a61f31c6..08c0308ada 100644
--- a/src/core/prompts/system.ts
+++ b/src/core/prompts/system.ts
@@ -107,8 +107,8 @@ Common tool calls and explanations
- You know the exact text that should be replaced and its updated form.
**When NOT to use**:
-- If you have **2 or more edits** to make (even to the same file), use \`multi_file_edit\` instead.
-- Never call \`file_edit\` multiple times in sequence. Batch your edits with \`multi_file_edit\`.
+- If you have **2 or more independent edits** that are all confirmed and ready right now, use \`multi_file_edit\` instead.
+- Do not hold an edit back to accumulate a larger batch. If one edit is ready, make it now and gather the next edits afterwards.
**Parameters**:
1. \`file_path\` — Absolute path to the file you want to modify (e.g., /Users/username/project/src/file.ts).
@@ -118,11 +118,11 @@ Common tool calls and explanations
## multi_file_edit
-**Description**: Make multiple text replacements across one or more files in a single tool call. This is the **preferred** tool for editing when you have 2+ changes to make.
+**Description**: Make multiple text replacements across one or more files in a single tool call. Use it when several edits are already confirmed and ready in the same step.
**When to use**:
-- You have **2 or more edits** to make, whether to the same file or different files.
-- You want to batch edits efficiently instead of making multiple separate tool calls.
+- You have **2 or more edits** that are all confirmed and ready now, whether to the same file or different files.
+- Keep each batch small and cohesive: the edits that belong to the current step of the task, not the whole task.
**Parameters**:
1. \`edits\` — An array of edit objects. Each edit has:
@@ -148,7 +148,8 @@ Common tool calls and explanations
**Guidance for choosing between file_edit and multi_file_edit**:
- 1 edit → \`file_edit\`
-- 2+ edits → \`multi_file_edit\` (always)
+- 2+ edits ready now → \`multi_file_edit\`
+- Never accumulate edits across the whole task into one giant batch. Edit granularity follows the steps of the task.
**Editing discipline (CRITICAL)**:
- ALWAYS copy \`old_string\` verbatim from a read_file result obtained in the same turn. NEVER reconstruct indentation or whitespace from memory — this is especially important in tab-indented files, where a reconstructed \`old_string\` will silently mismatch.
@@ -234,10 +235,18 @@ Use zero context for discovery, then read the relevant file region. If results a
- After EVERY tool call, verify the output actually matches the parameters you sent (correct file, correct line range, correct directory). A result that does not reflect your parameters means the call was malformed — fix the call, do not reason from the bad output.
- If two consecutive identical tool calls produce identical results, you are in a loop. Change the call or change the strategy. NEVER repeat the same call a third time.
-## Plan before editing
+## Edit early, iterate in small steps
-- Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything.
-- Then execute the edits in one pass (batched via \`multi_file_edit\`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking.
+- Make the first edit as soon as the change for one file is confirmed. Do not map the whole codebase before touching anything — gather context per step, on demand, between edits.
+- Alternate editing and checking: make an edit, run the relevant check (typecheck, test, or a targeted read), then continue. This is the intended workflow, not a planning failure.
+- Track remaining work with \`update_todo_list\` (one step in progress at a time, updated after each sub-task) instead of holding a full multi-file plan in context.
+- Keep each batch of edits small and cohesive — the edits that belong to the current step. A change spanning many files is executed as a sequence of small verified steps, not one giant multi-file edit.
+- Do not re-read a file just to confirm an edit succeeded; the tool result already reports success or failure. (Re-reading before a NEW edit in the same area is still required.)
+
+## Multi-repo workspaces
+
+- When several repositories or workspace roots are open, work inside the one that owns the code being changed. Do not read sibling repos to "understand the ecosystem."
+- Cross into another repo only when the task explicitly requires it (e.g., mirroring a change in a consumer). Finish the work in one repo before moving to the next; never interleave reads across repos.
## Investigation efficiency
From f82bfd8cb64de628572ef26f0729388d52d4a251 Mon Sep 17 00:00:00 2001
From: "matterai-app[bot]"
Date: Wed, 2 Sep 2026 15:38:49 +0530
Subject: [PATCH 5/9] chore(release): v6.8.4
- Bump the extension version from 6.8.2 to 6.8.4.
- Add the v6.8.3 changelog entry documenting the OSS model catalog that
replaces the Axon models, including per-model pricing and the new
deepseek/deepseek-v4-flash-0731 default.
- Consume the applied changesets (axon default model, pasted image paths,
Fireworks provider removal).
---
.changeset/axon-auto-default-model.md | 10 ----------
.changeset/pasted-image-paths.md | 11 -----------
.changeset/remove-fireworks-provider.md | 13 -------------
CHANGELOG.md | 6 ++++++
src/package.json | 2 +-
5 files changed, 7 insertions(+), 35 deletions(-)
delete mode 100644 .changeset/axon-auto-default-model.md
delete mode 100644 .changeset/pasted-image-paths.md
delete mode 100644 .changeset/remove-fireworks-provider.md
diff --git a/.changeset/axon-auto-default-model.md b/.changeset/axon-auto-default-model.md
deleted file mode 100644
index 7aea09693d..0000000000
--- a/.changeset/axon-auto-default-model.md
+++ /dev/null
@@ -1,10 +0,0 @@
----
-"kilo-code": minor
----
-
-Make Axon Auto the default model
-
-- Changed `openRouterDefaultModelId` in `@roo-code/types` from `axon-eido-3-code-mini-232k` to `axon-auto-232k`, so the Orbital/KiloCode provider defaults to the Axon Auto model.
-- Updated the `openRouterDefaultModelInfo` fallback description to describe Axon Auto (dynamic Flash/Mini/Pro selection).
-- Updated `getAxonPlanFallback` in `model-plan-access` to fall back to `axon-auto-232k` instead of `axon-eido-3-code-mini-232k` for plan-restricted users.
-- Added the `axon-eido-3-flash-400k` model variant to the KiloCode model catalog (extension and webview), gated behind the same Pro Plus/Ultra 400k plan check as the other 400k variants; `is400kAxonModel` now recognizes `axon-eido-3-flash-*` ids and the 232k fallback maps it to `axon-eido-3-flash`.
diff --git a/.changeset/pasted-image-paths.md b/.changeset/pasted-image-paths.md
deleted file mode 100644
index 062e1d7ab4..0000000000
--- a/.changeset/pasted-image-paths.md
+++ /dev/null
@@ -1,11 +0,0 @@
-## "kilo-code": patch
-
-Attach files pasted or dropped into the chat input
-
-- Pasting a supported file path (absolute, relative to the workspace, or a `file://` URI) into the chat textarea now attaches the file instead of inserting the path as plain text, so the content is actually sent to the LLM.
-- Copying a file directly (e.g. from Finder/Explorer) and pasting it now attaches non-image documents (PDF, DOCX, XLSX, CSV, JSON, MD, TXT, TSV) too — previously only image blobs were handled and other files vanished silently.
-- Supports images (`.png`, `.jpg`, `.jpeg`, `.webp`) and documents (`.csv`, `.docx`, `.json`, `.md`, `.pdf`, `.txt`, `.text`, `.tsv`, `.xlsx`), reusing the same extraction/size limits and the existing `selectedAttachments` response channel as the attachment picker.
-- Refactored `process-attachments.ts` to share the per-file processing logic between the file picker, the pasted-path flow, and the pasted-blob flow.
-- Document attachment chips now show the extension-specific material icon (e.g. PDF, Excel, Word) instead of a generic file icon, reusing the same `vscode-material-icons` mapping as mention chips.
-- Unified the image and document attachment chips into a single shared flex-wrap container with matching pill styling, so images and files render at the same size and flow together across rows.
-- Drag-and-drop now attaches image and document files too: paths dropped from the VS Code explorer are routed to attachments (instead of mentions) when they match a supported type, and non-image file blobs dropped from the OS are processed by the extension. `handleDrop` only intercepts drags that look like file/path drops (external files or path-like text); plain text drags within the editor fall through to native behavior. Drag-and-drop still requires holding Shift because VS Code intercepts native file drags otherwise.
diff --git a/.changeset/remove-fireworks-provider.md b/.changeset/remove-fireworks-provider.md
deleted file mode 100644
index 54ac902544..0000000000
--- a/.changeset/remove-fireworks-provider.md
+++ /dev/null
@@ -1,13 +0,0 @@
----
-"kilo-code": minor
----
-
-Remove the Fireworks AI provider
-
-- Removed the `FireworksHandler` from `src/api/providers`, the `fireworks` entry from the third-party model fetcher, the `fireworks` provider schema (`fireworksSchema`, `fireworksModels`, `fireworksDefaultModelId`) from `@roo-code/types`, and the `Fireworks` settings UI component.
-- Removed Fireworks from `MODELS_BY_PROVIDER`, `modelIdKeysByProvider`, `PROVIDER_CONFIGS`, `thirdPartyProviderRequiresApiKey`, `ProfileValidator`, `taskMetadata`, `webviewMessageHandler`, `ClineProvider`, `Task`, `useProviderModels`, `useSelectedModel`, and the `OpenAI` provider's Fireworks detection branch in `src/api/providers/openai.ts`.
-- Removed the Fireworks section from the `ThirdPartyProviders` settings panel and the Fireworks `HardcodedModelRecord` for `fireworks:accounts/fireworks/routers/kimi-k2p5-turbo`.
-- Removed Fireworks entries from the CLI: `cli/src/constants/providers/{settings,models,validation,labels}.ts`, `cli/src/types/messages.ts`, `cli/src/config/schema.json`, `cli/docs/PROVIDER_CONFIGURATION.md`, and `cli/src/constants/providers/__tests__/models.test.ts`.
-- Removed the Fireworks provider docs page and sidebar entry from `apps/kilocode-docs`.
-- Removed `fireworksApiKey` and `getFireworksApiKey` i18n strings from all 22 webview locales (ar, ca, cs, de, en, es, fr, hi, id, it, ja, ko, nl, pl, pt-BR, ru, th, tr, uk, vi, zh-CN, zh-TW).
-- Removed the unused `fireworks-ic.png` icon.
diff --git a/CHANGELOG.md b/CHANGELOG.md
index da995ee7e6..7f20bfcae2 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,11 @@
# Changelog
+## [v6.8.3] - 2026-09-02
+
+### Changed
+
+- **OSS model catalog replaces Axon models.** The KiloCode model catalog (extension `kilocode-models.ts` and the webview `useOpenRouterModelProviders` copy) now exposes seven OSS models — `meta/muse-spark-1.2-contributor` (Muse Spark 1.2 Contributor), `deepseek/deepseek-v4-flash-0731` (DeepSeek V4 Flash), `zai/glm-5.3` (GLM 5.3), `zai/glm-5.3-flash` (GLM 5.3 Flash), `gpt-5.6-sol` (GPT-5.6 Sol), `gpt-5.6-luna` (GPT-5.6 Luna), and `gemini-3.7-flash` (Gemini 3.7 Flash) — in place of the Axon context-window variants. Each OSS model carries its published per-token pricing (Muse Spark 1.2 Contributor $0.10/M input, $0.002/M cache read, $0.20/M output; DeepSeek V4 Flash $0.14/M input, $0.028/M cache read, $0.28/M output; GLM 5.3 $1.40/M input, $0.14/M cache read, $4.40/M output; GLM 5.3 Flash $0.15/M input, $0.03/M cache read, $0.50/M output; GPT-5.6 Sol $5/M input, $0.50/M cache read, $30/M output; GPT-5.6 Luna $0.20/M input, $0.02/M cache read, $1.20/M output; Gemini 3.7 Flash $0.75/M input, $0.075/M cache read, $3.75/M output). The default model is `deepseek/deepseek-v4-flash-0731` (`openRouterDefaultModelId` in `@roo-code/types`, plus the CLI `kilocodeModel` defaults and web-evals `MODEL_DEFAULT`). Stored Axon model selections are now stale and reset to the default on next launch via the existing `isValidKilocodeModel` stale-model check.
+
## [v6.8.2] - 2026-08-28
### Added
diff --git a/src/package.json b/src/package.json
index 5191be3023..a9efcd95f6 100644
--- a/src/package.json
+++ b/src/package.json
@@ -3,7 +3,7 @@
"displayName": "%extension.displayName%",
"description": "%extension.description%",
"publisher": "matterai",
- "version": "6.8.2",
+ "version": "6.8.4",
"icon": "assets/icons/matterai-ic.png",
"galleryBanner": {
"color": "#FFFFFF",
From 1233955e153eb65217dc54c1db0c5174110cf4ac Mon Sep 17 00:00:00 2001
From: "matterai-app[bot]"
Date: Thu, 3 Sep 2026 12:22:16 +0530
Subject: [PATCH 6/9] feat(models): dynamic backend model catalog and per-model
usage visibility
- Fetch the model catalog from the MatterAI backend and register it via registerDynamicKilocodeModels, replacing static hardcoded lists; refresh on window focus, manual refresh, and a 10-minute poller with forceRefresh cache bypass
- Show per-model usage (weekly/monthly share of the shared plan pool, cost-multiplier badges) in the usage dialog and a new read-only Settings Model Usage section; /axoncode/profile now returns modelUsage
- Fix refreshKilocodeModels wiping OpenRouter models by fetching openrouter and kilocode-openrouter in parallel and merging results
- Fetch the catalog at provider startup so stale axon model selections are reset via isValidKilocodeModel against a populated catalog
---
CHANGELOG.md | 13 +-
.../__tests__/kilocode-models.spec.ts | 169 +++++++++--
src/api/providers/fetchers/modelCache.ts | 21 +-
src/api/providers/fetchers/openrouter.ts | 236 ++++++++-------
src/api/providers/kilocode-models.ts | 271 ++++++++++-------
src/core/webview/ClineProvider.ts | 111 +++++++
.../webview/__tests__/ClineProvider.spec.ts | 9 +
src/core/webview/webviewMessageHandler.ts | 9 +-
src/shared/WebviewMessage.ts | 15 +
src/shared/api.ts | 1 +
.../src/components/chat/UsageDialog.tsx | 52 +++-
.../kilocode/chat/ModelSelector.tsx | 47 ++-
.../kilocode/hooks/useProviderModels.ts | 1 +
.../settings/ModelUsageSettings.tsx | 205 +++++++++++++
.../src/components/settings/SettingsView.tsx | 9 +-
.../ui/hooks/useOpenRouterModelProviders.ts | 273 ++++++++----------
.../components/ui/hooks/useRouterModels.ts | 45 ++-
webview-ui/src/i18n/locales/en/settings.json | 1 +
18 files changed, 1077 insertions(+), 411 deletions(-)
create mode 100644 webview-ui/src/components/settings/ModelUsageSettings.tsx
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 7f20bfcae2..eef5475856 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,11 +1,22 @@
# Changelog
-## [v6.8.3] - 2026-09-02
+## [v6.8.4] - 2026-09-03
+
+### Added
+
+- **Per-model usage visibility.** The usage dialog and a new Settings → Model Usage section now show each tracked OSS model's share of the shared plan pool as weekly/monthly percentages (no credit amounts), alongside the model's plan-cost multiplier badge (e.g. `5x cost`). The new settings section also renders the weekly/monthly plan windows with reset times and a refresh action, and is excluded from the save-button flow since it is read-only. Backend: `/axoncode/profile` now returns a `modelUsage` array (`model`, `multiplier`, `weeklyPercentage`, `monthlyPercentage`); charges for tracked OSS models are multiplied by their plan-cost multiplier before draining the shared pool.
### Changed
+- **Dynamic model catalog synchronization.** Models are now dynamically fetched from the MatterAI backend `/v1/models` (or `/v1/web/models`) and registered into the client model registry (`registerDynamicKilocodeModels`), replacing static hardcoded lists while continuing to exclude deprecated axon models from active selection. Models automatically refresh on window focus, via the refresh button in the model selector, and through a 10-minute background poller. Cache bypass (`forceRefresh`) ensures database updates reflect immediately without restarting the extension.
+
- **OSS model catalog replaces Axon models.** The KiloCode model catalog (extension `kilocode-models.ts` and the webview `useOpenRouterModelProviders` copy) now exposes seven OSS models — `meta/muse-spark-1.2-contributor` (Muse Spark 1.2 Contributor), `deepseek/deepseek-v4-flash-0731` (DeepSeek V4 Flash), `zai/glm-5.3` (GLM 5.3), `zai/glm-5.3-flash` (GLM 5.3 Flash), `gpt-5.6-sol` (GPT-5.6 Sol), `gpt-5.6-luna` (GPT-5.6 Luna), and `gemini-3.7-flash` (Gemini 3.7 Flash) — in place of the Axon context-window variants. Each OSS model carries its published per-token pricing (Muse Spark 1.2 Contributor $0.10/M input, $0.002/M cache read, $0.20/M output; DeepSeek V4 Flash $0.14/M input, $0.028/M cache read, $0.28/M output; GLM 5.3 $1.40/M input, $0.14/M cache read, $4.40/M output; GLM 5.3 Flash $0.15/M input, $0.03/M cache read, $0.50/M output; GPT-5.6 Sol $5/M input, $0.50/M cache read, $30/M output; GPT-5.6 Luna $0.20/M input, $0.02/M cache read, $1.20/M output; Gemini 3.7 Flash $0.75/M input, $0.075/M cache read, $3.75/M output). The default model is `deepseek/deepseek-v4-flash-0731` (`openRouterDefaultModelId` in `@roo-code/types`, plus the CLI `kilocodeModel` defaults and web-evals `MODEL_DEFAULT`). Stored Axon model selections are now stale and reset to the default on next launch via the existing `isValidKilocodeModel` stale-model check.
+### Fixed
+
+- **OpenRouter models no longer wiped on background refresh.** `refreshKilocodeModels` (window-focus refresh, 10-minute poller, startup fetch) previously posted `openrouter: {}` to the webview, blanking the OpenRouter model list until the next manual reload. It now fetches both `openrouter` and `kilocode-openrouter` in parallel (mirroring the `requestRouterModels` handler) and posts the merged payload, logging and skipping only the provider that failed.
+- **Stale model selections reset on launch.** The dynamic catalog is now fetched once at provider startup (after `ContextProxy` initialization) and the webview state re-posted, so the `isValidKilocodeModel` stale-model check runs against a populated catalog instead of the empty-catalog bypass. Previously, a stored axon model survived until the first focus- or webview-triggered fetch completed.
+
## [v6.8.2] - 2026-08-28
### Added
diff --git a/src/api/providers/__tests__/kilocode-models.spec.ts b/src/api/providers/__tests__/kilocode-models.spec.ts
index dfaec86e25..d1ec044a42 100644
--- a/src/api/providers/__tests__/kilocode-models.spec.ts
+++ b/src/api/providers/__tests__/kilocode-models.spec.ts
@@ -1,15 +1,75 @@
-import { getKilocodeApiModelId, KILO_CODE_MODELS, type KiloCodeModel } from "../kilocode-models"
-
-const OSS_MODEL_IDS = [
- "meta/muse-spark-1.2-contributor",
- "deepseek/deepseek-v4-flash-0731",
- "zai/glm-5.3",
- "zai/glm-5.3-flash",
- "gpt-5.6-sol",
- "gpt-5.6-luna",
- "gemini-3.7-flash",
+import {
+ getKilocodeApiModelId,
+ isValidKilocodeModel,
+ KILO_CODE_MODELS,
+ registerDynamicKilocodeModels,
+ type KiloCodeModel,
+} from "../kilocode-models"
+
+// Raw catalog entries as served by the MatterAI backend /v1/web/models
+// endpoint (non-axon models only). The static catalog ships empty; models
+// arrive at runtime via registerDynamicKilocodeModels.
+const RAW_OSS_MODELS: Array> = [
+ {
+ id: "meta/muse-spark-1.3-contributor",
+ name: "Muse Spark 1.3 Contributor",
+ description: "Meta Muse Spark 1.3 Contributor is an open general purpose model for everyday coding tasks.",
+ context_length: 232000,
+ owned_by: "meta",
+ pricing: { prompt: "0.0000001", completion: "0.0000002", input_cache_reads: "0.000000002" },
+ },
+ {
+ id: "deepseek/deepseek-v4-flash-0731",
+ name: "DeepSeek V4 Flash",
+ description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.",
+ context_length: 232000,
+ owned_by: "deepseek",
+ pricing: { prompt: "0.00000014", completion: "0.00000028", input_cache_reads: "0.000000028" },
+ },
+ {
+ id: "zai/glm-5.3",
+ name: "GLM 5.3",
+ description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.",
+ context_length: 232000,
+ owned_by: "zai",
+ pricing: { prompt: "0.0000014", completion: "0.0000044", input_cache_reads: "0.00000014" },
+ },
+ {
+ id: "zai/glm-5.3-flash",
+ name: "GLM 5.3 Flash",
+ description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.",
+ context_length: 232000,
+ owned_by: "zai",
+ pricing: { prompt: "0.00000015", completion: "0.0000005", input_cache_reads: "0.00000003" },
+ },
+ {
+ id: "gpt-5.6-sol",
+ name: "GPT-5.6 Sol",
+ description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.",
+ context_length: 232000,
+ owned_by: "openai",
+ pricing: { prompt: "0.000005", completion: "0.00003", input_cache_reads: "0.0000005" },
+ },
+ {
+ id: "gpt-5.6-luna",
+ name: "GPT-5.6 Luna",
+ description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.",
+ context_length: 232000,
+ owned_by: "openai",
+ pricing: { prompt: "0.0000002", completion: "0.0000012", input_cache_reads: "0.00000002" },
+ },
+ {
+ id: "gemini-3.7-flash",
+ name: "Gemini 3.7 Flash",
+ description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.",
+ context_length: 232000,
+ owned_by: "google",
+ pricing: { prompt: "0.00000075", completion: "0.00000375", input_cache_reads: "0.000000075" },
+ },
]
+const OSS_MODEL_IDS = RAW_OSS_MODELS.map((raw) => raw.id as string)
+
const getSharedMetadata = ({
id: _id,
name: _name,
@@ -21,14 +81,22 @@ const getSharedMetadata = ({
...metadata
}: KiloCodeModel) => metadata
-describe("KiloCode OSS model catalog", () => {
- it("exposes exactly the seven OSS models", () => {
+describe("KiloCode dynamic model catalog", () => {
+ // KILO_CODE_MODELS is module-level mutable state shared across tests in
+ // this file; seed the backend catalog before each test.
+ beforeEach(() => {
+ registerDynamicKilocodeModels(RAW_OSS_MODELS)
+ })
+
+ it("registers exactly the backend catalog models", () => {
expect(Object.keys(KILO_CODE_MODELS).sort()).toEqual([...OSS_MODEL_IDS].sort())
})
- it("no longer exposes axon models", () => {
- expect(KILO_CODE_MODELS["axon-auto-232k"]).toBeUndefined()
- expect(KILO_CODE_MODELS["axon-eido-3.2-code-pro-400k"]).toBeUndefined()
+ it("skips axon models and entries without an id", () => {
+ registerDynamicKilocodeModels([{ id: "axon-eido-3.2-code-flash", name: "Axon" }, { name: "no id" }])
+
+ expect(KILO_CODE_MODELS["axon-eido-3.2-code-flash"]).toBeUndefined()
+ expect(Object.keys(KILO_CODE_MODELS).sort()).toEqual([...OSS_MODEL_IDS].sort())
})
it.each([...OSS_MODEL_IDS])("sends %s to the API unchanged", (modelId) => {
@@ -37,11 +105,16 @@ describe("KiloCode OSS model catalog", () => {
expect(model).toBeDefined()
expect(model?.id).toBe(modelId)
expect(model?.context_length).toBe(232000)
+ expect(model?.openrouter.slug).toBe(modelId)
expect(getKilocodeApiModelId(modelId)).toBe(modelId)
})
- it("prices each OSS model at its published rate", () => {
- expect(KILO_CODE_MODELS["meta/muse-spark-1.2-contributor"]?.pricing).toMatchObject({
+ it("passes unknown ids through unchanged", () => {
+ expect(getKilocodeApiModelId("acme/unknown")).toBe("acme/unknown")
+ })
+
+ it("prices each model at its published rate", () => {
+ expect(KILO_CODE_MODELS["meta/muse-spark-1.3-contributor"]?.pricing).toMatchObject({
prompt: "0.0000001",
completion: "0.0000002",
input_cache_reads: "0.000000002",
@@ -86,7 +159,7 @@ describe("KiloCode OSS model catalog", () => {
}
})
- it("shares identical base metadata across all OSS models", () => {
+ it("shares identical base metadata across all models", () => {
const models = OSS_MODEL_IDS.map((modelId) => KILO_CODE_MODELS[modelId]!)
const [first, ...rest] = models.map(getSharedMetadata)
@@ -94,4 +167,64 @@ describe("KiloCode OSS model catalog", () => {
expect(metadata).toEqual(first)
}
})
+
+ it("falls back to sane defaults for sparse entries", () => {
+ registerDynamicKilocodeModels([{ id: "acme/prototype" }])
+
+ expect(KILO_CODE_MODELS["acme/prototype"]).toMatchObject({
+ name: "acme/prototype",
+ description: "acme/prototype via MatterAI",
+ context_length: 232000,
+ max_output_length: 64000,
+ owned_by: "acme",
+ input_modalities: ["text", "image"],
+ output_modalities: ["text"],
+ })
+ expect(KILO_CODE_MODELS["acme/prototype"].pricing).toMatchObject({
+ prompt: "0",
+ completion: "0",
+ input_cache_reads: "0",
+ input_cache_writes: "0",
+ })
+ })
+
+ it("coerces numeric pricing to strings", () => {
+ registerDynamicKilocodeModels([
+ {
+ id: "acme/numeric",
+ pricing: { prompt: 0.0000001, completion: 0.0000002, input_cache_reads: 0.000000002 },
+ },
+ ])
+
+ // String() renders sub-1e-6 numbers in scientific notation; parsePrice
+ // consumes these via parseFloat, so the coercion stays lossless.
+ expect(KILO_CODE_MODELS["acme/numeric"].pricing).toMatchObject({
+ prompt: "1e-7",
+ completion: "2e-7",
+ input_cache_reads: "2e-9",
+ })
+ })
+
+ it("validates registered models and rejects stale non-auto axon selections", () => {
+ expect(isValidKilocodeModel("zai/glm-5.3")).toBe(true)
+ expect(isValidKilocodeModel("axon-eido-3.2-code-flash")).toBe(false)
+ expect(isValidKilocodeModel("unknown/model")).toBe(false)
+ })
+
+ it("always accepts axon-auto router ids", () => {
+ expect(isValidKilocodeModel("axon-auto")).toBe(true)
+ expect(isValidKilocodeModel("axon-auto-400k")).toBe(true)
+ })
+})
+
+describe("isValidKilocodeModel with an empty catalog", () => {
+ it("accepts any model before the first backend fetch", () => {
+ for (const key of Object.keys(KILO_CODE_MODELS)) {
+ delete KILO_CODE_MODELS[key]
+ }
+
+ expect(Object.keys(KILO_CODE_MODELS).length).toBe(0)
+ expect(isValidKilocodeModel("axon-eido-3.2-code-flash")).toBe(true)
+ expect(isValidKilocodeModel("anything")).toBe(true)
+ })
})
diff --git a/src/api/providers/fetchers/modelCache.ts b/src/api/providers/fetchers/modelCache.ts
index 4c68f17a22..b93f102164 100644
--- a/src/api/providers/fetchers/modelCache.ts
+++ b/src/api/providers/fetchers/modelCache.ts
@@ -57,9 +57,9 @@ export /*kilocode_change*/ async function readModels(router: RouterName): Promis
* @returns The models from the cache or the fetched models.
*/
export const getModels = async (options: GetModelsOptions): Promise => {
- const { provider } = options
+ const { provider, forceRefresh } = options
- let models = getModelsFromCache(provider)
+ let models = !forceRefresh ? getModelsFromCache(provider) : undefined
if (models) {
return models
@@ -71,7 +71,10 @@ export const getModels = async (options: GetModelsOptions): Promise
// forked_change start: base url and bearer token
models = await getOpenRouterModels({
openRouterBaseUrl: options.baseUrl,
- headers: options.apiKey ? { Authorization: `Bearer ${options.apiKey}` } : undefined,
+ headers: {
+ ...(options.apiKey ? { Authorization: `Bearer ${options.apiKey}` } : {}),
+ ...(forceRefresh ? { "Cache-Control": "no-cache" } : {}),
+ },
})
// forked_change end
break
@@ -96,9 +99,19 @@ export const getModels = async (options: GetModelsOptions): Promise
? `https://api.matterai.so/organizations/${options.kilocodeOrganizationId}`
: "https://api.matterai.so/v1/web"
const openRouterBaseUrl = getKiloUrlFromToken(backendUrl, options.kilocodeToken ?? "")
+ const headers: Record = {}
+ if (options.kilocodeToken) {
+ headers["Authorization"] = `Bearer ${options.kilocodeToken}`
+ }
+ if (options.kilocodeOrganizationId) {
+ headers["X-KILOCODE-ORGANIZATIONID"] = options.kilocodeOrganizationId
+ }
+ if (forceRefresh) {
+ headers["Cache-Control"] = "no-cache"
+ }
models = await getOpenRouterModels({
openRouterBaseUrl,
- headers: options.kilocodeToken ? { Authorization: `Bearer ${options.kilocodeToken}` } : undefined,
+ headers,
})
break
}
diff --git a/src/api/providers/fetchers/openrouter.ts b/src/api/providers/fetchers/openrouter.ts
index 943897cbda..576a671715 100644
--- a/src/api/providers/fetchers/openrouter.ts
+++ b/src/api/providers/fetchers/openrouter.ts
@@ -133,7 +133,7 @@ export async function getOpenRouterModels(
const models: Record = {}
// First, add static models from KILO_CODE_MODELS
- const { KILO_CODE_MODELS } = await import("../kilocode-models")
+ const { KILO_CODE_MODELS, registerDynamicKilocodeModels } = await import("../kilocode-models")
for (const [id, model] of Object.entries(KILO_CODE_MODELS)) {
models[id] = parseOpenRouterModel({
@@ -159,59 +159,67 @@ export async function getOpenRouterModels(
}
// Then, fetch dynamic models from MatterAI API
- // try {
- // const headers: Record = {
- // ...DEFAULT_HEADERS,
- // ...options?.headers,
- // }
-
- // const response = await axios.get("https://api.matterai.so/v1/models/openrouter", { headers })
-
- // const rawModels = response.data?.data || []
-
- // for (const rawModel of rawModels) {
- // // Filter out image generation models (only text output)
- // const outputModalities = rawModel.output_modalities || []
- // if (outputModalities.includes("image") && !outputModalities.includes("text")) {
- // continue
- // }
-
- // // Prefix the model ID with "openrouter/"
- // const modelId = `openrouter/${rawModel.id}`
-
- // // Don't override static models
- // if (models[modelId]) {
- // continue
- // }
-
- // models[modelId] = parseOpenRouterModel({
- // id: modelId,
- // model: {
- // name: rawModel.name || rawModel.id,
- // description: rawModel.description,
- // context_length: rawModel.context_length || 8192,
- // max_completion_tokens: rawModel.max_output_length,
- // pricing: rawModel.pricing
- // ? {
- // prompt: rawModel.pricing.prompt,
- // completion: rawModel.pricing.completion,
- // input_cache_read: rawModel.pricing.input_cache_reads,
- // input_cache_write: rawModel.pricing.input_cache_writes,
- // }
- // : undefined,
- // },
- // displayName: rawModel.name,
- // inputModality: rawModel.input_modalities,
- // outputModality: rawModel.output_modalities,
- // maxTokens: rawModel.max_output_length,
- // supportedParameters: rawModel.supported_sampling_parameters,
- // })
- // }
- // } catch (error) {
- // console.error(
- // `Error fetching OpenRouter models from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`,
- // )
- // }
+ try {
+ const headers: Record = {
+ ...DEFAULT_HEADERS,
+ ...options?.headers,
+ }
+
+ const base = (options?.openRouterBaseUrl || "https://api.matterai.so/v1/web").replace(/\/+$/, "")
+ let url = base.endsWith("/models") ? base : `${base}/models`
+ if (headers["Cache-Control"] === "no-cache") {
+ url += (url.includes("?") ? "&" : "?") + "forceRefresh=true"
+ }
+
+ const response = await axios.get(url, { headers })
+
+ const rawModels = response.data?.data || []
+
+ if (Array.isArray(rawModels) && rawModels.length > 0) {
+ registerDynamicKilocodeModels(rawModels)
+ }
+
+ for (const rawModel of rawModels) {
+ if (!rawModel?.id || rawModel.id.startsWith("axon-")) {
+ continue
+ }
+
+ // Filter out image generation models (only text output)
+ const outputModalities = rawModel.output_modalities || []
+ if (outputModalities.includes("image") && !outputModalities.includes("text")) {
+ continue
+ }
+
+ const modelId = rawModel.id
+
+ models[modelId] = parseOpenRouterModel({
+ id: modelId,
+ model: {
+ name: rawModel.name || rawModel.id,
+ description: rawModel.description,
+ context_length: rawModel.context_length || 232000,
+ max_completion_tokens: rawModel.max_output_length,
+ pricing: rawModel.pricing
+ ? {
+ prompt: rawModel.pricing.prompt,
+ completion: rawModel.pricing.completion,
+ input_cache_read: rawModel.pricing.input_cache_reads,
+ input_cache_write: rawModel.pricing.input_cache_writes,
+ }
+ : undefined,
+ },
+ displayName: rawModel.name,
+ inputModality: rawModel.input_modalities,
+ outputModality: rawModel.output_modalities,
+ maxTokens: rawModel.max_output_length,
+ supportedParameters: rawModel.supported_sampling_parameters,
+ })
+ }
+ } catch (error) {
+ console.error(
+ `Error fetching OpenRouter models from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`,
+ )
+ }
return models
}
@@ -227,7 +235,7 @@ export async function getOpenRouterModelEndpoints(
const models: Record = {}
// First, check static models from KILO_CODE_MODELS
- const { KILO_CODE_MODELS } = await import("../kilocode-models")
+ const { KILO_CODE_MODELS, registerDynamicKilocodeModels } = await import("../kilocode-models")
const staticModel = KILO_CODE_MODELS[modelId]
if (staticModel) {
@@ -255,60 +263,70 @@ export async function getOpenRouterModelEndpoints(
}
// If not found in static models, fetch from MatterAI API
- // try {
- // const headers: Record = {
- // ...DEFAULT_HEADERS,
- // ...options?.headers,
- // }
-
- // const response = await axios.get("https://api.matterai.so/v1/models/openrouter", { headers })
-
- // const rawModels = response.data?.data || []
-
- // // Strip "openrouter/" prefix if present for matching
- // const strippedModelId = modelId.replace(/^openrouter\//, "")
-
- // const rawModel = (rawModels as MatterAiOpenRouterModel[]).find((m) => m.id === strippedModelId)
-
- // if (!rawModel) {
- // return models
- // }
-
- // // Filter out image generation models
- // const outputModalities = rawModel.output_modalities || []
- // if (outputModalities.includes("image") && !outputModalities.includes("text")) {
- // return models
- // }
-
- // const prefixedModelId = `openrouter/${rawModel.id}`
-
- // models["MatterAI"] = parseOpenRouterModel({
- // id: prefixedModelId,
- // model: {
- // name: rawModel.name || rawModel.id,
- // description: rawModel.description,
- // context_length: rawModel.context_length || 8192,
- // max_completion_tokens: rawModel.max_output_length,
- // pricing: rawModel.pricing
- // ? {
- // prompt: rawModel.pricing.prompt,
- // completion: rawModel.pricing.completion,
- // input_cache_read: rawModel.pricing.input_cache_reads,
- // input_cache_write: rawModel.pricing.input_cache_writes,
- // }
- // : undefined,
- // },
- // displayName: rawModel.name,
- // inputModality: rawModel.input_modalities,
- // outputModality: rawModel.output_modalities,
- // maxTokens: rawModel.max_output_length,
- // supportedParameters: rawModel.supported_sampling_parameters,
- // })
- // } catch (error) {
- // console.error(
- // `Error fetching OpenRouter model endpoints from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`,
- // )
- // }
+ try {
+ const headers: Record = {
+ ...DEFAULT_HEADERS,
+ ...options?.headers,
+ }
+
+ const base = (options?.openRouterBaseUrl || "https://api.matterai.so/v1/web").replace(/\/+$/, "")
+ let url = base.endsWith("/models") ? base : `${base}/models`
+ if (headers["Cache-Control"] === "no-cache") {
+ url += (url.includes("?") ? "&" : "?") + "forceRefresh=true"
+ }
+
+ const response = await axios.get(url, { headers })
+
+ const rawModels = response.data?.data || []
+
+ if (Array.isArray(rawModels) && rawModels.length > 0) {
+ registerDynamicKilocodeModels(rawModels)
+ }
+
+ // Strip "openrouter/" prefix if present for matching
+ const strippedModelId = modelId.replace(/^openrouter\//, "")
+
+ const rawModel = (rawModels as MatterAiOpenRouterModel[]).find(
+ (m) => m.id === strippedModelId || m.id === modelId,
+ )
+
+ if (!rawModel) {
+ return models
+ }
+
+ // Filter out image generation models
+ const outputModalities = rawModel.output_modalities || []
+ if (outputModalities.includes("image") && !outputModalities.includes("text")) {
+ return models
+ }
+
+ models["MatterAI"] = parseOpenRouterModel({
+ id: rawModel.id,
+ model: {
+ name: rawModel.name || rawModel.id,
+ description: rawModel.description,
+ context_length: rawModel.context_length || 232000,
+ max_completion_tokens: rawModel.max_output_length,
+ pricing: rawModel.pricing
+ ? {
+ prompt: rawModel.pricing.prompt,
+ completion: rawModel.pricing.completion,
+ input_cache_read: rawModel.pricing.input_cache_reads,
+ input_cache_write: rawModel.pricing.input_cache_writes,
+ }
+ : undefined,
+ },
+ displayName: rawModel.name,
+ inputModality: rawModel.input_modalities,
+ outputModality: rawModel.output_modalities,
+ maxTokens: rawModel.max_output_length,
+ supportedParameters: rawModel.supported_sampling_parameters,
+ })
+ } catch (error) {
+ console.error(
+ `Error fetching OpenRouter model endpoints from MatterAI: ${JSON.stringify(error, Object.getOwnPropertyNames(error), 2)}`,
+ )
+ }
return models
}
diff --git a/src/api/providers/kilocode-models.ts b/src/api/providers/kilocode-models.ts
index 443e3a1bbc..878077fad3 100644
--- a/src/api/providers/kilocode-models.ts
+++ b/src/api/providers/kilocode-models.ts
@@ -61,118 +61,118 @@ const OSS_MODEL_BASE: KiloCodeModelVariant = {
}
export const KILO_CODE_MODELS: Record = {
- "meta/muse-spark-1.2-contributor": {
- ...OSS_MODEL_BASE,
- id: "meta/muse-spark-1.2-contributor",
- name: "Muse Spark 1.2 Contributor",
- description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.",
- context_length: 232000,
- owned_by: "meta",
- openrouter: { slug: "meta/muse-spark-1.2-contributor" },
- // $0.10/M input, $0.002/M cache read, $0.20/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.0000001",
- completion: "0.0000002",
- input_cache_reads: "0.000000002",
- },
- },
- "deepseek/deepseek-v4-flash-0731": {
- ...OSS_MODEL_BASE,
- id: "deepseek/deepseek-v4-flash-0731",
- name: "DeepSeek V4 Flash",
- description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.",
- context_length: 232000,
- owned_by: "deepseek",
- openrouter: { slug: "deepseek/deepseek-v4-flash-0731" },
- // $0.14/M input, $0.028/M cache read, $0.28/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.00000014",
- completion: "0.00000028",
- input_cache_reads: "0.000000028",
- },
- },
- "zai/glm-5.3-flash": {
- ...OSS_MODEL_BASE,
- id: "zai/glm-5.3-flash",
- name: "GLM 5.3 Flash",
- description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.",
- context_length: 232000,
- owned_by: "zai",
- openrouter: { slug: "zai/glm-5.3-flash" },
- // $0.15/M input, $0.03/M cache read, $0.50/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.00000015",
- completion: "0.0000005",
- input_cache_reads: "0.00000003",
- },
- },
- "zai/glm-5.3": {
- ...OSS_MODEL_BASE,
- id: "zai/glm-5.3",
- name: "GLM 5.3",
- description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.",
- context_length: 232000,
- owned_by: "zai",
- openrouter: { slug: "zai/glm-5.3" },
- // $1.40/M input, $0.14/M cache read, $4.40/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.0000014",
- completion: "0.0000044",
- input_cache_reads: "0.00000014",
- },
- },
- "gpt-5.6-luna": {
- ...OSS_MODEL_BASE,
- id: "gpt-5.6-luna",
- name: "GPT-5.6 Luna",
- description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.",
- context_length: 232000,
- owned_by: "openai",
- openrouter: { slug: "gpt-5.6-luna" },
- // $0.20/M input, $0.02/M cache read, $1.20/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.0000002",
- completion: "0.0000012",
- input_cache_reads: "0.00000002",
- },
- },
- "gpt-5.6-sol": {
- ...OSS_MODEL_BASE,
- id: "gpt-5.6-sol",
- name: "GPT-5.6 Sol",
- description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.",
- context_length: 232000,
- owned_by: "openai",
- openrouter: { slug: "gpt-5.6-sol" },
- // $5/M input, $0.50/M cache read, $30/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.000005",
- completion: "0.00003",
- input_cache_reads: "0.0000005",
- },
- },
- "gemini-3.7-flash": {
- ...OSS_MODEL_BASE,
- id: "gemini-3.7-flash",
- name: "Gemini 3.7 Flash",
- description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.",
- context_length: 232000,
- owned_by: "google",
- openrouter: { slug: "gemini-3.7-flash" },
- // $0.75/M input, $0.075/M cache read, $3.75/M output
- pricing: {
- ...OSS_MODEL_BASE.pricing,
- prompt: "0.00000075",
- completion: "0.00000375",
- input_cache_reads: "0.000000075",
- },
- },
+ // "meta/muse-spark-1.2-contributor": {
+ // ...OSS_MODEL_BASE,
+ // id: "meta/muse-spark-1.2-contributor",
+ // name: "Muse Spark 1.2 Contributor",
+ // description: "Meta Muse Spark 1.2 Contributor is an open general purpose model for everyday coding tasks.",
+ // context_length: 232000,
+ // owned_by: "meta",
+ // openrouter: { slug: "meta/muse-spark-1.2-contributor" },
+ // // $0.10/M input, $0.002/M cache read, $0.20/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.0000001",
+ // completion: "0.0000002",
+ // input_cache_reads: "0.000000002",
+ // },
+ // },
+ // "deepseek/deepseek-v4-flash-0731": {
+ // ...OSS_MODEL_BASE,
+ // id: "deepseek/deepseek-v4-flash-0731",
+ // name: "DeepSeek V4 Flash",
+ // description: "DeepSeek V4 Flash is a fast, low cost open model for low-effort day-to-day coding tasks.",
+ // context_length: 232000,
+ // owned_by: "deepseek",
+ // openrouter: { slug: "deepseek/deepseek-v4-flash-0731" },
+ // // $0.14/M input, $0.028/M cache read, $0.28/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.00000014",
+ // completion: "0.00000028",
+ // input_cache_reads: "0.000000028",
+ // },
+ // },
+ // "zai/glm-5.3-flash": {
+ // ...OSS_MODEL_BASE,
+ // id: "zai/glm-5.3-flash",
+ // name: "GLM 5.3 Flash",
+ // description: "GLM 5.3 Flash is a fast, low cost open model for everyday coding tasks.",
+ // context_length: 232000,
+ // owned_by: "zai",
+ // openrouter: { slug: "zai/glm-5.3-flash" },
+ // // $0.15/M input, $0.03/M cache read, $0.50/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.00000015",
+ // completion: "0.0000005",
+ // input_cache_reads: "0.00000003",
+ // },
+ // },
+ // "zai/glm-5.3": {
+ // ...OSS_MODEL_BASE,
+ // id: "zai/glm-5.3",
+ // name: "GLM 5.3",
+ // description: "GLM 5.3 is Z.ai's frontier open model for complex coding tasks and long running agents.",
+ // context_length: 232000,
+ // owned_by: "zai",
+ // openrouter: { slug: "zai/glm-5.3" },
+ // // $1.40/M input, $0.14/M cache read, $4.40/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.0000014",
+ // completion: "0.0000044",
+ // input_cache_reads: "0.00000014",
+ // },
+ // },
+ // "gpt-5.6-luna": {
+ // ...OSS_MODEL_BASE,
+ // id: "gpt-5.6-luna",
+ // name: "GPT-5.6 Luna",
+ // description: "GPT-5.6 Luna is a fast, low cost open model for everyday coding tasks.",
+ // context_length: 232000,
+ // owned_by: "openai",
+ // openrouter: { slug: "gpt-5.6-luna" },
+ // // $0.20/M input, $0.02/M cache read, $1.20/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.0000002",
+ // completion: "0.0000012",
+ // input_cache_reads: "0.00000002",
+ // },
+ // },
+ // "gpt-5.6-sol": {
+ // ...OSS_MODEL_BASE,
+ // id: "gpt-5.6-sol",
+ // name: "GPT-5.6 Sol",
+ // description: "GPT-5.6 Sol is an open reasoning model for complex coding tasks and long running agents.",
+ // context_length: 232000,
+ // owned_by: "openai",
+ // openrouter: { slug: "gpt-5.6-sol" },
+ // // $5/M input, $0.50/M cache read, $30/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.000005",
+ // completion: "0.00003",
+ // input_cache_reads: "0.0000005",
+ // },
+ // },
+ // "gemini-3.7-flash": {
+ // ...OSS_MODEL_BASE,
+ // id: "gemini-3.7-flash",
+ // name: "Gemini 3.7 Flash",
+ // description: "Gemini 3.7 Flash is a fast, low cost open model for everyday coding tasks.",
+ // context_length: 232000,
+ // owned_by: "google",
+ // openrouter: { slug: "gemini-3.7-flash" },
+ // // $0.75/M input, $0.075/M cache read, $3.75/M output
+ // pricing: {
+ // ...OSS_MODEL_BASE.pricing,
+ // prompt: "0.00000075",
+ // completion: "0.00000375",
+ // input_cache_reads: "0.000000075",
+ // },
+ // },
}
/**
@@ -180,6 +180,12 @@ export const KILO_CODE_MODELS: Record = {
* Used to detect stale model selections after extension updates remove models.
*/
export function isValidKilocodeModel(modelId: string): boolean {
+ if (Object.keys(KILO_CODE_MODELS).length === 0) {
+ return true
+ }
+ if (modelId.startsWith("axon-auto")) {
+ return true
+ }
return modelId in KILO_CODE_MODELS
}
@@ -190,3 +196,44 @@ export function isValidKilocodeModel(modelId: string): boolean {
export function getKilocodeApiModelId(modelId: string): string {
return KILO_CODE_MODELS[modelId]?.id ?? modelId
}
+
+/**
+ * Registers models fetched dynamically from the MatterAI backend catalog into KILO_CODE_MODELS.
+ */
+export function registerDynamicKilocodeModels(rawModels: Array>): void {
+ for (const raw of rawModels) {
+ if (!raw?.id || raw.id.startsWith("axon-")) continue
+ const pricingObj = raw.pricing || {}
+ KILO_CODE_MODELS[raw.id] = {
+ ...OSS_MODEL_BASE,
+ id: raw.id,
+ name: raw.name || raw.id,
+ description: raw.description || `${raw.name || raw.id} via MatterAI`,
+ context_length: raw.context_length || 232000,
+ max_output_length: raw.max_output_length || 64000,
+ input_modalities: raw.input_modalities || ["text", "image"],
+ output_modalities: raw.output_modalities || ["text"],
+ supported_sampling_parameters:
+ raw.supported_sampling_parameters || OSS_MODEL_BASE.supported_sampling_parameters,
+ supported_features: raw.supported_features || OSS_MODEL_BASE.supported_features,
+ owned_by: raw.owned_by || raw.id.split("/")[0] || "matterai",
+ openrouter: { slug: raw.id },
+ pricing: {
+ ...OSS_MODEL_BASE.pricing,
+ prompt: typeof pricingObj.prompt === "string" ? pricingObj.prompt : String(pricingObj.prompt ?? "0"),
+ completion:
+ typeof pricingObj.completion === "string"
+ ? pricingObj.completion
+ : String(pricingObj.completion ?? "0"),
+ input_cache_reads:
+ typeof pricingObj.input_cache_reads === "string"
+ ? pricingObj.input_cache_reads
+ : String(pricingObj.input_cache_reads ?? "0"),
+ input_cache_writes:
+ typeof pricingObj.input_cache_writes === "string"
+ ? pricingObj.input_cache_writes
+ : String(pricingObj.input_cache_writes ?? "0"),
+ },
+ }
+ }
+}
diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts
index 73ff9cf04f..c5ec1c13ca 100644
--- a/src/core/webview/ClineProvider.ts
+++ b/src/core/webview/ClineProvider.ts
@@ -107,6 +107,8 @@ import { stringifyError } from "../../shared/kilocode/errorUtils"
import isWsl from "is-wsl"
import { getKilocodeDefaultModel } from "../../api/providers/kilocode/getKilocodeDefaultModel"
import { isValidKilocodeModel } from "../../api/providers/kilocode-models"
+import { getModels, flushModels } from "../../api/providers/fetchers/modelCache"
+import { type RouterName, type ModelRecord, type GetModelsOptions } from "../../shared/api"
import { getKiloCodeWrapperProperties } from "../../core/kilocode/wrapper"
import { getKiloUrlFromToken } from "@roo-code/types" // kilocode_change
import { getKilocodeConfig, getWorkspaceProjectId, KilocodeConfig } from "../../utils/kilo-config-file" // kilocode_change
@@ -167,6 +169,7 @@ export class ClineProvider
private usageGitHead?: string
private usageGitReportInFlight = false
private settingsStateSyncTimer?: NodeJS.Timeout
+ private modelsRefreshInterval?: NodeJS.Timeout
private recentTasksCache?: string[]
private pendingOperations: Map = new Map()
@@ -326,15 +329,46 @@ export class ClineProvider
}
// Multi-window synchronization: refresh secrets and global state when window gains focus
+ let lastModelFocusRefresh = 0
const windowStateDisposable = vscode.window.onDidChangeWindowState(async (e) => {
if (e.focused && this.contextProxy.isInitialized) {
await this.contextProxy.refreshSecrets()
await this.contextProxy.refreshGlobalState()
await this.postStateToWebview()
+
+ const now = Date.now()
+ if (now - lastModelFocusRefresh > 5000) {
+ lastModelFocusRefresh = now
+ this.refreshKilocodeModels({ forceRefresh: true }).catch((err) =>
+ this.log(`Error refreshing models on focus: ${err}`),
+ )
+ }
}
})
this.disposables.push(windowStateDisposable)
+ // Background 10-minute periodic model check
+ this.modelsRefreshInterval = setInterval(
+ () => {
+ this.refreshKilocodeModels({ forceRefresh: true }).catch((err) =>
+ this.log(`Error in 10-min background models refresh: ${err}`),
+ )
+ },
+ 10 * 60 * 1000,
+ )
+ this.modelsRefreshInterval.unref?.()
+
+ // kilocode_change: populate the dynamic model catalog once at startup so
+ // stale model selections (e.g. removed axon models) are validated and reset
+ // by getState() as soon as the catalog is available. Without this,
+ // isValidKilocodeModel() treats every model as valid until the first fetch
+ // completes, letting stale selections survive the launch.
+ if (this.contextProxy.isInitialized) {
+ this.refreshKilocodeModels()
+ .then(() => this.postStateToWebview())
+ .catch((err) => this.log(`Error during startup models refresh: ${err}`))
+ }
+
// kilocode_change: Setup git HEAD watcher for real-time branch updates
this.setupGitHeadWatcher()
@@ -504,6 +538,68 @@ export class ClineProvider
}
}
+ public async refreshKilocodeModels(options: { forceRefresh?: boolean } = {}): Promise {
+ try {
+ const { apiConfiguration } = await this.getState()
+ if (options.forceRefresh) {
+ await flushModels("kilocode-openrouter")
+ await flushModels("openrouter")
+ }
+
+ // forked_change start: fetch both openrouter and kilocode-openrouter
+ // Mirrors the requestRouterModels handler so the webview receives the
+ // full routerModels payload; posting an empty openrouter map would
+ // wipe the OpenRouter model list in the webview on every refresh.
+ const modelFetchPromises: Array<{ key: RouterName; options: GetModelsOptions }> = [
+ {
+ key: "openrouter",
+ options: {
+ provider: "openrouter",
+ apiKey: apiConfiguration?.openRouterApiKey,
+ baseUrl: apiConfiguration?.openRouterBaseUrl,
+ forceRefresh: options.forceRefresh,
+ },
+ },
+ {
+ key: "kilocode-openrouter",
+ options: {
+ provider: "kilocode-openrouter",
+ kilocodeToken: apiConfiguration?.kilocodeToken,
+ kilocodeOrganizationId: apiConfiguration?.kilocodeOrganizationId,
+ forceRefresh: options.forceRefresh,
+ },
+ },
+ ]
+
+ const results = await Promise.allSettled(
+ modelFetchPromises.map(async ({ options }) => await getModels(options)),
+ )
+
+ const routerModels: Record = {
+ openrouter: {},
+ "kilocode-openrouter": {},
+ }
+
+ results.forEach((result, index) => {
+ if (result.status === "fulfilled") {
+ routerModels[modelFetchPromises[index].key] = result.value
+ } else {
+ this.log(
+ `Failed to refresh ${modelFetchPromises[index].key} models: ${result.reason instanceof Error ? result.reason.message : String(result.reason)}`,
+ )
+ }
+ })
+
+ this.postMessageToWebview({
+ type: "routerModels",
+ routerModels,
+ })
+ // forked_change end
+ } catch (error) {
+ this.log(`Failed to refresh kilocode models: ${error}`)
+ }
+ }
+
/**
* Synchronize cloud profiles with local profiles.
*/
@@ -905,6 +1001,8 @@ export class ClineProvider
this.usageGitRefsWatcher = undefined
if (this.usageGitPoller) clearInterval(this.usageGitPoller)
this.usageGitPoller = undefined
+ if (this.modelsRefreshInterval) clearInterval(this.modelsRefreshInterval)
+ this.modelsRefreshInterval = undefined
await this.mcpHub?.unregisterClient()
this.mcpHub = undefined
this.marketplaceManager?.cleanup()
@@ -2700,6 +2798,19 @@ ${prompt}
providerSettings.apiProvider = apiProvider
}
+ // Validate global kilocodeModel against available models.
+ if (providerSettings?.apiProvider === "kilocode" && providerSettings?.kilocodeModel) {
+ if (!isValidKilocodeModel(providerSettings.kilocodeModel)) {
+ const staleModel = providerSettings.kilocodeModel
+ const defaultModel = await getKilocodeDefaultModel()
+ providerSettings.kilocodeModel = defaultModel
+ await this.contextProxy.setProviderSettings(providerSettings)
+ this.log(
+ `[ModelValidation] Reset stale kilocodeModel "${staleModel}" to default "${defaultModel}" — model no longer available`,
+ )
+ }
+ }
+
let organizationAllowList = ORGANIZATION_ALLOW_ALL
try {
diff --git a/src/core/webview/__tests__/ClineProvider.spec.ts b/src/core/webview/__tests__/ClineProvider.spec.ts
index c6207a4af9..c8c3f7a390 100644
--- a/src/core/webview/__tests__/ClineProvider.spec.ts
+++ b/src/core/webview/__tests__/ClineProvider.spec.ts
@@ -2035,6 +2035,15 @@ describe("ClineProvider", () => {
beforeEach(async () => {
await provider.resolveWebviewView(mockWebviewView)
logSpy = vi.spyOn(provider, "log").mockImplementation(() => {})
+
+ // Seed the dynamic catalog the way the backend /v1/web/models fetch
+ // does at runtime, so isValidKilocodeModel validates against real
+ // entries instead of hitting the empty-catalog bypass.
+ const { registerDynamicKilocodeModels } = await import("../../../api/providers/kilocode-models")
+ registerDynamicKilocodeModels([
+ { id: "deepseek/deepseek-v4-flash-0731", name: "DeepSeek V4 Flash" },
+ { id: "zai/glm-5.3", name: "GLM 5.3" },
+ ])
})
afterEach(() => {
diff --git a/src/core/webview/webviewMessageHandler.ts b/src/core/webview/webviewMessageHandler.ts
index 01f733b414..721d7f63c5 100644
--- a/src/core/webview/webviewMessageHandler.ts
+++ b/src/core/webview/webviewMessageHandler.ts
@@ -1905,11 +1905,17 @@ ${comment.suggestion}
// forked_change start: openrouter auth, kilocode provider
const openRouterApiKey = apiConfiguration.openRouterApiKey || message?.values?.openRouterApiKey
const openRouterBaseUrl = apiConfiguration.openRouterBaseUrl || message?.values?.openRouterBaseUrl
+ const forceRefresh = Boolean(message?.values?.forceRefresh)
const modelFetchPromises: Array<{ key: RouterName; options: GetModelsOptions }> = [
{
key: "openrouter",
- options: { provider: "openrouter", apiKey: openRouterApiKey, baseUrl: openRouterBaseUrl },
+ options: {
+ provider: "openrouter",
+ apiKey: openRouterApiKey,
+ baseUrl: openRouterBaseUrl,
+ forceRefresh,
+ },
},
{
key: "kilocode-openrouter",
@@ -1917,6 +1923,7 @@ ${comment.suggestion}
provider: "kilocode-openrouter",
kilocodeToken: apiConfiguration.kilocodeToken,
kilocodeOrganizationId: apiConfiguration.kilocodeOrganizationId,
+ forceRefresh,
},
},
]
diff --git a/src/shared/WebviewMessage.ts b/src/shared/WebviewMessage.ts
index 6b6fb9331e..f19ad27e4c 100644
--- a/src/shared/WebviewMessage.ts
+++ b/src/shared/WebviewMessage.ts
@@ -514,6 +514,10 @@ export type ProfileData = {
// Tiered usage windows (weekly / monthly). Each window is expressed
// as a fraction of the user's monthly plan limit.
tieredUsage?: AxonCodeTieredUsage
+ // Per-model usage for the tracked OSS models. Each entry attributes the
+ // model's share of the shared plan pool as percentages of the same
+ // weekly/monthly windows (no credit amounts are exposed).
+ modelUsage?: AxonCodeModelUsage[]
weeklyReset?: AxonCodeWeeklyResetAvailability
// Overage lets the plan keep running on shared org API credits once the
// plan windows hit 98%. `enabled` is true only when the org has turned
@@ -558,6 +562,17 @@ export interface AxonCodeTieredUsage {
monthly: AxonCodeWindowUsage
}
+export interface AxonCodeModelUsage {
+ // OSS catalog model id (e.g. "zai/glm-5.3").
+ model: string
+ // Plan-cost multiplier applied to this model's charges (e.g. 5 for
+ // deepseek, 1 for untracked/axon models).
+ multiplier: number
+ // Share of the shared weekly / monthly plan windows, as percentages.
+ weeklyPercentage: number
+ monthlyPercentage: number
+}
+
export interface ProfileDataResponsePayload {
success: boolean
data?: ProfileData
diff --git a/src/shared/api.ts b/src/shared/api.ts
index 78b047edd7..846a2384b0 100644
--- a/src/shared/api.ts
+++ b/src/shared/api.ts
@@ -148,6 +148,7 @@ export const getModelMaxOutputTokens = ({
type CommonFetchParams = {
apiKey?: string
baseUrl?: string
+ forceRefresh?: boolean
}
// Exhaustive, value-level map for all dynamic providers.
diff --git a/webview-ui/src/components/chat/UsageDialog.tsx b/webview-ui/src/components/chat/UsageDialog.tsx
index 0c2f9cc64e..b734560eda 100644
--- a/webview-ui/src/components/chat/UsageDialog.tsx
+++ b/webview-ui/src/components/chat/UsageDialog.tsx
@@ -3,7 +3,7 @@ import { Dialog, DialogContent, DialogHeader, DialogTitle, DialogDescription } f
import { vscode } from "@/utils/vscode"
import { useExtensionState } from "@/context/ExtensionStateContext"
import { ProfileData, WebviewMessage, AxonCodeTieredUsage } from "@roo/WebviewMessage"
-import { Activity, Gauge, Sparkles, Wallet, ShieldCheck, Layers } from "lucide-react"
+import { Activity, Cpu, Gauge, Sparkles, Wallet, ShieldCheck, Layers } from "lucide-react"
interface UsageDialogProps {
open: boolean
@@ -306,6 +306,56 @@ export const UsageDialog: React.FC = ({
)}
+
+ {/* 3. MODEL USAGE */}
+ {profileData?.modelUsage && profileData.modelUsage.length > 0 && (
+
+
+
+ Model Usage
+
+
+ Each model's share of your plan windows (weekly / monthly).
+
- {entry.model}
+
+
+ {entry.model}
+
{entry.multiplier}x limit
diff --git a/webview-ui/src/components/ui/index.ts b/webview-ui/src/components/ui/index.ts
index 1edc71d6ab..98f11fd539 100644
--- a/webview-ui/src/components/ui/index.ts
+++ b/webview-ui/src/components/ui/index.ts
@@ -10,6 +10,7 @@ export * from "./dropdown-menu"
export * from "./input"
export * from "./labeled-progress"
export * from "./popover"
+export * from "./provider-logo"
export * from "./progress"
export * from "./searchable-select"
export * from "./separator"
diff --git a/webview-ui/src/components/ui/provider-logo.tsx b/webview-ui/src/components/ui/provider-logo.tsx
new file mode 100644
index 0000000000..d3c0aaa028
--- /dev/null
+++ b/webview-ui/src/components/ui/provider-logo.tsx
@@ -0,0 +1,18 @@
+import { cn } from "@src/lib/utils"
+
+/**
+ * Provider logo rendered on a white circular background. Catalog icons are
+ * transparent SVGs (some monochrome), so they need the white backing to stay
+ * visible on dark VS Code themes.
+ */
+export const ProviderLogo = ({ src, className }: { src?: string | null; className?: string }) => {
+ if (!src) {
+ return null
+ }
+
+ return (
+
+
+
+ )
+}
From 5b6144509de1a7e31aa3043f4dcf3e91121bb8f5 Mon Sep 17 00:00:00 2001
From: "matterai-app[bot]"
Date: Thu, 3 Sep 2026 14:56:36 +0530
Subject: [PATCH 9/9] docs(readme): update AI Models section to the current OSS
catalog
Replace the retired axon-mini/axon-code/axon-code-2 table with the seven
OSS models served dynamically from the MatterAI backend catalog, with
providers, credit multipliers, and use cases.
---
README.md | 18 +++++++++++-------
1 file changed, 11 insertions(+), 7 deletions(-)
diff --git a/README.md b/README.md
index 6d6a97cd42..b2b2265d92 100644
--- a/README.md
+++ b/README.md
@@ -85,13 +85,17 @@ Orbital provides a comprehensive suite of intelligent tools:
## 🤖 AI Models
-Choose the right model for your workflow — switch seamlessly between efficiency and power:
-
-| Model | Credit Usage | Use Case | Capabilities |
-| ----------------- | ------------ | --------------------- | --------------------------------------------------------------------------- |
-| **`axon-mini`** | 0.5x | Quick tasks | Lightweight, fast, cost-effective for high-volume usage |
-| **`axon-code`** | 0.8x | Daily coding tasks | High intelligence, balanced performance, best for regular coding |
-| **`axon-code-2`** | 1x | Complex agentic tasks | Maximum intelligence, complex context harness, best for complex development |
+Choose the right model for your workflow — switch seamlessly between efficiency and power. The model catalog is served dynamically from the MatterAI backend, so new models appear automatically in the model selector:
+
+| Model | Provider | Credit Usage | Use Case | Capabilities |
+| ------------------------------ | -------- | ------------ | --------------------- | --------------------------------------------------------------- |
+| **Muse Spark 1.3 Contributor** | Meta | 2x | Everyday coding | Open general-purpose model for everyday coding tasks |
+| **DeepSeek V4 Flash** | DeepSeek | 5x | Fast, low-cost coding | Fast, low-cost open model for day-to-day coding tasks |
+| **GLM 5.3** | Z.ai | 4x | Complex agentic tasks | Frontier open model for complex coding and long-running agents |
+| **GLM 5.3 Flash** | Z.ai | 4x | Everyday coding | Fast, low-cost open model for everyday coding tasks |
+| **GPT-5.6 Luna** | OpenAI | 2x | Everyday coding | Fast, low-cost open model for everyday coding tasks |
+| **GPT-5.6 Sol** | OpenAI | 5x | Complex reasoning | Open reasoning model for complex coding and long-running agents |
+| **Gemini 3.8 Flash** | Google | 3x | Everyday coding | Fast, low-cost model for everyday coding tasks |
## 📦 Installation