diff --git a/CHANGELOG.md b/CHANGELOG.md index eef5475856..4e7c342166 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,17 @@ # Changelog +## [v6.8.5] - 2026-09-04 + +### Added + +- **Generalized malformed tool-call JSON repair (`src/utils/jsonRepair.ts`).** A new best-effort repair layer (`parseToolCallArguments`) is now the single entry point for every tool-call argument ingestion path — live streaming, stream finalization, and MCP argument validation. When a model emits almost-JSON arguments, the call is repaired and executed instead of being dropped (weaker models repeat the exact same mistake on retry, dead-looping the task). Repairs include: placeholder tags in structural positions (e.g. `"limit": 180` keeps the value; a tag-only value becomes `null` so the tool's default applies; a tag inside a key is stripped), unquoted keys and scalar values (`{path: "src"}`, `{"file_pattern": *.ts}`), single-quoted keys/values, Python literals (`True`/`False`/`None`), `//` and `/* */` comments, trailing commas, missing commas between members, and lost key quotes (`{"offset: 600, ...}`). At stream finalization only, truncated arguments are recovered by closing dangling strings and open containers innermost-first, appending `null` after a dangling colon, and wrapping braceless bodies (`path: "src"`); during streaming an incomplete buffer stays unparseable so deltas keep accumulating. A single-object array is unwrapped when the model wrapped the tool object by mistake. Valid JSON is always strict-parsed first and returned verbatim (`repaired: false`) — the only exception is stripping a leaked placeholder tag from a key, since keys are structural parameter names; tags inside string values are legitimate content and are never touched. +- **Repair transparency.** `ToolUse` blocks now carry a `repaired` flag; when a repaired tool call completes, the executed arguments are appended to the tool result as a note so the model sees what actually ran instead of repeating its malformed form on the next turn. `useMcp_tool` applies the same repair when validating MCP tool arguments and appends the same note. +- **Executor parameter hardening.** `read_file` (native JSON, single `file_path`, and XML paths) and the kilocode read-file path now parse `offset`/`limit` through a `parsePositiveInteger` helper that floors floats, clamps to a minimum of 1, and falls back to the tool default on non-numeric input; `search_files` floors fractional `max_results`/`context_lines` into their allowed ranges. + +### Fixed + +- **Placeholder-tagged tool arguments are repaired instead of dropped.** When a model emits malformed native tool-call JSON with a placeholder tag in value position (e.g. `"limit": 180` or `"limit": `), the tag is stripped and the trailing value kept, or the value becomes `null` when no value follows so the tool's default applies. The repair runs only after strict `JSON.parse` fails, so valid JSON is never altered. Streaming partial previews are tag-cleaned as well. + ## [v6.8.4] - 2026-09-03 ### Added diff --git a/src/core/assistant-message/AssistantMessageParser.ts b/src/core/assistant-message/AssistantMessageParser.ts index 6010f406f6..4d42605d25 100644 --- a/src/core/assistant-message/AssistantMessageParser.ts +++ b/src/core/assistant-message/AssistantMessageParser.ts @@ -3,6 +3,7 @@ import { TextContent, ToolUse, ToolParamName, toolParamNames } from "../../share import { AssistantMessageContent } from "./parseAssistantMessage" import { NativeToolCall, parseDoubleEncodedParams } from "./kilocode/native-tool-call" import Anthropic from "@anthropic-ai/sdk" // kilocode_change +import { parseToolCallArguments } from "../../utils/jsonRepair" // forked_change /** * Callback function type to check if a tool name is a valid MCP tool. @@ -10,6 +11,37 @@ import Anthropic from "@anthropic-ai/sdk" // kilocode_change */ export type McpToolChecker = (toolName: string) => { isMcpTool: boolean; serverName?: string } | undefined +// forked_change start +/** + * Placeholder tags some models leak into tool-call JSON arguments, e.g. + * `"limit": 180`. The tag is an artifact of the provider's + * internal argument templating and makes the arguments invalid JSON. + */ +const PLACEHOLDER_TAG = "<\\/?[a-zA-Z_][a-zA-Z0-9_.\\-]*>" + +/** + * Matches a placeholder tag in value position — immediately after a + * `"key":` — so tags inside legitimate string values are never touched. + */ +const VALUE_POSITION_PLACEHOLDER_TAG = new RegExp(`("[^"]*"\\s*:\\s*)${PLACEHOLDER_TAG}`, "g") + +/** + * Remove placeholder tags that prefix a real value: + * `"limit": 180` becomes `"limit": 180`. Repeated until stable in case a + * model emits several tags in a row. Used for partial-params display only; + * actual argument parsing goes through `parseToolCallArguments`. + */ +function stripValuePositionTags(args: string): string { + let stripped = args + let previous: string + do { + previous = stripped + stripped = stripped.replace(VALUE_POSITION_PLACEHOLDER_TAG, "$1") + } while (stripped !== previous) + return stripped +} +// forked_change end + /** * Parse a native tool-call `arguments` JSON string robustly. * @@ -24,41 +56,33 @@ export type McpToolChecker = (toolName: string) => { isMcpTool: boolean; serverN * * Strategy: * 1. Only attempt parsing once the buffer looks like JSON (`{`/`[`). - * 2. Try strict `JSON.parse` first. This is the ONLY path valid JSON ever - * takes, so well-formed arguments are never mutated. - * 3. Only if strict parsing throws do we attempt a conservative repair for - * genuinely-malformed JSON (unquoted scalar values like - * `{"file_pattern": *.js}`) and parse once more. + * At finalization (`repairTruncated`) a braceless body such as + * `path: "src"` is still worth a repair attempt — the scanner rejects + * prose because a bare key must be followed by a colon. + * 2. `parseToolCallArguments` strict-parses first, so well-formed arguments + * are never mutated. Only on failure does it apply the best-effort repair + * pass (placeholder tags, unquoted keys/values, single quotes, Python + * literals, comments, trailing commas, dropped key quotes). With + * `repairTruncated` it additionally closes dangling strings/containers + * and appends `null` after a dangling colon. * - * @returns `{ parsed }` when complete & valid, or `undefined` when the buffer is - * not yet valid JSON (caller should keep accumulating streaming deltas). + * @returns `{ parsed, repaired }` when complete & valid (or repaired), or + * `undefined` when the buffer is not yet valid JSON (caller should keep + * accumulating streaming deltas). */ -export function tryParseToolArguments(rawArgs: string): { parsed: any } | undefined { +export function tryParseToolArguments( + rawArgs: string, + options?: { repairTruncated?: boolean }, +): { parsed: any; repaired?: boolean } | undefined { + const repairTruncated = options?.repairTruncated ?? false const trimmed = rawArgs.trim() // During streaming, arguments may begin as natural-language text; wait until // the buffer actually looks like a JSON object/array before parsing. - if (!trimmed.startsWith("{") && !trimmed.startsWith("[")) { - return undefined - } - - try { - // Strict, verbatim parse — the happy path for all well-formed JSON. - return { parsed: JSON.parse(rawArgs) } - } catch { - // Strict parse failed. This is either (a) an incomplete buffer that is - // still streaming, or (b) genuinely malformed JSON. Attempt a single - // conservative repair (quote bare scalar values) and parse again. Because - // this only runs after a strict failure, it can never corrupt valid JSON. - try { - const repaired = rawArgs.replace(/("([^"]+)"\s*:\s*)([a-zA-Z0-9_.*\/\\-]+)(?=\s*[,\]}])/g, '$1"$3"') - if (repaired !== rawArgs) { - return { parsed: JSON.parse(repaired) } - } - } catch { - // Repair did not yield valid JSON either — fall through. - } + if (!trimmed.startsWith("{") && !trimmed.startsWith("[") && !repairTruncated) { return undefined } + const result = parseToolCallArguments(rawArgs, { repairTruncated }) + return result ? { parsed: result.args, repaired: result.repaired } : undefined } /** @@ -157,6 +181,10 @@ export class AssistantMessageParser { return partialParams } + // forked_change: strip placeholder tags (e.g. "limit": 180) + // so the partial display reflects the real value behind the tag. + const cleaned = stripValuePositionTags(argsString) + // Match patterns like "key": "value" or "key": "partial value (without closing quote) // Also match "key": unquoted_value for simple values // Handle escaped quotes within values: \" @@ -165,11 +193,11 @@ export class AssistantMessageParser { let match // Extract quoted values - while ((match = quotedValueRegex.exec(argsString)) !== null) { + while ((match = quotedValueRegex.exec(cleaned)) !== null) { partialParams[match[1]] = match[2] } // Extract unquoted values (only if not already captured as quoted) - while ((match = unquotedValueRegex.exec(argsString)) !== null) { + while ((match = unquotedValueRegex.exec(cleaned)) !== null) { if (!(match[1] in partialParams)) { partialParams[match[1]] = match[2] } @@ -178,7 +206,7 @@ export class AssistantMessageParser { // Also try to extract partial quoted values (without closing quote) // e.g., "file_path": "src/componen -> extract "src/componen" const partialQuotedRegex = /"((?:[^"\\]|\\.)*)"\s*:\s*"((?:[^"\\]|\\.)*)$/g - while ((match = partialQuotedRegex.exec(argsString)) !== null) { + while ((match = partialQuotedRegex.exec(cleaned)) !== null) { partialParams[match[1]] = match[2] } @@ -338,6 +366,7 @@ export class AssistantMessageParser { // throws — it returns undefined while the buffer is still incomplete. let isComplete = false let parsedArgs: Record = {} + let argumentsRepaired = false const parseResult = accumulatedCall.function!.arguments.trim() ? tryParseToolArguments(accumulatedCall.function!.arguments) @@ -347,6 +376,10 @@ export class AssistantMessageParser { // Fix any double-encoded parameters parsedArgs = parseDoubleEncodedParams(parseResult.parsed) isComplete = true + argumentsRepaired = parseResult.repaired ?? false + if (argumentsRepaired) { + console.warn("[AssistantMessageParser] Repaired malformed tool arguments") + } } else { // Arguments are not yet complete valid JSON, continue accumulating. // forked_change: Update the partial tool use block with new partial params @@ -391,6 +424,7 @@ export class AssistantMessageParser { }, partial: false, toolUseId: accumulatedCall.id, + repaired: argumentsRepaired, } console.log("[MCP Debug] Converted to use_mcp_tool:", toolUse.params) } else { @@ -401,6 +435,7 @@ export class AssistantMessageParser { params: parsedArgs, partial: false, // Now complete after accumulation toolUseId: accumulatedCall.id, + repaired: argumentsRepaired, } } @@ -421,6 +456,7 @@ export class AssistantMessageParser { existingBlock.params = toolUse.params existingBlock.partial = false existingBlock.toolUseId = toolUse.toolUseId + existingBlock.repaired = toolUse.repaired } else { // Partial block not found, add as new (shouldn't happen normally) this.contentBlocks.push(toolUse) @@ -704,9 +740,12 @@ export class AssistantMessageParser { // Try to parse the arguments one final time. Use the same raw-first // helper as the streaming path so valid JSON is decoded verbatim and - // never mutated by the lenient repair regex. + // never mutated by the repair pass. The stream has now ended, so + // repairTruncated closes dangling strings/containers and appends null + // after a dangling colon — recovering calls whose JSON was cut off + // mid-argument instead of dropping them. const finalParse = accumulatedCall.function?.arguments?.trim() - ? tryParseToolArguments(accumulatedCall.function.arguments) + ? tryParseToolArguments(accumulatedCall.function.arguments, { repairTruncated: true }) : { parsed: {} } if (!finalParse) { @@ -727,6 +766,10 @@ export class AssistantMessageParser { } const parsedArgs: Record = parseDoubleEncodedParams(finalParse.parsed) + const argumentsRepaired = finalParse.repaired ?? false + if (argumentsRepaired) { + console.warn("[AssistantMessageParser] Repaired malformed tool arguments at finalization") + } // Finalize any current text content before adding tool use if (this.currentTextContent) { @@ -749,6 +792,7 @@ export class AssistantMessageParser { }, partial: false, toolUseId: accumulatedCall.id, + repaired: argumentsRepaired, } } else { // Create a ToolUse block from the native tool call @@ -758,6 +802,7 @@ export class AssistantMessageParser { params: parsedArgs, partial: false, toolUseId: accumulatedCall.id, + repaired: argumentsRepaired, } } diff --git a/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts b/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts index 4bf22eed57..b22ae2c2fc 100644 --- a/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts +++ b/src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts @@ -2,7 +2,7 @@ import { beforeEach, describe, expect, it } from "vitest" // npx vitest src/core/assistant-message/__tests__/AssistantMessageParser.spec.ts -import { AssistantMessageParser } from "../AssistantMessageParser" +import { AssistantMessageParser, tryParseToolArguments } from "../AssistantMessageParser" import { AssistantMessageContent } from "../parseAssistantMessage" import { TextContent, ToolUse } from "../../../shared/tools" @@ -615,6 +615,206 @@ describe("AssistantMessageParser (streaming)", () => { }) }) + describe("placeholder-tag repair (forked_change)", () => { + it("keeps the value that follows a placeholder tag", () => { + const raw = '{"files": [{"file_path": "/tmp/a.ts", "offset": 70, "limit": 180}]}' + const result = tryParseToolArguments(raw) + expect(result).toBeDefined() + expect(result!.repaired).toBe(true) + expect(result!.parsed.files[0].limit).toBe(180) + expect(result!.parsed.files[0].offset).toBe(70) + expect(result!.parsed.files[0].file_path).toBe("/tmp/a.ts") + }) + + it("nulls a tag-only value so the tool default applies", () => { + const raw = '{"files": [{"file_path": "/tmp/a.ts", "offset": 70, "limit": }]}' + const result = tryParseToolArguments(raw) + expect(result).toBeDefined() + expect(result!.parsed.files[0]).toEqual({ file_path: "/tmp/a.ts", offset: 70, limit: null }) + }) + + it("nulls a tag-only key followed by another member", () => { + const result = tryParseToolArguments('{"path": , "regex": "foo"}') + expect(result).toBeDefined() + expect(result!.parsed).toEqual({ path: null, regex: "foo" }) + }) + + it("nulls a tag-only first key", () => { + const result = tryParseToolArguments('{"limit": , "offset": 5}') + expect(result).toBeDefined() + expect(result!.parsed).toEqual({ limit: null, offset: 5 }) + }) + + it("keeps a string value that follows a placeholder tag", () => { + const result = tryParseToolArguments('{"file_path": "/tmp/a.ts"}') + expect(result).toBeDefined() + expect(result!.parsed).toEqual({ file_path: "/tmp/a.ts" }) + }) + + it("composes with the bare-scalar repair", () => { + const result = tryParseToolArguments('{"file_pattern": *.ts}') + expect(result).toBeDefined() + expect(result!.parsed).toEqual({ file_pattern: "*.ts" }) + }) + + it("still repairs bare scalars when no tags are present (regression)", () => { + const result = tryParseToolArguments('{"file_pattern": *.ts}') + expect(result).toBeDefined() + expect(result!.parsed).toEqual({ file_pattern: "*.ts" }) + }) + + it("never touches tags inside valid JSON string values", () => { + const raw = '{"content": "uses bold and inside a string"}' + const result = tryParseToolArguments(raw) + expect(result).toBeDefined() + expect(result!.parsed.content).toBe("uses bold and inside a string") + }) + + it("returns undefined for an incomplete buffer containing a tag (keep accumulating)", () => { + const raw = '{"files": [{"file_path": "/tmp/a.ts", "limit": ' + expect(tryParseToolArguments(raw)).toBeUndefined() + }) + + it("reports repaired: false for well-formed JSON", () => { + const result = tryParseToolArguments('{"path": "src"}') + expect(result).toEqual({ parsed: { path: "src" }, repaired: false }) + }) + + it("reports repaired: true when the repair pass ran", () => { + const result = tryParseToolArguments('{"path": "src"}') + expect(result!.repaired).toBe(true) + }) + + it("recovers a truncated buffer only at finalization", () => { + const truncated = '{"path": "src' + expect(tryParseToolArguments(truncated)).toBeUndefined() + expect(tryParseToolArguments(truncated, { repairTruncated: true })).toEqual({ + parsed: { path: "src" }, + repaired: true, + }) + }) + + it("recovers a braceless body only at finalization", () => { + expect(tryParseToolArguments('path: "src"')).toBeUndefined() + expect(tryParseToolArguments('path: "src"', { repairTruncated: true })).toEqual({ + parsed: { path: "src" }, + repaired: true, + }) + }) + }) + + describe("native tool calls with placeholder tags", () => { + const tagWithValueArgs = + '{"files": [{"file_path": "/tmp/a.ts", "offset": 70, "limit": 180}]}' + const tagOnlyArgs = '{"files": [{"file_path": "/tmp/a.ts", "offset": 70, "limit": }]}' + + it("repairs a read_file call whose limit is prefixed by a placeholder tag", () => { + const yielded = [ + ...parser.processNativeToolCalls([ + { + index: 0, + id: "read_file:0", + type: "function", + function: { name: "read_file", arguments: tagWithValueArgs }, + }, + ]), + ] + // Partial (with tag-stripped display params) + complete. + expect(yielded).toHaveLength(2) + expect((yielded[0].input as Record).limit).toBe("180") + const toolUse = parser.getContentBlocks().find((b) => b.type === "tool_use") as ToolUse + expect(toolUse.partial).toBe(false) + expect(toolUse.repaired).toBe(true) + expect((toolUse.params.files as unknown as Array>)[0]).toEqual({ + file_path: "/tmp/a.ts", + offset: 70, + limit: 180, + }) + }) + + it("nulls a tag-only limit so the read_file default applies", () => { + const yielded = [ + ...parser.processNativeToolCalls([ + { + index: 0, + id: "read_file:0", + type: "function", + function: { name: "read_file", arguments: tagOnlyArgs }, + }, + ]), + ] + expect(yielded).toHaveLength(2) + const toolUse = parser.getContentBlocks().find((b) => b.type === "tool_use") as ToolUse + expect(toolUse.partial).toBe(false) + expect(toolUse.repaired).toBe(true) + expect((toolUse.params.files as unknown as Array>)[0]).toEqual({ + file_path: "/tmp/a.ts", + offset: 70, + limit: null, + }) + }) + + it.each([1, 7, 64, tagWithValueArgs.length])( + "completes a tag-repaired call streamed in small deltas (chunkSize=%i)", + (chunkSize) => { + let first = true + for (let i = 0; i < tagWithValueArgs.length; i += chunkSize) { + const call = first + ? { + index: 0, + id: "call_1", + type: "function", + function: { name: "read_file", arguments: tagWithValueArgs.slice(i, i + chunkSize) }, + } + : { index: 0, function: { arguments: tagWithValueArgs.slice(i, i + chunkSize) } } + first = false + for (const _ of parser.processNativeToolCalls([call as any])) { + /* consume */ + } + } + parser.finalizeContentBlocks() + const toolUses = parser.getContentBlocks().filter((b) => b.type === "tool_use") as ToolUse[] + expect(toolUses).toHaveLength(1) + expect(toolUses[0].partial).toBe(false) + expect((toolUses[0].params.files as unknown as Array>)[0]).toEqual({ + file_path: "/tmp/a.ts", + offset: 70, + limit: 180, + }) + }, + ) + + it.each([1, 7, 64, tagOnlyArgs.length])( + "completes a tag-only streamed call with the value nulled (chunkSize=%i)", + (chunkSize) => { + let first = true + for (let i = 0; i < tagOnlyArgs.length; i += chunkSize) { + const call = first + ? { + index: 0, + id: "call_1", + type: "function", + function: { name: "read_file", arguments: tagOnlyArgs.slice(i, i + chunkSize) }, + } + : { index: 0, function: { arguments: tagOnlyArgs.slice(i, i + chunkSize) } } + first = false + for (const _ of parser.processNativeToolCalls([call as any])) { + /* consume */ + } + } + parser.finalizeContentBlocks() + const toolUses = parser.getContentBlocks().filter((b) => b.type === "tool_use") as ToolUse[] + expect(toolUses).toHaveLength(1) + expect(toolUses[0].partial).toBe(false) + expect((toolUses[0].params.files as unknown as Array>)[0]).toEqual({ + file_path: "/tmp/a.ts", + offset: 70, + limit: null, + }) + }, + ) + }) + describe("size limit handling", () => { it("should throw an error when MAX_ACCUMULATOR_SIZE is exceeded", () => { // Create a message that exceeds 1MB (MAX_ACCUMULATOR_SIZE) diff --git a/src/core/assistant-message/presentAssistantMessage.ts b/src/core/assistant-message/presentAssistantMessage.ts index 79775b8038..b518d2ad67 100644 --- a/src/core/assistant-message/presentAssistantMessage.ts +++ b/src/core/assistant-message/presentAssistantMessage.ts @@ -46,6 +46,7 @@ import { webSearchTool } from "../tools/webSearchTool" import { figmaFetchTool } from "../tools/figmaFetchTool" import { askFollowupQuestionTool } from "../tools/askFollowupQuestionTool" import { MAX_PARALLEL_READ_ONLY_TOOLS, MAX_TOOL_REPETITION_AUTO_RETRIES } from "../tools/toolExecutionPolicy" +import { formatArgumentRepairNote } from "../../utils/jsonRepair" // forked_change type PresentAssistantMessageOptions = { /** Explicit content-block index used by the read-only batch scheduler. */ @@ -954,6 +955,25 @@ export async function presentAssistantMessage(cline: Task, options: PresentAssis await removeStaleToolPreview() } + // forked_change: transparency for auto-repaired arguments. The parser + // repaired the malformed JSON before execution; appending what actually + // ran to the tool result stops the model from repeating the same + // malformed form on the next turn. + if (block.repaired && !block.partial && toolResultPushed && !cline.didRejectTool) { + try { + const executedArguments = + block.name === "use_mcp_tool" + ? String(block.params.arguments ?? "{}") + : JSON.stringify(block.params) + pushToolResult_withToolUseId_kilocode({ + type: "text", + text: formatArgumentRepairNote(executedArguments), + }) + } catch (error) { + console.error("[presentAssistantMessage] Failed to append argument repair note:", error) + } + } + // CRITICAL: every non-partial tool_use with a toolUseId MUST have a // matching tool_result pushed, even on failure. The assistant message // already contains the tool_use block, so without a paired tool_result diff --git a/src/core/tools/kilocode.ts b/src/core/tools/kilocode.ts index cd6e76b1c4..ae0647593a 100644 --- a/src/core/tools/kilocode.ts +++ b/src/core/tools/kilocode.ts @@ -55,6 +55,23 @@ type FileEntry = { limit?: number } +/** + * Parse a positive integer tool parameter. The model may send numbers as + * strings, null, or fractional values; anything unparseable falls back to + * the caller's default, and out-of-range values clamp to the minimum + * instead of failing the tool call. + */ +export function parsePositiveInteger(value: unknown): number | undefined { + if (value === null || value === undefined) { + return undefined + } + const parsed = typeof value === "number" ? value : parseInt(String(value), 10) + if (!Number.isFinite(parsed)) { + return undefined + } + return Math.max(1, Math.floor(parsed)) +} + export function parseNativeFiles( nativeFiles: { file_path?: string @@ -70,14 +87,13 @@ export function parseNativeFiles( const filePath = file.file_path || file.path if (!filePath) continue - // Parse offset and limit as integers - LLM may send them as strings - const parsedOffset = file.offset !== undefined ? parseInt(String(file.offset), 10) : undefined - const parsedLimit = file.limit !== undefined ? parseInt(String(file.limit), 10) : undefined - + // Parse offset and limit as positive integers - the model may send + // them as strings, null, or fractional values; unparseable values + // fall back to the tool defaults. const fileEntry: FileEntry = { path: filePath, - offset: !isNaN(parsedOffset as number) ? parsedOffset : 1, - limit: !isNaN(parsedLimit as number) ? parsedLimit : undefined, // undefined means read complete file + offset: parsePositiveInteger(file.offset) ?? 1, + limit: parsePositiveInteger(file.limit), // undefined means read complete file } // Legacy support: convert line_ranges to offset+limit if provided diff --git a/src/core/tools/readFileTool.ts b/src/core/tools/readFileTool.ts index 3f99365233..22be407c43 100644 --- a/src/core/tools/readFileTool.ts +++ b/src/core/tools/readFileTool.ts @@ -15,7 +15,12 @@ import { readLines } from "../../integrations/misc/read-lines" import { extractTextFromFile, addLineNumbers, getSupportedBinaryFormats } from "../../integrations/misc/extract-text" import { parseSourceCodeDefinitionsForFile } from "../../services/tree-sitter" import { parseXml } from "../../utils/xml" -import { blockFileReadWhenTooLarge, getNativeReadFileToolDescription, parseNativeFiles } from "./kilocode" +import { + blockFileReadWhenTooLarge, + getNativeReadFileToolDescription, + parseNativeFiles, + parsePositiveInteger, +} from "./kilocode" import { DEFAULT_MAX_IMAGE_FILE_SIZE_MB, DEFAULT_MAX_TOTAL_IMAGE_SIZE_MB, @@ -143,11 +148,13 @@ export async function readFileTool( const newFilePath: string | undefined = (block.params as any).file_path // New: support file_path directly const legacyStartLineStr: string | undefined = block.params.start_line const legacyEndLineStr: string | undefined = block.params.end_line - // Parse offset and limit as integers - LLM may send them as strings causing string concatenation bugs + // Parse offset and limit as positive integers - the model may send them as + // strings, null, or fractional values; unparseable values fall back to the + // tool defaults instead of failing the read. const rawOffset = (block.params as any).offset const rawLimit = (block.params as any).limit - const offsetParam: number | undefined = rawOffset !== undefined ? parseInt(String(rawOffset), 10) : undefined - const limitParam: number | undefined = rawLimit !== undefined ? parseInt(String(rawLimit), 10) : undefined + const offsetParam: number | undefined = parsePositiveInteger(rawOffset) + const limitParam: number | undefined = parsePositiveInteger(rawLimit) const nativeFiles: any[] | undefined = (block.params as any).files // kilocode_change: Native JSON format from OpenAI-style tool calls @@ -218,14 +225,12 @@ export async function readFileTool( const filePath = file.file_path || file.path if (!filePath) continue // Skip if no path in a file entry - // Parse offset and limit as integers - XML parsing may produce strings - const parsedOffset = file.offset !== undefined ? parseInt(String(file.offset), 10) : undefined - const parsedLimit = file.limit !== undefined ? parseInt(String(file.limit), 10) : undefined - + // Parse offset and limit as positive integers - XML parsing may + // produce strings, and repaired arguments may carry nulls. const fileEntry: FileEntry = { path: filePath, - offset: !isNaN(parsedOffset as number) ? parsedOffset : 1, - limit: !isNaN(parsedLimit as number) ? parsedLimit : undefined, // undefined means read complete file + offset: parsePositiveInteger(file.offset) ?? 1, + limit: parsePositiveInteger(file.limit), // undefined means read complete file } // Legacy support: convert line_range to offset+limit diff --git a/src/core/tools/useMcpToolTool.ts b/src/core/tools/useMcpToolTool.ts index 591c50302f..ec01933bb7 100644 --- a/src/core/tools/useMcpToolTool.ts +++ b/src/core/tools/useMcpToolTool.ts @@ -6,6 +6,7 @@ import { McpExecutionStatus } from "@roo-code/types" import { t } from "../../i18n" import { McpToolCallResponse, McpAuthError } from "../../shared/mcp" // kilocode_change import { summarizeSuccessfulMcpOutputWhenTooLong } from "./kilocode" // kilocode_change +import { formatArgumentRepairNote, parseToolCallArguments } from "../../utils/jsonRepair" // forked_change interface McpToolParams { server_name?: string @@ -20,6 +21,7 @@ type ValidationResult = serverName: string toolName: string parsedArguments?: Record + argumentsRepaired?: boolean } async function handlePartialRequest( @@ -72,14 +74,23 @@ async function validateParams( } let parsedArguments: Record = {} + let argumentsRepaired = false if (params.arguments) { try { // Handle both string (from XML) and object (from native function calling) if (typeof params.arguments === "string") { - const parsed = JSON.parse(params.arguments) - if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) { - parsedArguments = parsed + // forked_change: repair malformed JSON arguments (placeholder tags, + // unquoted values, single quotes, truncation) instead of failing the + // whole tool call. XML-mode arguments arrive complete, so truncated + // input is closed here too. + const parsed = parseToolCallArguments(params.arguments, { repairTruncated: true }) + if (!parsed) { + throw new Error("Invalid JSON in tool arguments") + } + if (parsed.args && typeof parsed.args === "object" && !Array.isArray(parsed.args)) { + parsedArguments = parsed.args as Record + argumentsRepaired = parsed.repaired } } else if (typeof params.arguments === "object") { // Already parsed (from native function calling) @@ -105,6 +116,7 @@ async function validateParams( serverName: params.server_name, toolName: params.tool_name, parsedArguments, + argumentsRepaired, } } @@ -253,6 +265,7 @@ async function executeToolAndProcessResult( parsedArguments: Record | undefined, executionId: string, pushToolResult: PushToolResult, + argumentsRepaired?: boolean, ): Promise { await cline.say("mcp_server_request_started") @@ -306,6 +319,11 @@ async function executeToolAndProcessResult( } await cline.say("mcp_server_response", toolResultPretty) + if (argumentsRepaired) { + // forked_change: transparency — tell the model what actually executed + // so it does not repeat the malformed form. + toolResultPretty += `\n\n${formatArgumentRepairNote(JSON.stringify(parsedArguments ?? {}))}` + } pushToolResult(formatResponse.toolResult(toolResultPretty)) } catch (error) { // Handle authentication errors specially @@ -364,7 +382,7 @@ export async function useMcpToolTool( return } - const { serverName, toolName, parsedArguments } = validation + const { serverName, toolName, parsedArguments, argumentsRepaired } = validation // Validate that the tool exists on the server const toolValidation = await validateToolExists(cline, serverName, toolName, pushToolResult) @@ -404,7 +422,15 @@ export async function useMcpToolTool( } // Execute the tool and process results - await executeToolAndProcessResult(cline, serverName!, toolName!, parsedArguments, executionId, pushToolResult) + await executeToolAndProcessResult( + cline, + serverName!, + toolName!, + parsedArguments, + executionId, + pushToolResult, + argumentsRepaired, + ) } catch (error) { await handleError("executing MCP tool", error) } diff --git a/src/package.json b/src/package.json index a9efcd95f6..36fc2767a9 100644 --- a/src/package.json +++ b/src/package.json @@ -3,7 +3,7 @@ "displayName": "%extension.displayName%", "description": "%extension.description%", "publisher": "matterai", - "version": "6.8.4", + "version": "6.8.5", "icon": "assets/icons/matterai-ic.png", "galleryBanner": { "color": "#FFFFFF", diff --git a/src/services/search-files/types.ts b/src/services/search-files/types.ts index f15962a3d8..3a73c11ec0 100644 --- a/src/services/search-files/types.ts +++ b/src/services/search-files/types.ts @@ -46,8 +46,10 @@ export function clampSearchOptions(options: SearchFilesOptions): Required> partial: boolean toolUseId?: string // kilocode_change + repaired?: boolean // forked_change: arguments were malformed JSON and were auto-repaired before execution } export interface ExecuteCommandToolUse extends ToolUse { diff --git a/src/utils/__tests__/jsonRepair.spec.ts b/src/utils/__tests__/jsonRepair.spec.ts new file mode 100644 index 0000000000..f6fdbdea1f --- /dev/null +++ b/src/utils/__tests__/jsonRepair.spec.ts @@ -0,0 +1,287 @@ +import { describe, expect, it } from "vitest" + +// npx vitest src/utils/__tests__/jsonRepair.spec.ts + +import { formatArgumentRepairNote, parseToolCallArguments } from "../jsonRepair" + +const parse = (raw: string, options?: { repairTruncated?: boolean }) => parseToolCallArguments(raw, options) + +describe("parseToolCallArguments — strict path", () => { + it("parses valid JSON verbatim and reports repaired: false", () => { + const result = parse('{"a": 1, "b": [1, 2, {"c": null}], "d": "text"}') + expect(result).toEqual({ args: { a: 1, b: [1, 2, { c: null }], d: "text" }, repaired: false }) + }) + + it("never touches tags inside valid JSON string values", () => { + const raw = '{"content": "uses bold and inside a string"}' + const result = parse(raw) + expect(result!.repaired).toBe(false) + expect(result!.args).toEqual({ content: "uses bold and inside a string" }) + }) + + it("keeps keys that contain colons when the JSON is otherwise valid", () => { + const raw = '{"Content-Type": "text/html", "X-Custom": "1"}' + expect(parse(raw)).toEqual({ args: { "Content-Type": "text/html", "X-Custom": "1" }, repaired: false }) + }) + + it("parses a valid array payload without unwrapping it", () => { + const result = parse('[{"a": 1}, {"b": 2}]') + expect(result!.repaired).toBe(false) + expect(result!.args).toEqual([{ a: 1 }, { b: 2 }]) + }) + + it("returns null for empty input", () => { + expect(parse("")).toBeNull() + expect(parse(" ")).toBeNull() + }) +}) + +describe("parseToolCallArguments — placeholder tags", () => { + it("keeps the value that follows a placeholder tag", () => { + const raw = '{"files": [{"file_path": "/tmp/a.ts", "offset": 70, "limit": 180}]}' + const result = parse(raw) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ + files: [{ file_path: "/tmp/a.ts", offset: 70, limit: 180 }], + }) + }) + + it("nulls a tag-only value so the tool default applies", () => { + const raw = '{"files": [{"file_path": "/tmp/a.ts", "offset": 70, "limit": }]}' + const result = parse(raw) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ + files: [{ file_path: "/tmp/a.ts", offset: 70, limit: null }], + }) + }) + + it("nulls a tag-only key followed by another member", () => { + const result = parse('{"path": , "regex": "foo"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: null, regex: "foo" }) + }) + + it("nulls a tag-only first key", () => { + const result = parse('{"limit": , "offset": 5}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ limit: null, offset: 5 }) + }) + + it("keeps a string value that follows a placeholder tag", () => { + const result = parse('{"file_path": "/tmp/a.ts"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ file_path: "/tmp/a.ts" }) + }) + + it("quotes a bare scalar that follows a placeholder tag", () => { + const result = parse('{"file_pattern": *.ts}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ file_pattern: "*.ts" }) + }) + + it("strips a tag that prefixes a key", () => { + const result = parse('{"path": "src"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src" }) + }) + + it("drops a key whose closing quote was replaced by a tag", () => { + const result = parse('{"offset, "regex": "foo"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ regex: "foo" }) + }) + + it("preserves tags inside string values of otherwise-broken JSON", () => { + const result = parse('{"content": "uses bold", path: "src"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ content: "uses bold", path: "src" }) + }) + + it("returns null for an incomplete buffer containing a tag (keep accumulating)", () => { + const raw = '{"files": [{"file_path": "/tmp/a.ts", "limit": ' + expect(parse(raw)).toBeNull() + }) +}) + +describe("parseToolCallArguments — unquoted keys and values", () => { + it("quotes bare keys", () => { + const result = parse('{path: "src", regex: "foo"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src", regex: "foo" }) + }) + + it("quotes bare scalar values", () => { + const result = parse('{"file_pattern": *.ts}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ file_pattern: "*.ts" }) + }) + + it("keeps strict numbers as numbers", () => { + const result = parse("{a: 1, b: 2.5, c: -3, d: 1e3}") + expect(result!.args).toEqual({ a: 1, b: 2.5, c: -3, d: 1000 }) + }) + + it("keeps JSON booleans and null unquoted", () => { + const result = parse("{a: true, b: false, c: null}") + expect(result!.args).toEqual({ a: true, b: false, c: null }) + }) + + it("keeps colons inside bare values so URLs survive", () => { + const result = parse('{"url": https://example.com/x}') + expect(result!.args).toEqual({ url: "https://example.com/x" }) + }) + + it("repairs nested structures", () => { + const result = parse('{"files": [{path: "a.ts", offset: 5}]}') + expect(result!.args).toEqual({ files: [{ path: "a.ts", offset: 5 }] }) + }) +}) + +describe("parseToolCallArguments — single quotes and Python literals", () => { + it("converts single-quoted keys and values", () => { + const result = parse("{'path': 'src'}") + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src" }) + }) + + it("unescapes escaped single quotes inside values", () => { + const result = parse("{'text': 'it\\'s'}") + expect(result!.args).toEqual({ text: "it's" }) + }) + + it("maps Python literals to JSON", () => { + const result = parse("{show_line_numbers: True, include_summary: False, cursor: None}") + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ show_line_numbers: true, include_summary: false, cursor: null }) + }) +}) + +describe("parseToolCallArguments — comments and separators", () => { + it("strips line and block comments", () => { + const raw = [ + "{", + " // search for the parser", + ' "path": "src",', + ' "regex": "foo" /* trailing */', + "}", + ].join("\n") + const result = parse(raw) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src", regex: "foo" }) + }) + + it("drops trailing commas before closers", () => { + const result = parse('{"a": 1, "b": [1, 2,],}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ a: 1, b: [1, 2] }) + }) + + it("inserts a missing comma between members", () => { + const result = parse('{"a": 1 "b": 2}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ a: 1, b: 2 }) + }) +}) + +describe("parseToolCallArguments — dropped key quotes", () => { + it("recovers a key whose closing quote was lost to a colon split", () => { + const result = parse('{"offset: 600, "regex": "foo"}') + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ offset: 600, regex: "foo" }) + }) + + it("recovers a dangling key with its value on truncation", () => { + const result = parse('{"offset: 600', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ offset: 600 }) + }) +}) + +describe("parseToolCallArguments — truncation repair", () => { + it("closes a dangling string value", () => { + const result = parse('{"path": "src/foo', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src/foo" }) + }) + + it("appends null after a dangling colon", () => { + const result = parse('{"path": "src", "offset":', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src", offset: null }) + }) + + it("closes unclosed containers innermost-first", () => { + const result = parse('{"files": [{"file_path": "/tmp/a.ts"', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ files: [{ file_path: "/tmp/a.ts" }] }) + }) + + it("drops a dangling key and its trailing comma", () => { + const result = parse('{"a": 1, "b"', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ a: 1 }) + }) + + it("closes an unclosed array", () => { + const result = parse("[1, 2", { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual([1, 2]) + }) + + it("drops a trailing comma at end of input", () => { + const result = parse('{"a": 1,', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ a: 1 }) + }) + + it("returns null for incomplete input without repairTruncated", () => { + expect(parse('{"path": "src/foo')).toBeNull() + expect(parse('{"a": 1')).toBeNull() + expect(parse("[1, 2")).toBeNull() + }) +}) + +describe("parseToolCallArguments — braceless bodies", () => { + it("wraps a braceless body when repairing a finalized buffer", () => { + const result = parse('path: "src", regex: "foo"', { repairTruncated: true }) + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src", regex: "foo" }) + }) + + it("returns null for a braceless buffer while still streaming", () => { + expect(parse('path: "src"')).toBeNull() + }) +}) + +describe("parseToolCallArguments — single-object arrays", () => { + it("unwraps a repaired single-object array", () => { + const result = parse("[{path: 'src'}]") + expect(result!.repaired).toBe(true) + expect(result!.args).toEqual({ path: "src" }) + }) +}) + +describe("parseToolCallArguments — unrecoverable input", () => { + it("returns null for prose", () => { + expect(parse("just some prose")).toBeNull() + expect(parse("just some prose", { repairTruncated: true })).toBeNull() + }) + + it("returns null for a key not followed by a colon", () => { + expect(parse('{"a" 1}')).toBeNull() + }) +}) + +describe("formatArgumentRepairNote", () => { + it("includes the executed arguments", () => { + const note = formatArgumentRepairNote('{"path": "src"}') + expect(note).toContain("malformed JSON") + expect(note).toContain('{"path": "src"}') + }) + + it("truncates very long arguments", () => { + const note = formatArgumentRepairNote("x".repeat(3000)) + expect(note).toContain("...(truncated)") + expect(note.length).toBeLessThan(2200) + }) +}) diff --git a/src/utils/jsonRepair.ts b/src/utils/jsonRepair.ts new file mode 100644 index 0000000000..5f363a2c99 --- /dev/null +++ b/src/utils/jsonRepair.ts @@ -0,0 +1,641 @@ +/** + * Best-effort repair layer for malformed tool-call JSON arguments. + * + * Models — especially smaller OSS ones — emit almost-JSON tool-call + * arguments: unquoted values, dropped quotes, internal markup tags, Python + * literals, comments, trailing commas, or truncated bodies. Rejecting the + * call burns a full round trip, and weaker models repeat the exact same + * mistake on retry, dead-looping. + * + * `parseToolCallArguments` is the single entry point used by every + * tool-call ingestion path (live streaming, stream finalization, MCP + * argument validation): + * + * 1. Strict parse first. Valid payloads are returned verbatim with + * `repaired: false` — zero overhead, zero risk of corrupting good + * input. The one exception: a leaked placeholder tag inside a key + * (e.g. `{"path": "src"}`) parses as valid JSON but + * names an unknown parameter, so tags are stripped from keys only — + * tags inside string values may be legitimate content and survive. + * Everything below only runs on input that already failed. + * 2. A single left-to-right repair scan rewrites the source into + * parseable JSON: placeholder tags in structural positions are + * skipped (tags inside quoted strings are preserved), unquoted + * keys/values quoted, single quotes converted, Python literals + * mapped, comments removed, trailing commas dropped, and lost key + * quotes recovered. + * 3. With `repairTruncated` (stream finalization only), dangling strings + * are closed, a trailing comma dropped, `null` appended after a + * dangling colon, and open braces/brackets closed innermost-first. + * + * If the repaired source still fails to parse, `null` is returned — the + * caller then surfaces the corrective error with the raw arguments. + */ + +export interface ToolCallArgumentsParse { + args: unknown + repaired: boolean +} + +export interface ParseToolCallArgumentsOptions { + /** + * Repair truncated input (dangling strings, unclosed containers, missing + * values) by closing what is open. Only enable once the stream has + * ended — during streaming an incomplete buffer must stay unparseable + * so the parser keeps accumulating deltas. + */ + repairTruncated?: boolean +} + +/** Placeholder tags some models leak into tool-call JSON (e.g. ``). */ +const PLACEHOLDER_TAG = /<\/?[A-Za-z_][A-Za-z0-9_.\-]*>/g + +/** Bare tokens that stay unquoted when repaired: strict JSON numbers and literals. */ +const STRICT_JSON_NUMBER = /^-?(?:0|[1-9]\d*)(?:\.\d+)?(?:[eE][+-]?\d+)?$/ + +const JSON_LITERALS: Record = { + true: "true", + false: "false", + null: "null", + True: "true", + False: "false", + None: "null", +} + +/** Bare keys must look like identifiers — this is what rejects prose. */ +const BARE_KEY = /^[A-Za-z_$][A-Za-z0-9_$.\-]*$/ + +const REPAIR_NOTE_MAX_ARGS_LENGTH = 2000 + +export function parseToolCallArguments( + raw: string, + options?: ParseToolCallArgumentsOptions, +): ToolCallArgumentsParse | null { + const trimmed = raw.trim() + if (!trimmed) { + return null + } + + // 1. Strict parse first — the ONLY path valid JSON ever takes, so + // well-formed arguments are never mutated. + try { + const args = JSON.parse(raw) + const cleaned = stripTagsFromKeys(args) + return { args: cleaned.value, repaired: cleaned.changed } + } catch { + // Fall through to the repair pass. + } + + // 2. Repair pass. + const repairedSource = repairJsonSource(trimmed, options?.repairTruncated ?? false) + if (repairedSource === null) { + return null + } + + try { + let args: unknown = JSON.parse(repairedSource) + // A single-object array is an object the model wrapped by mistake. + if (Array.isArray(args) && args.length === 1 && typeof args[0] === "object" && args[0] !== null) { + args = args[0] + } + return { args, repaired: true } + } catch { + return null + } +} + +/** + * Human-readable note appended to a tool result when its arguments were + * auto-repaired, so the model sees what actually executed instead of its + * broken original — this is what stops it from repeating the mistake. + */ +export function formatArgumentRepairNote(executedArguments: string): string { + const bounded = + executedArguments.length > REPAIR_NOTE_MAX_ARGS_LENGTH + ? `${executedArguments.slice(0, REPAIR_NOTE_MAX_ARGS_LENGTH)}...(truncated)` + : executedArguments + return `The arguments were malformed JSON and were auto-repaired before execution. Executed arguments: ${bounded}` +} + +/** + * Strips placeholder tags from object keys of a strictly-parsed payload. + * Keys are structural (parameter names), so a tag inside a key is always a + * leaked artifact; tags inside string values may be legitimate content and + * are never touched. A member whose key is only a tag is dropped — the + * parameter name was elided entirely. + */ +function stripTagsFromKeys(value: unknown): { value: unknown; changed: boolean } { + if (Array.isArray(value)) { + let changed = false + const items = value.map((item) => { + const result = stripTagsFromKeys(item) + changed = changed || result.changed + return result.value + }) + return { value: changed ? items : value, changed } + } + if (typeof value === "object" && value !== null) { + let changed = false + const output: Record = {} + for (const [key, item] of Object.entries(value)) { + const nested = stripTagsFromKeys(item) + const cleanedKey = key.replace(PLACEHOLDER_TAG, "") + changed = changed || nested.changed || cleanedKey !== key + if (cleanedKey) { + output[cleanedKey] = nested.value + } + } + return { value: changed ? output : value, changed } + } + return { value, changed: false } +} + +function repairJsonSource(source: string, repairTruncated: boolean): string | null { + const trimmed = source.trim() + if (!trimmed) { + return null + } + + // Braceless bodies (e.g. `path: "src"`) are wrapped in {}. Only attempted + // when repairing a finalized buffer: during streaming there is no way to + // know the body is complete. + if (!trimmed.startsWith("{") && !trimmed.startsWith("[")) { + if (!repairTruncated) { + return null + } + return scanJson(`{${trimmed}}`, repairTruncated) + } + + return scanJson(trimmed, repairTruncated) +} + +type Container = "{" | "[" +type ScanState = "expect-key" | "expect-value" | "expect-comma-or-close" + +type KeyToken = + | { kind: "key-with-colon"; text: string; nextIndex: number } + | { kind: "key"; text: string; nextIndex: number } + | { kind: "dropped"; nextIndex: number } + +function scanJson(source: string, repairTruncated: boolean): string | null { + let out = "" + let index = 0 + const stack: Container[] = [] + let state: ScanState = "expect-value" + // How we entered expect-value: after a colon (a missing value gets null), + // after a comma, or right after an opening bracket. + let valueOrigin: "colon" | "comma" | "open" = "open" + + const top = (): Container | undefined => stack[stack.length - 1] + + while (index < source.length) { + // The top-level container closed and something else follows — trailing + // junk. Keep the repaired value and stop. + if (stack.length === 0 && state === "expect-comma-or-close") { + break + } + + const char = source[index] + + // Whitespace and comments between tokens. + if (isWhitespace(char)) { + index++ + continue + } + if (char === "/" && source[index + 1] === "/") { + index = skipLineComment(source, index) + continue + } + if (char === "/" && source[index + 1] === "*") { + index = skipBlockComment(source, index) + continue + } + + // Placeholder tags (e.g. ``) in structural + // positions are dropped. Tags inside quoted strings are preserved by + // the string readers, so markup in string values survives. + if (isTagStart(source, index)) { + index = skipTag(source, index) + continue + } + + // A token where a comma or closer was expected means the model dropped + // the comma. Insert one and reprocess the same character. + if (state === "expect-comma-or-close" && char !== "," && char !== "}" && char !== "]") { + out += "," + state = top() === "{" ? "expect-key" : "expect-value" + valueOrigin = "comma" + continue + } + + if (char === "{") { + if (state === "expect-value") { + out += "{" + stack.push("{") + state = "expect-key" + } + // A stray opener elsewhere is dropped. + index++ + continue + } + + if (char === "[") { + if (state === "expect-value") { + out += "[" + stack.push("[") + state = "expect-value" + valueOrigin = "open" + } + index++ + continue + } + + if (char === "}" || char === "]") { + if (stack.length === 0) { + // Unmatched top-level closer — drop. + index++ + continue + } + if (state === "expect-value" && valueOrigin === "colon") { + // `"key": }` — the value is missing; null keeps the key so the + // tool's default applies. + out += "null" + } + out += top() === "{" ? "}" : "]" + stack.pop() + index++ + state = "expect-comma-or-close" + continue + } + + if (char === ",") { + if (state === "expect-value") { + if (valueOrigin === "colon") { + // `"key": ,` — missing value before the separator. + out += "null" + state = "expect-comma-or-close" + } else { + // Stray comma right after an opener or another comma. + index++ + continue + } + } else if (state === "expect-key") { + // Stray comma where a key belongs — drop. + index++ + continue + } + // Trailing comma before a closer (or end of input) is dropped. + const peek = peekSignificant(source, index + 1) + if (peek >= source.length || source[peek] === "}" || source[peek] === "]") { + index++ + continue + } + out += "," + index++ + state = top() === "{" ? "expect-key" : "expect-value" + valueOrigin = "comma" + continue + } + + // A colon reaching the loop is always stray — legitimate colons are + // consumed by key processing, and colons inside bare values (URLs) + // never terminate a token. + if (char === ":") { + index++ + continue + } + + if (state === "expect-key") { + const keyToken = readKeyToken(source, index) + if (keyToken === null) { + return null + } + if (keyToken.kind === "dropped") { + index = keyToken.nextIndex + continue + } + if (keyToken.kind === "key-with-colon") { + out += keyToken.text + index = keyToken.nextIndex + state = "expect-value" + valueOrigin = "colon" + continue + } + // The colon must follow, else the key is dangling. + const colonIndex = peekSignificant(source, keyToken.nextIndex) + if (colonIndex < source.length && source[colonIndex] === ":") { + out += `${keyToken.text}:` + index = colonIndex + 1 + state = "expect-value" + valueOrigin = "colon" + continue + } + if (colonIndex >= source.length || source[colonIndex] === "}" || source[colonIndex] === "]") { + // Dangling key without a colon — drop it; the closer (or end + // of input) is handled by the loop. + index = keyToken.nextIndex + continue + } + // A key followed by anything other than a colon is unrepairable. + return null + } + + // state === "expect-value" + if (char === '"') { + const stringValue = readDoubleQuotedString(source, index, repairTruncated) + if (stringValue === null) { + return null + } + out += stringValue.text + index = stringValue.nextIndex + state = "expect-comma-or-close" + continue + } + + if (char === "'") { + const stringValue = readSingleQuotedString(source, index, repairTruncated) + if (stringValue === null) { + return null + } + out += stringValue.text + index = stringValue.nextIndex + state = "expect-comma-or-close" + continue + } + + // Bare token value. Colons and slashes do NOT terminate values, so + // URLs and glob patterns survive intact. + let end = index + while (end < source.length && !isBareTokenTerminator(source[end])) { + end++ + } + const token = source.slice(index, end) + if (!token) { + index++ + continue + } + out += classifyBareToken(token) + index = end + state = "expect-comma-or-close" + } + + // End of input. + if (!repairTruncated) { + // Incomplete unless the top-level container closed cleanly. + if (stack.length === 0 && state === "expect-comma-or-close") { + return out + } + return null + } + + // Truncation repair: append the missing value, drop a trailing comma + // left by a dropped dangling key, and close open containers + // innermost-first. + if (state === "expect-value" && valueOrigin === "colon") { + out += "null" + } + out = out.replace(/,$/, "") + while (stack.length > 0) { + out += stack.pop() === "{" ? "}" : "]" + } + return out +} + +function readKeyToken(source: string, start: number): KeyToken | null { + const char = source[start] + + if (char === '"') { + const close = findUnescaped(source, start + 1, '"') + const contentEnd = close === -1 ? source.length : close + const content = source.slice(start + 1, contentEnd) + + // A properly closed key followed by a colon is a legitimate key, even + // when the key itself contains a colon (e.g. "Content-Type"). + if (close !== -1) { + const after = peekSignificant(source, close + 1) + if (after < source.length && source[after] === ":") { + const key = content.replace(PLACEHOLDER_TAG, "") + if (!key) { + return null + } + return { kind: "key", text: JSON.stringify(key), nextIndex: close + 1 } + } + } + + // The string swallowed structure (lost closing quote) or never closed. + // Recover the key from the head, splitting at the first colon, comma, + // or whitespace outside placeholder tags. + const split = findSplitOutsideTags(content) + if (split) { + const key = content.slice(0, split.index).trim().replace(PLACEHOLDER_TAG, "") + if (!key) { + return null + } + const splitIndex = start + 1 + split.index + if (split.char === ",") { + // The member never got a value — drop the key and let the + // scanner reprocess the comma. + return { kind: "dropped", nextIndex: splitIndex } + } + return { + kind: "key-with-colon", + text: `${JSON.stringify(key)}:`, + nextIndex: skipWhitespace(source, split.char === ":" ? splitIndex + 1 : splitIndex), + } + } + + // No split character: a closed string is a plain key; a dangling one + // never completed. + if (close !== -1) { + const key = content.replace(PLACEHOLDER_TAG, "") + if (!key) { + return null + } + return { kind: "key", text: JSON.stringify(key), nextIndex: close + 1 } + } + return { kind: "dropped", nextIndex: source.length } + } + + if (char === "'") { + const close = findUnescaped(source, start + 1, "'") + if (close === -1) { + return null + } + const key = source.slice(start + 1, close).replace(PLACEHOLDER_TAG, "") + if (!key) { + return null + } + return { kind: "key", text: JSON.stringify(key), nextIndex: close + 1 } + } + + // Bare key. + let end = start + while (end < source.length && !isKeyTerminator(source[end])) { + end++ + } + const token = source.slice(start, end) + if (!token || !BARE_KEY.test(token)) { + // Prose or a number where a key belongs — unrepairable. + return null + } + return { kind: "key", text: JSON.stringify(token), nextIndex: end } +} + +function readDoubleQuotedString( + source: string, + start: number, + repairTruncated: boolean, +): { text: string; nextIndex: number } | null { + const close = findUnescaped(source, start + 1, '"') + if (close === -1) { + if (!repairTruncated) { + // Still streaming — the string has not closed yet. + return null + } + // Close the dangling string. Content is emitted verbatim so intended + // JSON escapes survive. + return { text: `"${source.slice(start + 1)}"`, nextIndex: source.length } + } + return { text: source.slice(start, close + 1), nextIndex: close + 1 } +} + +function readSingleQuotedString( + source: string, + start: number, + repairTruncated: boolean, +): { text: string; nextIndex: number } | null { + const close = findUnescaped(source, start + 1, "'") + if (close === -1) { + if (!repairTruncated) { + return null + } + return { text: JSON.stringify(unescapeSingleQuoted(source.slice(start + 1))), nextIndex: source.length } + } + return { text: JSON.stringify(unescapeSingleQuoted(source.slice(start + 1, close))), nextIndex: close + 1 } +} + +function classifyBareToken(token: string): string { + const literal = JSON_LITERALS[token] + if (literal) { + return literal + } + if (STRICT_JSON_NUMBER.test(token)) { + return token + } + return JSON.stringify(token) +} + +function unescapeSingleQuoted(content: string): string { + return content.replace(/\\(['\\])/g, "$1") +} + +function isWhitespace(char: string): boolean { + return char === " " || char === "\t" || char === "\n" || char === "\r" +} + +function skipWhitespace(source: string, start: number): number { + let index = start + while (index < source.length && isWhitespace(source[index])) { + index++ + } + return index +} + +function findUnescaped(source: string, start: number, quote: string): number { + for (let index = start; index < source.length; index++) { + if (source[index] === "\\") { + index++ + continue + } + if (source[index] === quote) { + return index + } + } + return -1 +} + +function skipLineComment(source: string, start: number): number { + let index = start + 2 + while (index < source.length && source[index] !== "\n") { + index++ + } + return index +} + +function skipBlockComment(source: string, start: number): number { + const end = source.indexOf("*/", start + 2) + return end === -1 ? source.length : end + 2 +} + +function peekSignificant(source: string, start: number): number { + let index = start + while (index < source.length) { + const char = source[index] + if (isWhitespace(char)) { + index++ + continue + } + if (char === "/" && source[index + 1] === "/") { + index = skipLineComment(source, index) + continue + } + if (char === "/" && source[index + 1] === "*") { + index = skipBlockComment(source, index) + continue + } + if (isTagStart(source, index)) { + index = skipTag(source, index) + continue + } + return index + } + return index +} + +/** True when the character at `index` opens a placeholder tag such as `` or ``. */ +function isTagStart(source: string, index: number): boolean { + if (source[index] !== "<") { + return false + } + const next = source[index + 1] + if (next !== undefined && /[A-Za-z_]/.test(next)) { + return true + } + return next === "/" && /[A-Za-z_]/.test(source[index + 2] ?? "") +} + +function skipTag(source: string, start: number): number { + const close = source.indexOf(">", start + 1) + return close === -1 ? source.length : close + 1 +} + +/** + * Find the first colon, comma, or whitespace in `content` that is not inside + * a placeholder tag, so index math stays valid against the original source. + */ +function findSplitOutsideTags(content: string): { index: number; char: string } | null { + let index = 0 + while (index < content.length) { + const char = content[index] + if (char === ":" || char === "," || isWhitespace(char)) { + return { index, char } + } + if (char === "<") { + const tagMatch = /^<\/?[A-Za-z_][A-Za-z0-9_.\-]*>/.exec(content.slice(index)) + if (tagMatch) { + index += tagMatch[0].length + continue + } + } + index++ + } + return null +} + +function isBareTokenTerminator(char: string): boolean { + return isWhitespace(char) || char === "," || char === "}" || char === "]" || char === '"' || char === "'" +} + +function isKeyTerminator(char: string): boolean { + return isBareTokenTerminator(char) || char === ":" +}