diff --git a/CHANGELOG.md b/CHANGELOG.md index e5c3d26943..9b3e0640ed 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,16 @@ # Changelog +## [Unreleased] + +### Changed + +- **Investigation efficiency guidance in system prompt.** Added an "Investigation efficiency" section to the shared tool-usage instructions (`applyDiffToolDescription` in `src/core/prompts/system.ts`) that directs the agent to classify comprehension questions separately from implementation tasks, form a one-line hypothesis before searching, read call sites rather than implementation internals, avoid reading prose/content when the question is about control flow, and stop exploring as soon as it can answer. +- **Overlapping read detection in `read_file`.** `readFileTool.ts` now detects when a requested file region partially overlaps with a region already read earlier in the same task (same file, same mtime). The exact-match short-circuit (unchanged) still blocks identical re-reads; the new overlap check prepends a notice to the served content telling the model which lines are already in context, so it does not waste a follow-up call re-reading the overlapping portion. This catches the common failure of reading lines 600-849 then 800-1059 of the same file. +- **Zero-result guidance in `search_files`.** `searchFilesTool.ts` now appends actionable guidance when a search returns 0 matches, directing the model to tighten or simplify the regex, widen the path scope, try a different glob, or stop searching after 2+ failed attempts. This prevents the search to 0 results to slightly different regex to 0 results loop. +- **Native tool description improvements.** The `read_file` native tool description now tells the model not to read file contents (prompt text, config values, prose) when investigating control flow, and not to re-read regions already read earlier. The `search_files` native tool description now tells the model to scope the path to the narrowest plausible directory and to stop after 2+ zero-result searches. + +--- + ## [v6.9.0] - 2026-08-04 ### Added diff --git a/src/core/prompts/system.ts b/src/core/prompts/system.ts index 1bca3251f1..cd2aa9593a 100644 --- a/src/core/prompts/system.ts +++ b/src/core/prompts/system.ts @@ -240,6 +240,16 @@ Use zero context for discovery, then read the relevant file region. Reuse a curs - Investigate first, edit second. Once the root cause is confirmed, write out the full change plan — which files, the exact locations, and the edit order — BEFORE touching anything. - Then execute the edits in one pass (batched via \`multi_file_edit\`) and verify with a single typecheck/build at the end, rather than alternating between editing and checking. +## Investigation efficiency + +Before every tool call, ask: "Will this result change my answer or my implementation?" If no, do not make the call. + +- **Classify the question first.** Is this a comprehension question ("how does X work?", "is this by design or a bug?") or an implementation task? Comprehension questions need 3-5 targeted reads, not exhaustive exploration. +- **Form a hypothesis, then verify.** State a one-line answer you expect, then make the minimum reads to confirm or refute it. Do not explore speculatively. +- **Read the call site, not the implementation.** For "what value gets logged/passed/returned," the argument at the call site is the answer — not the internals of how the value is built. +- **Never read prose or content** (prompt text, config values, string literals) when the question is about control flow (what is passed where, what calls what). +- **Stop when you can answer.** Once you have enough to answer the user's question, stop exploring. Do not read additional files "for completeness." + ## update_todo_list **Description:** diff --git a/src/core/prompts/tools/native-tools/read_file.ts b/src/core/prompts/tools/native-tools/read_file.ts index 7b57437e97..3fe51c2e49 100644 --- a/src/core/prompts/tools/native-tools/read_file.ts +++ b/src/core/prompts/tools/native-tools/read_file.ts @@ -5,7 +5,7 @@ export const read_file = { function: { name: "read_file", description: - "Read one or more files and return line-numbered contents. Batch every independent file or region needed for the current investigation into this single call. Prefer 200-1000 lines per source-code region; for files up to 1000 lines, omit offset and limit to read the file once. Do not walk adjacent regions through many small calls. Each requested region is capped at 1000 lines.", + "Read one or more files and return line-numbered contents. Batch every independent file or region needed for the current investigation into this single call. Prefer 200-1000 lines per source-code region; for files up to 1000 lines, omit offset and limit to read the file once. Do not walk adjacent regions through many small calls. Each requested region is capped at 1000 lines. Do not read file contents (prompt text, config values, prose) when investigating control flow — read the call site, not the implementation. If a region was already read earlier in the conversation, do not re-read it; use the earlier content.", strict: true, parameters: { type: "object", diff --git a/src/core/prompts/tools/native-tools/search_files.ts b/src/core/prompts/tools/native-tools/search_files.ts index 44b1be450f..38a8a58f90 100644 --- a/src/core/prompts/tools/native-tools/search_files.ts +++ b/src/core/prompts/tools/native-tools/search_files.ts @@ -5,7 +5,7 @@ export default { function: { name: "search_files", description: - "Search file contents recursively under a directory using a Rust-compatible regex and optional file glob. Returns a compact, paginated page with at most three matches per file. To continue, call search_files again with the returned cursor and the same path, regex, and file_pattern; pass JSON null without quotes for the first page.", + "Search file contents recursively under a directory using a Rust-compatible regex and optional file glob. Returns a compact, paginated page with at most three matches per file. To continue, call search_files again with the returned cursor and the same path, regex, and file_pattern; pass JSON null without quotes for the first page. Scope path to the narrowest plausible directory instead of searching from the repository root. If a search returns 0 matches, tighten or simplify the regex rather than retrying with a slightly different pattern. After 2+ searches with no results, stop and reason from what you already know.", strict: true, parameters: { type: "object", diff --git a/src/core/tools/readFileTool.ts b/src/core/tools/readFileTool.ts index f8aba87ef1..82de2626bd 100644 --- a/src/core/tools/readFileTool.ts +++ b/src/core/tools/readFileTool.ts @@ -89,6 +89,7 @@ interface FileResult { feedbackImages?: any[] // User feedback images from approval/denial mtimeMs?: number // forked_change: file mtime at read time, for repeated-read detection wasRepeated?: boolean // The unchanged region was already served earlier in this task + overlapNotice?: string // forked_change: notice prepended when the read overlaps a prior read totalLines?: number startLine?: number endLine?: number @@ -99,6 +100,36 @@ function readRegionKey(fullPath: string, offset?: number, limit?: number): strin return `${fullPath}|${offset ?? 1}|${limit ?? "all"}` } +// forked_change: find a previously-read region of the same file version that +// overlaps with the requested range. Returns the overlapping range or null. +// Unlike readRegionKey (exact match), this catches partial overlaps — e.g. +// reading lines 600-849 then 800-1059 — so the model can be told part of the +// content is already in context. +function findOverlappingReadRegion( + cline: Task, + fullPath: string, + mtimeMs: number, + offset: number, + limit: number | undefined, +): { startLine: number; endLine: number } | null { + const newStart = offset + const newEnd = limit !== undefined ? offset + limit - 1 : Infinity + const prefix = `${fullPath}|` + for (const [key, storedMtime] of cline.readRegionHistory) { + if (storedMtime !== mtimeMs) continue + if (!key.startsWith(prefix)) continue + const parts = key.split("|") + const oldStart = parseInt(parts[1], 10) + const oldLimitStr = parts[2] + const oldEnd = oldLimitStr === "all" ? Infinity : oldStart + parseInt(oldLimitStr, 10) - 1 + // Overlap check: ranges [newStart, newEnd] and [oldStart, oldEnd] + if (newStart <= oldEnd && oldStart <= newEnd) { + return { startLine: oldStart, endLine: oldEnd === Infinity ? 0 : oldEnd } + } + } + return null +} + export async function readFileTool( cline: Task, block: ToolUse, @@ -458,6 +489,25 @@ Do not stop or ask the user because of this skipped read; proceed with the best } // forked_change end + // forked_change: detect overlapping reads of the same file version. + // The exact-match check above catches identical (path, offset, limit) + // repeats. This catches partial overlaps — e.g. reading 600-849 then + // 800-1059 — and prepends a notice so the model knows part of the + // content is already in context and should not re-read it. + if (fileResult.mtimeMs !== undefined && !fileResult.wasRepeated) { + const overlap = findOverlappingReadRegion( + cline, + fullPath, + fileResult.mtimeMs, + fileResult.offset ?? 1, + fileResult.limit, + ) + if (overlap) { + const endLabel = overlap.endLine === 0 ? "end" : String(overlap.endLine) + fileResult.overlapNotice = `[overlap notice] Lines ${overlap.startLine}-${endLabel} of this file were already read earlier in this conversation. Only lines outside that range are new. Use the earlier content for the overlapping portion instead of re-reading it.` + } + } + // Process approved files try { const [totalLines, isBinary] = await Promise.all([countFileLines(fullPath), isBinaryFile(fullPath)]) @@ -637,6 +687,14 @@ Do not stop or ask the user because of this skipped read; proceed with the best } } + // forked_change: prepend overlap notices to served content so the model + // is aware it is partially re-reading a region. + for (const result of fileResults) { + if (result.overlapNotice && result.xmlContent && !result.wasRepeated) { + result.xmlContent = `${result.overlapNotice}\n${result.xmlContent}` + } + } + // forked_change start: record regions that were actually served so repeated // reads of the same unchanged region can be short-circuited next time. // Only successful reads count — denials and errors must stay retryable. diff --git a/src/core/tools/searchFilesTool.ts b/src/core/tools/searchFilesTool.ts index 1a1b7217bf..aabb19383a 100644 --- a/src/core/tools/searchFilesTool.ts +++ b/src/core/tools/searchFilesTool.ts @@ -58,11 +58,27 @@ export async function searchFilesTool( cline.consecutiveMistakeCount = 0 - const results = await searchFiles(cline.cwd, absolutePath, regex, filePattern, cline.rooIgnoreController, { - cursor, - maxResults: Number.isFinite(maxResults) ? maxResults : undefined, - contextLines: Number.isFinite(contextLines) ? contextLines : undefined, - }) + const { text: results, matchCount } = await searchFiles( + cline.cwd, + absolutePath, + regex, + filePattern, + cline.rooIgnoreController, + { + cursor, + maxResults: Number.isFinite(maxResults) ? maxResults : undefined, + contextLines: Number.isFinite(contextLines) ? contextLines : undefined, + }, + ) + + // forked_change: append guidance when a search returns no matches, + // steering the model toward tightening/loosening the regex or scoping + // the path instead of blindly retrying with a slightly different pattern. + let output = results + if (matchCount === 0) { + output += + "\n\nNo matches found. Before retrying:\n- Tighten or simplify the regex (e.g. use a shorter, more specific pattern).\n- Widen the path scope (e.g. search from the repo root instead of a subdirectory).\n- Try a different file_pattern glob.\n- If you have already searched 2+ times with no results, stop searching and reason from what you already know." + } const completeMessage = JSON.stringify({ ...sharedMessageProps, content: results } satisfies ClineSayTool) const didApprove = await askApproval("tool", completeMessage) @@ -71,7 +87,7 @@ export async function searchFilesTool( return } - pushToolResult(results) + pushToolResult(output) return } diff --git a/src/package.json b/src/package.json index 005eb62d7f..fe0534878d 100644 --- a/src/package.json +++ b/src/package.json @@ -3,7 +3,7 @@ "displayName": "%extension.displayName%", "description": "%extension.description%", "publisher": "matterai", - "version": "6.8.0", + "version": "6.8.1", "icon": "assets/icons/matterai-ic.png", "galleryBanner": { "color": "#FFFFFF", diff --git a/src/services/search-files/__tests__/index.spec.ts b/src/services/search-files/__tests__/index.spec.ts index 3c22be1985..6ea7046ffb 100644 --- a/src/services/search-files/__tests__/index.spec.ts +++ b/src/services/search-files/__tests__/index.spec.ts @@ -18,7 +18,7 @@ describe("search_files engine selection", () => { it("uses FFF by default", async () => { fffMock.mockResolvedValue({ engine: "fff", matches: [], nextCursor: null }) - const output = await searchFiles("/workspace", "/workspace/src", "needle") + const { text: output } = await searchFiles("/workspace", "/workspace/src", "needle") expect(output).toContain("Engine: fff") expect(fffMock).toHaveBeenCalledOnce() @@ -29,7 +29,7 @@ describe("search_files engine selection", () => { fffMock.mockRejectedValue(new Error("native unavailable")) ripgrepMock.mockResolvedValue({ engine: "ripgrep", matches: [], nextCursor: null }) - const output = await searchFiles("/workspace", "/workspace/src", "needle") + const { text: output } = await searchFiles("/workspace", "/workspace/src", "needle") expect(output).toContain("Engine: ripgrep") expect(output).toContain("FFF failed; used ripgrep fallback") @@ -43,6 +43,8 @@ describe("search_files engine selection", () => { cursor: { engine: "ripgrep", offset: 50 }, }) + // Return value is unused in this test; just verifying engine selection. + expect(fffMock).not.toHaveBeenCalled() expect(ripgrepMock).toHaveBeenCalledOnce() }) diff --git a/src/services/search-files/index.ts b/src/services/search-files/index.ts index 3952e6ffe7..d7ec200f5a 100644 --- a/src/services/search-files/index.ts +++ b/src/services/search-files/index.ts @@ -4,6 +4,11 @@ import { searchFilesWithRipgrep } from "../ripgrep" import { formatSearchPage } from "./format" import { SearchFilesOptions, SearchPage } from "./types" +export interface SearchFilesResult { + text: string + matchCount: number +} + export async function searchFiles( cwd: string, directoryPath: string, @@ -11,12 +16,12 @@ export async function searchFiles( filePattern?: string, rooIgnoreController?: RooIgnoreController, options: SearchFilesOptions = {}, -): Promise { +): Promise { let page: SearchPage if (options.cursor?.engine === "ripgrep") { page = await searchFilesWithRipgrep(cwd, directoryPath, regex, filePattern, rooIgnoreController, options) - return formatSearchPage(page) + return { text: formatSearchPage(page), matchCount: page.matches.length } } try { @@ -31,7 +36,7 @@ export async function searchFiles( page.warning = `FFF failed; used ripgrep fallback (${message})` } - return formatSearchPage(page) + return { text: formatSearchPage(page), matchCount: page.matches.length } } export * from "./format"