diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json new file mode 100644 index 00000000..c97d3acb --- /dev/null +++ b/.claude-plugin/marketplace.json @@ -0,0 +1,14 @@ +{ + "name": "claude-devtools", + "description": "Audit Claude Code sessions for token waste: per-session findings and a cross-session inventory.", + "owner": { + "name": "axisrow", + "url": "https://github.com/axisrow/claude-devtools" + }, + "plugins": [ + { + "name": "session-audit", + "source": "./" + } + ] +} diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json new file mode 100644 index 00000000..96363d12 --- /dev/null +++ b/.claude-plugin/plugin.json @@ -0,0 +1,7 @@ +{ + "name": "session-audit", + "description": "Audit Claude Code session JSONLs for token waste: per-session ledger with findings (pnpm analyze:session) and a cross-session inventory (pnpm analyze:sessions).", + "author": { + "name": "axisrow" + } +} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index fb03825e..c8395d84 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -3,31 +3,11 @@ name: CI on: push: branches: [main] - paths: - - 'src/**' - - 'test/**' - - 'package.json' - - 'pnpm-lock.yaml' - - 'tsconfig*.json' - - 'vite*.config.*' - - 'vitest*.config.*' - - 'tailwind.config.*' - - 'eslint.config.*' pull_request: branches: [main] - paths: - - 'src/**' - - 'test/**' - - 'package.json' - - 'pnpm-lock.yaml' - - 'tsconfig*.json' - - 'vite*.config.*' - - 'vitest*.config.*' - - 'tailwind.config.*' - - 'eslint.config.*' jobs: - validate: + ci: runs-on: ubuntu-latest steps: @@ -44,7 +24,7 @@ jobs: cache: pnpm - name: Install dependencies - run: pnpm install --no-frozen-lockfile + run: pnpm install --frozen-lockfile - name: Typecheck run: pnpm typecheck @@ -52,31 +32,5 @@ jobs: - name: Lint run: pnpm lint - - name: Build - run: pnpm build - - test: - strategy: - fail-fast: false - matrix: - os: [ubuntu-latest, windows-latest] - runs-on: ${{ matrix.os }} - - steps: - - name: Checkout - uses: actions/checkout@v4 - - - name: Setup pnpm - uses: pnpm/action-setup@v4 - - - name: Setup Node.js - uses: actions/setup-node@v4 - with: - node-version: 20 - cache: pnpm - - - name: Install dependencies - run: pnpm install --no-frozen-lockfile - - name: Test run: pnpm test diff --git a/CHANGELOG.md b/CHANGELOG.md index 1365ccfa..033e7ff8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,20 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) For the full list of merged PRs per release, see [GitHub Releases](https://github.com/matt1398/claude-devtools/releases). +## Fork + +Development continues in this fork — see the [Upstream sync policy](README.md#upstream-sync-policy). Upstream entries below are kept unchanged. + +## [0.1.0-fork.1] — 2026-09-21 + +### Added +- Session-audit CLI: `pnpm analyze:session` (deep per-session ledger, waste findings, slow subagents, cost estimate) and `pnpm analyze:sessions` (session inventory), with a unified ccusage-style flag grammar (#1). +- Session-audit CLI packaged as a Claude Code plugin with marketplace manifest and skills (#12). + +### Fixed +- Treat the `Agent` tool as a subagent spawn (renamed from `Task` in Claude Code 2.1.63). +- Short model ids parsed correctly; removed the `priceFamily` stopgap (#13). + ## [Unreleased] ### Added diff --git a/README.md b/README.md index 3f979ea2..c40c827b 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,6 @@ +> [!NOTE] +> **Actively maintained fork** of [matt1398/claude-devtools](https://github.com/matt1398/claude-devtools) (upstream dormant since 2026-05). This fork carries new development: session-audit CLI and fixes. See the [Upstream sync policy](#upstream-sync-policy). +

Your Claude is coding blind

@@ -171,6 +174,15 @@ claude-devtools does **not** wrap, modify, or interfere with Claude Code. It rea --- +## Upstream Sync Policy + +This repository is the actively maintained fork of [matt1398/claude-devtools](https://github.com/matt1398/claude-devtools); upstream has been dormant since 2026-05. Upstream is still fetched periodically, but it is not expected to move. + +- Local development happens in `main` and feature branches of this fork. +- Open upstream PRs (e.g. [#235](https://github.com/matt1398/claude-devtools/pull/235)) are closed as "carried in fork" once their changes are applied locally. + +--- + ## Docker / Standalone Deployment Run without Electron — in Docker, on a remote server, or anywhere Node.js runs. @@ -197,6 +209,53 @@ The standalone server has **zero** outbound network calls. For maximum isolation --- +## CLI: Token Analytics + +Two scripts answer "where did the billed tokens go" with exact numbers from your JSONL sessions — no app, no rebuild: + +```bash +# Deep audit of one session: per-turn/per-round ledger (input / cache_read / +# cache_write / output), waste findings, slow subagents, cost estimate +pnpm analyze:session +pnpm analyze:session --project --last + +# Inventory of all sessions: duration, models, token totals, billing scheme +pnpm analyze:sessions +pnpm analyze:sessions --min-minutes 120 --sort tokens +``` + +Both commands share one flag grammar: + +| Flag | Effect | +|------|--------| +| `--json` | Machine-readable output (the single machine-readable mode) | +| `--breakdown` | Per-model token/cost split (`analyze:session`: BY MODEL table + JSON `breakdown`; `analyze:sessions`: model share in the models column + JSON `tokensByModel` per session) | +| `--since DATE` / `--until DATE` | Date-range filter (`YYYY-MM-DD` or `YYYYMMDD`; activity window for `analyze:session`, session last-activity dates for `analyze:sessions` — a session that ran past `--until` drops out) | +| `--last N` | Relative shortcut for `--since`: last N calendar days, local midnight N−1 days back (ccusage-style). On `analyze:session` a bare `--last` (no value) keeps its original meaning: pick the newest session of `--project` | +| `--no-cost` | Omit cost estimates (hides the cost line, JSON `costUsd`/`costPartial`, breakdown costs) | +| `--project`, `--min-minutes`, `--sort`, `--limit`, `--subagent-min-minutes` | Existing per-command flags, unchanged | + +More examples: + +```bash +# Token cost breakdown by model for the latest session of a project +pnpm analyze:session --project my-project --last --breakdown + +# Audit only yesterday's rounds, without cost estimates +pnpm analyze:session session.jsonl --since 2026-09-20 --until 2026-09-20 --no-cost + +# Sessions active in the last 7 days, newest first +pnpm analyze:sessions --last 7 --sort date --limit 20 +``` + +### Positioning vs ccusage + +[ccusage](https://github.com/ryoppippi/ccusage) computes **macro usage aggregates** — daily/monthly/blocks reports across many agent sources, plus a statusline. claude-devtools does the opposite: a **deep per-session audit** of Claude Code sessions (ledger, waste findings, subagents, inventory). Complementary tools, zero feature mixing: no daily/monthly/blocks aggregates, no multi-agent sources, no statusline, no pricing-sync network logic here — and no per-session waste audit in ccusage. The commands above deliberately reuse ccusage's flag *conventions* (`--json`, `--since`/`--until`, `--last N`, `--breakdown`, `--no-cost`) so the two tools feel consistent side by side. + +`--help` on either command prints the full flag list. + +--- + ## Development
diff --git a/knip.json b/knip.json index 8f8f03a3..a40c6a84 100644 --- a/knip.json +++ b/knip.json @@ -3,6 +3,8 @@ "entry": [ "src/main/index.ts", "src/main/standalone.ts", + "src/cli/analyzeSession.ts", + "src/cli/sessionInventory.ts", "src/preload/index.ts", "src/renderer/main.tsx", "electron.vite.config.ts", diff --git a/package.json b/package.json index a99f83e8..70cc3b2a 100644 --- a/package.json +++ b/package.json @@ -44,6 +44,10 @@ "test:coverage": "vitest run --coverage", "test:coverage:critical": "vitest run --coverage --config vitest.critical.config.ts", "standalone": "tsx src/main/standalone.ts", + "analyze:session": "tsx src/cli/analyzeSession.ts", + "analyze:sessions": "tsx src/cli/sessionInventory.ts", + "loops:import": "tsx src/cli/importLoops.ts", + "turn-spend:stats": "tsx src/cli/turnSpendStats.ts", "standalone:build": "electron-vite build && vite build --config vite.standalone.config.ts", "standalone:start": "node dist-standalone/index.cjs" }, @@ -178,6 +182,8 @@ "publish": [ { "provider": "github", + "owner": "axisrow", + "repo": "claude-devtools", "releaseType": "draft" } ] diff --git a/scripts/turn-budget-hook.mjs b/scripts/turn-budget-hook.mjs new file mode 100644 index 00000000..83f1bfa8 --- /dev/null +++ b/scripts/turn-budget-hook.mjs @@ -0,0 +1,216 @@ +#!/usr/bin/env node +/** + * Turn input-budget limiter — PreToolUse hook for Claude Code. + * + * Counts input-side tokens of the current turn (last real user message -> EOF) + * in the session transcript; denies the next tool call once the budget is + * spent, so the agent wraps up and reports instead of looping on. + * + * Fail-open: any error exits 0 silently — a broken limiter must not break + * sessions, and a spend counted without a found turn boundary allows too. + * Note: Claude Code writes the transcript asynchronously, so the very last + * round may be missing — the deny fires on the NEXT call based on rounds + * already on disk. Acceptable undercount. + */ + +import { + appendFileSync, + openSync, + readSync, + closeSync, + readFileSync, + statSync, + existsSync, +} from 'node:fs'; +import { homedir } from 'node:os'; +import { join } from 'node:path'; + +const CONFIG_PATH = join(homedir(), '.claude', 'claude-devtools-config.json'); +// corpus-calibrated (pnpm turn-spend:stats, 10 080 turns): p95 = 12.56M +const DEFAULT_BUDGET = 15_000_000; +const CHUNK = 1 << 20; // backwards-read window + +/** Simplified teammate-message wrapper detection. */ +function isTeammateText(t) { + return t.startsWith(' 0) { + const len = Math.min(CHUNK, remaining); + remaining -= len; + readSync(fd, buf, 0, len, remaining); + const text = buf.toString('utf8', 0, len) + carry; + const parts = text.split('\n'); + carry = parts.shift() ?? ''; + for (let i = parts.length - 1; i >= 0; i--) { + if (parts[i]) yield parts[i]; + } + } + if (carry) yield carry; +} + +/** Read turnBudget config; defaults when missing. */ +export function readConfig(raw) { + try { + const cfg = JSON.parse(raw ?? '{}'); + // ConfigManager writes it under notifications.turnBudget, not top level + const tb = cfg.notifications?.turnBudget ?? {}; + return { + enabled: tb.enabled !== false, + budget: Number.isInteger(tb.maxInputTokensPerTurn) ? tb.maxInputTokensPerTurn : DEFAULT_BUDGET, + }; + } catch { + return { enabled: true, budget: DEFAULT_BUDGET }; + } +} + +// ============================================================================= +// Entry +// ============================================================================= + +export function main() { + let raw; + try { + raw = readFileSync(0, 'utf8'); + } catch { + return; + } + let hook; + try { + hook = JSON.parse(raw); + } catch { + return; + } + const { enabled, budget } = readConfig(readConfigSafely()); + if (!enabled) return; + + const transcript = hook.transcript_path; + if (typeof transcript !== 'string' || !existsSync(transcript)) return; + + let fd; + let spent = 0; + let boundaryFound = false; + try { + fd = openSync(transcript, 'r'); + const size = statSync(transcript).size; + for (const line of linesBackward(fd, size)) { + const r = analyzeTurn([line]); + spent += r.spent; + if (r.boundaryFound) { + boundaryFound = true; + break; + } + } + } catch { + // fail-open + } finally { + if (fd !== undefined) closeSync(fd); + } + + if (!boundaryFound) { + // a spend counted without a turn boundary is not trustworthy (that was + // the 872M incident) — allow, but leave an anomaly line in the log + logDecision(hook.session_id, spent, budget, false, 'no-boundary'); + return; + } + if (spent >= budget) { + logDecision(hook.session_id, spent, budget, true, 'deny'); + deny(spent, budget); + } +} + +const LOG_PATH = join(homedir(), '.claude', 'claude-devtools-turnbudget.log'); + +/** Append deny/anomaly decisions only — routine allows stay silent. */ +function logDecision(sessionId, spent, budget, denied, kind) { + try { + appendFileSync( + LOG_PATH, + `${new Date().toISOString()} kind=${kind} session=${sessionId ?? '?'} spent=${spent} budget=${budget} denied=${denied}\n` + ); + } catch { + // logging must never break the hook + } +} + +function readConfigSafely() { + try { + return readFileSync(CONFIG_PATH, 'utf8'); + } catch { + return ''; + } +} + +export function deny(spent, budget) { + process.stdout.write( + JSON.stringify({ + hookSpecificOutput: { + hookEventName: 'PreToolUse', + permissionDecision: 'deny', + permissionDecisionReason: `Turn input budget exhausted: ~${Math.round(spent / 1e6)}M / ${Math.round(budget / 1e6)}M tokens re-read this turn. Finish the turn now: report results and stop.`, + }, + }) + ); +} + +if (process.argv[1] && process.argv[1].endsWith('turn-budget-hook.mjs')) { + try { + main(); + } catch { + // fail-open: a broken limiter must never block a session + } +} \ No newline at end of file diff --git a/skills/audit-session/SKILL.md b/skills/audit-session/SKILL.md new file mode 100644 index 00000000..102d0004 --- /dev/null +++ b/skills/audit-session/SKILL.md @@ -0,0 +1,38 @@ +--- +name: audit-session +description: Audit one Claude Code session JSONL for token waste — per-turn context growth, cache reread share, duplicate/failed/oversized tool calls, context spikes, dead prompt cache, thinking-heavy turns, slow subagents. Use when the user asks to audit a session, find wasted tokens or cost, or asks where a session's tokens went. +allowed-tools: Bash(pnpm analyze:session:*) +--- + +# Session audit + +Runs the token-audit CLI from a claude-devtools checkout (plugin root = repo root). If `pnpm analyze:session` fails with "no such script", cd to this plugin's checkout first and run `pnpm install` once. Those two steps are outside the `allowed-tools` prefix, so a permission prompt there is expected — approve it, it is not a failure. + +## Pick the session + +- Newest session of a project: `pnpm analyze:session --project --last` — `` is the plain filesystem path of the project (e.g. `~/Projects/foo`) or its encoded dir name (`-Users-name-Projects-foo`). `--project` always requires `--last`. +- Explicit file: `pnpm analyze:session `. +- Narrow the window: `--since/--until YYYY-MM-DD` or `--last N` (last N calendar days). + +Other flags: `--breakdown` (per-model table), `--min-severity medium|high`, `--subagent-min-minutes N` (default 5), `--rounds N` (rounds table length, default 20), `--no-cost`, `--json`. + +## Read the ledger + +- TOKENS: `cache_read (reread)` is the whole context re-billed every round; `reread share` near 100% is normal for long sessions — the lever is shorter turns and subagents, not the cache. +- BY TURN: growing `context` column = parent context getting fat; jumps usually trace to the tools listed on that row. +- ROUNDS: `delta` is per-round context change; `+30k` in one round means one big tool result landed in context. +- `est. cost (partial …)` = some rounds are unpriced models; costs are estimates from a built-in price table, `--no-cost` for pure token counts. + +## Findings and what to recommend + +| finding | meaning | recommendation | +| --- | --- | --- | +| duplicate_call | same tool+input called repeatedly, results re-read each time | add a deny rule for the culprit command in Claude Code settings; state the result in the prompt instead of re-reading | +| failed_call | tool errored (normal user rejections excluded) | fix the invocation; repeated failures of one command → deny rule | +| oversized_output | one tool result over ~8k tok | narrower flags on the command; write to a file and Read slices | +| context_spike | context jumped +30k in one round | move that work into a subagent so the parent only sees the summary | +| cache_dead | whole context billed as fresh input (no cache reads/writes) | check the provider/router setup — first-party Anthropic API should never look like this | +| thinking_heavy | >8k estimated thinking tok in one turn | split the long turn, scope the task tighter | +| !SLOW (SUBAGENTS table) | subagent ran past the threshold | split its prompt; pass less context in | + +Finish with the numbers: total billed tokens, reread share, top findings by wasted tokens, and a concrete fix for each. The CLI observes, it does not prevent — prevention is deny rules in Claude Code settings. diff --git a/skills/sessions-inventory/SKILL.md b/skills/sessions-inventory/SKILL.md new file mode 100644 index 00000000..edc4bf23 --- /dev/null +++ b/skills/sessions-inventory/SKILL.md @@ -0,0 +1,25 @@ +--- +name: sessions-inventory +description: List Claude Code sessions across all projects with duration, models, token totals, and billing scheme (first-party Anthropic vs router vs no cache). Use when the user asks which sessions exist, which ran longest or used the most tokens, or wants an overview table of sessions. +allowed-tools: Bash(pnpm analyze:sessions:*) +--- + +# Sessions inventory + +Runs the inventory CLI from a claude-devtools checkout (plugin root = repo root). If `pnpm analyze:sessions` fails with "no such script", cd to this plugin's checkout first and run `pnpm install` once. Those two steps are outside the `allowed-tools` prefix, so a permission prompt there is expected — approve it, it is not a failure. + +## Filters + +- `--project ` — one project (plain path or encoded dir name); omit for all projects +- `--min-minutes N` — only sessions running at least N minutes +- `--sort duration|tokens|date` (default duration) and `--limit N` +- `--since/--until YYYY-MM-DD` or `--last N` — by last-activity date +- `--breakdown` — per-session model token split (models column shows shares) +- `--json` — machine-readable (`tokensByModel` only appears with `--breakdown`) +- `--no-cost` accepted for grammar parity; the inventory has no cost figures + +## Presenting + +Render the table as printed: duration, date, file (first 8 chars of the session id), project, models, tokens, msgs, billing. Sort by duration unless the user asks otherwise; reach for `--min-minutes` to cut the noise instead of filtering the table by hand. + +Billing scheme column: `anthropic-style` = cache writes seen (first-party Anthropic API), `router-style` = cache reads only, no writes (proxy/router), `mixed` = both, `no-cache` = neither. Point it out when the user wonders why some sessions cost differently than others. diff --git a/src/cli/analyzeSession.ts b/src/cli/analyzeSession.ts new file mode 100644 index 00000000..f57b57ec --- /dev/null +++ b/src/cli/analyzeSession.ts @@ -0,0 +1,1143 @@ +/** + * Session token audit CLI — where do billed tokens go, and what was wasted. + * + * Numbers only: exact per-turn/per-round usage from JSONL (input / cache_read / + * cache_write / output), waste findings, slow-subagent table. No transcript. + * + * Usage: + * pnpm analyze:session [flags] + * pnpm analyze:session --project --last [flags] + * Flags: + * --rounds N rounds table length (default 20) + * --subagent-min-minutes N slow-subagent threshold (default 5) + * --min-severity S low | medium | high (default low = all) + * --breakdown per-model token/cost breakdown + * --since / --until DATE only activity within the range (YYYY-MM-DD or YYYYMMDD) + * --last [N] no value: newest session of --project; N: last N calendar days + * --no-cost omit cost estimates + * --json machine-readable output + */ + +import { ProjectScanner, SubagentResolver } from '@main/services/discovery'; +import { isParsedUserChunkMessage } from '@main/types'; +import { deduplicateByRequestId, getTaskCalls, parseJsonlFile } from '@main/utils/jsonl'; +import { encodePath, extractSessionId, getProjectsBasePath } from '@main/utils/pathDecoder'; +import { isQuietTick, isStalledRound } from '@shared/constants/loopPolicy'; +import { asText, normalizeCallKey } from '@shared/utils/callKey'; +import { parseModelString } from '@shared/utils/modelParser'; +import { + estimateTokens, + formatTokensCompact, + formatTokensDetailed, +} from '@shared/utils/tokenFormatting'; +import * as fs from 'fs'; +import * as path from 'path'; +import { pathToFileURL } from 'url'; + +import { inDateRange, lastDaysSince, parseDayBound, takeFlagValue, wantsHelp } from './args'; + +import type { ParsedMessage, Process } from '@main/types'; + +// ponytail: single-file pricing table; extract to a module when something else needs it +// glm: public rates (OpenRouter/z.ai, verified 2026-09-21); cache write assumed = input rate +const PRICE_PER_MTOK: Record = { + opus: [15, 75, 1.5, 18.75], + sonnet: [3, 15, 0.3, 3.75], + haiku: [1, 5, 0.1, 1.25], + glm: [0.075, 0.25, 0.015, 0.075], +}; + +// pricing family: claude via parser; other models — first id segment (glm-5.3-flash → glm) +function priceFamily(model: string): string { + return parseModelString(model)?.family ?? model.toLowerCase().split('-')[0]; +} + +export type BillingScheme = 'anthropic-style' | 'router-style' | 'no-cache' | 'mixed'; + +// anthropic-style billing has cache_write > 0 on rounds; routers typically report +// cache_read without any write counter — cw=0 with cr>0 is the router signature +export function detectBillingScheme( + rounds: { cacheReadTokens: number; cacheCreationTokens: number }[] +): BillingScheme { + let sawWrite = false; + let sawRead = false; + for (const r of rounds) { + if (r.cacheCreationTokens > 0) sawWrite = true; + else if (r.cacheReadTokens > 0) sawRead = true; + } + return billingFromFlags(sawWrite, sawRead); +} + +export function billingFromFlags(sawWrite: boolean, sawRead: boolean): BillingScheme { + if (sawWrite && sawRead) return 'mixed'; + if (sawWrite) return 'anthropic-style'; + if (sawRead) return 'router-style'; + return 'no-cache'; +} + +export const WASTE_THRESHOLDS = { + oversizedOutputTokens: 8000, + contextSpikeTokens: 30000, + cacheDeadContextTokens: 20000, + thinkingHeavyTokens: 8000, + longTurnActiveMinutes: 45, + loopStreakMin: 3, + waitLoopTicks: 5, +} as const; + +// gaps between a turn's rounds longer than this are idle, not work +// ponytail: calibration knob — tune after live runs +export const TURN_IDLE_GAP_CAP_MINUTES = 10; + +// wait-loop tick thresholds live in @shared/constants/loopPolicy — the +// renderer's Visible Context wait-loop category uses the same numbers +export { isQuietTick }; + +// Tool results that look like errors but are normal flow (user said no / aborted) +const REJECTION_PATTERNS = [ + "The user doesn't want to proceed with this tool use", + '[Request interrupted by user', +]; + +export type FindingType = + | 'duplicate_call' + | 'failed_call' + | 'oversized_output' + | 'context_spike' + | 'cache_dead' + | 'thinking_heavy' + | 'long_turn' + | 'loop_streak' + | 'stall_streak' + | 'wait_loop'; + +export interface Finding { + type: FindingType; + severity: 'low' | 'medium' | 'high'; + tokensWasted: number; + turnIndex?: number; + summary: string; +} + +// run of back-to-back identical calls being tracked for loop_streak +interface StreakState { + count: number; + tokens: number; + errors: number; + turn?: number; + start: Date; + end: Date; +} + +export interface RoundRow { + index: number; + timestamp: Date; + turnIndex: number; + model: string; + inputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; + outputTokens: number; + contextSize: number; + contextDelta: number; + thinkingTokens: number; + tools: string[]; + /** router-retry copy of the previous round — counted in sums, excluded from findings (#15) */ + isRetryCopy?: boolean; +} + +export interface TurnRow { + index: number; + start: Date; + /** active work time: inter-round gaps capped at TURN_IDLE_GAP_CAP_MINUTES */ + activeMinutes: number; +} + +export interface SessionLedger { + turns: TurnRow[]; + rounds: RoundRow[]; + totals: { + inputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; + outputTokens: number; + billedTokens: number; + rereadShare: number; + thinkingTokens: number; + noUsageRounds: number; + retryCopies: number; + longestTurn?: { turn: number; activeMinutes: number; rounds: number }; + costUsd?: number; + costPartial?: boolean; + }; + models: string[]; + durationMs: number; + /** sum of turn activeMinutes — API work time, idle excluded (the honest "how long did it run") */ + activeMinutes: number; + billing: BillingScheme; +} + +// ============================================================================= +// Ledger (port of the validated prototype) +// ============================================================================= + +function thinkingTokensOf(msg: ParsedMessage): number { + if (!Array.isArray(msg.content)) return 0; + let chars = 0; + for (const block of msg.content) { + if (block.type === 'thinking') chars += (block.thinking ?? '').length; + } + return Math.ceil(chars / 4); +} + +// cost of one round at the built-in price table; null = unpriced model +export function roundCostUsd(r: RoundRow): number | null { + const price = PRICE_PER_MTOK[priceFamily(r.model)] ?? null; + if (!price) return null; + return ( + (r.inputTokens * price[0] + + r.outputTokens * price[1] + + r.cacheReadTokens * price[2] + + r.cacheCreationTokens * price[3]) / + 1e6 + ); +} + +// active work time of a turn: gaps between consecutive rounds, each capped — +// hours of orchestrator silence between pings count as zero, not as a "turn" +export function turnActiveMinutes(rounds: RoundRow[]): number { + let ms = 0; + for (let i = 1; i < rounds.length; i++) { + const gap = rounds[i].timestamp.getTime() - rounds[i - 1].timestamp.getTime(); + ms += Math.min(Math.max(gap, 0), TURN_IDLE_GAP_CAP_MINUTES * 60000); + } + return Math.round(ms / 60000); +} + +// one grouping of rounds by turn, shared by totals, findings and the report +function roundsByTurn(rounds: RoundRow[]): Map { + const byTurn = new Map(); + for (const r of rounds) { + const list = byTurn.get(r.turnIndex); + if (list) list.push(r); + else byTurn.set(r.turnIndex, [r]); + } + return byTurn; +} + +const countTools = (rs: RoundRow[]): Map => { + const counts = new Map(); + for (const r of rs) { + for (const name of r.tools) counts.set(name, (counts.get(name) ?? 0) + 1); + } + return counts; +}; + +export function totalsFromRounds(rounds: RoundRow[]): SessionLedger['totals'] { + const t = { + inputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + outputTokens: 0, + billedTokens: 0, + rereadShare: 0, + thinkingTokens: 0, + }; + let costUsd = 0; + let unpriced = false; + for (const r of rounds) { + t.inputTokens += r.inputTokens; + t.cacheReadTokens += r.cacheReadTokens; + t.cacheCreationTokens += r.cacheCreationTokens; + t.outputTokens += r.outputTokens; + t.thinkingTokens += r.thinkingTokens; + const c = roundCostUsd(r); + if (c === null) unpriced = true; + else costUsd += c; + } + t.billedTokens = t.inputTokens + t.cacheReadTokens + t.cacheCreationTokens + t.outputTokens; + t.rereadShare = t.billedTokens > 0 ? t.cacheReadTokens / t.billedTokens : 0; + const noUsageRounds = rounds.filter((r) => r.contextSize === 0).length; + const retryCopies = rounds.filter((r) => r.isRetryCopy === true).length; + let longestTurn: SessionLedger['totals']['longestTurn']; + for (const [turn, rs] of roundsByTurn(rounds)) { + const active = turnActiveMinutes(rs); + if (!longestTurn || active > longestTurn.activeMinutes) { + longestTurn = { turn, activeMinutes: active, rounds: rs.length }; + } + } + return { + ...t, + noUsageRounds, + retryCopies, + ...(longestTurn ? { longestTurn } : {}), + ...(costUsd > 0 ? { costUsd, costPartial: unpriced } : {}), + }; +} + +export interface ModelBreakdownRow { + model: string; + inputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; + outputTokens: number; + billedTokens: number; + costUsd?: number; +} + +// --breakdown: per-model token/cost totals; costUsd dropped when any round of +// that model is unpriced (no partial per-model figures). withCost=false is the +// --no-cost mode: no cost figures at all. +export function breakdownFromRounds(rounds: RoundRow[], withCost = true): ModelBreakdownRow[] { + const byModel = new Map(); + for (const r of rounds) { + let row = byModel.get(r.model); + if (!row) { + row = { + model: r.model, + inputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + outputTokens: 0, + billedTokens: 0, + }; + byModel.set(r.model, row); + } + row.inputTokens += r.inputTokens; + row.cacheReadTokens += r.cacheReadTokens; + row.cacheCreationTokens += r.cacheCreationTokens; + row.outputTokens += r.outputTokens; + row.billedTokens += r.inputTokens + r.cacheReadTokens + r.cacheCreationTokens + r.outputTokens; + if (withCost) { + const c = roundCostUsd(r); + if (c === null) delete row.costUsd; + else row.costUsd = (row.costUsd ?? 0) + c; + } + } + return [...byModel.values()].sort((a, b) => b.billedTokens - a.billedTokens); +} + +// --since/--until: keep rounds inside the window, recompute everything derived +export function filterLedgerByDate( + ledger: SessionLedger, + since?: Date, + until?: Date +): SessionLedger { + if (!since && !until) return ledger; + const rounds = ledger.rounds.filter((r) => inDateRange(r.timestamp, since, until)); + const keptTurns = new Set(rounds.map((r) => r.turnIndex)); + let minTs = Number.POSITIVE_INFINITY; + let maxTs = Number.NEGATIVE_INFINITY; + for (const r of rounds) { + minTs = Math.min(minTs, r.timestamp.getTime()); + maxTs = Math.max(maxTs, r.timestamp.getTime()); + } + // turns are copies: activeMinutes must reflect the filtered window, not the + // whole session + const byTurn = roundsByTurn(rounds); + const turns = ledger.turns + .filter((t) => keptTurns.has(t.index)) + .map((t) => ({ ...t, activeMinutes: turnActiveMinutes(byTurn.get(t.index) ?? []) })); + return { + turns, + rounds, + totals: totalsFromRounds(rounds), + models: [...new Set(rounds.map((r) => r.model))], + durationMs: Number.isFinite(minTs) ? Math.max(0, maxTs - minTs) : 0, + activeMinutes: turns.reduce((s, t) => s + t.activeMinutes, 0), + billing: detectBillingScheme(rounds), + }; +} + +// router retries re-log the same response without a requestId (#15): identical +// counters, seconds apart. Comparison anchors on the last real round — ghosts +// (#14) and copies themselves never anchor, so ghost runs don't false-match and +// a ghost between original and copy doesn't hide the copy. Used by both the +// ledger (round flags) and findings (skip the copies' tool calls). +export function getRetryCopyMessageIds(allMessages: ParsedMessage[]): Set { + const copies = new Set(); + let anchor: { + model: string; + input: number; + cr: number; + cw: number; + out: number; + ts: number; + } | null = null; + for (const msg of deduplicateByRequestId(allMessages)) { + if (isParsedUserChunkMessage(msg)) continue; + if (msg.type !== 'assistant' || msg.isSidechain) continue; + if (!msg.usage || msg.model === '') continue; + const u = msg.usage; + const input = u.input_tokens ?? 0; + const cacheRead = u.cache_read_input_tokens ?? 0; + const cacheWrite = u.cache_creation_input_tokens ?? 0; + const output = u.output_tokens ?? 0; + const isCopy = + anchor !== null && + !msg.requestId && + anchor.model === (msg.model ?? 'unknown') && + anchor.input === input && + anchor.cr === cacheRead && + anchor.cw === cacheWrite && + anchor.out === output && + msg.timestamp.getTime() - anchor.ts <= 120_000; + if (isCopy) copies.add(msg.uuid); + if (input + cacheRead + cacheWrite > 0 && !isCopy) { + anchor = { + model: msg.model ?? 'unknown', + input, + cr: cacheRead, + cw: cacheWrite, + out: output, + ts: msg.timestamp.getTime(), + }; + } + } + return copies; +} + +export function buildLedger(allMessages: ParsedMessage[]): SessionLedger { + const messages = deduplicateByRequestId(allMessages); + const retryCopies = getRetryCopyMessageIds(messages); + + const turns: TurnRow[] = []; + const rounds: RoundRow[] = []; + const models = new Set(); + let currentTurn: TurnRow | null = null; + let prevContext = 0; + let minTs = Number.POSITIVE_INFINITY; + let maxTs = Number.NEGATIVE_INFINITY; + + const newTurn = (ts: Date): TurnRow => { + const turn: TurnRow = { index: turns.length + 1, start: ts, activeMinutes: 0 }; + turns.push(turn); + currentTurn = turn; + return turn; + }; + + for (const msg of messages) { + minTs = Math.min(minTs, msg.timestamp.getTime()); + maxTs = Math.max(maxTs, msg.timestamp.getTime()); + + if (isParsedUserChunkMessage(msg)) { + newTurn(msg.timestamp); + continue; + } + if (msg.type !== 'assistant' || msg.isSidechain) continue; + if (!msg.usage || msg.model === '') continue; + + const turn = currentTurn ?? newTurn(msg.timestamp); + const u = msg.usage; + const input = u.input_tokens ?? 0; + const cacheRead = u.cache_read_input_tokens ?? 0; + const cacheWrite = u.cache_creation_input_tokens ?? 0; + const output = u.output_tokens ?? 0; + const contextSize = input + cacheRead + cacheWrite; + const think = thinkingTokensOf(msg); + const model = msg.model ?? 'unknown'; + const tools = msg.toolCalls.map((tc) => tc.name); + + const isRetryCopy = retryCopies.has(msg.uuid); + + const round: RoundRow = { + index: rounds.length + 1, + timestamp: msg.timestamp, + turnIndex: turn.index, + model, + inputTokens: input, + cacheReadTokens: cacheRead, + cacheCreationTokens: cacheWrite, + outputTokens: output, + contextSize, + // empty rounds (provider ghosts, #14) must not drag the baseline to 0 + contextDelta: rounds.length === 0 || contextSize === 0 ? 0 : contextSize - prevContext, + thinkingTokens: think, + tools, + isRetryCopy: isRetryCopy || undefined, + }; + if (contextSize > 0) prevContext = contextSize; + rounds.push(round); + models.add(model); + } + + const byTurn = roundsByTurn(rounds); + for (const turn of turns) { + turn.activeMinutes = turnActiveMinutes(byTurn.get(turn.index) ?? []); + } + + return { + turns, + rounds, + totals: totalsFromRounds(rounds), + models: [...models], + durationMs: Number.isFinite(minTs) ? Math.max(0, maxTs - minTs) : 0, + activeMinutes: turns.reduce((s, t) => s + t.activeMinutes, 0), + billing: detectBillingScheme(rounds), + }; +} + +// ============================================================================= +// Findings +// ============================================================================= + +export { bashStem, normalizeCallKey } from '@shared/utils/callKey'; + +function resultText(content: string | unknown[]): string { + return typeof content === 'string' ? content : JSON.stringify(content); +} + +// since/until scope the tool-call walk (duplicate/failed/oversized) to the same +// window as the ledger; the results map is still built from ALL messages, so a +// call inside the window resolves a result that landed after --until +export function computeFindings( + messages: ParsedMessage[], + ledger: SessionLedger, + since?: Date, + until?: Date +): Finding[] { + const findings: Finding[] = []; + const th = WASTE_THRESHOLDS; + + const results = new Map(); + for (const msg of messages) { + if (msg.isSidechain) continue; + for (const r of msg.toolResults) { + results.set(r.toolUseId, { content: r.content, isError: r.isError }); + } + } + + // duplicate / failed / oversized — walk tool calls in order + // re-logged router copies (#15) are skipped: their calls were already walked + // with the original — counting them doubles duplicate/failed/oversized + const retryCopies = getRetryCopyMessageIds(messages); + const seen = new Map(); + + // loop_streak: the same call repeated back-to-back — model no-op loops + // (Bash true x114) and env retry loops (same failure hammered). Turn + // attribution walks the ledger's turn starts (sorted by construction). + let turnCursor = 0; + const turnOf = (ts: number): number | undefined => { + const turns = ledger.turns; + if (turns.length === 0 || ts < turns[0].start.getTime()) return undefined; + while (turnCursor + 1 < turns.length && turns[turnCursor + 1].start.getTime() <= ts) { + turnCursor += 1; + } + return turns[turnCursor].index; + }; + let streakKey: string | null = null; + let streak: StreakState | null = null; + const flushStreak = (): void => { + if (streak && streakKey && streak.count >= th.loopStreakMin) { + const env = streak.errors === streak.count; + findings.push({ + type: 'loop_streak', + severity: streak.count >= 5 ? 'high' : 'medium', + tokensWasted: streak.tokens, + turnIndex: streak.turn, + summary: `${short(streakKey, 60)} — x${streak.count} back-to-back (${env ? 'env loop — same failure each time' : 'no-op loop'}) ${hhmm(streak.start)}–${hhmm(streak.end)}`, + }); + } + streak = null; + streakKey = null; + }; + + for (const msg of messages) { + if (msg.isSidechain) continue; + if (retryCopies.has(msg.uuid)) continue; + if (!inDateRange(msg.timestamp, since, until)) continue; + for (const call of msg.toolCalls) { + const key = normalizeCallKey(call.name, call.input); + const result = results.get(call.id); + const text = result ? resultText(result.content) : ''; + const resultTok = estimateTokens(text); + + if (streak && streakKey === key) { + streak.count += 1; + streak.tokens += resultTok; + if (result?.isError) streak.errors += 1; + streak.end = msg.timestamp; + } else { + flushStreak(); + streakKey = key; + streak = { + count: 1, + tokens: 0, + errors: result?.isError ? 1 : 0, + turn: turnOf(msg.timestamp.getTime()), + start: msg.timestamp, + end: msg.timestamp, + }; + } + + if (result?.isError) { + const rejected = REJECTION_PATTERNS.some((p) => text.includes(p)); + if (!rejected) { + const inputTok = estimateTokens(JSON.stringify(call.input)); + findings.push({ + type: 'failed_call', + severity: 'medium', + tokensWasted: inputTok + resultTok, + summary: `${call.name} failed: ${short(text, 80)}`, + }); + } + } + if (resultTok > th.oversizedOutputTokens) { + findings.push({ + type: 'oversized_output', + severity: 'medium', + tokensWasted: resultTok, + summary: `${call.name} returned ~${formatTokensCompact(resultTok)} tok: ${short(call.name === 'Bash' ? asText(call.input.command) : JSON.stringify(call.input), 70)}`, + }); + } + + const prev = seen.get(key); + if (prev) { + prev.count += 1; + prev.tokens += resultTok; + } else { + seen.set(key, { count: 1, tokens: resultTok, first: resultTok }); + } + } + } + flushStreak(); + for (const [key, { count, tokens, first }] of seen) { + if (count > 1) { + const reread = tokens - first; // repeats only — the first read was legitimate + findings.push({ + type: 'duplicate_call', + severity: 'medium', + tokensWasted: reread, + summary: `${short(key, 90)} — called ${count}x (~${formatTokensCompact(reread)} tok of results re-read)`, + }); + } + } + + // context-side findings from the ledger + for (const r of ledger.rounds) { + if (r.isRetryCopy) continue; // findings already reported for the original round + if (r.contextDelta > th.contextSpikeTokens) { + findings.push({ + type: 'context_spike', + severity: 'high', + tokensWasted: r.contextDelta, + turnIndex: r.turnIndex, + summary: `context +${formatTokensCompact(r.contextDelta)} in one round (${formatTokensCompact(r.contextSize - r.contextDelta)} → ${formatTokensCompact(r.contextSize)}), tools: ${r.tools.join(', ') || 'none'}`, + }); + } + if ( + r.contextSize > th.cacheDeadContextTokens && + r.cacheReadTokens === 0 && + r.cacheCreationTokens === 0 + ) { + findings.push({ + type: 'cache_dead', + severity: 'high', + tokensWasted: r.contextSize, + turnIndex: r.turnIndex, + summary: `no prompt caching: ${formatTokensCompact(r.contextSize)} tok billed as fresh input (model ${r.model})`, + }); + } + } + const byTurn = roundsByTurn(ledger.rounds); + for (const turn of ledger.turns) { + const rs = byTurn.get(turn.index) ?? []; + const think = rs.reduce((s, r) => s + r.thinkingTokens, 0); + if (think > th.thinkingHeavyTokens) { + findings.push({ + type: 'thinking_heavy', + severity: 'low', + tokensWasted: think, + turnIndex: turn.index, + summary: `thinking ~${formatTokensCompact(think)} tok in turn ${turn.index}`, + }); + } + if (turn.activeMinutes >= th.longTurnActiveMinutes) { + const calls = rs.reduce((s, r) => s + r.tools.length, 0); + const top = [...countTools(rs)] + .sort((a, b) => b[1] - a[1]) + .slice(0, 5) + .map(([n, c]) => `${n} ${c}`) + .join(', '); + findings.push({ + type: 'long_turn', + severity: 'high', + // observation, not waste — unlike other findings this books no redundant + // tokens (the summary carries the activity numbers), so consumers summing + // tokensWasted don't count a healthy turn's whole billing as waste + tokensWasted: 0, + turnIndex: turn.index, + summary: `active ${turn.activeMinutes} min, ${calls} tool calls (${top || 'no tools'})`, + }); + } + // A tick is an IDLE round (no tool call) — see isQuietTick in loopPolicy: + // with a large baseline context, ordinary working rounds (short tool + // calls) would otherwise satisfy the context/output thresholds too + const ticks = rs.filter( + (r) => !r.isRetryCopy && isQuietTick(r.contextSize, r.outputTokens, r.tools.length) + ); + if (ticks.length >= th.waitLoopTicks) { + const wasted = ticks.reduce((s, r) => s + r.contextSize, 0); + findings.push({ + type: 'wait_loop', + severity: ticks.length >= 20 ? 'high' : 'medium', + tokensWasted: wasted, + turnIndex: turn.index, + summary: `wait-loop: ${ticks.length} quiet rounds re-read ~${formatTokensCompact(wasted)} tok (≤300 tok of output each)`, + }); + } + + // A stall is the tool-call counterpart of a quiet tick: rounds that MAKE + // calls yet stop growing the context (echo-marker loops like `echo w/v/u` + // — distinct args, so the repeat-key walk above sees no streak). See + // isStalledRound in loopPolicy. + let stallStart: RoundRow | null = null; + let stallEnd: RoundRow | null = null; + let stallCount = 0; + let stallWasted = 0; + const flushStall = (): void => { + if (stallCount >= th.loopStreakMin && stallStart && stallEnd) { + findings.push({ + type: 'stall_streak', + severity: stallCount >= 5 ? 'high' : 'medium', + tokensWasted: stallWasted, + turnIndex: turn.index, + summary: `stall: ${stallCount} rounds with no context growth re-read ~${formatTokensCompact(stallWasted)} tok (${short(stallStart.tools.join(', ') || 'tool calls', 40)}) ${hhmm(stallStart.timestamp)}–${hhmm(stallEnd.timestamp)}`, + }); + } + stallStart = null; + stallEnd = null; + stallCount = 0; + stallWasted = 0; + }; + let stallPrev = 0; + for (const r of rs) { + if (r.isRetryCopy) continue; // copies already walked with their original + if (isStalledRound(stallPrev, r.contextSize, r.outputTokens, r.tools.length)) { + if (stallCount === 0) stallStart = r; + stallCount += 1; + stallWasted += r.contextSize; + stallEnd = r; + } else { + flushStall(); + } + // ghost rounds must not drag the baseline (same rule as buildLedger) + if (r.contextSize > 0) stallPrev = r.contextSize; + } + flushStall(); + } + + const order = { high: 0, medium: 1, low: 2 } as const; + findings.sort((a, b) => order[b.severity] - order[a.severity] || b.tokensWasted - a.tokensWasted); + return findings; +} + +export function short(s: string, n: number): string { + const flat = s.replace(/\s+/g, ' ').trim(); + return flat.length <= n ? flat : `${flat.slice(0, n - 1)}…`; +} + +// ============================================================================= +// CLI plumbing +// ============================================================================= + +interface CliOpts { + sessionPath?: string; + projectArg?: string; + useLast: boolean; + lastDays?: number; + rounds: number; + subagentMinMinutes: number; + minSeverity: 'low' | 'medium' | 'high'; + json: boolean; + breakdown: boolean; + noCost: boolean; + since?: Date; + until?: Date; + error?: string; +} + +export function parseArgs(argv: string[]): CliOpts { + const opts: CliOpts = { + useLast: false, + rounds: 20, + subagentMinMinutes: 5, + minSeverity: 'low', + json: false, + breakdown: false, + noCost: false, + }; + let i = 0; + while (i < argv.length) { + const a = argv[i]; + const { value, next } = takeFlagValue(argv, i); + if (a === '--project') { + opts.projectArg = value; + i = next; + continue; + } + if (a === '--rounds') { + opts.rounds = parseInt(value, 10) || 20; + i = next; + continue; + } + if (a === '--subagent-min-minutes') { + opts.subagentMinMinutes = parseInt(value, 10) || 5; + i = next; + continue; + } + if (a === '--min-severity') { + opts.minSeverity = value === 'medium' || value === 'high' ? value : 'low'; + i = next; + continue; + } + if (a === '--since' || a === '--until') { + const d = parseDayBound(value, a === '--until'); + if (!d) opts.error = `invalid ${a} date (expected YYYY-MM-DD or YYYYMMDD): '${value}'`; + else if (a === '--since') opts.since = d; + else opts.until = d; + i = next; + continue; + } + // numeric --last N = ccusage-style day window; bare --last keeps its older + // meaning here: pick the newest session file of --project + if (a === '--last') { + if (/^[1-9]\d*$/.test(value)) { + opts.lastDays = parseInt(value, 10); + opts.since = lastDaysSince(opts.lastDays); + i = next; + } else { + // bare --last, or a token that is not --last's value (e.g. a positional + // path) — do not swallow it + opts.useLast = true; + i += 1; + } + continue; + } + if (a === '--breakdown') { + opts.breakdown = true; + i += 1; + continue; + } + if (a === '--no-cost') { + opts.noCost = true; + i += 1; + continue; + } + if (a === '--json') { + opts.json = true; + i += 1; + continue; + } + if (!a.startsWith('--')) opts.sessionPath = a; + i += 1; + } + return opts; +} + +export function pickNewestSessionFile(dir: string): string | null { + if (!fs.existsSync(dir)) return null; + let newest: { file: string; mtime: number } | null = null; + for (const e of fs.readdirSync(dir, { withFileTypes: true })) { + if (!e.isFile() || !e.name.endsWith('.jsonl') || e.name.startsWith('agent-')) continue; + const file = path.join(dir, e.name); + let mtime: number; + try { + mtime = fs.statSync(file).mtimeMs; + } catch { + continue; // vanished between readdir and stat + } + if (!newest || mtime > newest.mtime) newest = { file, mtime }; + } + return newest?.file ?? null; +} + +export function resolveProjectDir(arg: string): string { + const projectsRoot = getProjectsBasePath(); + // already encoded (full path or bare name) → use under the projects root + if (path.basename(arg).startsWith('-')) { + return path.isAbsolute(arg) ? arg : path.join(projectsRoot, arg); + } + // plain filesystem path of a project → its encoded sessions dir + return path.join(projectsRoot, encodePath(path.resolve(arg))); +} + +function splitSessionPath(file: string): { projectId: string; sessionId: string } { + const rel = path.relative(getProjectsBasePath(), path.resolve(file)); + const [projectId, sessionId] = rel.split(path.sep); + return { projectId, sessionId: sessionId ? extractSessionId(sessionId) : '' }; +} + +async function resolveSubagentsFor( + projectId: string, + sessionId: string, + messages: ParsedMessage[] +): Promise { + const scanner = new ProjectScanner(); + const resolver = new SubagentResolver(scanner); + return resolver.resolveSubagents(projectId, sessionId, getTaskCalls(messages), messages); +} + +// ============================================================================= +// Output +// ============================================================================= + +export const pad = (s: string, n: number): string => + s.length >= n ? s : s + ' '.repeat(n - s.length); +export const padL = (s: string, n: number): string => + s.length >= n ? s : ' '.repeat(n - s.length) + s; +const fmt = (n: number): string => formatTokensDetailed(n); +const hhmm = (d: Date): string => + `${String(d.getHours()).padStart(2, '0')}:${String(d.getMinutes()).padStart(2, '0')}`; +export const dur = (ms: number): string => + `${Math.floor(ms / 3600000)}h ${String(Math.floor((ms % 3600000) / 60000)).padStart(2, '0')}m`; + +function printReport( + file: string, + ledger: SessionLedger, + findings: Finding[], + subagents: Process[], + opts: CliOpts +): void { + const t = ledger.totals; + const activeTurns = new Set(ledger.rounds.map((r) => r.turnIndex)).size; + console.log('=== SESSION ==='); + console.log('file :', path.basename(file)); + console.log( + `turns: ${activeTurns} rounds: ${ledger.rounds.length} active: ${dur(ledger.activeMinutes * 60000)} (wall ${dur(ledger.durationMs)})` + ); + console.log('models:', ledger.models.join(', ') || 'n/a'); + console.log('billing:', ledger.billing); + if (t.noUsageRounds > 0) { + console.log( + `⚠ ${t.noUsageRounds} rounds have no usage stats — root cause under investigation (#14)` + ); + } + if (t.retryCopies > 0) { + console.log(`ℹ ${t.retryCopies} router-retry copies detected (sums untouched — #15)`); + } + if (t.longestTurn) { + console.log( + `longest turn: #${t.longestTurn.turn} (active ${t.longestTurn.activeMinutes}m, ${t.longestTurn.rounds} rounds)` + ); + } + if (t.costUsd !== undefined && !opts.noCost) { + const partial = t.costPartial ? ' (partial — unpriced models excluded)' : ''; + console.log('est. cost: $' + t.costUsd.toFixed(2) + partial); + } + console.log(); + console.log('=== TOKENS (billed) ==='); + console.log(`input (uncached) : ${fmt(t.inputTokens)}`); + console.log( + `cache_read (reread) : ${fmt(t.cacheReadTokens)} <- whole context re-read EVERY round` + ); + console.log(`cache_write (new) : ${fmt(t.cacheCreationTokens)}`); + console.log(`output : ${fmt(t.outputTokens)}`); + console.log(`thinking (estimate) : ~${fmt(t.thinkingTokens)}`); + console.log(`TOTAL : ${fmt(t.billedTokens)}`); + console.log(`reread share : ${Math.round(t.rereadShare * 100)}%`); + console.log(); + if (opts.breakdown) { + const rows = breakdownFromRounds(ledger.rounds); + console.log('=== BY MODEL ==='); + console.log( + `${pad('model', 28)} ${padL('in', 8)} ${padL('cr', 8)} ${padL('cw', 8)} ${padL('out', 8)} ${padL('total', 9)}${opts.noCost ? '' : padL('cost', 10)}` + ); + for (const b of rows) { + const cost = + opts.noCost || b.costUsd === undefined ? '' : padL('$' + b.costUsd.toFixed(2), 10); + console.log( + `${pad(short(b.model, 28), 28)} ${padL(fmt(b.inputTokens), 8)} ${padL(fmt(b.cacheReadTokens), 8)} ${padL(fmt(b.cacheCreationTokens), 8)} ${padL(fmt(b.outputTokens), 8)} ${padL(fmt(b.billedTokens), 9)}${cost}` + ); + } + console.log(); + } + console.log('=== BY TURN ==='); + console.log( + `${pad('#', 3)} ${pad('time', 6)} ${padL('dur', 5)} ${padL('context', 9)} ${padL('reread', 9)} ${padL('new', 8)} ${padL('out', 7)} ${pad('think%', 7)} tools` + ); + const byTurn = roundsByTurn(ledger.rounds); + for (const turn of ledger.turns) { + const rs = byTurn.get(turn.index) ?? []; + if (rs.length === 0) continue; // trailing user msg / empty implicit turn + const ctx = rs.filter((r) => r.contextSize > 0).at(-1)?.contextSize ?? 0; // ghosts (#14) don't hide the real context + const reread = rs.reduce((s, r) => s + r.cacheReadTokens, 0); + const fresh = rs.reduce((s, r) => s + r.inputTokens + r.cacheCreationTokens, 0); + const out = rs.reduce((s, r) => s + r.outputTokens, 0); + const think = rs.reduce((s, r) => s + r.thinkingTokens, 0); + const genTotal = think + out; + const thinkPct = genTotal > 0 ? Math.round((think / genTotal) * 100) : 0; + const toolCounts = countTools(rs); + const tools = [...toolCounts].map(([n, c]) => `${n} x${c}`).join(', '); + console.log( + `${pad(String(turn.index), 3)} ${pad(hhmm(turn.start), 6)} ${padL(turn.activeMinutes + 'm', 5)} ${padL(fmt(ctx), 9)} ${padL(fmt(reread), 9)} ${padL(fmt(fresh), 8)} ${padL(fmt(out), 7)} ${pad(thinkPct + '%', 7)} ${short(tools, 60)}` + ); + } + console.log(); + console.log(`=== ROUNDS (last ${opts.rounds}) ===`); + console.log( + `${pad('#', 4)} ${pad('time', 6)} ${padL('context', 9)} ${padL('delta', 9)} ${padL('in', 7)} ${padL('cr', 8)} ${padL('cw', 6)} ${padL('out', 6)} model` + ); + for (const r of ledger.rounds.slice(-opts.rounds)) { + console.log( + `${pad(String(r.index), 4)} ${pad(hhmm(r.timestamp), 6)} ${padL(fmt(r.contextSize), 9)} ${padL((r.contextDelta >= 0 ? '+' : '') + fmt(r.contextDelta), 9)} ${padL(fmt(r.inputTokens), 7)} ${padL(fmt(r.cacheReadTokens), 8)} ${padL(fmt(r.cacheCreationTokens), 6)} ${padL(fmt(r.outputTokens), 6)} ${short(r.model, 28)}` + ); + } + console.log(); + + if (subagents.length > 0) { + console.log('=== SUBAGENTS ==='); + for (const s of subagents) { + const slow = s.durationMs > opts.subagentMinMinutes * 60000; + const flag = slow ? ` !SLOW >${opts.subagentMinMinutes}min` : ''; + const ongoing = s.isOngoing ? ' (ongoing)' : ''; + console.log( + `${pad(short(s.description ?? s.subagentType ?? s.id, 52), 52)} ${pad(s.subagentType ?? '', 10)} ${padL(dur(s.durationMs), 8)}${flag}${ongoing} ~${formatTokensCompact(s.metrics?.totalTokens ?? 0)} tok` + ); + } + console.log(); + } + + const severityRank = { low: 0, medium: 1, high: 2 } as const; + const visible = findings.filter( + (f) => severityRank[f.severity] >= severityRank[opts.minSeverity] + ); + console.log('=== FINDINGS ==='); + if (visible.length === 0) { + console.log('(none — clean session)'); + } else { + for (const f of visible) { + const turnTag = f.turnIndex === undefined ? '' : ', turn ' + String(f.turnIndex); + console.log(`[${f.type}${turnTag}] ${f.summary}`); + } + } + console.log(); + console.log( + 'note: this is observation, not prevention — add deny rules for culprit commands in Claude Code settings.' + ); +} + +async function main(): Promise { + const raw = process.argv.slice(2); + if (wantsHelp(raw)) { + console.log( + [ + 'usage: pnpm analyze:session | --project --last [N] [flags]', + '', + 'flags:', + ' --rounds N rounds table length (default 20)', + ' --subagent-min-minutes N slow-subagent threshold in minutes (default 5)', + ' --min-severity S low | medium | high (default low = all)', + ' --breakdown per-model token/cost breakdown', + ' --since DATE only activity on/after this date (YYYY-MM-DD or YYYYMMDD)', + ' --until DATE only activity on/before this date', + ' --last [N] no value: analyze newest session of --project;', + ' N: only activity of the last N calendar days', + ' --no-cost omit cost estimates', + ' --json machine-readable output', + ].join('\n') + ); + return; + } + const opts = parseArgs(raw); + if (opts.error) { + console.error(opts.error); + process.exitCode = 1; + return; + } + + let sessionFile = opts.sessionPath ? path.resolve(opts.sessionPath) : null; + if (!sessionFile && opts.projectArg) { + if (!opts.useLast && opts.lastDays === undefined) { + console.error('--project requires --last (bare, or with a day count: --last N)'); + process.exitCode = 1; + return; + } + const dir = resolveProjectDir(opts.projectArg); + if (!fs.existsSync(dir)) { + console.error(`project dir not found: ${dir}`); + process.exitCode = 1; + return; + } + sessionFile = pickNewestSessionFile(dir); + if (!sessionFile) { + console.error(`no .jsonl sessions found in ${dir}`); + process.exitCode = 1; + return; + } + } + if (!sessionFile || !fs.existsSync(sessionFile)) { + console.error( + 'usage: pnpm analyze:session | pnpm analyze:session --project --last' + ); + process.exitCode = 1; + return; + } + + const { projectId, sessionId } = splitSessionPath(sessionFile); + const messages = await parseJsonlFile(sessionFile); + const ledger = filterLedgerByDate(buildLedger(messages), opts.since, opts.until); + const findings = computeFindings(messages, ledger, opts.since, opts.until); + + let subagents: Process[] = []; + try { + subagents = await resolveSubagentsFor(projectId, sessionId, messages); + } catch (err) { + console.error(`(subagent resolution unavailable: ${String(err)})`); + } + // under --since/--until keep only subagents whose [startTime, endTime] + // overlaps the same window as the ledger and findings + if (opts.since || opts.until) { + subagents = subagents.filter((s) => { + if (opts.since && s.endTime < opts.since) return false; + if (opts.until && s.startTime > opts.until) return false; + return true; + }); + } + + if (opts.json) { + // Process carries the full parsed transcript — strip it for the JSON dump + const subagentSummaries = subagents.map((p) => ({ + id: p.id, + description: p.description, + subagentType: p.subagentType, + durationMs: p.durationMs, + totalTokens: p.metrics?.totalTokens ?? 0, + isOngoing: p.isOngoing, + })); + // --no-cost: drop cost fields without mutating the live ledger + const totalsOut = { ...ledger.totals }; + if (opts.noCost) { + delete totalsOut.costUsd; + delete totalsOut.costPartial; + } + const ledgerOut = opts.noCost ? { ...ledger, totals: totalsOut } : ledger; + const breakdown = opts.breakdown ? breakdownFromRounds(ledger.rounds, !opts.noCost) : undefined; + console.log( + JSON.stringify( + { + file: sessionFile, + projectId, + sessionId, + ledger: ledgerOut, + findings, + subagents: subagentSummaries, + ...(breakdown ? { breakdown } : {}), + }, + null, + 2 + ) + ); + return; + } + printReport(sessionFile, ledger, findings, subagents, opts); +} + +// run only when executed directly: sessionInventory imports this module for +// helpers, vitest imports it for the pure functions — neither may trigger main +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + void main().catch((err) => { + console.error(err); + process.exitCode = 1; + }); +} diff --git a/src/cli/args.ts b/src/cli/args.ts new file mode 100644 index 00000000..94821820 --- /dev/null +++ b/src/cli/args.ts @@ -0,0 +1,42 @@ +/** + * Shared CLI plumbing for the analyze:* commands — flag-value scanning, + * ccusage-style date bounds (--since/--until/--last N), range checks. + * No dependencies; pure functions so tests import them directly. + */ + +// the next token is a flag's value unless missing or itself a flag +// (--rounds --json must not swallow --json) +export function takeFlagValue(argv: string[], i: number): { value: string; next: number } { + const hasArg = i + 1 < argv.length; + const isValue = hasArg && !argv[i + 1].startsWith('--'); + return isValue ? { value: argv[i + 1], next: i + 2 } : { value: '', next: i + 1 }; +} + +export const wantsHelp = (argv: string[]): boolean => + argv.includes('--help') || argv.includes('-h'); + +// `YYYY-MM-DD` or `YYYYMMDD` → local start (or end) of that day; null when malformed +export function parseDayBound(s: string, endOfDay: boolean): Date | null { + const m = /^(\d{4})-?(\d{2})-?(\d{2})$/.exec(s.trim()); + if (!m) return null; + const [y, mo, d] = [Number(m[1]), Number(m[2]), Number(m[3])]; + const date = new Date(y, mo - 1, d); + if (date.getFullYear() !== y || date.getMonth() !== mo - 1 || date.getDate() !== d) return null; + if (endOfDay) date.setHours(23, 59, 59, 999); + return date; +} + +// --last N → ccusage-style calendar window: local midnight of (today − N + 1), +// so N = 1 means "today". Callers must reject N < 1. +export function lastDaysSince(days: number): Date { + const d = new Date(); + d.setHours(0, 0, 0, 0); + d.setDate(d.getDate() - (days - 1)); + return d; +} + +export function inDateRange(ts: Date, since?: Date, until?: Date): boolean { + if (since && ts < since) return false; + if (until && ts > until) return false; + return true; +} diff --git a/src/cli/importLoops.ts b/src/cli/importLoops.ts new file mode 100644 index 00000000..f255ec56 --- /dev/null +++ b/src/cli/importLoops.ts @@ -0,0 +1,167 @@ +/** + * importLoops CLI — backfill the notification bell with historical loop + * incidents from the transcript corpus. + * + * Light scan (sessionInventory.scanSessionFile) -> filter by run length -> + * merge into ~/.claude/claude-devtools-notifications.json. Idempotent: + * previous imports carry triggerId 'historical-loop' and are replaced on + * each run, so threshold changes re-import cleanly. + * + * Flags: --min-cycle N (default 20), --dry-run. + * The bell reads the file at app startup — restart the app after import. + */ + +import { extractProjectName } from '@main/utils/pathDecoder'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { pathToFileURL } from 'url'; + +import { wantsHelp } from './args'; +import { scanSessionFile } from './sessionInventory'; + +const NOTIFICATIONS_PATH = path.join(os.homedir(), '.claude', 'claude-devtools-notifications.json'); +const HISTORICAL_TAG = 'historical-loop'; +const BELL_CAP = 100; + +interface StoredLike { + id: string; + timestamp: number; + triggerId?: string; +} + +interface ImportedLoop { + id: string; + timestamp: number; + sessionId: string; + projectId: string; + filePath: string; + source: string; + message: string; + toolUseId?: string; + triggerId: string; + triggerName: string; + context: { projectName: string; cwd?: string }; + isRead: boolean; + createdAt: number; +} + +// =========================================================================== +// Pure core (exported for tests) +// =========================================================================== + +/** Scan results -> importable loop notifications, filtered by run length. */ +export function buildImportEntries( + found: { + sessionId: string; + projectId: string; + filePath: string; + cwd?: string; + cycles: { key: string; count: number; startTs: string; toolUseId: string }[]; + }[], + minCycle: number +): ImportedLoop[] { + const out: ImportedLoop[] = []; + for (const f of found) { + for (const c of f.cycles) { + if (c.count < minCycle) continue; + const startMs = Date.parse(c.startTs); + if (!Number.isFinite(startMs)) continue; + out.push({ + id: `historical-loop-${f.sessionId}-${c.count}x-${c.startTs}`, + timestamp: startMs, + sessionId: f.sessionId, + projectId: f.projectId, + filePath: f.filePath, + source: 'loop', + message: `${c.key} ×${c.count} — possible stuck loop`, + toolUseId: c.toolUseId || undefined, + triggerId: HISTORICAL_TAG, + triggerName: 'Loop (historical)', + context: { projectName: extractProjectName(f.projectId, f.cwd) }, + isRead: false, + createdAt: Date.now(), + }); + } + } + return out.sort((a, b) => b.timestamp - a.timestamp); +} + +/** + * Merge incoming entries into the bell store: previous imports (triggerId + * 'historical-loop') are replaced wholesale, live notifications stay, newest + * first, capped. + */ +export function mergeImport( + existing: StoredLike[], + incoming: ImportedLoop[], + cap = BELL_CAP +): StoredLike[] { + const kept = existing.filter((n) => n.triggerId !== HISTORICAL_TAG); + const merged = [...kept, ...(incoming as StoredLike[])].sort((a, b) => b.timestamp - a.timestamp); + return merged.slice(0, cap); +} + +// =========================================================================== +// CLI +// =========================================================================== + +async function main(): Promise { + const argv = process.argv.slice(2); + if (wantsHelp(argv)) { + console.log('Usage: pnpm loops:import [--min-cycle N] [--dry-run]'); + return; + } + + let minCycle = 20; + const minCycleIdx = argv.indexOf('--min-cycle'); + if (minCycleIdx !== -1) { + minCycle = parseInt(argv[minCycleIdx + 1] ?? '', 10) || 20; + } + const dryRun = argv.includes('--dry-run'); + + console.log(`Scanning corpus (min cycle ${minCycle})...`); + const found: Parameters[0] = []; + let fileCount = 0; + const base = path.join(os.homedir(), '.claude', 'projects'); + + for (const dir of fs.readdirSync(base, { withFileTypes: true })) { + if (!dir.isDirectory()) continue; + const projectPath = path.join(base, dir.name); + let files: fs.Dirent[] = []; + try { + files = fs.readdirSync(projectPath, { withFileTypes: true }); + } catch { + continue; + } + for (const f of files) { + if (!f.isFile() || !f.name.endsWith('.jsonl') || f.name.startsWith('agent-')) continue; + const entry = await scanSessionFile(path.join(projectPath, f.name)).catch(() => null); + fileCount++; + if (!entry) continue; + if (entry.cycles.some((c) => c.count >= minCycle)) found.push(entry); + } + } + + const incoming = buildImportEntries(found, minCycle); + console.log(`scanned ${fileCount} files, ${incoming.length} loops at >= ${minCycle}x`); + + if (dryRun) { + console.log('dry run — nothing written'); + return; + } + + let existing: StoredLike[] = []; + try { + existing = JSON.parse(fs.readFileSync(NOTIFICATIONS_PATH, 'utf8')) as StoredLike[]; + } catch { + existing = []; + } + const merged = mergeImport(existing, incoming); + fs.writeFileSync(NOTIFICATIONS_PATH, JSON.stringify(merged, null, 2), 'utf8'); + console.log('Restart the app to see imported loops in the bell.'); +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + void main(); +} diff --git a/src/cli/sessionInventory.ts b/src/cli/sessionInventory.ts new file mode 100644 index 00000000..3ef3159a --- /dev/null +++ b/src/cli/sessionInventory.ts @@ -0,0 +1,588 @@ +/** + * Sessions inventory CLI — duration, models, token totals across all sessions. + * Answers "which sessions ran 2h+, on which model" without opening the app. + * + * Usage: + * pnpm analyze:sessions [--project ] [flags] + * Flags: + * --min-minutes N only sessions with at least N minutes of active (API) time + * --min-turn-minutes N only sessions whose longest turn ran at least N active minutes + * --min-streak N only sessions where some back-to-back cycle ran at least N times (runs < 3x are never recorded) + * --sort FIELD duration | active | turn | streak | tokens | date (default duration) + * --limit N show first N rows + * --breakdown per-session model token split (models column + JSON tokensByModel) + * --since / --until DATE filter by session last-activity date (YYYY-MM-DD or YYYYMMDD) + * --last N last-activity date within the last N calendar days + * --no-cost accepted for the shared grammar; inventory has no cost figures + * --json machine-readable output + */ + +import { isParsedUserChunkMessage, type ParsedMessage } from '@main/types'; +import { + decodePath, + extractSessionId, + getProjectsBasePath, + normalizeDriveLetter, + translateWslMountPath, +} from '@main/utils/pathDecoder'; +import { formatTokensCompact } from '@shared/utils/tokenFormatting'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import * as readline from 'readline'; +import { pathToFileURL } from 'url'; + +import { + bashStem, + billingFromFlags, + type BillingScheme, + dur, + normalizeCallKey, + pad, + padL, + resolveProjectDir, + short, + TURN_IDLE_GAP_CAP_MINUTES, + WASTE_THRESHOLDS, +} from './analyzeSession'; +import { inDateRange, lastDaysSince, parseDayBound, takeFlagValue, wantsHelp } from './args'; + +interface RawUsage { + input_tokens?: number; + output_tokens?: number; + cache_read_input_tokens?: number; + cache_creation_input_tokens?: number; +} + +interface ScanEntry { + type?: string; + timestamp?: string; + requestId?: string; + isSidechain?: boolean; + isMeta?: boolean; + cwd?: string; + message?: { model?: string; usage?: RawUsage; content?: unknown }; +} + +function mergeUsage(a: RawUsage, b: RawUsage): RawUsage { + return { + input_tokens: (a.input_tokens ?? 0) + (b.input_tokens ?? 0), + output_tokens: (a.output_tokens ?? 0) + (b.output_tokens ?? 0), + cache_read_input_tokens: (a.cache_read_input_tokens ?? 0) + (b.cache_read_input_tokens ?? 0), + cache_creation_input_tokens: + (a.cache_creation_input_tokens ?? 0) + (b.cache_creation_input_tokens ?? 0), + }; +} + +export interface InventoryEntry { + projectId: string; + sessionId: string; + filePath: string; + durationMs: number; + /** API work time: capped gaps between main-chain usage lines (one anchor per requestId) — same cap as turnActiveMinutes */ + activeMs: number; + /** longest single turn's active time — a 96h-active session may have no turn over 20 min */ + longestTurnMs: number; + /** every maximal run of >= loopStreakMin back-to-back identical calls, longest first (cap 10) */ + cycles: { key: string; count: number; startTs: string; toolUseId: string }[]; + lastTs: Date | null; + models: string[]; + messageCount: number; + inputTokens: number; + outputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; + totalTokens: number; + tokensByModel: Record; + sizeBytes: number; + billing: BillingScheme; + /** real project path from the session's cwd field — decodePath is lossy when the name has dashes */ + cwd?: string; +} + +// ponytail: own streaming pass instead of parseJsonlFile — it materializes every +// ParsedMessage and that measurably blows up on 1200+ files incl. 66MB ones +export async function scanSessionFile(filePath: string): Promise { + const rl = readline.createInterface({ + input: fs.createReadStream(filePath, { encoding: 'utf8' }), + crlfDelay: Infinity, + }); + + let firstTs: number | null = null; + let lastTs: number | null = null; + let activeMs = 0; + let turnActiveMs = 0; + let longestTurnMs = 0; + let prevUsageTs: number | null = null; + let messageCount = 0; + const seenCallIds = new Set(); + // back-to-back run tracking: one entry per maximal run of >= loopStreakMin + // identical calls. Buckets Bash by stem (pipe-cut) — analyze:session's + // loop_streak keys on the full input, so the two may split one session + let lastStreakKey = ''; + let curStreak = 0; + let curStreakStartTs = ''; + let curStreakToolUseId = ''; + const runs: { key: string; count: number; startTs: string; toolUseId: string }[] = []; + const flushStreak = (): void => { + if (lastStreakKey && curStreak >= WASTE_THRESHOLDS.loopStreakMin) { + runs.push({ + key: lastStreakKey, + count: curStreak, + startTs: curStreakStartTs, + toolUseId: curStreakToolUseId, + }); + } + }; + const countCalls = (line: ScanEntry): void => { + const m = line.message; + if (!m || !Array.isArray(m.content)) return; + for (const block of m.content) { + const b = block as { + type?: string; + name?: string; + id?: string; + input?: Record; + }; + if (b?.type !== 'tool_use' || !b.name) continue; + const dedupId = `${line.requestId ?? ''}|${b.id ?? ''}`; + if (seenCallIds.has(dedupId)) continue; + seenCallIds.add(dedupId); + const key = bashStem(normalizeCallKey(b.name, b.input ?? {})); + if (key === lastStreakKey) { + curStreak += 1; + } else { + flushStreak(); + lastStreakKey = key; + curStreak = 1; + curStreakStartTs = line.timestamp ?? ''; + } + curStreakToolUseId = b.id ?? curStreakToolUseId; + } + }; + let cwd: string | undefined; + const models = new Set(); + // anthropic-style rounds report cache writes; routers report cache_read with cw=0 + let sawWrite = false; + let sawRead = false; + const usageByRequestId = new Map(); + const directByModel = new Map(); + + for await (const line of rl) { + if (line === '') continue; + let e: ScanEntry; + try { + e = JSON.parse(line) as ScanEntry; + } catch { + continue; + } + if (!e.timestamp) continue; + const ts = new Date(e.timestamp).getTime(); + if (Number.isNaN(ts)) continue; + if (firstTs === null || ts < firstTs) firstTs = ts; + if (lastTs === null || ts > lastTs) lastTs = ts; + if (e.type === 'user' || e.type === 'assistant') messageCount++; + // normalize like extractCwd: uppercase drive letter, WSL mount → Windows path + if (!cwd && e.cwd) cwd = normalizeDriveLetter(translateWslMountPath(e.cwd)); + + // turn boundary = real user message, via the same isParsedUserChunkMessage + // guard buildLedger uses (system-output tags, array content, teammate pings + // all behave identically — the divergence this way is structural). !isSidechain + // stays local: sidechain lines never split the main-chain turn accounting. + // cast: the guard reads only type/isMeta/content, all present on the raw line + if ( + e.type === 'user' && + !e.isSidechain && + isParsedUserChunkMessage({ + type: e.type, + isMeta: e.isMeta, + content: e.message?.content, + } as unknown as ParsedMessage) + ) { + longestTurnMs = Math.max(longestTurnMs, turnActiveMs); + turnActiveMs = 0; + prevUsageTs = null; + } + + // calls are counted on ALL main-chain assistant lines — router sessions + // log usage on a separate final line, so usage-gating would miss most; + // seenCallIds keeps streaming snapshots from double-counting + if ( + e.type === 'assistant' && + !e.isSidechain && + e.message && + e.message.model !== '' + ) { + countCalls(e); + } + + // sidechain (subagent) entries stay out of totals/models/billing — same + // accounting as buildLedger; timestamps and message count cover the file + if ( + e.type === 'assistant' && + !e.isSidechain && + e.message?.usage && + e.message.model !== '' + ) { + // API time: gap to the previous main-chain usage line, capped — same + // accounting as turnActiveMinutes, hours of silence cost zero. Streaming + // snapshots of an already-seen requestId add no gap (analyze:session keeps + // one round per request, anchored at the last line) but still advance the + // anchor, so the next gap starts from the newest line of the request. + const snapshot = e.requestId !== undefined && usageByRequestId.has(e.requestId); + if (prevUsageTs !== null && !snapshot) { + const gap = Math.min(Math.max(ts - prevUsageTs, 0), TURN_IDLE_GAP_CAP_MINUTES * 60000); + activeMs += gap; + turnActiveMs += gap; + } + prevUsageTs = ts; + if (e.message.model) models.add(e.message.model); + if (e.message.usage.cache_creation_input_tokens) sawWrite = true; + else if (e.message.usage.cache_read_input_tokens) sawRead = true; + // streaming writes several entries per request — the last one has final counts + if (e.requestId) { + usageByRequestId.set(e.requestId, { + model: e.message.model ?? 'unknown', + usage: e.message.usage, + }); + } else { + const m = e.message.model ?? 'unknown'; + directByModel.set(m, mergeUsage(directByModel.get(m) ?? {}, e.message.usage)); + } + } + } + + if (firstTs === null || lastTs === null) return null; + longestTurnMs = Math.max(longestTurnMs, turnActiveMs); + flushStreak(); + + let totals: RawUsage = {}; + const byModel = new Map(); + const acc = (model: string, u: RawUsage): void => { + totals = mergeUsage(totals, u); + byModel.set(model, mergeUsage(byModel.get(model) ?? {}, u)); + }; + for (const { model, usage } of usageByRequestId.values()) acc(model, usage); + for (const [model, usage] of directByModel) acc(model, usage); + + const tokensByModel: Record = {}; + for (const [model, u] of byModel) { + tokensByModel[model] = + (u.input_tokens ?? 0) + + (u.output_tokens ?? 0) + + (u.cache_read_input_tokens ?? 0) + + (u.cache_creation_input_tokens ?? 0); + } + + const rel = path.relative(getProjectsBasePath(), path.resolve(filePath)); + const [projectId = '', sessionId = ''] = rel.split(path.sep); + + const inputTokens = totals.input_tokens ?? 0; + const outputTokens = totals.output_tokens ?? 0; + const cacheReadTokens = totals.cache_read_input_tokens ?? 0; + const cacheCreationTokens = totals.cache_creation_input_tokens ?? 0; + + return { + projectId, + sessionId: extractSessionId(sessionId), + filePath, + durationMs: Math.max(0, lastTs - firstTs), + activeMs, + longestTurnMs, + cycles: runs.toSorted((a, b) => b.count - a.count).slice(0, 10), + lastTs: new Date(lastTs), + models: [...models], + messageCount, + inputTokens, + outputTokens, + cacheReadTokens, + cacheCreationTokens, + totalTokens: inputTokens + outputTokens + cacheReadTokens + cacheCreationTokens, + tokensByModel, + sizeBytes: fs.statSync(filePath).size, + billing: billingFromFlags(sawWrite, sawRead), + cwd, + }; +} + +async function listSessionFiles(projectDir: string): Promise { + const out: string[] = []; + for (const e of await fs.promises.readdir(projectDir, { withFileTypes: true })) { + if (e.isFile() && e.name.endsWith('.jsonl') && !e.name.startsWith('agent-')) { + out.push(path.join(projectDir, e.name)); + } + } + return out; +} + +export async function mapWithConcurrency( + items: T[], + limit: number, + fn: (x: T) => Promise +): Promise<(R | null)[]> { + // ponytail: result order is nondeterministic — rows are sorted downstream anyway + const results: (R | null)[] = []; + let next = 0; + const workers = Array.from({ length: Math.min(limit, items.length) }, async () => { + while (next < items.length) { + const item = items[next]; + next += 1; + try { + results.push(await fn(item)); + } catch (err) { + // one unreadable file must not abort a 1200+-file scan + console.error(`(skipped ${String(item)}: ${String(err)})`); + results.push(null); + } + } + }); + await Promise.all(workers); + return results; +} + +async function collect(projectsRoot: string, projectArg?: string): Promise { + let dirs: string[]; + if (projectArg) { + dirs = [projectArg]; + } else { + const es = await fs.promises.readdir(projectsRoot, { withFileTypes: true }); + dirs = es.filter((e) => e.isDirectory()).map((e) => path.join(projectsRoot, e.name)); + } + const fileGroups = await mapWithConcurrency(dirs, 8, listSessionFiles); + const files = fileGroups.filter((g): g is string[] => g !== null).flat(); + // ponytail: no mtime cache yet — first full scan is IO-bound seconds, add one if it hurts + const scanned = await mapWithConcurrency(files, 8, scanSessionFile); + return scanned.filter((e): e is InventoryEntry => e !== null); +} + +const shortenHome = (p: string): string => { + const home = os.homedir(); + return p.startsWith(home) ? `~${p.slice(home.length)}` : p; +}; + +const ymd = (d: Date): string => + `${d.getFullYear()}-${String(d.getMonth() + 1).padStart(2, '0')}-${String(d.getDate()).padStart(2, '0')}`; + +export interface InventoryOpts { + projectArg?: string; + minMinutes: number; + sort: 'duration' | 'active' | 'turn' | 'streak' | 'tokens' | 'date'; + limit: number; + json: boolean; + breakdown: boolean; + minTurnMinutes: number; + minStreak: number; + since?: Date; + until?: Date; + error?: string; +} + +export function parseInventoryArgs(argv: string[]): InventoryOpts { + const opts: InventoryOpts = { + minMinutes: 0, + sort: 'duration', + limit: Number.POSITIVE_INFINITY, + json: false, + breakdown: false, + minTurnMinutes: 0, + minStreak: 0, + }; + let i = 0; + while (i < argv.length) { + const a = argv[i]; + const { value, next } = takeFlagValue(argv, i); + if (a === '--project') { + opts.projectArg = value; + i = next; + continue; + } + if (a === '--min-minutes') { + opts.minMinutes = parseInt(value, 10) || 0; + i = next; + continue; + } + if (a === '--min-turn-minutes') { + opts.minTurnMinutes = parseInt(value, 10) || 0; + i = next; + continue; + } + if (a === '--min-streak') { + opts.minStreak = parseInt(value, 10) || 0; + i = next; + continue; + } + if (a === '--sort') { + opts.sort = + value === 'tokens' || + value === 'date' || + value === 'active' || + value === 'turn' || + value === 'streak' + ? value + : 'duration'; + i = next; + continue; + } + if (a === '--limit') { + const n = parseInt(value, 10); + opts.limit = n > 0 ? n : Number.POSITIVE_INFINITY; + i = next; + continue; + } + if (a === '--since' || a === '--until') { + const d = parseDayBound(value, a === '--until'); + if (!d) opts.error = `invalid ${a} date (expected YYYY-MM-DD or YYYYMMDD): '${value}'`; + else if (a === '--since') opts.since = d; + else opts.until = d; + i = next; + continue; + } + if (a === '--last') { + if (/^[1-9]\d*$/.test(value)) { + opts.since = lastDaysSince(parseInt(value, 10)); + i = next; + } else { + opts.error = `--last expects a number of days >= 1, got '${value || '(missing)'}'`; + i += 1; + } + continue; + } + if (a === '--breakdown') { + opts.breakdown = true; + i += 1; + continue; + } + // accepted for the shared grammar; the inventory carries no cost figures + if (a === '--no-cost') { + i += 1; + continue; + } + if (a === '--json') { + opts.json = true; + i += 1; + continue; + } + i += 1; + } + return opts; +} + +async function main(): Promise { + const argv = process.argv.slice(2); + if (wantsHelp(argv)) { + console.log( + [ + 'usage: pnpm analyze:sessions [--project ] [flags]', + '', + 'flags:', + ' --min-minutes N only sessions with at least N minutes of active (API) time', + ' --min-turn-minutes N only sessions whose longest turn ran at least N active minutes', + ` --min-streak N only sessions where some back-to-back cycle ran at least N times (floor: ${String(WASTE_THRESHOLDS.loopStreakMin)}x — shorter runs are not recorded)`, + ' --sort FIELD duration | active | turn | streak | tokens | date (default duration)', + ' --limit N show first N rows', + ' --breakdown per-session model token split', + ' --since DATE only sessions whose last activity is on/after this date (YYYY-MM-DD or YYYYMMDD)', + ' --until DATE only sessions whose last activity is on/before this date', + ' --last N sessions whose last activity falls in the last N calendar days', + ' --no-cost accepted for the shared grammar; inventory has no cost figures', + ' --json machine-readable output', + ].join('\n') + ); + return; + } + const opts = parseInventoryArgs(argv); + if (opts.error) { + console.error(opts.error); + process.exitCode = 1; + return; + } + + let projectDir: string | undefined; + if (opts.projectArg) { + projectDir = resolveProjectDir(opts.projectArg); + if (!fs.existsSync(projectDir)) { + console.error(`project dir not found: ${opts.projectArg}`); + process.exitCode = 1; + return; + } + } + + const entries = await collect(getProjectsBasePath(), projectDir); + entries.sort((a, b) => { + if (opts.sort === 'tokens') return b.totalTokens - a.totalTokens; + if (opts.sort === 'date') return (b.lastTs?.getTime() ?? 0) - (a.lastTs?.getTime() ?? 0); + if (opts.sort === 'active') return b.activeMs - a.activeMs; + if (opts.sort === 'turn') return b.longestTurnMs - a.longestTurnMs; + if (opts.sort === 'streak') { + return (b.cycles[0]?.count ?? 0) - (a.cycles[0]?.count ?? 0); + } + return b.durationMs - a.durationMs; + }); + const hasDateFilter = opts.since !== undefined || opts.until !== undefined; + // heuristic: a session's date = its LAST activity; a session that started + // before --since still matches if it ended inside the window, but one that + // ran past --until drops out + const shown = entries + .filter((e) => e.activeMs >= opts.minMinutes * 60000) + .filter((e) => e.longestTurnMs >= opts.minTurnMinutes * 60000) + .filter((e) => (e.cycles[0]?.count ?? 0) >= opts.minStreak) + .filter((e) => + hasDateFilter ? e.lastTs !== null && inDateRange(e.lastTs, opts.since, opts.until) : true + ) + .slice(0, opts.limit); + + if (opts.json) { + // tokensByModel surfaces only with --breakdown; default JSON stays as before + const out = shown.map((e) => { + if (opts.breakdown) return e; + const copy = { ...e }; + delete (copy as Partial).tokensByModel; + return copy; + }); + console.log(JSON.stringify(out, null, 2)); + return; + } + + const notes: string[] = []; + if (opts.minMinutes > 0) notes.push(`>= ${String(opts.minMinutes)} min active`); + if (opts.minTurnMinutes > 0) notes.push(`turn >= ${String(opts.minTurnMinutes)} min`); + if (opts.minStreak > 0) notes.push(`cycle >= ${String(opts.minStreak)}x`); + if (opts.since && opts.until) notes.push(`${ymd(opts.since)}..${ymd(opts.until)}`); + else if (opts.since) notes.push(`since ${ymd(opts.since)}`); + else if (opts.until) notes.push(`until ${ymd(opts.until)}`); + const note = notes.length > 0 ? ` (${notes.join(', ')})` : ''; + console.log(`sessions: ${entries.length} total, showing ${shown.length}${note}`); + console.log(); + console.log( + `${padL('active', 9)} ${padL('wall', 9)} ${padL('max turn', 8)} ${padL('cycle', 9)} ${pad('date', 11)} ${pad('file', 9)} ${pad('project', 42)} ${pad(opts.breakdown ? 'by model (share)' : 'models', 30)} ${padL('tokens', 9)} ${padL('msgs', 6)} ${pad('billing', 15)}` + ); + for (const e of shown) { + // real path from the session beats decoding the encoded dir name (lossy on dashes) + const project = e.cwd ? shortenHome(e.cwd) : shortenHome(decodePath(e.projectId)); + const models = opts.breakdown ? modelShareCell(e) : e.models.join(', '); + console.log( + `${padL(dur(e.activeMs), 9)} ${padL(dur(e.durationMs), 9)} ${padL(dur(e.longestTurnMs), 8)} ${padL(cycleCell(e), 9)} ${pad(e.lastTs ? e.lastTs.toISOString().slice(0, 10) : 'n/a', 11)} ${pad(e.sessionId.slice(0, 8), 9)} ${pad(short(project, 42), 42)} ${pad(short(models, 30), 30)} ${padL(formatTokensCompact(e.totalTokens), 9)} ${padL(String(e.messageCount), 6)} ${pad(e.billing, 15)}` + ); + } +} + +// "2c/40x" — number of distinct back-to-back cycles and the longest one +const cycleCell = (e: InventoryEntry): string => + e.cycles.length > 0 ? `${String(e.cycles.length)}c/${String(e.cycles[0].count)}x` : '-'; + +function modelShareCell(e: InventoryEntry): string { + const total = Object.values(e.tokensByModel).reduce((s, v) => s + v, 0); + if (total === 0) return 'n/a'; + return Object.entries(e.tokensByModel) + .sort((a, b) => b[1] - a[1]) + .map(([m, v]) => `${m} ${Math.round((v / total) * 100)}%`) + .join(', '); +} + +// run only when executed directly — vitest imports this file for scanSessionFile +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + void main().catch((err) => { + console.error(err); + process.exitCode = 1; + }); +} diff --git a/src/cli/turnSpendStats.ts b/src/cli/turnSpendStats.ts new file mode 100644 index 00000000..e0453ed5 --- /dev/null +++ b/src/cli/turnSpendStats.ts @@ -0,0 +1,146 @@ +/** + * turnSpendStats CLI — corpus calibration for the turn-input budget hook. + * + * Walks ~/.claude/projects/** session transcripts with the SAME accounting the + * turn-budget hook enforces: per turn (from one real user message to the next), + * sum the input-side tokens (input + cache_read + cache_creation) of every + * assistant round. Prints percentiles so the default budget comes from data, + * not guesswork. Flags: --p N (percentile to highlight, default 99). + */ + +import { isParsedUserChunkMessage } from '@main/types'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import * as readline from 'readline'; + +import { wantsHelp } from './args'; + +export interface TurnSpend { + file: string; + turnIndex: number; + inputSide: number; + rounds: number; +} + +/** One JSONL line -> state mutation for the turn-spend walker. */ +export function feedLine( + line: string, + state: { current: TurnSpend | null; spends: TurnSpend[]; file: string } +): void { + let msg: { + type?: string; + isMeta?: boolean; + message?: { + usage?: { + input_tokens?: number; + cache_read_input_tokens?: number; + cache_creation_input_tokens?: number; + }; + content?: unknown; + }; + }; + try { + msg = JSON.parse(line) as typeof msg; + } catch { + return; + } + // raw lines wrap content/usage in .message (ParsedMessage flattens it) + const inner = msg.message ?? {}; + + // real user message = turn boundary — same predicate as analyzeSession + if ( + isParsedUserChunkMessage({ + type: msg.type, + isMeta: msg.isMeta, + content: inner.content, + } as never) + ) { + if (state.current && state.current.rounds > 0) state.spends.push(state.current); + state.current = { file: state.file, turnIndex: state.spends.length, inputSide: 0, rounds: 0 }; + return; + } + + if (msg.type === 'assistant' && inner.usage && state.current) { + const u = inner.usage; + state.current.inputSide += + (u.input_tokens ?? 0) + + (u.cache_read_input_tokens ?? 0) + + (u.cache_creation_input_tokens ?? 0); + state.current.rounds += 1; + } +} + +/** Nearest-rank percentile of an ascending-sorted array. */ +export function percentile(sorted: number[], q: number): number { + if (sorted.length === 0) return 0; + const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((q / 100) * sorted.length) - 1)); + return sorted[idx]; +} + +// ============================================================================= +// CLI +// ============================================================================= + +async function main(): Promise { + const argv = process.argv.slice(2); + if (wantsHelp(argv)) { + console.log('Usage: pnpm turn-spend:stats [--p N]'); + return; + } + const pIdx = argv.indexOf('--p'); + const p = pIdx !== -1 ? parseFloat(argv[pIdx + 1] ?? '') || 99 : 99; + + const base = path.join(os.homedir(), '.claude', 'projects'); + const spends: TurnSpend[] = []; + let fileCount = 0; + + for (const dir of fs.readdirSync(base, { withFileTypes: true })) { + if (!dir.isDirectory()) continue; + const projectPath = path.join(base, dir.name); + let files: fs.Dirent[] = []; + try { + files = fs.readdirSync(projectPath, { withFileTypes: true }); + } catch { + continue; + } + for (const f of files) { + if (!f.isFile() || !f.name.endsWith('.jsonl') || f.name.startsWith('agent-')) continue; + const filePath = path.join(projectPath, f.name); + const state = { current: null as TurnSpend | null, spends, file: filePath }; + try { + const rl = readline.createInterface({ + input: fs.createReadStream(filePath), + crlfDelay: Infinity, + }); + for await (const line of rl) feedLine(line, state); + } catch { + continue; + } + if (state.current && state.current.rounds > 0) spends.push(state.current); + fileCount++; + } + } + + const values = spends.map((s) => s.inputSide).sort((a, b) => a - b); + const fmt = (n: number): string => n.toLocaleString('en-US'); + console.log(`scanned ${fileCount} files, ${values.length} turns`); + for (const q of [50, 75, 90, 95, 99, 99.9]) { + console.log(`p${q}: ${fmt(percentile(values, q))}`); + } + console.log(`max: ${fmt(values[values.length - 1] ?? 0)}`); + console.log(`p${p} suggestion: ${fmt(percentile(values, p))}`); + + const buckets = [0, 100_000, 500_000, 1_000_000, 2_000_000, 5_000_000, Infinity]; + const labels = ['<100k', '100-500k', '500k-1M', '1-2M', '2-5M', '5M+']; + for (let i = 0; i < labels.length; i++) { + const n = values.filter((v) => v >= buckets[i] && v < buckets[i + 1]).length; + console.log( + `${labels[i].padEnd(9)} ${String(n).padStart(6)} (${((n / values.length) * 100).toFixed(1)}%)` + ); + } +} + +if (process.argv[1] && import.meta.url === `file://${process.argv[1]}`) { + void main(); +} diff --git a/src/main/index.ts b/src/main/index.ts index 5140f968..c91929e9 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -32,6 +32,25 @@ const totalMB = Math.floor(totalmem() / (1024 * 1024)); const heapMB = Math.min(4096, Math.max(2048, Math.floor(totalMB * 0.5))); app.commandLine.appendSwitch('js-flags', `--max-old-space-size=${heapMB}`); +// Single instance: a dockless (UIElement) app gives no way to discover zombie +// duplicates — a second launch must summon the existing window instead. +let appReady = false; +if (!app.requestSingleInstanceLock()) { + app.quit(); +} else { + app.on('second-instance', () => { + if (mainWindow && !mainWindow.isDestroyed()) { + if (mainWindow.isMinimized()) mainWindow.restore(); + mainWindow.show(); + app.focus({ steal: true }); + } else if (appReady) { + // last window was closed (macOS keeps the app alive) — recreate it + createWindow(); + } + // while initialization is still pending, startup's own createWindow() covers us + }); +} + // Window icon path for non-mac platforms. const getWindowIconPath = (): string | undefined => { const isDev = process.env.NODE_ENV === 'development'; @@ -461,6 +480,11 @@ function createWindow(): void { title: 'claude-devtools', }); + // Dockless apps (showDockIcon: false → app.dock.hide()) never take front on + // their own: summon the window on every creation. + mainWindow.show(); + app.focus({ steal: true }); + // Load the renderer if (process.env.NODE_ENV === 'development') { void mainWindow.loadURL(`http://localhost:${DEV_SERVER_PORT}`); @@ -618,10 +642,15 @@ void app.whenReady().then(async () => { createWindow(); } } + appReady = true; app.on('activate', () => { if (BrowserWindow.getAllWindows().length === 0) { createWindow(); + } else if (mainWindow && !mainWindow.isDestroyed()) { + // reopen (open -a, notification click) with a live window: summon it + mainWindow.show(); + app.focus({ steal: true }); } }); }); diff --git a/src/main/ipc/configValidation.ts b/src/main/ipc/configValidation.ts index 94f27205..bda17679 100644 --- a/src/main/ipc/configValidation.ts +++ b/src/main/ipc/configValidation.ts @@ -110,6 +110,8 @@ function validateNotificationsSection( 'snoozedUntil', 'snoozeMinutes', 'triggers', + 'loopDetection', + 'turnBudget', ]; const result: Partial = {}; @@ -180,6 +182,54 @@ function validateNotificationsSection( } result.triggers = value; break; + case 'loopDetection': { + if (!isPlainObject(value)) { + return { valid: false, error: 'notifications.loopDetection must be an object' }; + } + const { enabled, cycleThreshold } = value as { + enabled?: unknown; + cycleThreshold?: unknown; + }; + if (typeof enabled !== 'boolean') { + return { valid: false, error: 'notifications.loopDetection.enabled must be a boolean' }; + } + if ( + !isFiniteNumber(cycleThreshold) || + !Number.isInteger(cycleThreshold) || + cycleThreshold < 1 + ) { + return { + valid: false, + error: 'notifications.loopDetection.cycleThreshold must be an integer >= 1', + }; + } + result.loopDetection = { enabled, cycleThreshold }; + break; + } + case 'turnBudget': { + if (!isPlainObject(value)) { + return { valid: false, error: 'notifications.turnBudget must be an object' }; + } + const { enabled, maxInputTokensPerTurn } = value as { + enabled?: unknown; + maxInputTokensPerTurn?: unknown; + }; + if (typeof enabled !== 'boolean') { + return { valid: false, error: 'notifications.turnBudget.enabled must be a boolean' }; + } + if ( + !isFiniteNumber(maxInputTokensPerTurn) || + !Number.isInteger(maxInputTokensPerTurn) || + maxInputTokensPerTurn < 100_000 + ) { + return { + valid: false, + error: 'notifications.turnBudget.maxInputTokensPerTurn must be an integer >= 100000', + }; + } + result.turnBudget = { enabled, maxInputTokensPerTurn }; + break; + } default: return { valid: false, error: `Unsupported notifications key: ${key}` }; } diff --git a/src/main/services/analysis/ChunkBuilder.ts b/src/main/services/analysis/ChunkBuilder.ts index 528c3edd..aa8cf4fe 100644 --- a/src/main/services/analysis/ChunkBuilder.ts +++ b/src/main/services/analysis/ChunkBuilder.ts @@ -76,11 +76,13 @@ export class ChunkBuilder { * All chunk types are INDEPENDENT - no pairing between User and AI. */ buildChunks(messages: ParsedMessage[], subagents: Process[] = []): EnhancedChunk[] { + // Note: streaming fragment lines are merged at parse time + // (parseJsonlFile -> mergeAssistantFragments) — one request arrives as + // one message everywhere downstream const chunks: EnhancedChunk[] = []; // Filter to main thread messages (non-sidechain) const mainMessages = messages.filter((m) => !m.isSidechain); - logger.debug(`Total messages: ${messages.length}, Main thread: ${mainMessages.length}`); // Classify each message into categories using MessageClassifier const classified = classifyMessages(mainMessages); diff --git a/src/main/services/discovery/ProjectScanner.ts b/src/main/services/discovery/ProjectScanner.ts index 32e4b551..e978c91b 100644 --- a/src/main/services/discovery/ProjectScanner.ts +++ b/src/main/services/discovery/ProjectScanner.ts @@ -762,6 +762,7 @@ export class ProjectScanner { createdAt: Math.floor(createdAt), updatedAt: Math.floor(effectiveMtime), firstMessage: metadata.firstUserMessage?.text, + name: metadata.name ?? undefined, messageTimestamp: metadata.firstUserMessage?.timestamp, hasSubagents, messageCount: metadata.messageCount, @@ -771,6 +772,8 @@ export class ProjectScanner { contextConsumption: metadata.contextConsumption, compactionCount: metadata.compactionCount, phaseBreakdown: metadata.phaseBreakdown, + totalTokens: metadata.totalTokens, + turnCount: metadata.turnCount, }; } @@ -814,6 +817,8 @@ export class ProjectScanner { messageCount: 0, isOngoing: false, gitBranch: null, + totalTokens: 0, + turnCount: 0, hasDisplayableContent: false, }; } @@ -833,9 +838,12 @@ export class ProjectScanner { createdAt: Math.floor(createdAt), updatedAt: Math.floor(effectiveMtime), firstMessage: metadata.firstUserMessage?.text, + name: metadata.name ?? undefined, messageTimestamp: metadata.firstUserMessage?.timestamp, hasSubagents: false, messageCount: metadata.messageCount, + totalTokens: metadata.totalTokens, + turnCount: metadata.turnCount, metadataLevel, }; } diff --git a/src/main/services/discovery/SessionSearcher.ts b/src/main/services/discovery/SessionSearcher.ts index c3d26ea0..502ee29a 100644 --- a/src/main/services/discovery/SessionSearcher.ts +++ b/src/main/services/discovery/SessionSearcher.ts @@ -12,7 +12,7 @@ */ import { LocalFileSystemProvider } from '@main/services/infrastructure/LocalFileSystemProvider'; -import { parseJsonlFile } from '@main/utils/jsonl'; +import { parseJsonlFile, readSessionName } from '@main/utils/jsonl'; import { extractBaseDir, extractSessionId } from '@main/utils/pathDecoder'; import { createLogger } from '@shared/utils/logger'; import * as path from 'path'; @@ -219,6 +219,20 @@ export class SessionSearcher { // Cache miss — parse and extract const messages = await parseJsonlFile(filePath, this.fsProvider); const extracted = extractSearchableEntries(messages); + // /name: the real session name becomes every result's title and a + // searchable entry itself, so a name query finds the session + const name = await readSessionName(filePath, this.fsProvider); + if (name) { + extracted.entries.unshift({ + text: name, + groupId: 'session-name', + messageType: 'user', + itemType: 'user', + timestamp: mtimeMs, + messageUuid: 'session-name', + }); + extracted.sessionTitle = name; + } this.searchCache.set(filePath, mtimeMs, extracted.entries, extracted.sessionTitle); cached = extracted; } diff --git a/src/main/services/infrastructure/ConfigManager.ts b/src/main/services/infrastructure/ConfigManager.ts index d5ec6f83..33e287de 100644 --- a/src/main/services/infrastructure/ConfigManager.ts +++ b/src/main/services/infrastructure/ConfigManager.ts @@ -42,6 +42,10 @@ export interface NotificationConfig { includeSubagentErrors: boolean; /** Notification triggers - define when to generate notifications */ triggers: NotificationTrigger[]; + /** Live tool-call loop detection (FileWatcher) */ + loopDetection: { enabled: boolean; cycleThreshold: number }; + /** Per-turn input-token budget enforced by the PreToolUse hook */ + turnBudget: { enabled: boolean; maxInputTokensPerTurn: number }; } /** @@ -243,6 +247,9 @@ const DEFAULT_CONFIG: AppConfig = { snoozeMinutes: 30, includeSubagentErrors: true, triggers: DEFAULT_TRIGGERS, + loopDetection: { enabled: true, cycleThreshold: 4 }, + // corpus-calibrated (pnpm turn-spend:stats, 10 080 turns): p95 = 12.56M + turnBudget: { enabled: true, maxInputTokensPerTurn: 15_000_000 }, }, general: { launchAtLogin: false, @@ -429,6 +436,16 @@ export class ConfigManager { private mergeWithDefaults(loaded: Partial): AppConfig { const loadedNotifications = loaded.notifications ?? ({} as Partial); const loadedTriggers = loadedNotifications.triggers ?? []; + // nested field — shallow spread would let a partial hand-edited object + // drop the other subfield's default + const mergedLoopDetection = { + ...DEFAULT_CONFIG.notifications.loopDetection, + ...(loadedNotifications.loopDetection ?? {}), + }; + const mergedTurnBudget = { + ...DEFAULT_CONFIG.notifications.turnBudget, + ...(loadedNotifications.turnBudget ?? {}), + }; const mergedGeneral: GeneralConfig = { ...DEFAULT_CONFIG.general, ...(loaded.general ?? {}), @@ -443,6 +460,8 @@ export class ConfigManager { ...DEFAULT_CONFIG.notifications, ...loadedNotifications, triggers: mergedTriggers, + loopDetection: mergedLoopDetection, + turnBudget: mergedTurnBudget, }, general: mergedGeneral, display: { diff --git a/src/main/services/infrastructure/FileWatcher.ts b/src/main/services/infrastructure/FileWatcher.ts index 96fd0162..a3d8048a 100644 --- a/src/main/services/infrastructure/FileWatcher.ts +++ b/src/main/services/infrastructure/FileWatcher.ts @@ -12,7 +12,8 @@ import { type FileChangeEvent, type ParsedMessage } from '@main/types'; import { parseJsonlFile, parseJsonlLine } from '@main/utils/jsonl'; -import { getProjectsBasePath, getTodosBasePath } from '@main/utils/pathDecoder'; +import { LoopDetector, StallDetector } from '@main/utils/loopDetection'; +import { extractProjectName, getProjectsBasePath, getTodosBasePath } from '@main/utils/pathDecoder'; import { createLogger } from '@shared/utils/logger'; import { EventEmitter } from 'events'; import * as fs from 'fs'; @@ -21,6 +22,7 @@ import * as path from 'path'; import { projectPathResolver } from '../discovery/ProjectPathResolver'; import { type ProjectScanner } from '../discovery/ProjectScanner'; import { errorDetector } from '../error/ErrorDetector'; +import { createDetectedError } from '../error/ErrorMessageBuilder'; import { ConfigManager } from './ConfigManager'; import { type DataCache } from './DataCache'; @@ -86,6 +88,9 @@ export class FileWatcher extends EventEmitter { private processingInProgress = new Set(); /** Files that need reprocessing after current processing completes */ private pendingReprocess = new Set(); + /** Live tool-call loop detection state, fed from detectErrorsInSessionFile */ + private loopDetector = new LoopDetector(); + private stallDetector = new StallDetector(); /** Flag to prevent reuse after disposal */ private disposed = false; @@ -202,6 +207,8 @@ export class FileWatcher extends EventEmitter { this.activeSessionFiles.clear(); this.processingInProgress.clear(); this.pendingReprocess.clear(); + this.loopDetector.resetAll(); + this.stallDetector.resetAll(); logger.info('Stopped watching'); } @@ -653,6 +660,8 @@ export class FileWatcher extends EventEmitter { processedSize = lastSize + appended.consumedBytes; } else { // Fallback for first-read, truncation, or rewrite scenarios + this.loopDetector.reset(filePath); + this.stallDetector.reset(filePath); const messages = await parseJsonlFile(filePath); currentLineCount = messages.length; newMessages = messages.slice(lastLineCount); @@ -686,6 +695,73 @@ export class FileWatcher extends EventEmitter { await this.notificationManager.addError(error); } + // Live tool-call loop detection — stateful, main sessions only (agent + // files arrive with subagentId and are excluded), incremental appends + // only: a first-read/catch-up batch replays whole-file history and + // would re-notify loops that already ended (live detector, not a + // report — the CLI analyzers cover the offline case). + const loopCfg = ConfigManager.getInstance().getConfig().notifications.loopDetection; + logger.debug( + `loop gate ${path.basename(filePath)}: enabled=${loopCfg.enabled} ` + + `threshold=${loopCfg.cycleThreshold} ` + + `incremental=${canUseIncrementalAppend} newMessages=${newMessages.length} ` + + `notificationManager=${this.notificationManager ? 'set' : 'null'}` + ); + if ( + loopCfg.enabled && + canUseIncrementalAppend && + !subagentId && + !path.basename(filePath).startsWith('agent-') + ) { + const incident = this.loopDetector.feed(filePath, newMessages, loopCfg.cycleThreshold); + const incidentText = incident ? `${incident.key} x${incident.count}` : 'none'; + logger.debug(`loop feed ${path.basename(filePath)}: incident=${incidentText}`); + if (incident) { + await this.notificationManager.addError( + createDetectedError({ + sessionId, + projectId, + filePath, + projectName: extractProjectName(projectId, incident.cwd), + // approximate — deep link targets toolUseId, line is a fallback + lineNumber: lastLineCount + incident.batchIndex + 1, + source: 'loop', + message: `${incident.key} ×${incident.count} — possible stuck loop`, + timestamp: new Date(), + cwd: incident.cwd, + toolUseId: incident.toolUseId || undefined, + triggerName: 'Loop detected', + }) + ); + } + + // Context-stall detection — the tool-call counterpart of the key-based + // walk above: rounds that make calls yet stop growing the context + // (echo-marker loops with distinct args). Same gate, same threshold. + const stallIncident = this.stallDetector.feed( + filePath, + newMessages, + loopCfg.cycleThreshold + ); + if (stallIncident) { + await this.notificationManager.addError( + createDetectedError({ + sessionId, + projectId, + filePath, + projectName: extractProjectName(projectId, stallIncident.cwd), + lineNumber: lastLineCount + stallIncident.batchIndex + 1, + source: 'loop', + message: `${stallIncident.key} ×${stallIncident.count} — context not growing (echo-marker loop)`, + timestamp: new Date(), + cwd: stallIncident.cwd, + toolUseId: stallIncident.toolUseId || undefined, + triggerName: 'Stall detected', + }) + ); + } + } + // Update the last processed line count this.lastProcessedLineCount.set(filePath, currentLineCount); this.lastProcessedSize.set(filePath, processedSize); @@ -716,6 +792,8 @@ export class FileWatcher extends EventEmitter { this.lastProcessedLineCount.delete(filePath); this.lastProcessedSize.delete(filePath); this.activeSessionFiles.delete(filePath); + this.loopDetector.reset(filePath); + this.stallDetector.reset(filePath); } /** @@ -725,6 +803,8 @@ export class FileWatcher extends EventEmitter { this.lastProcessedLineCount.clear(); this.lastProcessedSize.clear(); this.activeSessionFiles.clear(); + this.loopDetector.resetAll(); + this.stallDetector.resetAll(); } /** @@ -914,12 +994,60 @@ export class FileWatcher extends EventEmitter { * Only checks files modified within the last hour. */ private async runCatchUpScan(): Promise { - if (!this.notificationManager || this.activeSessionFiles.size === 0) { + if (!this.notificationManager) { return; } const now = Date.now(); + // Discovery sweep: fs.watch can drop events for brand-new files (macOS + // coalesces directory creation and may deliver a null filename, which is + // discarded), and only event-seen files ever enter activeSessionFiles. + // Walk the projects tree for untracked session files so nothing is missed; + // stale files are evicted by the mtime guard in the loop below. + try { + const dirs = await this.fsProvider.readdir(this.projectsPath); + for (const dir of dirs) { + if (!dir.isDirectory()) continue; + let entries: FsDirent[]; + try { + entries = await this.fsProvider.readdir(path.join(this.projectsPath, dir.name)); + } catch { + continue; + } + for (const entry of entries) { + if (!entry.isFile() || !entry.name.endsWith('.jsonl')) continue; + if (entry.name.startsWith('agent-')) continue; + const fullPath = path.join(this.projectsPath, dir.name, entry.name); + if (this.activeSessionFiles.has(fullPath)) continue; + this.activeSessionFiles.set(fullPath, { + projectId: dir.name, + sessionId: path.basename(entry.name, '.jsonl'), + }); + // Baseline silently: the file's history predates this watcher, so + // the bell must only ring for calls that happen after discovery. + // Pin the size cursor; line count is a >0 placeholder — the byte + // offset is the real cursor for incremental appends. + try { + const observed = + typeof entry.size === 'number' + ? entry.size + : (await this.fsProvider.stat(fullPath)).size; + this.lastProcessedSize.set(fullPath, observed); + this.lastProcessedLineCount.set(fullPath, 1); + } catch { + this.activeSessionFiles.delete(fullPath); + } + } + } + } catch (err) { + logger.error('FileWatcher: Error discovering session files during catch-up:', err); + } + + if (this.activeSessionFiles.size === 0) { + return; + } + for (const [filePath, info] of this.activeSessionFiles) { try { const stats = await this.fsProvider.stat(filePath); diff --git a/src/main/services/infrastructure/NotificationManager.ts b/src/main/services/infrastructure/NotificationManager.ts index a92db953..90249e27 100644 --- a/src/main/services/infrastructure/NotificationManager.ts +++ b/src/main/services/infrastructure/NotificationManager.ts @@ -454,8 +454,14 @@ export class NotificationManager extends EventEmitter { // Deduplicate by toolUseId: the same tool call can appear in both the // subagent JSONL file and the parent session JSONL (as a progress event). // Keep the subagent-annotated version (with subagentId) when possible. - if (error.toolUseId) { - const existingIndex = this.notifications.findIndex((n) => n.toolUseId === error.toolUseId); + // Loop incidents (source 'loop') are exempt in both directions: they + // carry the run's latest call's toolUseId, which per-call error + // notifications also carry, but the two are different kinds — neither + // should suppress the other. + if (error.toolUseId && error.source !== 'loop') { + const existingIndex = this.notifications.findIndex( + (n) => n.toolUseId === error.toolUseId && n.source !== 'loop' + ); if (existingIndex !== -1) { const existing = this.notifications[existingIndex]; if (!existing.subagentId && error.subagentId) { diff --git a/src/main/services/parsing/MessageClassifier.ts b/src/main/services/parsing/MessageClassifier.ts index fd1aab93..7eb95757 100644 --- a/src/main/services/parsing/MessageClassifier.ts +++ b/src/main/services/parsing/MessageClassifier.ts @@ -38,8 +38,10 @@ export function classifyMessages(messages: ParsedMessage[]): ClassifiedMessage[] /** * Categorize a single message into one of five categories. + * Exported so lightweight scanners (e.g. analyzeSessionFileMetadata) apply + * the exact same rules as chunk building — counts must not drift. */ -function categorizeMessage(message: ParsedMessage): MessageCategory { +export function categorizeMessage(message: ParsedMessage): MessageCategory { // Check hard noise first (filtered out) if (isParsedHardNoiseMessage(message)) { return 'hardNoise'; diff --git a/src/main/types/domain.ts b/src/main/types/domain.ts index c50f1a44..fb35e588 100644 --- a/src/main/types/domain.ts +++ b/src/main/types/domain.ts @@ -93,6 +93,8 @@ export interface Session { updatedAt?: number; /** First user message text (for preview) */ firstMessage?: string; + /** Session name from /name (last agent-name line; ai-title fallback) */ + name?: string; /** Timestamp of first user message (RFC3339) */ messageTimestamp?: string; /** Whether this session has subagents */ @@ -111,6 +113,12 @@ export interface Session { compactionCount?: number; /** Per-phase token breakdown for tooltip display */ phaseBreakdown?: PhaseTokenBreakdown[]; + /** Total spend: sum of all assistant usage in the transcript (in+cache+out) */ + totalTokens?: number; + /** Completed user→assistant exchanges (turns / AI groups) */ + turnCount?: number; + /** Worktree name tag set by the renderer when listing a whole repository */ + worktreeName?: string; } /** diff --git a/src/main/types/jsonl.ts b/src/main/types/jsonl.ts index 6435a707..34c4f7ed 100644 --- a/src/main/types/jsonl.ts +++ b/src/main/types/jsonl.ts @@ -18,7 +18,9 @@ type EntryType = | 'system' | 'summary' | 'file-history-snapshot' - | 'queue-operation'; + | 'queue-operation' + | 'agent-name' + | 'ai-title'; type ContentType = 'text' | 'thinking' | 'tool_use' | 'tool_result' | 'image'; @@ -209,13 +211,29 @@ export interface QueueOperationEntry extends BaseEntry { operation: string; } +/** Session name set via /name — rewritten (last wins) as the session evolves. */ +export interface AgentNameEntry extends BaseEntry { + type: 'agent-name'; + agentName: string; + sessionId: string; +} + +/** Auto-generated session title — fallback display name when /name is unset. */ +export interface AiTitleEntry extends BaseEntry { + type: 'ai-title'; + aiTitle: string; + sessionId: string; +} + export type ChatHistoryEntry = | UserEntry | AssistantEntry | SystemEntry | SummaryEntry | FileHistorySnapshotEntry - | QueueOperationEntry; + | QueueOperationEntry + | AgentNameEntry + | AiTitleEntry; /** * Conversational entries - entries that represent chat messages. diff --git a/src/main/types/messages.ts b/src/main/types/messages.ts index cc745c0d..170f0749 100644 --- a/src/main/types/messages.ts +++ b/src/main/types/messages.ts @@ -105,6 +105,8 @@ export interface ParsedMessage { isCompactSummary?: boolean; /** API request ID for deduplicating streaming entries */ requestId?: string; + /** API message ID (one request = one id, even when the proxy streams it as several lines) */ + messageId?: string; } // ============================================================================= diff --git a/src/main/utils/jsonl.ts b/src/main/utils/jsonl.ts index 8c90181e..23a0b0fb 100644 --- a/src/main/utils/jsonl.ts +++ b/src/main/utils/jsonl.ts @@ -13,6 +13,7 @@ import * as readline from 'readline'; import { SessionContentFilter } from '../services/discovery/SessionContentFilter'; import { LocalFileSystemProvider } from '../services/infrastructure/LocalFileSystemProvider'; +import { categorizeMessage } from '../services/parsing/MessageClassifier'; import { type ChatHistoryEntry, type ContentBlock, @@ -78,7 +79,11 @@ export async function parseJsonlFile( } } - return messages; + // One API request = one message: a proxy streaming one line per content + // block produces fragments (same message.id, no requestId) that must not + // reach any consumer as separate messages — they would inflate every + // accounting surface (rounds, spend, wait-loop ticks, search hits) + return mergeAssistantFragments(messages); } /** @@ -118,6 +123,7 @@ function parseChatHistoryEntry(entry: ChatHistoryEntry): ParsedMessage | null { let usage: TokenUsage | undefined; let model: string | undefined; let requestId: string | undefined; + let messageId: string | undefined; let cwd: string | undefined; let gitBranch: string | undefined; let agentId: string | undefined; @@ -157,6 +163,7 @@ function parseChatHistoryEntry(entry: ChatHistoryEntry): ParsedMessage | null { model = entry.message.model; agentId = entry.agentId; requestId = entry.requestId; + messageId = entry.message.id; } else if (entry.type === 'system') { isMeta = entry.isMeta ?? false; } @@ -190,6 +197,7 @@ function parseChatHistoryEntry(entry: ChatHistoryEntry): ParsedMessage | null { sourceToolAssistantUUID, toolUseResult, requestId, + messageId, }; } @@ -220,6 +228,16 @@ function parseMessageType(type?: string): MessageType | null { // Streaming Deduplication // ============================================================================= +/** + * Which key identifies one billed request on this transcript: the backend's + * requestId when present, else the API message.id (proxies that stream one + * line per content block omit requestId but repeat message.id). Every + * once-per-request accounting site keys on this. + */ +export function billedRequestKey(msg: ParsedMessage): string | undefined { + return msg.requestId ?? msg.messageId; +} + /** * Deduplicate streaming assistant entries by requestId. * @@ -231,10 +249,10 @@ function parseMessageType(type?: string): MessageType | null { * Returns a new array with only the last entry per requestId kept. */ export function deduplicateByRequestId(messages: ParsedMessage[]): ParsedMessage[] { - // Map from requestId -> index of last occurrence + // Map from billedRequestKey -> index of last occurrence const lastIndexByRequestId = new Map(); for (let i = 0; i < messages.length; i++) { - const rid = messages[i].requestId; + const rid = billedRequestKey(messages[i]); if (rid) { lastIndexByRequestId.set(rid, i); } @@ -246,11 +264,63 @@ export function deduplicateByRequestId(messages: ParsedMessage[]): ParsedMessage } return messages.filter((msg, i) => { - if (!msg.requestId) return true; - return lastIndexByRequestId.get(msg.requestId) === i; + const key = billedRequestKey(msg); + if (!key) return true; + return lastIndexByRequestId.get(key) === i; }); } +/** Fragment content concat: block arrays concatenate, strings join directly + * (proxy fragment lines partition one response's text); a string beside a + * block array becomes a text block, so neither side is dropped. */ +// eslint-disable-next-line sonarjs/function-return-type -- preserves the input shape by contract: string fragments stay string +function concatContent( + a: ParsedMessage['content'], + b: ParsedMessage['content'] +): ParsedMessage['content'] { + const aBlocks = typeof a === 'string' ? [{ type: 'text' as const, text: a }] : (a ?? []); + const bBlocks = typeof b === 'string' ? [{ type: 'text' as const, text: b }] : (b ?? []); + return typeof a === 'string' && typeof b === 'string' ? a + b : [...aBlocks, ...bBlocks]; +} + +/** + * Merge assistant streaming fragment lines into one message per API request. + * + * Some backends (OpenAI-compatible proxies) write one JSONL line per content + * block of a response, each line carrying the same message.id and the same + * full usage — the lines PARTITION the response. Lines without requestId that + * share a messageId are therefore concatenated here: one request becomes one + * message, one accounting round, one stream divider. Lines with requestId are + * streaming snapshots (the last line holds the final state) and stay on the + * keep-last dedupe path of deduplicateByRequestId. + */ +export function mergeAssistantFragments(messages: ParsedMessage[]): ParsedMessage[] { + const firstIndexByMessageId = new Map(); + const result: ParsedMessage[] = []; + for (const msg of messages) { + if (msg.type !== 'assistant' || msg.requestId || !msg.messageId) { + result.push(msg); + continue; + } + const messageId = msg.messageId; + const firstIdx = firstIndexByMessageId.get(messageId); + if (firstIdx === undefined) { + firstIndexByMessageId.set(messageId, result.length); + result.push(msg); + continue; + } + const first = result[firstIdx]; + result[firstIdx] = { + ...first, + content: concatContent(first.content, msg.content), + toolCalls: [...first.toolCalls, ...msg.toolCalls], + toolResults: [...first.toolResults, ...msg.toolResults], + usage: msg.usage ?? first.usage, + }; + } + return result; +} + // ============================================================================= // Metrics Calculation // ============================================================================= @@ -341,6 +411,8 @@ export function getTaskCalls(messages: ParsedMessage[]): ToolCall[] { export interface SessionFileMetadata { firstUserMessage: { text: string; timestamp: string } | null; + /** /name session name — last agent-name wins, ai-title as fallback; null if unnamed */ + name?: string | null; messageCount: number; isOngoing: boolean; gitBranch: string | null; @@ -350,6 +422,10 @@ export interface SessionFileMetadata { compactionCount?: number; /** Per-phase token breakdown */ phaseBreakdown?: PhaseTokenBreakdown[]; + /** Total spend: sum of all assistant usage in this transcript (in+cache+out) */ + totalTokens: number; + /** AI response groups — same count as the "Turn N" chips in the chat */ + turnCount: number; hasDisplayableContent: boolean; } @@ -367,6 +443,8 @@ export async function analyzeSessionFileMetadata( messageCount: 0, isOngoing: false, gitBranch: null, + totalTokens: 0, + turnCount: 0, hasDisplayableContent: false, }; } @@ -379,10 +457,17 @@ export async function analyzeSessionFileMetadata( let firstUserMessage: { text: string; timestamp: string } | null = null; let firstCommandMessage: { text: string; timestamp: string } | null = null; + // /name session name: last agent-name wins, ai-title as fallback + let lastName: string | null = null; + const lastAiTitle: string | null = null; let messageCount = 0; let hasDisplayableContent = false; // After a UserGroup, await the first main-thread assistant message to count the AIGroup let awaitingAIGroup = false; + // Turn counting mirrors ChunkBuilder.buildChunks exactly: an AI run closed by + // a user/system/compact boundary (or EOF) == one AI group == one "Turn N". + let aiRunOpen = false; + let turnCount = 0; let gitBranch: string | null = null; let activityIndex = 0; @@ -399,6 +484,13 @@ export async function analyzeSessionFileMetadata( let awaitingPostCompaction = false; + // Total spend: every assistant round in this transcript (input + cache + output), + // billed once per request — streaming writes several lines per request, + // each carrying the full usage (requestId when the backend provides one, + // otherwise the API message.id) + let totalTokens = 0; + const billedRequests = new Set(); + for await (const line of rl) { const trimmed = line.trim(); if (!trimmed) { @@ -412,6 +504,13 @@ export async function analyzeSessionFileMetadata( continue; } + // /name session name — non-conversational lines the entry parser drops, + // so capture from the raw entry before that parser sees it + const entryName = sessionNameFromEntry(entry); + if (entryName) { + lastName = entryName; + } + const parsed = parseChatHistoryEntry(entry); if (!parsed) { continue; @@ -436,6 +535,19 @@ export async function analyzeSessionFileMetadata( awaitingAIGroup = false; } + // Same rules as the chunk pipeline: sidechain never reaches the main + // thread (SessionParser splits it out), hardNoise is skipped entirely, + // a user/system/compact boundary closes the current AI run. + if (!parsed.isSidechain) { + const category = categorizeMessage(parsed); + if (category === 'ai') { + aiRunOpen = true; + } else if (category !== 'hardNoise' && aiRunOpen) { + turnCount++; + aiRunOpen = false; + } + } + if (!gitBranch && 'gitBranch' in entry && entry.gitBranch) { gitBranch = entry.gitBranch; } @@ -569,6 +681,20 @@ export async function analyzeSessionFileMetadata( } } + // Total spend: sum every assistant usage block (sidechain included — this + // is the cost of the transcript), synthetic lines carry no usage + if (parsed.type === 'assistant' && parsed.usage) { + const requestKey = parsed.requestId ?? parsed.messageId; + if (!requestKey || !billedRequests.has(requestKey)) { + if (requestKey) billedRequests.add(requestKey); + totalTokens += + (parsed.usage.input_tokens ?? 0) + + (parsed.usage.cache_read_input_tokens ?? 0) + + (parsed.usage.cache_creation_input_tokens ?? 0) + + (parsed.usage.output_tokens ?? 0); + } + } + // Context consumption: detect compaction events if (parsed.isCompactSummary) { compactionPhases.push({ pre: lastMainAssistantInputTokens, post: 0 }); @@ -635,14 +761,55 @@ export async function analyzeSessionFileMetadata( } } + if (aiRunOpen) { + turnCount++; + } + return { firstUserMessage: firstUserMessage ?? firstCommandMessage, + name: lastName ?? lastAiTitle, messageCount, isOngoing: lastEndingIndex === -1 ? hasAnyOngoingActivity : hasActivityAfterLastEnding, gitBranch, contextConsumption, compactionCount: compactionPhases.length > 0 ? compactionPhases.length : undefined, phaseBreakdown, + totalTokens, + turnCount, hasDisplayableContent, }; } + +/** /name session name from a raw entry — agent-name wins over ai-title. */ +function sessionNameFromEntry(entry: ChatHistoryEntry): string | null { + if (entry.type === 'agent-name') return entry.agentName; + if (entry.type === 'ai-title') return entry.aiTitle; + return null; +} + +/** + * Read a session's /name (last agent-name, ai-title as fallback) in one + * streaming pass. The LAST name line wins, so the file is scanned to EOF — + * callers should cache the result (search caches it via SearchTextCache). + */ +export async function readSessionName( + filePath: string, + fsProvider: FileSystemProvider = defaultProvider +): Promise { + let name: string | null = null; + const fileStream = fsProvider.createReadStream(filePath, { encoding: 'utf8' }); + const rl = readline.createInterface({ input: fileStream, crlfDelay: Infinity }); + for await (const line of rl) { + const trimmed = line.trim(); + if (!trimmed) continue; + let entry: ChatHistoryEntry; + try { + entry = JSON.parse(trimmed) as ChatHistoryEntry; + } catch { + continue; + } + const entryName = sessionNameFromEntry(entry); + if (entryName) name = entryName; + } + return name; +} diff --git a/src/main/utils/loopDetection.ts b/src/main/utils/loopDetection.ts new file mode 100644 index 00000000..cdc8dac8 --- /dev/null +++ b/src/main/utils/loopDetection.ts @@ -0,0 +1,206 @@ +/** + * LoopDetector — live detection of tool-call loops in growing session files. + * + * Fed incrementally by FileWatcher with appended messages. Flags maximal + * back-to-back runs of identical tool calls, keyed like the inventory's + * cycles (bashStem(normalizeCallKey)) so `git show X | wc -l` variants and + * offset-only Read repeats bucket as one loop. Notification policy: first + * incident when a run reaches `threshold`, re-notify when it doubles — a + * 7-hour loop escalates (4 → 8 → 16 …) instead of pinging every round. + */ + +import { type ParsedMessage } from '@main/types'; +import { billedRequestKey } from '@main/utils/jsonl'; +import { isStalledRound } from '@shared/constants/loopPolicy'; +import { bashStem, normalizeCallKey } from '@shared/utils/callKey'; + +export interface LoopIncident { + /** normalized key of the looping call */ + key: string; + /** current run length */ + count: number; + /** toolUseId of the run's latest call — deep-link target */ + toolUseId: string; + /** cwd of the last fed message carrying one, for project naming */ + cwd?: string; + /** index of the incident's message within this batch (lineNumber is approximate) */ + batchIndex: number; +} + +interface FileLoopState { + lastKey: string; + streak: number; + lastToolUseId: string; + /** streak length at last notification; 0 = not yet notified for this run */ + notifiedCount: number; + cwd?: string; +} + +const freshState = (): FileLoopState => ({ + lastKey: '', + streak: 0, + lastToolUseId: '', + notifiedCount: 0, +}); + +export class LoopDetector { + private perFile = new Map(); + + /** Drop state — file was truncated/rewritten, counters no longer describe it. */ + reset(filePath: string): void { + this.perFile.delete(filePath); + } + + /** Drop state for every file (full tracking cleanup). */ + resetAll(): void { + this.perFile.clear(); + } + + /** + * Feed one batch of appended messages. Returns an incident when the current + * run reaches `threshold` for the first time or doubles the last notified + * length; null otherwise. + */ + feed(filePath: string, messages: ParsedMessage[], threshold: number): LoopIncident | null { + const state = this.perFile.get(filePath) ?? freshState(); + this.perFile.set(filePath, state); + + // One incident per batch (the first): keep scanning to the end so + // streak/lastToolUseId stay accurate — FileWatcher marks the whole + // batch processed, so an early return would strand the remainder. + let incident: LoopIncident | null = null; + + for (let i = 0; i < messages.length; i++) { + const msg = messages[i]; + // main-chain assistant lines only — same accounting as the inventory scan + if (msg.type !== 'assistant' || msg.isSidechain || msg.model === '') continue; + if (msg.cwd) state.cwd = msg.cwd; + + for (const call of msg.toolCalls) { + // ponytail: snapshot dedup by consecutive toolUseId only — real repeats + // always carry fresh ids (verified on loop forensics); add a Set if + // non-adjacent same-id lines ever show up + if (call.id && call.id === state.lastToolUseId) continue; + const key = bashStem(normalizeCallKey(call.name, call.input ?? {})); + if (key === state.lastKey) { + state.streak += 1; + } else { + state.lastKey = key; + state.streak = 1; + state.notifiedCount = 0; + } + state.lastToolUseId = call.id ?? ''; + if ( + !incident && + state.streak >= threshold && + (state.notifiedCount === 0 || state.streak >= state.notifiedCount * 2) + ) { + state.notifiedCount = state.streak; + incident = { + key, + count: state.streak, + toolUseId: call.id, + cwd: state.cwd, + batchIndex: i, + }; + } + } + } + return incident; + } +} + +interface FileStallState { + lastContext: number; + /** billedRequestKey of the last processed request — GLM-proxy fragment dedup */ + lastRequestKey?: string; + streak: number; + lastToolUseId: string; + /** streak length at last notification; 0 = not yet notified for this run */ + notifiedCount: number; + cwd?: string; +} + +const freshStallState = (): FileStallState => ({ + lastContext: 0, + streak: 0, + lastToolUseId: '', + notifiedCount: 0, +}); + +/** + * StallDetector — live detection of context-stall loops: rounds that MAKE + * tool calls while the context stops growing (echo-marker loops like + * `echo w/v/u` — distinct args, so the key-based LoopDetector above sees no + * streak). Criterion shared with the CLI/renderer: isStalledRound. GLM-proxy + * fragments of one request are billed once (billedRequestKey dedup) — raw + * appended lines are not merged at parse time. Same notification policy as + * LoopDetector: first incident at `threshold`, re-notify on doubling. + */ +export class StallDetector { + private perFile = new Map(); + + /** Drop state — file was truncated/rewritten, counters no longer describe it. */ + reset(filePath: string): void { + this.perFile.delete(filePath); + } + + /** Drop state for every file (full tracking cleanup). */ + resetAll(): void { + this.perFile.clear(); + } + + /** Same contract as LoopDetector.feed. */ + feed(filePath: string, messages: ParsedMessage[], threshold: number): LoopIncident | null { + const state = this.perFile.get(filePath) ?? freshStallState(); + this.perFile.set(filePath, state); + + let incident: LoopIncident | null = null; + + for (let i = 0; i < messages.length; i++) { + const msg = messages[i]; + if (msg.type !== 'assistant' || msg.isSidechain || msg.model === '') continue; + if (msg.cwd) state.cwd = msg.cwd; + // one request streamed as several lines (each with the full usage) is + // one round — skipping the later fragments keeps the streak honest + const requestKey = billedRequestKey(msg); + if (requestKey && requestKey === state.lastRequestKey) continue; + if (requestKey) state.lastRequestKey = requestKey; + + const u = msg.usage; + if (!u) continue; + const context = + (u.input_tokens ?? 0) + + (u.cache_read_input_tokens ?? 0) + + (u.cache_creation_input_tokens ?? 0); + const output = u.output_tokens ?? 0; + const stalled = isStalledRound(state.lastContext, context, output, msg.toolCalls.length); + // ghost rounds must not drag the baseline — same rule as buildLedger + if (context > 0) state.lastContext = context; + + const toolUseId = msg.toolCalls[msg.toolCalls.length - 1]?.id ?? ''; + if (stalled) { + state.streak += 1; + state.lastToolUseId = toolUseId; + if ( + !incident && + state.streak >= threshold && + (state.notifiedCount === 0 || state.streak >= state.notifiedCount * 2) + ) { + state.notifiedCount = state.streak; + incident = { + key: 'context stall', + count: state.streak, + toolUseId, + cwd: state.cwd, + batchIndex: i, + }; + } + } else { + state.streak = 0; + state.notifiedCount = 0; + } + } + return incident; + } +} diff --git a/src/main/utils/metadataExtraction.ts b/src/main/utils/metadataExtraction.ts index 92777f6d..ce9d8209 100644 --- a/src/main/utils/metadataExtraction.ts +++ b/src/main/utils/metadataExtraction.ts @@ -9,23 +9,12 @@ import * as readline from 'readline'; import { LocalFileSystemProvider } from '../services/infrastructure/LocalFileSystemProvider'; import { type ChatHistoryEntry, isTextContent, type UserEntry } from '../types'; -import { translateWslMountPath } from './pathDecoder'; +import { normalizeDriveLetter, translateWslMountPath } from './pathDecoder'; import type { FileSystemProvider } from '../services/infrastructure/FileSystemProvider'; const logger = createLogger('Util:metadataExtraction'); -/** - * Normalize Windows drive letter to uppercase for consistent path comparison. - * CLI uses uppercase (C:\...) while VS Code extension uses lowercase (c:\...). - */ -function normalizeDriveLetter(p: string): string { - if (p.length >= 2 && p[1] === ':') { - return p[0].toUpperCase() + p.slice(1); - } - return p; -} - const defaultProvider = new LocalFileSystemProvider(); interface MessagePreview { diff --git a/src/main/utils/pathDecoder.ts b/src/main/utils/pathDecoder.ts index 7f82dbe9..d175c175 100644 --- a/src/main/utils/pathDecoder.ts +++ b/src/main/utils/pathDecoder.ts @@ -94,6 +94,17 @@ export function extractProjectName(encodedName: string, cwdHint?: string): strin return segments[segments.length - 1] || encodedName; } +/** + * Normalize Windows drive letter to uppercase for consistent path comparison. + * CLI uses uppercase (C:\...) while VS Code extension uses lowercase (c:\...). + */ +export function normalizeDriveLetter(p: string): string { + if (p.length >= 2 && p[1] === ':') { + return p[0].toUpperCase() + p.slice(1); + } + return p; +} + /** * Translate WSL mount paths (/mnt/X/...) to Windows drive-letter paths (X:/...) * when running on Windows. No-op on other platforms. diff --git a/src/main/utils/toolExtraction.ts b/src/main/utils/toolExtraction.ts index de963083..1ce71bc6 100644 --- a/src/main/utils/toolExtraction.ts +++ b/src/main/utils/toolExtraction.ts @@ -17,7 +17,9 @@ export function extractToolCalls(content: ContentBlock[] | string): ToolCall[] { for (const block of content) { if (block.type === 'tool_use' && block.id && block.name) { const input = block.input ?? {}; - const isTask = block.name === 'Task'; + // Claude Code 2.1.63 renamed the Task tool to Agent (same input schema; + // docs: code.claude.com/docs/en/sub-agents.md) + const isTask = block.name === 'Task' || block.name === 'Agent'; const toolCall: ToolCall = { id: block.id, diff --git a/src/renderer/components/chat/AIChatGroup.tsx b/src/renderer/components/chat/AIChatGroup.tsx index 7bf24642..8fde02da 100644 --- a/src/renderer/components/chat/AIChatGroup.tsx +++ b/src/renderer/components/chat/AIChatGroup.tsx @@ -4,9 +4,10 @@ import { COLOR_TEXT_MUTED, COLOR_TEXT_SECONDARY } from '@renderer/constants/cssV import { useTabUI } from '@renderer/hooks/useTabUI'; import { useStore } from '@renderer/store'; import { enhanceAIGroup, type PrecedingSlashInfo } from '@renderer/utils/aiGroupEnhancer'; +import { TOOL_HIGHLIGHT_CLASSES, type TriggerColor } from '@shared/constants/triggerColors'; import { extractSlashInfo, isCommandContent } from '@shared/utils/contentSanitizer'; import { getModelColorClass } from '@shared/utils/modelParser'; -import { estimateTokens } from '@shared/utils/tokenFormatting'; +import { estimateTokens, formatTokensCompact } from '@shared/utils/tokenFormatting'; import { format } from 'date-fns'; import { Bot, ChevronDown, Clock } from 'lucide-react'; import { useShallow } from 'zustand/react/shallow'; @@ -24,7 +25,6 @@ import type { EnhancedAIGroup, UserGroup, } from '@renderer/types/groups'; -import type { TriggerColor } from '@shared/constants/triggerColors'; /** * Extract slash info from a UserGroup's message content. @@ -83,6 +83,10 @@ interface AIChatGroupProps { highlightToolUseId?: string; /** Custom highlight color from trigger */ highlightColor?: TriggerColor; + /** Red alarm on the header row (burn pills): aggregate is the alarm target */ + isHeaderHighlighted?: boolean; + /** Red tint on the whole turn's stream: this group is the red navigation target */ + isBodyHighlighted?: boolean; /** Register ref for individual tool items (for precise scroll targeting) */ registerToolRef?: (toolId: string, el: HTMLElement | null) => void; } @@ -124,6 +128,8 @@ const AIChatGroupInner = ({ aiGroup, highlightToolUseId, highlightColor, + isHeaderHighlighted, + isBodyHighlighted, registerToolRef, }: Readonly): React.JSX.Element => { // Per-tab UI state for expansion (completely isolated per tab) @@ -388,7 +394,10 @@ const AIChatGroupInner = ({ }; return ( -
+
{/* Header Row */} {hasToggleContent && (
@@ -413,6 +422,17 @@ const AIChatGroupInner = ({ Claude + {/* Turn number — matches the Visible Context panel's 1-based "Turn N" */} + + Turn {aiGroup.turnIndex + 1} + + {/* Main agent model */} {enhanced.mainModel && ( @@ -450,7 +470,14 @@ const AIChatGroupInner = ({
{/* Right side: Context badge, Token usage, Timestamp (non-clickable) */} -
+
+ {/* Burn pills: this turn's loop/wait-loop waste */} + {contextStats && } + {/* Context injection badge (CLAUDE.md, mentioned files, tool outputs) */} {contextStats && } @@ -509,6 +536,7 @@ const AIChatGroupInner = ({ highlightColor={highlightColor} notificationColorMap={notificationColorMap} registerToolRef={registerToolRef} + roundFlags={contextStats?.roundFlags} />
)} @@ -527,3 +555,45 @@ const AIChatGroupInner = ({ }; export const AIChatGroup = React.memo(AIChatGroupInner); + +/** Red waste pills for this turn: Wait-loop quiet-round re-read + Loop repeats. */ +const BurnPills = ({ stats }: Readonly<{ stats: ContextStats }>): React.ReactElement | null => { + let waitTokens = 0; + let waitRounds = 0; + let loopTokens = 0; + for (const inj of stats.newInjections) { + if (inj.category === 'wait-loop') { + waitTokens += inj.estimatedTokens; + waitRounds += inj.roundCount; + } else if (inj.category === 'loop') { + loopTokens += inj.estimatedTokens; + } + } + if (waitTokens === 0 && loopTokens === 0) return null; + + const pillStyle: React.CSSProperties = { + backgroundColor: 'rgba(239, 68, 68, 0.15)', + color: '#f87171', + }; + + return ( + + {waitTokens > 0 && ( + + Wait {formatTokensCompact(waitTokens)} · {waitRounds} rd + + )} + {loopTokens > 0 && ( + + Loop {formatTokensCompact(loopTokens)} + + )} + + ); +}; diff --git a/src/renderer/components/chat/ChatHistory.tsx b/src/renderer/components/chat/ChatHistory.tsx index 33159f14..6dabf465 100644 --- a/src/renderer/components/chat/ChatHistory.tsx +++ b/src/renderer/components/chat/ChatHistory.tsx @@ -1,5 +1,6 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react'; +import { isGroupHeaderAlarm } from '@renderer/hooks/navigation/utils'; import { isNearBottom, useAutoScrollBottom } from '@renderer/hooks/useAutoScrollBottom'; import { useTabNavigationController } from '@renderer/hooks/useTabNavigationController'; import { useTabUI } from '@renderer/hooks/useTabUI'; @@ -173,6 +174,10 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { const [isNavigationHighlight, setIsNavigationHighlight] = useState(false); const navigationHighlightTimerRef = useRef | null>(null); + // Red header flash (aggregate burn pills) — Loop/Wait-loop panel navigation + const [headerFlashGroupId, setHeaderFlashGroupId] = useState(null); + const headerFlashTimerRef = useRef | null>(null); + // Refs map for AI groups, chat items, and individual tool items (for scrolling) const aiGroupRefs = useRef>(new Map()); const chatItemRefs = useRef>(new Map()); @@ -212,7 +217,7 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { }); const ensureGroupVisible = useCallback( - async (groupId: string) => { + async (groupId: string, align: 'start' | 'center' | 'end' | 'auto' = 'center') => { if (!shouldVirtualize) { return; } @@ -220,7 +225,7 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { if (index === undefined) { return; } - rowVirtualizer.scrollToIndex(index, { align: 'center' }); + rowVirtualizer.scrollToIndex(index, { align }); // Wait 2 RAF frames so the virtualizer has time to render the target row await waitForDoubleRaf(); }, @@ -261,10 +266,34 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { // Local tool highlight for context panel navigation (separate from controller) const [contextNavToolUseId, setContextNavToolUseId] = useState(null); + // Red body target: Wait/Loop entries keep the whole turn tinted until the + // next such navigation (unlike the 2s header flash) + const [bodyHighlightGroupId, setBodyHighlightGroupId] = useState(null); const effectiveHighlightToolUseId = controllerToolUseId ?? contextNavToolUseId ?? undefined; // Use blue for context panel tool navigation, otherwise use controller's color const effectiveHighlightColor = contextNavToolUseId ? ('blue' as const) : highlightColor; + // Red alarm on the turn header: 2s panel flash or persistent loop-notification + // alarm (red group navigation without a tool target) + const isHeaderHighlightedFor = (groupId: string): boolean => + headerFlashGroupId === groupId || + isGroupHeaderAlarm(groupId, { + highlightedGroupId, + highlightColor, + highlightToolUseId: effectiveHighlightToolUseId, + }); + + // Red body: the whole turn's stream stays tinted while it remains the red + // group navigation target (Wait/Loop entries) — not a flash, clears when + // another navigation lands + const isBodyHighlightedFor = (groupId: string): boolean => + bodyHighlightGroupId === groupId || + isGroupHeaderAlarm(groupId, { + highlightedGroupId, + highlightColor, + highlightToolUseId: effectiveHighlightToolUseId, + }); + // Keep search match indices aligned with this tab's rendered conversation. // This avoids stale/global match lists after tab switches or in-place refreshes. useEffect(() => { @@ -401,7 +430,7 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { // Handler to navigate to a specific turn (AI group) from CLAUDE.md panel const handleNavigateToTurn = useCallback( - (turnIndex: number) => { + (turnIndex: number, opts?: { flashHeader?: boolean }) => { if (!conversation) return; const targetItem = conversation.items.find( (item) => item.type === 'ai' && item.group.turnIndex === turnIndex @@ -410,13 +439,28 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { const run = async (): Promise => { const groupId = targetItem.group.id; - await ensureGroupVisible(groupId); + // navigating to a turn means reading it — expand the collapsed group + expandAIGroup(groupId); + // the header with burn pills must be on screen — align the group top + await ensureGroupVisible(groupId, opts?.flashHeader ? 'start' : 'center'); const element = aiGroupRefs.current.get(groupId); if (!element) return; - element.scrollIntoView({ behavior: 'smooth', block: 'center' }); + element.scrollIntoView({ + behavior: 'smooth', + block: opts?.flashHeader ? 'start' : 'center', + }); setHighlightedGroupId(groupId); setIsNavigationHighlight(true); + if (opts?.flashHeader) { + setBodyHighlightGroupId(groupId); + setHeaderFlashGroupId(groupId); + if (headerFlashTimerRef.current) clearTimeout(headerFlashTimerRef.current); + headerFlashTimerRef.current = setTimeout(() => { + setHeaderFlashGroupId(null); + headerFlashTimerRef.current = null; + }, 2000); + } if (navigationHighlightTimerRef.current) { clearTimeout(navigationHighlightTimerRef.current); } @@ -428,7 +472,7 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { }; void run(); }, - [conversation, ensureGroupVisible, setHighlightedGroupId] + [conversation, ensureGroupVisible, expandAIGroup, setHighlightedGroupId] ); // Handler to navigate to a user message group (preceding the AI group at turnIndex) @@ -817,6 +861,8 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { isSearchHighlight={isSearchHighlight} isNavigationHighlight={isNavigationHighlight} highlightColor={effectiveHighlightColor} + isHeaderHighlighted={isHeaderHighlightedFor(item.group.id)} + isBodyHighlighted={isBodyHighlightedFor(item.group.id)} registerChatItemRef={registerChatItemRef} registerAIGroupRef={registerAIGroupRefCombined} registerToolRef={registerToolRef} @@ -835,6 +881,8 @@ export const ChatHistory = ({ tabId }: ChatHistoryProps): JSX.Element => { isSearchHighlight={isSearchHighlight} isNavigationHighlight={isNavigationHighlight} highlightColor={effectiveHighlightColor} + isHeaderHighlighted={isHeaderHighlightedFor(item.group.id)} + isBodyHighlighted={isBodyHighlightedFor(item.group.id)} registerChatItemRef={registerChatItemRef} registerAIGroupRef={registerAIGroupRefCombined} registerToolRef={registerToolRef} diff --git a/src/renderer/components/chat/ChatHistoryItem.tsx b/src/renderer/components/chat/ChatHistoryItem.tsx index 77733ca3..f848ce83 100644 --- a/src/renderer/components/chat/ChatHistoryItem.tsx +++ b/src/renderer/components/chat/ChatHistoryItem.tsx @@ -21,6 +21,9 @@ interface ChatHistoryItemProps { readonly isSearchHighlight: boolean; readonly isNavigationHighlight: boolean; readonly highlightColor?: TriggerColor; + /** Red alarm on the AI group header (burn pills): 2s panel flash or persistent loop alarm */ + readonly isHeaderHighlighted?: boolean; + readonly isBodyHighlighted?: boolean; readonly registerChatItemRef: (groupId: string) => (el: HTMLElement | null) => void; readonly registerAIGroupRef: (groupId: string) => (el: HTMLElement | null) => void; /** Register ref for individual tool items (for precise scroll targeting) */ @@ -54,6 +57,8 @@ const ChatHistoryItemInner = ({ isSearchHighlight, isNavigationHighlight, highlightColor, + isHeaderHighlighted, + isBodyHighlighted, registerChatItemRef, registerAIGroupRef, registerToolRef, @@ -117,6 +122,8 @@ const ChatHistoryItemInner = ({ aiGroup={item.group} highlightToolUseId={toolUseIdForGroup} highlightColor={highlightColor} + isHeaderHighlighted={isHeaderHighlighted} + isBodyHighlighted={isBodyHighlighted} registerToolRef={registerToolRef} />
diff --git a/src/renderer/components/chat/ContextBadge.tsx b/src/renderer/components/chat/ContextBadge.tsx index 2b4d5991..f133dced 100644 --- a/src/renderer/components/chat/ContextBadge.tsx +++ b/src/renderer/components/chat/ContextBadge.tsx @@ -25,11 +25,13 @@ import type { ClaudeMdContextInjection, ContextInjection, ContextStats, + LoopInjection, MentionedFileInjection, TaskCoordinationInjection, ThinkingTextInjection, ToolOutputInjection, UserMessageInjection, + WaitLoopInjection, } from '@renderer/types/contextInjection'; interface ContextBadgeProps { @@ -79,6 +81,20 @@ function isUserMessageInjection(inj: ContextInjection): inj is UserMessageInject return inj.category === 'user-message'; } +/** + * Type guard for LoopInjection. + */ +function isLoopInjection(inj: ContextInjection): inj is LoopInjection { + return inj.category === 'loop'; +} + +/** + * Type guard for WaitLoopInjection. + */ +function isWaitLoopInjection(inj: ContextInjection): inj is WaitLoopInjection { + return inj.category === 'wait-loop'; +} + /** * Section component for expandable groups in the popover. */ @@ -148,7 +164,9 @@ export const ContextBadge = ({ stats.newCounts.toolOutputs + stats.newCounts.thinkingText + stats.newCounts.taskCoordination + - stats.newCounts.userMessages, + stats.newCounts.userMessages + + stats.newCounts.loop + + stats.newCounts.waitLoop, [stats.newCounts] ); @@ -183,6 +201,16 @@ export const ContextBadge = ({ [stats.newInjections] ); + const newLoopInjections = useMemo( + () => stats.newInjections.filter(isLoopInjection), + [stats.newInjections] + ); + + const newWaitLoopInjections = useMemo( + () => stats.newInjections.filter(isWaitLoopInjection), + [stats.newInjections] + ); + // Calculate total new tokens const totalNewTokens = useMemo( () => stats.newInjections.reduce((sum, inj) => sum + inj.estimatedTokens, 0), @@ -231,6 +259,26 @@ export const ContextBadge = ({ [newUserMessageInjections] ); + const loopTokens = useMemo( + () => newLoopInjections.reduce((sum, inj) => sum + inj.estimatedTokens, 0), + [newLoopInjections] + ); + + const loopItemCount = useMemo( + () => newLoopInjections.reduce((sum, inj) => sum + inj.breakdown.length, 0), + [newLoopInjections] + ); + + const waitLoopTokens = useMemo( + () => newWaitLoopInjections.reduce((sum, inj) => sum + inj.estimatedTokens, 0), + [newWaitLoopInjections] + ); + + const waitLoopRoundCount = useMemo( + () => newWaitLoopInjections.reduce((sum, inj) => sum + inj.roundCount, 0), + [newWaitLoopInjections] + ); + // Linear-style neutral badge — uses theme-aware CSS variables const badgeStyle: React.CSSProperties = { backgroundColor: COLOR_SURFACE_RAISED, @@ -531,6 +579,47 @@ export const ContextBadge = ({ )} + {/* Loop section */} + {newLoopInjections.length > 0 && ( + + {newLoopInjections.map((injection) => + injection.breakdown.map((item) => ( +
+ + {item.key} ×{item.count} + + + ~{formatTokens(item.tokenCount)} tokens + +
+ )) + )} +
+ )} + + {/* Wait-loop section */} + {newWaitLoopInjections.length > 0 && ( + + {newWaitLoopInjections.map((injection) => ( +
+ + Turn {injection.turnIndex + 1} · {injection.roundCount} quiet rounds + + + ~{formatTokens(injection.estimatedTokens)} tokens + +
+ ))} +
+ )} + {/* Thinking + Text section */} {newThinkingTextInjections.length > 0 && ( ; /** Optional callback to register tool element refs for scroll targeting */ registerToolRef?: (toolId: string, el: HTMLDivElement | null) => void; + /** Per-round flags (quiet/repeat/billed) keyed by response uuid — drives round dividers */ + roundFlags?: Map; } /** @@ -70,6 +73,7 @@ export const DisplayItemList = React.memo(function DisplayItemList({ highlightColor, notificationColorMap, registerToolRef, + roundFlags, }: Readonly): React.JSX.Element { // Reply-link highlight: when hovering a reply badge, dim everything except the linked pair const [replyLinkToolId, setReplyLinkToolId] = useState(null); @@ -95,9 +99,52 @@ export const DisplayItemList = React.memo(function DisplayItemList({ ); } + // Round dividers: the first item of each round (roundId) emits a divider + // chip between line segments — red for burn rounds (quiet/repeat/stall), + // neutral for normal rounds. Precomputed once per render — no render-time mutation. + const roundDividers = new Map(); + for (const item of items) { + const rid = 'roundId' in item ? (item.roundId ?? null) : null; + if (!rid || roundDividers.has(rid)) continue; + const flag = roundFlags?.get(rid); + if (!flag) continue; + const isBurn = flag.quiet || flag.repeat || flag.stalled; + const label = flag.quiet + ? `R${flag.index} · quiet · ${formatTokensCompact(flag.billed)}` + : flag.repeat + ? `R${flag.index} · repeat` + : flag.stalled + ? `R${flag.index} · stall · ${formatTokensCompact(flag.billed)}` + : `R${flag.index}`; + roundDividers.set( + rid, +
+
+ + {label} + +
+
+ ); + } + return (
{items.map((item, index) => { + const rid = 'roundId' in item ? (item.roundId ?? null) : null; + const divider = rid ? (roundDividers.get(rid) ?? null) : null; let itemKey = ''; let element: React.ReactNode = null; @@ -326,16 +373,18 @@ export const DisplayItemList = React.memo(function DisplayItemList({ // Apply reply-link spotlight: dim items not in the highlighted pair const isDimmed = replyLinkToolId !== null && !isItemInReplyLink(item); return ( -
- {element} -
+ + {divider} +
+ {element} +
+
); })}
diff --git a/src/renderer/components/chat/SessionContextPanel/components/FlatInjectionList.tsx b/src/renderer/components/chat/SessionContextPanel/components/FlatInjectionList.tsx index cb2d9e39..1555e334 100644 --- a/src/renderer/components/chat/SessionContextPanel/components/FlatInjectionList.tsx +++ b/src/renderer/components/chat/SessionContextPanel/components/FlatInjectionList.tsx @@ -25,6 +25,8 @@ const CATEGORY_COLORS: Record diff --git a/src/renderer/components/chat/SessionContextPanel/components/LoopSection.tsx b/src/renderer/components/chat/SessionContextPanel/components/LoopSection.tsx new file mode 100644 index 00000000..578cead6 --- /dev/null +++ b/src/renderer/components/chat/SessionContextPanel/components/LoopSection.tsx @@ -0,0 +1,121 @@ +/** + * LoopSection - Section for displaying repeat-call (loop) injections. + * Each key row expands into the rounds that carried the repeats. + */ + +import React, { Fragment, useState } from 'react'; + +import { ChevronDown, ChevronRight } from 'lucide-react'; + +import { CollapsibleSection } from './CollapsibleSection'; + +import type { LoopInjection } from '@renderer/types/contextInjection'; + +interface LoopSectionProps { + injections: LoopInjection[]; + tokenCount: number; + isExpanded: boolean; + onToggle: () => void; + onNavigateToTool?: (turnIndex: number, toolUseId: string) => void; + onNavigateToTurn?: (turnIndex: number, opts?: { flashHeader?: boolean }) => void; +} + +export const LoopSection = ({ + injections, + tokenCount, + isExpanded, + onToggle, + onNavigateToTool, + onNavigateToTurn, +}: Readonly): React.ReactElement | null => { + const [expandedKeys, setExpandedKeys] = useState>(new Set()); + + const toggleKey = (key: string): void => { + setExpandedKeys((prev) => { + const next = new Set(prev); + if (next.has(key)) { + next.delete(key); + } else { + next.add(key); + } + return next; + }); + }; + + if (injections.length === 0) return null; + + return ( + + {injections.map((injection) => + injection.breakdown.map((item) => { + const expandKey = `${injection.id}-${item.key}`; + const isOpen = expandedKeys.has(expandKey); + const keyRounds = injection.rounds.filter((round) => round.keys.includes(item.key)); + return ( + +
+ + +
+ {isOpen && + keyRounds.map((round) => ( +
+ R{round.index} + repeat + {round.billed.toLocaleString()} tok +
+ ))} +
+ ); + }) + )} +
+ ); +}; diff --git a/src/renderer/components/chat/SessionContextPanel/components/RankedInjectionList.tsx b/src/renderer/components/chat/SessionContextPanel/components/RankedInjectionList.tsx index c7666541..4408aeab 100644 --- a/src/renderer/components/chat/SessionContextPanel/components/RankedInjectionList.tsx +++ b/src/renderer/components/chat/SessionContextPanel/components/RankedInjectionList.tsx @@ -28,6 +28,8 @@ const CATEGORY_COLORS: Record void; + onNavigateToTurn?: (turnIndex: number, opts?: { flashHeader?: boolean }) => void; +} + +export const WaitLoopSection = ({ + injections, + tokenCount, + isExpanded, + onToggle, + onNavigateToTurn, +}: Readonly): React.ReactElement | null => { + const [expandedIds, setExpandedIds] = useState>(new Set()); + + const toggleExpanded = (id: string): void => { + setExpandedIds((prev) => { + const next = new Set(prev); + if (next.has(id)) { + next.delete(id); + } else { + next.add(id); + } + return next; + }); + }; + + if (injections.length === 0) return null; + + return ( + + {injections.map((injection) => { + const isOpen = expandedIds.has(injection.id); + return ( + +
+ + +
+ {isOpen && + injection.rounds.map((round) => ( +
+ R{round.index} + quiet + out {round.outputTokens.toLocaleString()} + {round.billed.toLocaleString()} tok +
+ ))} +
+ ); + })} +
+ ); +}; diff --git a/src/renderer/components/chat/SessionContextPanel/index.tsx b/src/renderer/components/chat/SessionContextPanel/index.tsx index 19f8cc79..569aee76 100644 --- a/src/renderer/components/chat/SessionContextPanel/index.tsx +++ b/src/renderer/components/chat/SessionContextPanel/index.tsx @@ -14,6 +14,7 @@ import { import { ClaudeMdFilesSection } from './components/ClaudeMdFilesSection'; import { FlatInjectionList } from './components/FlatInjectionList'; +import { LoopSection } from './components/LoopSection'; import { MentionedFilesSection } from './components/MentionedFilesSection'; import { RankedInjectionList } from './components/RankedInjectionList'; import { SessionContextHeader } from './components/SessionContextHeader'; @@ -21,23 +22,28 @@ import { TaskCoordinationSection } from './components/TaskCoordinationSection'; import { ThinkingTextSection } from './components/ThinkingTextSection'; import { ToolOutputsSection } from './components/ToolOutputsSection'; import { UserMessagesSection } from './components/UserMessagesSection'; +import { WaitLoopSection } from './components/WaitLoopSection'; import { SECTION_CLAUDE_MD, + SECTION_LOOP, SECTION_MENTIONED_FILES, SECTION_TASK_COORDINATION, SECTION_THINKING_TEXT, SECTION_TOOL_OUTPUTS, SECTION_USER_MESSAGES, + SECTION_WAIT_LOOP, } from './types'; import type { ContextViewMode, SectionType, SessionContextPanelProps } from './types'; import type { ClaudeMdContextInjection, + LoopInjection, MentionedFileInjection, TaskCoordinationInjection, ThinkingTextInjection, ToolOutputInjection, UserMessageInjection, + WaitLoopInjection, } from '@renderer/types/contextInjection'; export const SessionContextPanel = ({ @@ -66,6 +72,8 @@ export const SessionContextPanel = ({ SECTION_TOOL_OUTPUTS, SECTION_TASK_COORDINATION, SECTION_THINKING_TEXT, + SECTION_LOOP, + SECTION_WAIT_LOOP, ]) ); @@ -77,6 +85,8 @@ export const SessionContextPanel = ({ thinkingTextInjections, taskCoordinationInjections, userMessageInjections, + loopInjections, + waitLoopInjections, } = useMemo(() => { const claudeMd: ClaudeMdContextInjection[] = []; const mentionedFiles: MentionedFileInjection[] = []; @@ -84,6 +94,8 @@ export const SessionContextPanel = ({ const thinkingText: ThinkingTextInjection[] = []; const taskCoordination: TaskCoordinationInjection[] = []; const userMessages: UserMessageInjection[] = []; + const loop: LoopInjection[] = []; + const waitLoop: WaitLoopInjection[] = []; for (const injection of injections) { switch (injection.category) { @@ -105,6 +117,12 @@ export const SessionContextPanel = ({ case 'user-message': userMessages.push(injection); break; + case 'loop': + loop.push(injection); + break; + case 'wait-loop': + waitLoop.push(injection); + break; } } @@ -117,6 +135,9 @@ export const SessionContextPanel = ({ thinkingText.sort((a, b) => a.turnIndex - b.turnIndex); // Sort user messages by turn index ascending userMessages.sort((a, b) => a.turnIndex - b.turnIndex); + // Loops and wait-loops: biggest burn first + loop.sort((a, b) => b.estimatedTokens - a.estimatedTokens); + waitLoop.sort((a, b) => b.estimatedTokens - a.estimatedTokens); return { claudeMdInjections: claudeMd, @@ -125,6 +146,8 @@ export const SessionContextPanel = ({ thinkingTextInjections: thinkingText, taskCoordinationInjections: taskCoordination, userMessageInjections: userMessages, + loopInjections: loop, + waitLoopInjections: waitLoop, }; }, [injections]); @@ -165,6 +188,16 @@ export const SessionContextPanel = ({ [userMessageInjections] ); + const loopTokens = useMemo( + () => loopInjections.reduce((sum, inj) => sum + inj.estimatedTokens, 0), + [loopInjections] + ); + + const waitLoopTokens = useMemo( + () => waitLoopInjections.reduce((sum, inj) => sum + inj.estimatedTokens, 0), + [waitLoopInjections] + ); + // Toggle section expansion const toggleSection = (section: SectionType): void => { setExpandedSections((prev) => { @@ -258,6 +291,23 @@ export const SessionContextPanel = ({ onToggle={() => toggleSection(SECTION_THINKING_TEXT)} onNavigateToTurn={onNavigateToTurn} /> + + toggleSection(SECTION_LOOP)} + onNavigateToTool={onNavigateToTool} + onNavigateToTurn={onNavigateToTurn} + /> + + toggleSection(SECTION_WAIT_LOOP)} + onNavigateToTurn={onNavigateToTurn} + /> ) : ( <> diff --git a/src/renderer/components/chat/SessionContextPanel/types.ts b/src/renderer/components/chat/SessionContextPanel/types.ts index b5222683..f6a2dc02 100644 --- a/src/renderer/components/chat/SessionContextPanel/types.ts +++ b/src/renderer/components/chat/SessionContextPanel/types.ts @@ -16,8 +16,8 @@ export interface SessionContextPanelProps { onClose?: () => void; /** Project root for relative path display */ projectRoot?: string; - /** Click Turn N to navigate to that turn */ - onNavigateToTurn?: (turnIndex: number) => void; + /** Click Turn N to navigate to that turn; flashHeader lights the group header (burn pills) */ + onNavigateToTurn?: (turnIndex: number, opts?: { flashHeader?: boolean }) => void; /** Navigate to a specific tool within a turn by toolUseId */ onNavigateToTool?: (turnIndex: number, toolUseId: string) => void; /** Navigate to the user message group preceding the AI group at turnIndex */ @@ -43,6 +43,8 @@ export const SECTION_TOOL_OUTPUTS = 'tool-outputs' as const; export const SECTION_THINKING_TEXT = 'thinking-text' as const; export const SECTION_TASK_COORDINATION = 'task-coordination' as const; export const SECTION_USER_MESSAGES = 'user-messages' as const; +export const SECTION_LOOP = 'loop' as const; +export const SECTION_WAIT_LOOP = 'wait-loop' as const; /** Section identifiers for collapsible panels */ export type SectionType = @@ -51,7 +53,9 @@ export type SectionType = | typeof SECTION_TOOL_OUTPUTS | typeof SECTION_THINKING_TEXT | typeof SECTION_TASK_COORDINATION - | typeof SECTION_USER_MESSAGES; + | typeof SECTION_USER_MESSAGES + | typeof SECTION_LOOP + | typeof SECTION_WAIT_LOOP; /** View mode for the context panel */ export type ContextViewMode = 'category' | 'ranked'; diff --git a/src/renderer/components/common/TokenUsageDisplay.tsx b/src/renderer/components/common/TokenUsageDisplay.tsx index 5ad96066..070afc27 100644 --- a/src/renderer/components/common/TokenUsageDisplay.tsx +++ b/src/renderer/components/common/TokenUsageDisplay.tsx @@ -83,6 +83,8 @@ const SessionContextSection = ({ const toolOutputsCount = counts.toolOutputs; const taskCoordinationCount = counts.taskCoordination; const userMessagesCount = counts.userMessages; + const loopCount = counts.loop; + const waitLoopCount = counts.waitLoop; // Calculate percentages for each category const claudeMdPercent = @@ -107,6 +109,12 @@ const SessionContextSection = ({ totalTokens > 0 ? Math.min((tokensByCategory.userMessages / totalTokens) * 100, 100).toFixed(1) : '0.0'; + const loopPercent = + totalTokens > 0 ? Math.min((tokensByCategory.loop / totalTokens) * 100, 100).toFixed(1) : '0.0'; + const waitLoopPercent = + totalTokens > 0 + ? Math.min((tokensByCategory.waitLoop / totalTokens) * 100, 100).toFixed(1) + : '0.0'; return (
@@ -195,6 +203,32 @@ const SessionContextSection = ({
)} + {/* Loop (repeat calls) */} + {tokensByCategory.loop > 0 && ( +
+ + Loop ×{loopCount} + + + {formatTokens(tokensByCategory.loop)}{' '} + ({loopPercent}%) + +
+ )} + + {/* Wait-loop (quiet rounds) */} + {tokensByCategory.waitLoop > 0 && ( +
+ + Wait-loop ×{waitLoopCount} + + + {formatTokens(tokensByCategory.waitLoop)}{' '} + ({waitLoopPercent}%) + +
+ )} + {/* User Messages */} {tokensByCategory.userMessages > 0 && (
diff --git a/src/renderer/components/dashboard/DashboardView.tsx b/src/renderer/components/dashboard/DashboardView.tsx index a427bb54..7fb9f205 100644 --- a/src/renderer/components/dashboard/DashboardView.tsx +++ b/src/renderer/components/dashboard/DashboardView.tsx @@ -13,6 +13,7 @@ import { api } from '@renderer/api'; import { useStore } from '@renderer/store'; import { formatShortcut } from '@renderer/utils/stringUtils'; import { createLogger } from '@shared/utils/logger'; +import { formatTokensCompact } from '@shared/utils/tokenFormatting'; import { useShallow } from 'zustand/react/shallow'; const logger = createLogger('Component:DashboardView'); @@ -76,7 +77,11 @@ const CommandSearch = ({ value, onChange }: Readonly): React
); @@ -261,15 +310,25 @@ const ProjectsGrid = ({ searchQuery, maxProjects = 12, }: Readonly): React.JSX.Element => { - const { repositoryGroups, repositoryGroupsLoading, fetchRepositoryGroups, selectRepository } = - useStore( - useShallow((s) => ({ - repositoryGroups: s.repositoryGroups, - repositoryGroupsLoading: s.repositoryGroupsLoading, - fetchRepositoryGroups: s.fetchRepositoryGroups, - selectRepository: s.selectRepository, - })) - ); + const { + repositoryGroups, + repositoryGroupsLoading, + repositorySpend, + repositoryTurns, + repositoryLastName, + fetchRepositoryGroups, + selectRepository, + } = useStore( + useShallow((s) => ({ + repositoryGroups: s.repositoryGroups, + repositoryGroupsLoading: s.repositoryGroupsLoading, + repositorySpend: s.repositorySpend, + repositoryTurns: s.repositoryTurns, + repositoryLastName: s.repositoryLastName, + fetchRepositoryGroups: s.fetchRepositoryGroups, + selectRepository: s.selectRepository, + })) + ); useEffect(() => { if (repositoryGroups.length === 0) { @@ -380,6 +439,9 @@ const ProjectsGrid = ({ selectRepository(repo.id)} isHighlighted={!!searchQuery.trim()} /> diff --git a/src/renderer/components/settings/SettingsView.tsx b/src/renderer/components/settings/SettingsView.tsx index ee4f0400..4c63c1d6 100644 --- a/src/renderer/components/settings/SettingsView.tsx +++ b/src/renderer/components/settings/SettingsView.tsx @@ -143,6 +143,8 @@ export const SettingsView = (): React.JSX.Element | null => { ignoredRepositoryItems={ignoredRepositoryItems} excludedRepositoryIds={excludedRepositoryIds} onNotificationToggle={handlers.handleNotificationToggle} + onLoopDetectionChange={handlers.handleLoopDetectionChange} + onTurnBudgetChange={handlers.handleTurnBudgetChange} onSnooze={handlers.handleSnooze} onClearSnooze={handlers.handleClearSnooze} onAddIgnoredRepository={handlers.handleAddIgnoredRepository} diff --git a/src/renderer/components/settings/hooks/useSettingsConfig.ts b/src/renderer/components/settings/hooks/useSettingsConfig.ts index 192346ae..730a8098 100644 --- a/src/renderer/components/settings/hooks/useSettingsConfig.ts +++ b/src/renderer/components/settings/hooks/useSettingsConfig.ts @@ -42,6 +42,8 @@ export interface SafeConfig { snoozeMinutes: number; includeSubagentErrors: boolean; triggers: AppConfig['notifications']['triggers']; + loopDetection: { enabled: boolean; cycleThreshold: number }; + turnBudget: { enabled: boolean; maxInputTokensPerTurn: number }; }; display: { showTimestamps: boolean; @@ -168,6 +170,15 @@ export function useSettingsConfig(): UseSettingsConfigReturn { snoozeMinutes: displayConfig?.notifications?.snoozeMinutes ?? 30, includeSubagentErrors: displayConfig?.notifications?.includeSubagentErrors ?? true, triggers: displayConfig?.notifications?.triggers ?? [], + loopDetection: { + enabled: displayConfig?.notifications?.loopDetection?.enabled ?? true, + cycleThreshold: displayConfig?.notifications?.loopDetection?.cycleThreshold ?? 4, + }, + turnBudget: { + enabled: displayConfig?.notifications?.turnBudget?.enabled ?? true, + maxInputTokensPerTurn: + displayConfig?.notifications?.turnBudget?.maxInputTokensPerTurn ?? 15_000_000, + }, }, display: { showTimestamps: displayConfig?.display?.showTimestamps ?? true, diff --git a/src/renderer/components/settings/hooks/useSettingsHandlers.ts b/src/renderer/components/settings/hooks/useSettingsHandlers.ts index 617e2c45..ec10c929 100644 --- a/src/renderer/components/settings/hooks/useSettingsHandlers.ts +++ b/src/renderer/components/settings/hooks/useSettingsHandlers.ts @@ -34,6 +34,8 @@ interface SettingsHandlers { // Notification handlers handleNotificationToggle: (key: keyof AppConfig['notifications'], value: boolean) => void; + handleLoopDetectionChange: (value: { enabled: boolean; cycleThreshold: number }) => void; + handleTurnBudgetChange: (value: { enabled: boolean; maxInputTokensPerTurn: number }) => void; handleSnooze: (minutes: number) => Promise; handleClearSnooze: () => Promise; handleAddIgnoredRepository: (item: RepositoryDropdownItem) => Promise; @@ -96,6 +98,22 @@ export function useSettingsHandlers({ [updateConfig] ); + // Loop detection: nested config field — merge with the current subobject + const handleLoopDetectionChange = useCallback( + (value: { enabled: boolean; cycleThreshold: number }) => { + void updateConfig('notifications', { loopDetection: value }); + }, + [updateConfig] + ); + + // Turn budget: nested config field, read by the PreToolUse hook directly + const handleTurnBudgetChange = useCallback( + (value: { enabled: boolean; maxInputTokensPerTurn: number }) => { + void updateConfig('notifications', { turnBudget: value }); + }, + [updateConfig] + ); + const handleSnooze = useCallback( async (minutes: number) => { try { @@ -280,6 +298,8 @@ export function useSettingsHandlers({ snoozeMinutes: 30, includeSubagentErrors: true, triggers: defaultTriggers, + loopDetection: { enabled: true, cycleThreshold: 4 }, + turnBudget: { enabled: true, maxInputTokensPerTurn: 15_000_000 }, }, general: { launchAtLogin: false, @@ -377,6 +397,8 @@ export function useSettingsHandlers({ handleThemeChange, handleDefaultTabChange, handleNotificationToggle, + handleLoopDetectionChange, + handleTurnBudgetChange, handleSnooze, handleClearSnooze, handleAddIgnoredRepository, diff --git a/src/renderer/components/settings/sections/NotificationsSection.tsx b/src/renderer/components/settings/sections/NotificationsSection.tsx index 5a3829b5..996e526e 100644 --- a/src/renderer/components/settings/sections/NotificationsSection.tsx +++ b/src/renderer/components/settings/sections/NotificationsSection.tsx @@ -33,6 +33,8 @@ interface NotificationsSectionProps { key: 'enabled' | 'soundEnabled' | 'includeSubagentErrors', value: boolean ) => void; + readonly onLoopDetectionChange: (value: { enabled: boolean; cycleThreshold: number }) => void; + readonly onTurnBudgetChange: (value: { enabled: boolean; maxInputTokensPerTurn: number }) => void; readonly onSnooze: (minutes: number) => Promise; readonly onClearSnooze: () => Promise; readonly onAddIgnoredRepository: (item: RepositoryDropdownItem) => Promise; @@ -52,6 +54,8 @@ export const NotificationsSection = ({ ignoredRepositoryItems, excludedRepositoryIds, onNotificationToggle, + onLoopDetectionChange, + onTurnBudgetChange, onSnooze, onClearSnooze, onAddIgnoredRepository, @@ -100,6 +104,94 @@ export const NotificationsSection = ({ disabled={saving || !safeConfig.notifications.enabled} /> + + {/* Live loop detection */} + + + + onLoopDetectionChange({ + ...safeConfig.notifications.loopDetection, + enabled: v, + }) + } + disabled={saving || !safeConfig.notifications.enabled} + /> + + +
+ + {safeConfig.notifications.loopDetection.cycleThreshold}× + + { + const n = parseInt(e.target.value, 10); + if (Number.isInteger(n) && n >= 1) { + onLoopDetectionChange({ + ...safeConfig.notifications.loopDetection, + cycleThreshold: n, + }); + } + }} + disabled={saving || !safeConfig.notifications.loopDetection.enabled} + className="w-20 rounded-md border border-claude-dark-border bg-surface px-2 py-1 text-sm text-claude-dark-text" + /> +
+
+ + {/* Per-turn input budget (PreToolUse hook) */} + + + + onTurnBudgetChange({ + ...safeConfig.notifications.turnBudget, + enabled: v, + }) + } + disabled={saving || !safeConfig.notifications.enabled} + /> + + +
+ { + const n = parseInt(e.target.value, 10); + if (Number.isInteger(n) && n >= 100000) { + onTurnBudgetChange({ + ...safeConfig.notifications.turnBudget, + maxInputTokensPerTurn: n, + }); + } + }} + disabled={saving || !safeConfig.notifications.turnBudget.enabled} + className="w-32 rounded-md border border-claude-dark-border bg-surface px-2 py-1 text-sm text-claude-dark-text" + /> +
+
+ - {session.firstMessage ?? 'Untitled'} + {session.name ?? session.firstMessage ?? 'Untitled'}
- {/* Second line: message count + time + context consumption */} + {/* Second line: messages · turns · spend · time · context · worktree — units as text, icon-only metrics proved unreadable */}
- - - {session.messageCount} - + {session.messageCount} msg + {session.turnCount != null && session.turnCount > 0 && ( + <> + · + + {formatTokensCompact(session.turnCount)} turns + + + )} + {session.totalTokens != null && session.totalTokens > 0 && ( + <> + · + + {formatTokensCompact(session.totalTokens)} + + + )} · {formatShortTime( @@ -299,6 +316,18 @@ export const SessionItem = React.memo(function SessionItem({ /> )} + {session.worktreeName && ( + <> + · + + {session.worktreeName} + + + )}
diff --git a/src/renderer/hooks/navigation/utils.ts b/src/renderer/hooks/navigation/utils.ts index c5ce035f..69e860e2 100644 --- a/src/renderer/hooks/navigation/utils.ts +++ b/src/renderer/hooks/navigation/utils.ts @@ -6,6 +6,25 @@ */ import type { ChatItem } from '@renderer/types/groups'; +import type { TriggerColor } from '@shared/constants/triggerColors'; + +/** + * Header-level alarm policy: a red error-kind navigation WITHOUT a tool + * target (e.g. a loop-notification deep link) is a group-level defect — + * light the turn header (where the burn pills live), not one tool card. + */ +export function isGroupHeaderAlarm( + groupId: string, + nav: { + highlightedGroupId: string | null; + highlightColor?: TriggerColor | null; + highlightToolUseId?: string | null; + } +): boolean { + return ( + nav.highlightedGroupId === groupId && nav.highlightColor === 'red' && !nav.highlightToolUseId + ); +} // ============================================================================= // Target Resolution diff --git a/src/renderer/hooks/useTabNavigationController.ts b/src/renderer/hooks/useTabNavigationController.ts index 65274fea..72adbf2e 100644 --- a/src/renderer/hooks/useTabNavigationController.ts +++ b/src/renderer/hooks/useTabNavigationController.ts @@ -38,6 +38,15 @@ import type { SessionConversation } from '@renderer/types/groups'; import type { TabNavigationRequest } from '@renderer/types/tabs'; import type { TriggerColor } from '@shared/constants/triggerColors'; +/** + * Error navigation highlights are alarm state, not a flash: they persist + * until the detection itself is revoked or another navigation replaces them. + * Search (and other kinds) keep the auto-clear flash. + */ +export function isPersistentHighlight(kind: TabNavigationRequest['kind']): boolean { + return kind === 'error'; +} + // ============================================================================= // Types // ============================================================================= @@ -392,18 +401,26 @@ export function useTabNavigationController( if (abortController.signal.aborted) return; if (success) { - // Schedule highlight end - highlightTimerRef.current = setTimeout(() => { - if (!abortController.signal.aborted) { - // Clear search state if it was a search navigation - if (request.kind === 'search') { - setSearchQuery(''); + if (isPersistentHighlight(request.kind)) { + // Alarm semantics: keep the highlight until the detection is + // revoked or another navigation replaces it. Return to idle so + // auto-scroll keeps working in live sessions. + setPhase('idle'); + activeRequestIdRef.current = null; + } else { + // Schedule highlight end + highlightTimerRef.current = setTimeout(() => { + if (!abortController.signal.aborted) { + // Clear search state if it was a search navigation + if (request.kind === 'search') { + setSearchQuery(''); + } + handleHighlightEnd(); } - handleHighlightEnd(); - } - }, highlightDuration); + }, highlightDuration); - setPhase('complete'); + setPhase('complete'); + } } else { // Navigation failed - reset setPhase('idle'); diff --git a/src/renderer/store/index.ts b/src/renderer/store/index.ts index d40ddd31..2e057bd2 100644 --- a/src/renderer/store/index.ts +++ b/src/renderer/store/index.ts @@ -122,6 +122,12 @@ export function initializeNotificationListeners(): () => void { const timer = setTimeout(() => { pendingProjectRefreshTimers.delete(projectId); const state = useStore.getState(); + // Repo-wide sidebar: refresh the merged all-worktrees list in place — + // the single-worktree refresh would drop the other worktrees' sessions + if (state.selectedRepositoryId) { + void state.refreshRepositorySessionsInPlace(); + return; + } void state.refreshSessionsInPlace(projectId); }, PROJECT_REFRESH_DEBOUNCE_MS); pendingProjectRefreshTimers.set(projectId, timer); diff --git a/src/renderer/store/slices/notificationSlice.ts b/src/renderer/store/slices/notificationSlice.ts index 320c71a7..1556bb62 100644 --- a/src/renderer/store/slices/notificationSlice.ts +++ b/src/renderer/store/slices/notificationSlice.ts @@ -211,7 +211,8 @@ export const createNotificationSlice: StateCreator total spend summed over all worktrees' sessions (best-effort) */ + repositorySpend: Record; + /** Repo-id -> total turns summed over all worktrees' sessions (best-effort) */ + repositoryTurns: Record; + /** Repo-id -> /name of the most recently updated named session (best-effort) */ + repositoryLastName: Record; viewMode: 'flat' | 'grouped'; // Actions fetchRepositoryGroups: () => Promise; + /** Background per-repo spend + turns aggregation (memoized main-side) */ + fetchRepositoryStats: () => void; selectRepository: (repositoryId: string) => void; selectWorktree: (worktreeId: string) => void; setViewMode: (mode: 'flat' | 'grouped') => void; @@ -47,6 +55,9 @@ export const createRepositorySlice: StateCreator { + for (const repo of get().repositoryGroups) { + void (async () => { + try { + const perWorktree = await Promise.all( + repo.worktrees.map((worktree) => api.getSessions(worktree.id)) + ); + const sessions = perWorktree.flat(); + const total = sessions.reduce((sum, s) => sum + (s.totalTokens ?? 0), 0); + const turns = sessions.reduce((sum, s) => sum + (s.turnCount ?? 0), 0); + const lastNamed = sessions + .filter((s) => s.name) + .sort((a, b) => (b.updatedAt ?? 0) - (a.updatedAt ?? 0))[0]; + set((state) => ({ + repositorySpend: { ...state.repositorySpend, [repo.id]: total }, + repositoryTurns: { ...state.repositoryTurns, [repo.id]: turns }, + repositoryLastName: { + ...state.repositoryLastName, + [repo.id]: lastNamed?.name, + }, + })); + } catch { + // leave this repo's total absent — the card just omits it + } + })(); + } + }, + // Select a repository group and auto-select a worktree selectRepository: (repositoryId: string) => { const { repositoryGroups } = get(); @@ -90,8 +133,8 @@ export const createRepositorySlice: StateCreator + Math.max(b.updatedAt ?? b.createdAt, b.createdAt) - + Math.max(a.updatedAt ?? a.createdAt, a.createdAt) + ); +} + const logger = createLogger('Store:session'); /** @@ -46,6 +59,10 @@ export interface SessionSlice { // Actions fetchSessions: (projectId: string) => Promise; fetchSessionsInitial: (projectId: string) => Promise; + /** Whole-repository listing: sessions of every worktree, merged and sorted */ + fetchSessionsForRepository: (repo: RepositoryGroup) => Promise; + /** Silent repo-wide refresh on file-change — no loading wipe, unlike fetchSessionsForRepository */ + refreshRepositorySessionsInPlace: () => Promise; fetchSessionsMore: () => Promise; resetSessionsPagination: () => void; selectSession: (id: string) => void; @@ -170,6 +187,68 @@ export const createSessionSlice: StateCreator = } }, + // Whole-repository listing: sessions of every worktree, merged and sorted. + // Non-main worktree rows carry a worktreeName tag so the sidebar can show origin. + fetchSessionsForRepository: async (repo) => { + set({ + sessionsLoading: true, + sessionsError: null, + sessions: [], + sessionsCursor: null, + sessionsHasMore: false, + sessionsTotalCount: 0, + }); + try { + const perWorktree = await Promise.all( + repo.worktrees.map(async (worktree) => { + const sessions = await api.getSessions(worktree.id); + // tag only non-default worktrees — untagged rows read as "main" + const tag = worktree.isMainWorktree ? undefined : worktree.name; + return sessions.map((s) => (tag ? { ...s, worktreeName: tag } : s)); + }) + ); + const merged = mergeWorktreeSessions(perWorktree); + set({ + sessions: merged, + sessionsLoading: false, + sessionsTotalCount: merged.length, + }); + void get().loadPinnedSessions(); + void get().loadHiddenSessions(); + } catch (error) { + set({ + sessionsError: error instanceof Error ? error.message : 'Failed to fetch sessions', + sessionsLoading: false, + }); + } + }, + + // Silent repo-wide refresh for file-change: same merge as + // fetchSessionsForRepository, but without the loading wipe — otherwise the + // upstream single-worktree refreshSessionsInPlace would shrink the merged + // list back to one worktree on every session update. + refreshRepositorySessionsInPlace: async () => { + const currentState = get(); + const repo = currentState.repositoryGroups.find( + (r) => r.id === currentState.selectedRepositoryId + ); + if (!repo) return; + + try { + const perWorktree = await Promise.all( + repo.worktrees.map(async (worktree) => { + const sessions = await api.getSessions(worktree.id); + const tag = worktree.isMainWorktree ? undefined : worktree.name; + return sessions.map((s) => (tag ? { ...s, worktreeName: tag } : s)); + }) + ); + const merged = mergeWorktreeSessions(perWorktree); + set({ sessions: merged, sessionsTotalCount: merged.length }); + } catch (error) { + logger.error('refreshRepositorySessionsInPlace error:', error); + } + }, + // Fetch more sessions (next page) fetchSessionsMore: async () => { const state = get(); diff --git a/src/renderer/types/contextInjection.ts b/src/renderer/types/contextInjection.ts index 7ee2b388..e8c243c9 100644 --- a/src/renderer/types/contextInjection.ts +++ b/src/renderer/types/contextInjection.ts @@ -192,6 +192,96 @@ export interface TaskCoordinationInjection { breakdown: TaskCoordinationBreakdown[]; } +// ============================================================================= +// Loop Types +// ============================================================================= + +/** + * Breakdown of tokens contributed by a single looping tool-call series. + */ +export interface LoopTokenBreakdown { + /** Canonical call key (same identity the live loop detector uses) */ + key: string; + /** How many times this repeat streak has fired so far */ + count: number; + /** Estimated token count for this repeat call (call + result + skill) */ + tokenCount: number; + /** Tool use ID for deep-link navigation to the specific repeat call */ + toolUseId?: string; +} + +/** + * Represents tokens burned by repeated back-to-back identical tool calls. + * Calls 2..N of a streak land here; the first call stays in tool-output. + */ +export interface LoopInjection { + /** Unique identifier (e.g., "loop-ai-0") */ + id: string; + /** Discriminator for type narrowing */ + category: 'loop'; + /** Turn index where these repeats occurred */ + turnIndex: number; + /** AI group ID for navigation (e.g., "ai-0") */ + aiGroupId: string; + /** Total estimated tokens from repeat calls in this turn */ + estimatedTokens: number; + /** Detailed breakdown of tokens by repeat series */ + breakdown: LoopTokenBreakdown[]; + /** Rounds carrying repeat calls, one row each (billed usage) */ + rounds: LoopRoundInfo[]; +} + +/** One assistant round (response) inside a turn, for round-level displays */ +export interface LoopRoundInfo { + /** Response message uuid (matches SemanticStep.sourceMessageId) */ + uuid: string; + /** 1-based round number within the turn */ + index: number; + /** Billed usage of the round (in + cache_read + cache_creation + output) */ + billed: number; + /** Repeat call key(s) present in this round */ + keys: string[]; +} + +// ============================================================================= +// Wait-Loop Types +// ============================================================================= + +/** + * Represents tokens burned by quiet rounds in a turn: rounds that billed a + * huge input-side context (>= WAIT_TICK_CONTEXT_TOKENS) while producing almost + * nothing (<= WAIT_TICK_OUTPUT_TOKENS out). Same criterion as the CLI's + * wait_loop findings. + */ +export interface WaitLoopInjection { + /** Unique identifier (e.g., "wait-loop-ai-0") */ + id: string; + /** Discriminator for type narrowing */ + category: 'wait-loop'; + /** Turn index where these quiet rounds occurred */ + turnIndex: number; + /** AI group ID for navigation (e.g., "ai-0") */ + aiGroupId: string; + /** Total billed context re-read by quiet rounds in this turn */ + estimatedTokens: number; + /** How many quiet rounds fired in this turn */ + roundCount: number; + /** The quiet rounds themselves (for round-level expansion) */ + rounds: WaitRoundInfo[]; +} + +/** One quiet round: when it fired and what it billed */ +export interface WaitRoundInfo { + /** Response message uuid (matches SemanticStep.sourceMessageId) */ + uuid: string; + /** 1-based round number within the turn */ + index: number; + /** Output tokens of the round (always <= WAIT_TICK_OUTPUT_TOKENS) */ + outputTokens: number; + /** Full billed usage of the round (in + cache_read + cache_creation + output) */ + billed: number; +} + // ============================================================================= // Union Types // ============================================================================= @@ -217,7 +307,9 @@ export type ContextInjection = | ToolOutputInjection | ThinkingTextInjection | TaskCoordinationInjection - | UserMessageInjection; + | UserMessageInjection + | LoopInjection + | WaitLoopInjection; // ============================================================================= // Statistics Types @@ -239,6 +331,10 @@ export interface TokensByCategory { taskCoordination: number; /** Tokens from user messages */ userMessages: number; + /** Tokens from repeated back-to-back identical tool calls (loop waste) */ + loop: number; + /** Tokens re-read by quiet wait-loop rounds */ + waitLoop: number; } /** @@ -257,6 +353,10 @@ export interface NewCountsByCategory { taskCoordination: number; /** Count of new user message injections */ userMessages: number; + /** Count of new repeat-call entries */ + loop: number; + /** Count of quiet wait-loop rounds */ + waitLoop: number; } /** @@ -283,6 +383,22 @@ export interface ContextStats { accumulatedCounts: NewCountsByCategory; /** Which context phase this stats belongs to (1-based) */ phaseNumber?: number; + /** Per-round classification for the group's stream (keyed by response uuid) */ + roundFlags?: Map; +} + +/** Stream-level flag of one assistant round inside a turn */ +export interface RoundFlag { + /** 1-based round number within the turn */ + index: number; + /** Quiet round: billed a big context while producing almost nothing */ + quiet: boolean; + /** Round carries repeat tool calls */ + repeat: boolean; + /** Tool round that stopped growing the context (echo-marker loop shape) */ + stalled: boolean; + /** Full billed usage of the round (in + cache_read + cache_creation + output) */ + billed: number; } // ============================================================================= diff --git a/src/renderer/types/groups.ts b/src/renderer/types/groups.ts index 639d0436..5b80cc2e 100644 --- a/src/renderer/types/groups.ts +++ b/src/renderer/types/groups.ts @@ -248,13 +248,19 @@ export interface TeammateMessage { * These are flattened and shown in chronological order. */ export type AIGroupDisplayItem = - | { type: 'thinking'; content: string; timestamp: Date; tokenCount?: number } - | { type: 'tool'; tool: LinkedToolItem } + | { type: 'thinking'; content: string; timestamp: Date; tokenCount?: number; roundId?: string } + | { type: 'tool'; tool: LinkedToolItem; roundId?: string } | { type: 'subagent'; subagent: Process } - | { type: 'output'; content: string; timestamp: Date; tokenCount?: number } - | { type: 'slash'; slash: SlashItem } + | { type: 'output'; content: string; timestamp: Date; tokenCount?: number; roundId?: string } + | { type: 'slash'; slash: SlashItem; roundId?: string } | { type: 'teammate_message'; teammateMessage: TeammateMessage } - | { type: 'subagent_input'; content: string; timestamp: Date; tokenCount?: number } + | { + type: 'subagent_input'; + content: string; + timestamp: Date; + tokenCount?: number; + roundId?: string; + } | { type: 'compact_boundary'; content: string; diff --git a/src/renderer/utils/contextTracker.ts b/src/renderer/utils/contextTracker.ts index 15da9e94..db1d55f5 100644 --- a/src/renderer/utils/contextTracker.ts +++ b/src/renderer/utils/contextTracker.ts @@ -9,6 +9,8 @@ * This builds on claudeMdTracker.ts and extends it to track all context sources. */ +import { isQuietTick, isStalledRound } from '@shared/constants/loopPolicy'; +import { bashStem, normalizeCallKey } from '@shared/utils/callKey'; import { estimateTokens } from '@shared/utils/tokenFormatting'; import { MAX_MENTIONED_FILE_TOKENS } from '../types/contextInjection'; @@ -32,9 +34,13 @@ import type { ContextPhase, ContextPhaseInfo, ContextStats, + LoopInjection, + LoopRoundInfo, + LoopTokenBreakdown, MentionedFileInfo, MentionedFileInjection, NewCountsByCategory, + RoundFlag, TaskCoordinationBreakdown, TaskCoordinationInjection, ThinkingTextBreakdown, @@ -43,8 +49,10 @@ import type { ToolOutputInjection, ToolTokenBreakdown, UserMessageInjection, + WaitLoopInjection, + WaitRoundInfo, } from '../types/contextInjection'; -import type { ClaudeMdFileInfo } from '../types/data'; +import type { ClaudeMdFileInfo, ParsedMessage } from '../types/data'; import type { AIGroup, AIGroupDisplayItem, @@ -118,6 +126,33 @@ function generateUserMessageId(turnIndex: number): string { return `user-msg-ai-${turnIndex}`; } +/** + * Generate unique ID for loop injection. + */ +function generateLoopId(turnIndex: number): string { + return `loop-ai-${turnIndex}`; +} + +/** + * Generate unique ID for wait-loop injection. + */ +function generateWaitLoopId(turnIndex: number): string { + return `wait-loop-ai-${turnIndex}`; +} + +/** + * Loop streak state, threaded across AI groups: back-to-back identical calls + * (same identity the live loop detector uses) keep the streak alive. + */ +export interface LoopStreakState { + lastKey: string; + streak: number; +} + +export function createLoopStreakState(): LoopStreakState { + return { lastKey: '', streak: 0 }; +} + // ============================================================================= // Injection Wrapping Functions // ============================================================================= @@ -177,19 +212,32 @@ function createMentionedFileInjection( // ============================================================================= /** - * Aggregate tool outputs from all linked tools in a turn. - * Also includes tokens from user-invoked skills (via /skill-name commands). - * Returns a ToolOutputInjection if there are any tool outputs with tokens. + * Aggregate tool outputs from all linked tools in a turn, splitting repeat + * calls (2..N of a back-to-back identical series, same key as the live loop + * detector) into the loop bucket. Slash/skill items stay in tool-output. + * Advances loopState for every non-coordination call, even zero-token ones. */ function aggregateToolOutputs( linkedTools: Map, turnIndex: number, aiGroupId: string, - displayItems?: AIGroupDisplayItem[] -): ToolOutputInjection | null { + displayItems: AIGroupDisplayItem[] | undefined, + loopState: LoopStreakState, + responses: ParsedMessage[] +): { + toolOutput: ToolOutputInjection | null; + loop: LoopInjection | null; + loopState: LoopStreakState; + repeatKeys: Map; +} { const toolBreakdown: ToolTokenBreakdown[] = []; + const loopBreakdown = new Map(); let totalTokens = 0; - + let loopTokens = 0; + // tool-use id → repeat-call key; membership defines the repeat set + const keyByToolId = new Map(); + // copy — no-param-reassign; state is threaded back via the return value + const state = { ...loopState }; for (const linkedTool of linkedTools.values()) { // Skip task coordination tools - they are tracked separately if (TASK_COORDINATION_TOOL_NAMES.has(linkedTool.name)) { @@ -207,21 +255,47 @@ function aggregateToolOutputs( const skillTokens = linkedTool.skillInstructionsTokenCount ?? 0; const toolTokenCount = callTokens + resultTokens + skillTokens; + // Classify BEFORE the token check — the streak advances for every + // non-coordination call, even zero-token ones (same as live LoopDetector) + const key = bashStem(normalizeCallKey(linkedTool.name, linkedTool.input ?? {})); + let repeat = false; + if (key === state.lastKey) { + state.streak += 1; + repeat = true; + } else { + state.lastKey = key; + state.streak = 1; + } + if (toolTokenCount > 0) { - // Rename "Task" to "Task (Subagent)" for clarity in the UI - const displayName = linkedTool.name === 'Task' ? 'Task (Subagent)' : linkedTool.name; - toolBreakdown.push({ - toolName: displayName, - tokenCount: toolTokenCount, - isError: linkedTool.result?.isError ?? false, - toolUseId: linkedTool.id, - }); - totalTokens += toolTokenCount; + if (repeat) { + keyByToolId.set(linkedTool.id, key); + const existing = loopBreakdown.get(key); + if (existing) { + existing.count += 1; + // per-key tokens are filled by round billing below + existing.toolUseId = linkedTool.id; + } else { + loopBreakdown.set(key, { + key, + count: 1, + tokenCount: 0, + toolUseId: linkedTool.id, + }); + } + } else { + // Rename "Task" to "Task (Subagent)" for clarity in the UI + const displayName = linkedTool.name === 'Task' ? 'Task (Subagent)' : linkedTool.name; + toolBreakdown.push({ + toolName: displayName, + tokenCount: toolTokenCount, + isError: linkedTool.result?.isError ?? false, + toolUseId: linkedTool.id, + }); + totalTokens += toolTokenCount; + } } } - - // Include user-invoked slash tokens from display items - // These are slashes invoked via /xxx commands if (displayItems) { for (const item of displayItems) { if (item.type === 'slash' && item.slash.instructionsTokenCount) { @@ -235,19 +309,162 @@ function aggregateToolOutputs( } } - // Return null if no tokens from tools + // Loop tokens = billed usage of rounds carrying repeat calls, each round + // counted once. A round shared by two different repeat keys bills both + // breakdowns, but the turn total counts it once. + const loopRounds: LoopRoundInfo[] = []; + for (const round of classifyRounds(responses, keyByToolId)) { + if (!round.repeat) continue; + loopTokens += round.billed; + loopRounds.push({ + uuid: round.uuid, + index: round.index, + billed: round.billed, + keys: round.keys, + }); + for (const key of round.keys) { + const entry = loopBreakdown.get(key); + if (entry) entry.tokenCount += round.billed; + } + } + + let loop: LoopInjection | null = null; + if (loopTokens > 0) { + loop = { + id: generateLoopId(turnIndex), + category: 'loop', + turnIndex, + aiGroupId, + estimatedTokens: loopTokens, + breakdown: [...loopBreakdown.values()], + rounds: loopRounds, + }; + } + if (totalTokens === 0) { - return null; + return { toolOutput: null, loop, loopState: state, repeatKeys: keyByToolId }; } return { - id: generateToolOutputId(turnIndex), - category: 'tool-output', + toolOutput: { + id: generateToolOutputId(turnIndex), + category: 'tool-output', + turnIndex, + aiGroupId, + estimatedTokens: totalTokens, + toolCount: toolBreakdown.length, + toolBreakdown, + }, + loop, + loopState: state, + repeatKeys: keyByToolId, + }; +} + +// ============================================================================= +// Round Classification +// ============================================================================= + +/** One assistant round (response) of a turn, classified for accounting and stream markers */ +export interface ClassifiedRound { + uuid: string; + /** 1-based round number within the turn */ + index: number; + quiet: boolean; + repeat: boolean; + /** Tool round that stopped growing the context (echo-marker loop shape) */ + stalled: boolean; + /** Full billed usage (in + cache_read + cache_creation + output) */ + billed: number; + outputTokens: number; + /** Repeat call keys present in this round (empty for non-repeat rounds) */ + keys: string[]; +} + +/** + * Classify every assistant round of a turn: quiet (no tool call while the + * billed context >= WAIT_TICK_CONTEXT_TOKENS and output <= + * WAIT_TICK_OUTPUT_TOKENS — an idle tick, not a working round: with a large + * baseline context every ordinary tool round would otherwise qualify), repeat + * (carries a call whose id maps to a repeat key) and stalled (makes a tool + * call while the context stops growing — isStalledRound). The same + * classification feeds the wait-loop/loop aggregates and the stream round + * markers — one source, no drift. + */ +export function classifyRounds( + responses: ParsedMessage[], + keyByToolId?: Map +): ClassifiedRound[] { + const rounds: ClassifiedRound[] = []; + let prevContext = 0; + (responses ?? []).forEach((msg, i) => { + const usage = msg.usage; + const input = usage?.input_tokens ?? 0; + const cacheRead = usage?.cache_read_input_tokens ?? 0; + const cacheCreation = usage?.cache_creation_input_tokens ?? 0; + const output = usage?.output_tokens ?? 0; + const context = input + cacheRead + cacheCreation; + const roundToolIds = Array.isArray(msg.content) + ? msg.content.filter((b) => b.type === 'tool_use').map((b) => b.id) + : []; + const keys = keyByToolId + ? [ + ...new Set( + roundToolIds + .map((id) => keyByToolId.get(id)) + .filter((k): k is string => k !== undefined) + ), + ] + : []; + rounds.push({ + uuid: msg.uuid ?? `round-${i + 1}`, + index: i + 1, + quiet: isQuietTick(context, output, roundToolIds.length), + repeat: keys.length > 0, + stalled: isStalledRound(prevContext, context, output, roundToolIds.length), + billed: context + output, + outputTokens: output, + keys, + }); + // ghost rounds (no usage) must not drag the baseline — same rule as buildLedger + if (context > 0) prevContext = context; + }); + return rounds; +} + +// ============================================================================= +// Wait-Loop Aggregation +// ============================================================================= + +/** + * Sum the billed input-side context of quiet rounds in this turn — rounds that + * re-read the whole window (>= WAIT_TICK_CONTEXT_TOKENS) while producing + * almost nothing (<= WAIT_TICK_OUTPUT_TOKENS out). Same criterion as the CLI's + * wait_loop findings; rounds without usage are skipped (provider ghosts). + */ +function aggregateWaitLoopRounds( + aiGroup: AIGroup, + turnIndex: number, + aiGroupId: string +): WaitLoopInjection | null { + const rounds: WaitRoundInfo[] = classifyRounds(aiGroup.responses ?? []) + .filter((round) => round.quiet) + .map((round) => ({ + uuid: round.uuid, + index: round.index, + outputTokens: round.outputTokens, + billed: round.billed, + })); + if (rounds.length === 0) return null; + + return { + id: generateWaitLoopId(turnIndex), + category: 'wait-loop', turnIndex, aiGroupId, - estimatedTokens: totalTokens, - toolCount: toolBreakdown.length, - toolBreakdown, + estimatedTokens: rounds.reduce((sum, round) => sum + round.billed, 0), + roundCount: rounds.length, + rounds, }; } @@ -438,6 +655,8 @@ interface ComputeContextStatsParams { previousInjections: ContextInjection[]; /** Paths already seen in previous groups (threaded to avoid O(N) rebuild per group) */ previousPaths: Set; + /** Loop streak state threaded across groups (mutated in place) */ + loopState: LoopStreakState; /** Project root path for resolving relative paths */ projectRoot: string; /** Token data for CLAUDE.md files (global sources) */ @@ -452,6 +671,8 @@ interface ComputeContextStatsResult { stats: ContextStats; /** Updated previousPaths set — caller should thread this to the next group */ previousPaths: Set; + /** Updated loop streak state — caller should thread this to the next group */ + loopState: LoopStreakState; } /** @@ -606,6 +827,7 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS isFirstGroup, previousInjections, previousPaths, + loopState, projectRoot, claudeMdTokenData, mentionedFileTokenData, @@ -758,17 +980,33 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS } } - // d) Aggregate tool outputs (includes user-invoked skill tokens from displayItems) - // Task coordination tools are excluded here (tracked separately in step d2) - const toolOutputInjection = aggregateToolOutputs( + // d) Aggregate tool outputs + loop classification (task coordination tools + // are excluded here — tracked separately in step d2) + const { + toolOutput: toolOutputInjection, + loop: loopInjection, + loopState: updatedLoopState, + repeatKeys, + } = aggregateToolOutputs( linkedTools, aiGroup.turnIndex, turnGroupId, - displayItems + displayItems, + loopState, + aiGroup.responses ?? [] ); if (toolOutputInjection) { newInjections.push(toolOutputInjection); } + if (loopInjection) { + newInjections.push(loopInjection); + } + + // d1) Aggregate quiet-round wait-loop burn (billed re-read, not content) + const waitLoopInjection = aggregateWaitLoopRounds(aiGroup, aiGroup.turnIndex, turnGroupId); + if (waitLoopInjection) { + newInjections.push(waitLoopInjection); + } // d2) Aggregate task coordination tokens (SendMessage, TeamCreate, TaskCreate, etc.) const taskCoordinationInjection = aggregateTaskCoordination( @@ -819,6 +1057,8 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS thinkingText: 0, taskCoordination: 0, userMessages: 0, + loop: 0, + waitLoop: 0, }; const newCounts: NewCountsByCategory = { @@ -828,6 +1068,8 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS thinkingText: 0, taskCoordination: 0, userMessages: 0, + loop: 0, + waitLoop: 0, }; // Count new injections by category @@ -851,6 +1093,12 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS case 'user-message': newCounts.userMessages++; break; + case 'loop': + newCounts.loop += injection.breakdown.length; + break; + case 'wait-loop': + newCounts.waitLoop += injection.roundCount; + break; } } @@ -862,6 +1110,8 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS thinkingText: 0, taskCoordination: 0, userMessages: 0, + loop: 0, + waitLoop: 0, }; for (const injection of accumulatedInjections) { @@ -890,6 +1140,14 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS tokensByCategory.userMessages += injection.estimatedTokens; accumulatedCounts.userMessages++; break; + case 'loop': + tokensByCategory.loop += injection.estimatedTokens; + accumulatedCounts.loop += injection.breakdown.length; + break; + case 'wait-loop': + tokensByCategory.waitLoop += injection.estimatedTokens; + accumulatedCounts.waitLoop += injection.roundCount; + break; } } @@ -899,7 +1157,22 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS tokensByCategory.toolOutputs + tokensByCategory.thinkingText + tokensByCategory.taskCoordination + - tokensByCategory.userMessages; + tokensByCategory.userMessages + + tokensByCategory.loop + + tokensByCategory.waitLoop; + + // Per-round flags for the stream markers (quiet / repeat / billed), same + // classification the aggregates above use + const roundFlags = new Map(); + for (const round of classifyRounds(aiGroup.responses ?? [], repeatKeys)) { + roundFlags.set(round.uuid, { + index: round.index, + quiet: round.quiet, + repeat: round.repeat, + stalled: round.stalled, + billed: round.billed, + }); + } return { stats: { @@ -909,8 +1182,10 @@ function computeContextStats(params: ComputeContextStatsParams): ComputeContextS tokensByCategory, newCounts, accumulatedCounts, + roundFlags, }, previousPaths, + loopState: updatedLoopState, }; } @@ -974,6 +1249,7 @@ export function processSessionContextWithPhases( let previousPaths = new Set(); let isFirstAiGroup = true; let previousUserGroup: UserGroup | null = null; + let loopState = createLoopStreakState(); // Phase tracking state let currentPhaseNumber = 1; @@ -1063,6 +1339,7 @@ export function processSessionContextWithPhases( isFirstGroup: isFirstAiGroup, previousInjections: accumulatedInjections, previousPaths, + loopState, projectRoot, claudeMdTokenData, mentionedFileTokenData, @@ -1106,6 +1383,7 @@ export function processSessionContextWithPhases( // Update accumulated state for next iteration accumulatedInjections = stats.accumulatedInjections; previousPaths = result.previousPaths; + loopState = result.loopState; isFirstAiGroup = false; previousUserGroup = null; } diff --git a/src/renderer/utils/displayItemBuilder.ts b/src/renderer/utils/displayItemBuilder.ts index 07b4e786..86fff53b 100644 --- a/src/renderer/utils/displayItemBuilder.ts +++ b/src/renderer/utils/displayItemBuilder.ts @@ -141,6 +141,8 @@ export function buildDisplayItems( // Build display items for (const step of steps) { + // Round marker: the assistant response this item belongs to + const roundId = step.sourceMessageId; // Skip the last output step if (lastOutputStepId && step.id === lastOutputStepId) { continue; @@ -154,6 +156,7 @@ export function buildDisplayItems( content: step.content.thinkingText, timestamp: step.startTime, tokenCount: estimateTokens(step.content.thinkingText), + roundId, }); } break; @@ -169,6 +172,7 @@ export function buildDisplayItems( displayItems.push({ type: 'tool', tool: linkedTool, + roundId, }); } } @@ -198,6 +202,7 @@ export function buildDisplayItems( content: step.content.outputText, timestamp: step.startTime, tokenCount: estimateTokens(step.content.outputText), + roundId, }); } break; @@ -209,6 +214,7 @@ export function buildDisplayItems( content: step.content.interruptionText, timestamp: step.startTime, tokenCount: estimateTokens(step.content.interruptionText), + roundId, }); } break; @@ -425,8 +431,7 @@ export function buildDisplayItemsFromMessages( } // Only treat as subagent input if there are NO tool_result blocks in this message const hasToolResults = - Array.isArray(msg.content) && - msg.content.some((b) => b.type === 'tool_result'); + Array.isArray(msg.content) && msg.content.some((b) => b.type === 'tool_result'); if (rawText.trim() && !hasToolResults) { displayItems.push({ type: 'subagent_input', diff --git a/src/shared/constants/loopPolicy.ts b/src/shared/constants/loopPolicy.ts new file mode 100644 index 00000000..a54b433b --- /dev/null +++ b/src/shared/constants/loopPolicy.ts @@ -0,0 +1,57 @@ +/** + * Wait-loop policy shared by the CLI analyzer (analyzeSession findings) and the + * renderer's Visible Context wait-loop category (contextTracker). + * A "quiet round" re-reads the whole window (>= CONTEXT tokens billed on the + * input side) while producing almost nothing (<= OUTPUT tokens out). + */ + +/** Minimum input-side tokens (input + cache_read + cache_creation) for a quiet round to count */ +export const WAIT_TICK_CONTEXT_TOKENS = 50_000; + +/** Maximum output tokens for a round to qualify as quiet */ +export const WAIT_TICK_OUTPUT_TOKENS = 300; + +/** + * The one quiet-tick criterion, shared verbatim by both consumers so it + * cannot drift: a quiet round produces almost nothing out while re-reading + * a large context AND makes no tool call at all — with a large baseline + * context, ordinary working rounds (short tool calls) would otherwise + * satisfy the token thresholds too. + */ +export function isQuietTick( + billedContextTokens: number, + outputTokens: number, + toolCallCount: number +): boolean { + return ( + toolCallCount === 0 && + billedContextTokens >= WAIT_TICK_CONTEXT_TOKENS && + outputTokens <= WAIT_TICK_OUTPUT_TOKENS + ); +} + +/** Maximum round-to-round context growth for a round to qualify as stalled */ +export const STALL_CONTEXT_DELTA_TOKENS = 300; + +/** + * The one stall criterion, shared verbatim by all consumers (CLI findings, + * renderer round flags, main-process bell): a stalled round MAKES a tool + * call but the context stops growing (marker loops like `echo w/v/u` — + * distinct args, so the repeat-key walk sees no streak) and the output is + * tiny. Quiet ticks (isQuietTick) are the no-tool-call counterpart. + */ +export function isStalledRound( + prevContextTokens: number, + contextTokens: number, + outputTokens: number, + toolCallCount: number +): boolean { + const delta = contextTokens - prevContextTokens; + return ( + toolCallCount > 0 && + prevContextTokens > 0 && // first round / ghost baseline — nothing to compare + delta >= 0 && // a shrinking window (compaction) is not a stall + delta <= STALL_CONTEXT_DELTA_TOKENS && + outputTokens <= WAIT_TICK_OUTPUT_TOKENS + ); +} diff --git a/src/shared/types/notifications.ts b/src/shared/types/notifications.ts index a285a650..af76622a 100644 --- a/src/shared/types/notifications.ts +++ b/src/shared/types/notifications.ts @@ -249,6 +249,10 @@ export interface AppConfig { includeSubagentErrors: boolean; /** Notification triggers - define when to generate notifications */ triggers: NotificationTrigger[]; + /** Live tool-call loop detection (FileWatcher) */ + loopDetection: { enabled: boolean; cycleThreshold: number }; + /** Per-turn input-token budget enforced by the PreToolUse hook */ + turnBudget: { enabled: boolean; maxInputTokensPerTurn: number }; }; /** General application settings */ general: { diff --git a/src/shared/utils/callKey.ts b/src/shared/utils/callKey.ts new file mode 100644 index 00000000..693dcc5d --- /dev/null +++ b/src/shared/utils/callKey.ts @@ -0,0 +1,60 @@ +/** + * Tool-call identity for loop/repeat detection, shared by the CLI analyzers + * (analyzeSession, sessionInventory), the live LoopDetector (FileWatcher) and + * the renderer's Visible Context loop category (contextTracker). + */ + +/** + * Coerces a tool input value to its text form: strings pass through, nullish + * collapse to empty, everything else is JSON — so `5` and `"5"` stay distinct. + * ponytail: normalization is a heuristic — same command/file/pattern counts as a repeat + */ +export function asText(v: unknown): string { + if (typeof v === 'string') return v; + if (v === undefined || v === null) return ''; + return JSON.stringify(v); +} + +const squash = (v: unknown): string => asText(v).replace(/\s+/g, ' ').trim(); + +/** Canonical identity of a tool call: same command/file/pattern = same key. */ +export function normalizeCallKey(name: string, input: Record): string { + switch (name) { + case 'Bash': + return `Bash|${squash(input.command)}`; + case 'Read': + case 'Write': + case 'Edit': + case 'NotebookEdit': + return `${name}|${asText(input.file_path)}`; + case 'Grep': + case 'Glob': + return `${name}|${asText(input.pattern)}|${asText(input.path)}`; + case 'Skill': + return `Skill|${asText(input.skill)}`; + case 'Task': + case 'Agent': + return `Task|${asText(input.description) || asText(input.prompt)}`; + default: + // own top-level keys, sorted — a replacer array would recurse and flatten + // nested objects to {}, colliding keys of calls differing only in nesting + return `${name}|${JSON.stringify( + Object.fromEntries( + Object.keys(input) + .sort((a, b) => a.localeCompare(b)) + .map((k) => [k, input[k]]) + ) + )}`; + } +} + +/** + * Bash key without its pipe tail — hundreds of `git show X | wc -l`-style + * variants are ONE re-read loop. + * ponytail: naive pipe cut — pipes inside quoted patterns merge, accepted + */ +export function bashStem(key: string): string { + if (!key.startsWith('Bash|')) return key; + const pipe = key.indexOf('|', 5); + return pipe === -1 ? key : key.slice(0, pipe).trimEnd(); +} diff --git a/src/shared/utils/logger.ts b/src/shared/utils/logger.ts index a361b20b..aefeb26b 100644 --- a/src/shared/utils/logger.ts +++ b/src/shared/utils/logger.ts @@ -23,8 +23,14 @@ enum LogLevel { } class Logger { - private static level: LogLevel = - process.env.NODE_ENV === 'production' ? LogLevel.ERROR : LogLevel.WARN; + private static level: LogLevel = (() => { + // CLAUDE_DEVTOOLS_LOG_LEVEL=info|debug overrides the packaged default + // (ERROR) — the only way to see watcher/service flow in a release build. + const raw = process.env.CLAUDE_DEVTOOLS_LOG_LEVEL; + if (raw === 'debug') return LogLevel.DEBUG; + if (raw === 'info') return LogLevel.INFO; + return process.env.NODE_ENV === 'production' ? LogLevel.ERROR : LogLevel.WARN; + })(); constructor(private namespace: string) {} diff --git a/src/shared/utils/modelParser.ts b/src/shared/utils/modelParser.ts index 8d9fbf37..a3011702 100644 --- a/src/shared/utils/modelParser.ts +++ b/src/shared/utils/modelParser.ts @@ -28,6 +28,7 @@ const KNOWN_FAMILIES: KnownModelFamily[] = ['sonnet', 'opus', 'haiku']; * * Supported formats: * - New format: claude-{family}-{major}-{minor}-{date} (e.g., "claude-sonnet-4-5-20250929") + * - New format short: claude-{family}-{major} (e.g., "claude-sonnet-5") * - Old format: claude-{major}-{family}-{date} (e.g., "claude-3-opus-20240229") * - Old format with minor: claude-{major}-{minor}-{family}-{date} (e.g., "claude-3-5-sonnet-20241022") */ @@ -87,14 +88,12 @@ export function parseModelString(model: string | undefined): ModelInfo | null { // Determine format based on family position if (familyIndex === 1) { - // New format: claude-{family}-{major}-{minor}-{date} + // New format: claude-{family}-{major}[-{minor}][-{date}] // e.g., claude-sonnet-4-5-20250929 -> ["claude", "sonnet", "4", "5", "20250929"] - if (parts.length < 4) { - return null; - } - + // e.g., claude-sonnet-5 -> ["claude", "sonnet", "5"] majorVersion = parseInt(parts[2], 10); - if (isNaN(majorVersion)) { + // 8 digits in major position is a misplaced date (claude-sonnet-20250929), not a version + if (isNaN(majorVersion) || /^\d{8}$/.test(parts[2])) { return null; } diff --git a/test/main/cli/analyzeSession.test.ts b/test/main/cli/analyzeSession.test.ts new file mode 100644 index 00000000..67dcccf4 --- /dev/null +++ b/test/main/cli/analyzeSession.test.ts @@ -0,0 +1,800 @@ +/** + * Tests for the token-analytics CLI (src/cli/*). + * Covers the ported prototype logic: turn boundaries, requestId dedup, + * duplicate-call keys, waste findings, slow-subagent math, inventory scan. + */ +import { mkdtemp, rm, writeFile } from 'fs/promises'; +import { tmpdir } from 'os'; +import * as path from 'path'; + +import { afterEach, describe, expect, it, vi } from 'vitest'; + +import { + breakdownFromRounds, + buildLedger, + computeFindings, + detectBillingScheme, + filterLedgerByDate, + normalizeCallKey, + parseArgs, + turnActiveMinutes, +} from '../../../src/cli/analyzeSession'; +import { mapWithConcurrency, scanSessionFile } from '../../../src/cli/sessionInventory'; +import { estimateTokens } from '../../../src/shared/utils/tokenFormatting'; +import type { ParsedMessage } from '../../../src/main/types'; + +let seq = 0; +function makeMsg( + overrides: Partial & { type: ParsedMessage['type'] } +): ParsedMessage { + return { + uuid: `u${++seq}`, + parentUuid: null, + timestamp: new Date('2026-09-20T10:00:00Z'), + content: '', + toolCalls: [], + toolResults: [], + isSidechain: false, + isMeta: false, + ...overrides, + }; +} + +const usage = (input: number, cr: number, cw: number, out: number) => ({ + input_tokens: input, + output_tokens: out, + cache_read_input_tokens: cr, + cache_creation_input_tokens: cw, +}); + +afterEach(async () => { + seq = 0; +}); + +describe('buildLedger', () => { + it('starts turns only on real user messages, not isMeta tool-result carriers', () => { + const messages = [ + makeMsg({ type: 'user', content: 'hello' }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(100, 1000, 50, 10) }), + makeMsg({ + type: 'user', + isMeta: true, + content: [{ type: 'tool_result', tool_use_id: 't1', content: 'ok' } as never], + toolResults: [{ toolUseId: 't1', content: 'ok', isError: false }], + }), + makeMsg({ + type: 'assistant', + model: 'm1', + usage: usage(120, 1100, 0, 20), + toolCalls: [{ id: 't2', name: 'Bash', input: { command: 'ls' }, isTask: false }], + }), + ]; + + const ledger = buildLedger(messages); + expect(ledger.turns).toHaveLength(1); + expect(ledger.rounds).toHaveLength(2); + expect(ledger.rounds[1].contextDelta).toBe(120 + 1100 - (100 + 1000 + 50)); + expect(ledger.rounds[1].tools).toEqual(['Bash']); + expect(ledger.totals.outputTokens).toBe(30); + }); + + it('deduplicates streaming entries by requestId keeping the last usage', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'm1', + requestId: 'r1', + usage: usage(100, 0, 0, 5), + }), + makeMsg({ + type: 'assistant', + model: 'm1', + requestId: 'r1', + usage: usage(100, 0, 0, 25), + }), + ]; + + const ledger = buildLedger(messages); + expect(ledger.rounds).toHaveLength(1); + expect(ledger.totals.outputTokens).toBe(25); + }); + + it('excludes sidechain and assistant messages', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: '', usage: usage(9, 0, 0, 1) }), + makeMsg({ type: 'assistant', isSidechain: true, model: 'm1', usage: usage(50, 0, 0, 5) }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(10, 0, 0, 2) }), + ]; + + const ledger = buildLedger(messages); + expect(ledger.rounds).toHaveLength(1); + expect(ledger.totals.inputTokens).toBe(10); + }); +}); + +describe('normalizeCallKey', () => { + it('collapses whitespace in Bash commands', () => { + expect(normalizeCallKey('Bash', { command: 'pnpm test' })).toBe( + normalizeCallKey('Bash', { command: 'pnpm test' }) + ); + }); + + it('distinguishes different files for Read', () => { + expect(normalizeCallKey('Read', { file_path: '/a' })).not.toBe( + normalizeCallKey('Read', { file_path: '/b' }) + ); + }); + + it('keeps nested object inputs distinct in the default branch', () => { + expect(normalizeCallKey('TodoWrite', { todos: [{ content: 'a' }] })).not.toBe( + normalizeCallKey('TodoWrite', { todos: [{ content: 'b' }] }) + ); + }); + + it('does not swallow a following flag as a value', () => { + const opts = parseArgs(['--rounds', '--json', 'x.jsonl']); + expect(opts.rounds).toBe(20); + expect(opts.json).toBe(true); + expect(opts.sessionPath).toBe('x.jsonl'); + }); +}); + +describe('computeFindings', () => { + it('flags duplicate calls, real failures, and skips user rejections', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + usage: usage(10, 0, 0, 1), + toolCalls: [ + { id: 't1', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }, + { id: 't2', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }, + { id: 't3', name: 'Bash', input: { command: 'boom' }, isTask: false }, + { id: 't4', name: 'Bash', input: { command: 'declined' }, isTask: false }, + ], + }), + makeMsg({ + type: 'user', + isMeta: true, + toolResults: [ + { toolUseId: 't1', content: 'all passed', isError: false }, + { toolUseId: 't2', content: 'all passed', isError: false }, + { toolUseId: 't3', content: 'Traceback: boom', isError: true }, + { + toolUseId: 't4', + content: "The user doesn't want to proceed with this tool use.", + isError: true, + }, + ], + }), + ]; + const ledger = buildLedger(messages); + const findings = computeFindings(messages, ledger); + const types = findings.map((f) => f.type); + + expect(types).toContain('duplicate_call'); + expect(types).toContain('failed_call'); + const failed = findings.filter((f) => f.type === 'failed_call'); + expect(failed).toHaveLength(1); // only the real failure, not the rejection + const dup = findings.find((f) => f.type === 'duplicate_call'); + expect(dup?.tokensWasted).toBe(estimateTokens('all passed')); // one re-read, not both results + }); + + it('scopes tool-call findings to the window but resolves results from all messages', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go', timestamp: new Date('2026-09-01T10:00:00Z') }), + makeMsg({ + type: 'assistant', + timestamp: new Date('2026-09-01T10:01:00Z'), + usage: usage(10, 0, 0, 1), + toolCalls: [{ id: 'a', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }], + }), + makeMsg({ + type: 'assistant', + timestamp: new Date('2026-09-20T10:00:00Z'), + usage: usage(10, 0, 0, 1), + toolCalls: [ + { id: 'b1', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }, + { id: 'b2', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }, + { id: 'c', name: 'Bash', input: { command: 'boom' }, isTask: false }, + ], + }), + makeMsg({ + type: 'user', + isMeta: true, + timestamp: new Date('2026-09-25T10:00:00Z'), + toolResults: [ + { toolUseId: 'a', content: 'all passed', isError: false }, + { toolUseId: 'b1', content: 'all passed', isError: false }, + { toolUseId: 'b2', content: 'all passed', isError: false }, + { toolUseId: 'c', content: 'Traceback: boom', isError: true }, + ], + }), + ]; + const ledger = buildLedger(messages); + const findings = computeFindings( + messages, + ledger, + new Date(2026, 8, 15), + new Date(2026, 8, 20, 23, 59, 59, 999) + ); + + // only the two in-window 'pytest -q' calls count — the Sep 1 call is out of scope + const dup = findings.find((f) => f.type === 'duplicate_call'); + expect(dup).toBeDefined(); + expect(dup?.tokensWasted).toBe(estimateTokens('all passed')); + // 'boom' fails inside the window, its result lands after --until and still resolves + expect(findings.filter((f) => f.type === 'failed_call')).toHaveLength(1); + }); + + it('flags context spikes and dead caching from ledger rounds', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(1000, 0, 0, 1) }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(45000, 0, 0, 1) }), + ]; + + const ledger = buildLedger(messages); + const findings = computeFindings(messages, ledger); + expect(findings.some((f) => f.type === 'context_spike')).toBe(true); + expect(findings.some((f) => f.type === 'cache_dead')).toBe(true); + }); +}); + +describe('scanSessionFile', () => { + it('computes duration, dedups usage per requestId, excludes synthetic and sidechain usage', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-inv-')); + try { + const file = path.join(dir, 'session1.jsonl'); + await writeFile( + file, + [ + JSON.stringify({ type: 'user', uuid: '1', timestamp: '2026-09-20T10:00:00Z' }), + JSON.stringify({ + type: 'assistant', + uuid: '2', + timestamp: '2026-09-20T10:05:00Z', + requestId: 'r1', + message: { + model: 'claude-sonnet-5', + usage: { input_tokens: 10, output_tokens: 1, cache_read_input_tokens: 100 }, + }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '3', + timestamp: '2026-09-20T10:06:00Z', + requestId: 'r1', + message: { + model: 'claude-sonnet-5', + usage: { input_tokens: 10, output_tokens: 5, cache_read_input_tokens: 100 }, + }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '4', + timestamp: '2026-09-20T10:07:00Z', + message: { model: '', usage: { input_tokens: 9, output_tokens: 1 } }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '5', + timestamp: '2026-09-20T10:08:00Z', + isSidechain: true, + requestId: 'r2', + message: { + model: 'claude-haiku-4-5', + usage: { input_tokens: 500, output_tokens: 50, cache_creation_input_tokens: 30 }, + }, + }), + ].join('\n') + ); + + const entry = await scanSessionFile(file); + expect(entry).not.toBeNull(); + // sidechain timestamps span the file, but its tokens/models/billing stay out + expect(entry?.durationMs).toBe(8 * 60 * 1000); + expect(entry?.models).toEqual(['claude-sonnet-5']); + expect(entry?.inputTokens).toBe(10); + expect(entry?.outputTokens).toBe(5); + expect(entry?.cacheReadTokens).toBe(100); + expect(entry?.cacheCreationTokens).toBe(0); + expect(entry?.messageCount).toBe(5); + expect(entry?.billing).toBe('router-style'); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); + +describe('mapWithConcurrency', () => { + it('isolates per-item failures as null entries', async () => { + const errSpy = vi.spyOn(console, 'error').mockImplementation(() => {}); + try { + const results = await mapWithConcurrency([1, 2, 3], 2, async (n) => { + if (n === 2) throw new Error('EACCES: permission denied'); + return n * 10; + }); + expect(results).toHaveLength(3); + expect(results).toContain(10); + expect(results).toContain(30); + expect(results).toContain(null); + } finally { + errSpy.mockRestore(); + } + }); +}); + +describe('billing scheme', () => { + it('detects billing scheme from round signatures', () => { + const w = { cacheReadTokens: 0, cacheCreationTokens: 100 }; + const r = { cacheReadTokens: 500, cacheCreationTokens: 0 }; + const n = { cacheReadTokens: 0, cacheCreationTokens: 0 }; + expect(detectBillingScheme([w, r])).toBe('mixed'); + expect(detectBillingScheme([w])).toBe('anthropic-style'); + expect(detectBillingScheme([r, r])).toBe('router-style'); + expect(detectBillingScheme([n])).toBe('no-cache'); + }); + + it('computes cost for anthropic-style sessions logged with short model ids', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'claude-sonnet-5', + usage: usage(100, 1000, 200, 50), + }), + ]; + const ledger = buildLedger(messages); + expect(ledger.billing).toBe('anthropic-style'); + expect(ledger.totals.costUsd).toBeDefined(); + expect(ledger.totals.costUsd).toBeGreaterThan(0); + expect(ledger.totals.costPartial).toBe(false); + }); + + it('prices router-style glm rounds and leaves unpriced models without cost', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(100, 4000, 0, 10) }), + ]; + const ledger = buildLedger(messages); + expect(ledger.billing).toBe('router-style'); + // glm is priced now: (100*0.075 + 4000*0.015 + 10*0.25) / 1e6 = 70 / 1e6 + expect(ledger.totals.costUsd).toBeCloseTo(0.00007, 8); + }); + + it('marks cost partial when the session mixes priced and unpriced models', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: 'claude-sonnet-5', usage: usage(100, 0, 0, 10) }), + makeMsg({ type: 'assistant', model: 'deepseek-v4-flash', usage: usage(100, 0, 0, 10) }), + ]; + const ledger = buildLedger(messages); + expect(ledger.totals.costUsd).toBeDefined(); + expect(ledger.totals.costPartial).toBe(true); + }); +}); + +describe('unified flag grammar', () => { + it('parses --since/--until as local day bounds (dash and compact forms)', () => { + const opts = parseArgs(['--since', '2026-09-01', '--until', '20260920', 'x.jsonl']); + expect(opts.since).toEqual(new Date(2026, 8, 1)); + expect(opts.until).toEqual(new Date(2026, 8, 20, 23, 59, 59, 999)); + expect(opts.sessionPath).toBe('x.jsonl'); + }); + + it('reports an error for malformed date bounds', () => { + expect(parseArgs(['--since', 'nah']).error).toContain('--since'); + expect(parseArgs(['--until', '2026-13-01']).error).toContain('--until'); + }); + + it('keeps bare --last as pick-newest and numeric --last N as a calendar window', () => { + expect(parseArgs(['--project', 'p', '--last']).useLast).toBe(true); + const midnight = (back: number): Date => { + const d = new Date(); + d.setHours(0, 0, 0, 0); + d.setDate(d.getDate() - back); + return d; + }; + const w = parseArgs(['--last', '7']); + expect(w.useLast).toBe(false); + expect(w.lastDays).toBe(7); + expect(w.since).toEqual(midnight(6)); // ccusage-style: local midnight, N=1 → today + }); + + it('does not swallow a non-numeric --last value or a following flag', () => { + const a = parseArgs(['--last', 'x.jsonl']); + expect(a.useLast).toBe(true); + expect(a.sessionPath).toBe('x.jsonl'); + const b = parseArgs(['--last', '--json']); + expect(b.useLast).toBe(true); + expect(b.json).toBe(true); + }); + + it('parses --breakdown and --no-cost', () => { + const opts = parseArgs(['--breakdown', '--no-cost', '--json']); + expect(opts.breakdown).toBe(true); + expect(opts.noCost).toBe(true); + expect(opts.json).toBe(true); + }); +}); + +describe('filterLedgerByDate and breakdownFromRounds', () => { + const twoModelSession = () => [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'claude-sonnet-5', + timestamp: new Date('2026-09-01T10:00:00Z'), + usage: usage(100, 1000, 0, 10), + }), + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', + timestamp: new Date('2026-09-15T10:00:00Z'), + usage: usage(50, 0, 0, 5), + }), + makeMsg({ + type: 'assistant', + model: 'claude-sonnet-5', + timestamp: new Date('2026-09-20T10:00:00Z'), + usage: usage(30, 0, 0, 2), + }), + ]; + + it('keeps rounds inside the window and recomputes totals/models/duration', () => { + const ledger = buildLedger(twoModelSession()); + const filtered = filterLedgerByDate(ledger, new Date(2026, 8, 15), undefined); + expect(filtered.rounds).toHaveLength(2); + expect(filtered.totals.inputTokens).toBe(80); + expect(filtered.totals.outputTokens).toBe(7); + expect(filtered.models).toEqual(['glm-5.3-flash', 'claude-sonnet-5']); + expect(filtered.durationMs).toBe( + new Date('2026-09-20T10:00:00Z').getTime() - new Date('2026-09-15T10:00:00Z').getTime() + ); + }); + + it('recomputes cost from kept rounds only (glm now priced, no partial flag)', () => { + const ledger = buildLedger(twoModelSession()); + const filtered = filterLedgerByDate( + ledger, + new Date(2026, 8, 15), + new Date(2026, 8, 30, 23, 59, 59, 999) + ); + // kept: glm (50*0.075 + 5*0.25) + sonnet (30*3 + 2*15) → 125 / 1e6 + expect(filtered.totals.costUsd).toBeCloseTo(0.000125, 10); + expect(filtered.totals.costPartial).toBe(false); + }); + + it('returns the ledger untouched without bounds', () => { + const ledger = buildLedger(twoModelSession()); + expect(filterLedgerByDate(ledger)).toBe(ledger); + }); + + it('breakdown groups tokens per model with per-model cost', () => { + const rows = breakdownFromRounds(buildLedger(twoModelSession()).rounds); + expect(rows).toHaveLength(2); + const sonnet = rows.find((r) => r.model === 'claude-sonnet-5'); + expect(sonnet?.billedTokens).toBe(100 + 1000 + 10 + 30 + 2); + // (100*3 + 10*15 + 1000*0.3) + (30*3 + 2*15) = 750 + 120 per 1e6 + expect(sonnet?.costUsd).toBeCloseTo(0.00087, 10); + // glm: (50*0.075 + 5*0.25) / 1e6 + expect(rows.find((r) => r.model === 'glm-5.3-flash')?.costUsd).toBeCloseTo(0.000005, 10); + }); +}); + +describe('data quality (issues #14/#15)', () => { + it('zero-usage rounds: no false spike, baseline kept, counted as noUsageRounds', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(1000, 0, 0, 100) }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(0, 0, 0, 0) }), // ghost, #14 + makeMsg({ type: 'assistant', model: 'm1', usage: usage(2000, 0, 0, 100) }), + ]; + const ledger = buildLedger(messages); + // round 3 measures against round 1, not against the ghost + expect(ledger.rounds[2].contextDelta).toBe(1000); + expect(ledger.totals.noUsageRounds).toBe(1); + expect(ledger.totals.retryCopies).toBe(0); + expect(computeFindings(messages, ledger).some((f) => f.type === 'context_spike')).toBe(false); + }); + + it('flags router-retry copies without touching sums or doubling findings', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(335200, 0, 0, 100) }), + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(335200, 0, 0, 100) }), + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(335200, 0, 0, 100) }), + ]; + const ledger = buildLedger(messages); + expect(ledger.totals.retryCopies).toBe(2); + // variant A: sums stay untouched — gluing is postponed until billing is known + expect(ledger.totals.billedTokens).toBe(3 * (335200 + 100)); + const cacheDead = computeFindings(messages, ledger).filter((f) => f.type === 'cache_dead'); + expect(cacheDead).toHaveLength(1); // original round only + }); + + it('ghost runs are not copies; a copy after a ghost still matches its original', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(335200, 0, 0, 100) }), + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(0, 0, 0, 0) }), // ghost 1 + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(0, 0, 0, 0) }), // ghost 2 + makeMsg({ type: 'assistant', model: 'glm-5.3-flash', usage: usage(335200, 0, 0, 100) }), // copy after ghosts + ]; + const ledger = buildLedger(messages); + // ghosts never match the all-zero pattern against a non-zero anchor + expect(ledger.totals.noUsageRounds).toBe(2); + expect(ledger.totals.retryCopies).toBe(1); // only the re-logged copy + const cacheDead = computeFindings(messages, ledger).filter((f) => f.type === 'cache_dead'); + expect(cacheDead).toHaveLength(1); // copy after ghosts is still suppressed + }); + + it('skips re-logged copy tool calls in duplicate/failed/oversized findings', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', + requestId: 'r1', // original was logged with a requestId + usage: usage(335200, 0, 0, 100), + toolCalls: [{ id: 't1', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }], + }), + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', // copy: no requestId, identical counters + usage: usage(335200, 0, 0, 100), + toolCalls: [ + { id: 't1-copy', name: 'Bash', input: { command: 'pytest -q' }, isTask: false }, + ], + }), + makeMsg({ + type: 'user', + isMeta: true, + toolResults: [{ toolUseId: 't1', content: 'ok', isError: false }], + }), + ]; + const duplicates = computeFindings(messages, buildLedger(messages)).filter( + (f) => f.type === 'duplicate_call' + ); + expect(duplicates).toHaveLength(0); // the copy's call is not a real repeat + }); +}); + +describe('long turns and loop streaks', () => { + const at = (min: number): Date => new Date(Date.UTC(2026, 8, 20, 10, min)); + const call = (id: string, command = 'true') => ({ + id, + name: 'Bash', + input: { command }, + isTask: false, + }); + const ok = (id: string) => ({ toolUseId: id, content: 'ok', isError: false }); + const err = (id: string) => ({ toolUseId: id, content: 'Error: boom', isError: true }); + + it('flags back-to-back identical calls as a no-op loop streak', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'm1', + usage: usage(10, 0, 0, 1), + toolCalls: [call('a1'), call('a2'), call('a3')], + }), + makeMsg({ type: 'user', isMeta: true, toolResults: [ok('a1'), ok('a2'), ok('a3')] }), + ]; + const findings = computeFindings(messages, buildLedger(messages)); + const streak = findings.find((f) => f.type === 'loop_streak'); + expect(streak).toBeDefined(); + expect(streak?.severity).toBe('medium'); // x3 — not yet a hang + expect(streak?.summary).toContain('x3 back-to-back (no-op loop)'); + expect(streak?.tokensWasted).toBe(2 * estimateTokens('ok')); // repeats only + }); + + it('a streak where every result is an error is an env loop', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'm1', + usage: usage(10, 0, 0, 1), + toolCalls: [call('b1'), call('b2'), call('b3'), call('b4'), call('b5')], + }), + makeMsg({ + type: 'user', + isMeta: true, + toolResults: [err('b1'), err('b2'), err('b3'), err('b4'), err('b5')], + }), + ]; + const env = computeFindings(messages, buildLedger(messages)).find( + (f) => f.type === 'loop_streak' + ); + expect(env?.summary).toContain('env loop'); + expect(env?.severity).toBe('high'); // x5 + }); + + it('calls separated by a different call do not form a streak', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'm1', + usage: usage(10, 0, 0, 1), + toolCalls: [call('c1'), call('c2', 'ls -la'), call('c3'), call('c4')], + }), + makeMsg({ + type: 'user', + isMeta: true, + toolResults: [ok('c1'), ok('c2'), ok('c3'), ok('c4')], + }), + ]; + const findings = computeFindings(messages, buildLedger(messages)); + expect(findings.filter((f) => f.type === 'loop_streak')).toHaveLength(0); + expect(findings.some((f) => f.type === 'duplicate_call')).toBe(true); + }); + + it('retry copies do not grow a streak', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', + requestId: 'r1', // original was logged with a requestId + usage: usage(335200, 0, 0, 100), + toolCalls: [call('t1')], + }), + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', // copy: no requestId, identical counters + usage: usage(335200, 0, 0, 100), + toolCalls: [call('t1-copy')], + }), + makeMsg({ + type: 'user', + isMeta: true, + toolResults: [ok('t1'), ok('t1-copy')], + }), + ]; + const streaks = computeFindings(messages, buildLedger(messages)).filter( + (f) => f.type === 'loop_streak' + ); + expect(streaks).toHaveLength(0); // the copy's call is skipped from the walk + }); + + it('a dense turn (gaps under the idle cap) is flagged long_turn', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go', timestamp: at(0) }), + ...Array.from({ length: 14 }, (_, i) => + makeMsg({ + type: 'assistant', + model: 'm1', + timestamp: at(i * 5), + usage: usage(10, 0, 0, 1), + }) + ), + ]; + const ledger = buildLedger(messages); + const long = computeFindings(messages, ledger).find((f) => f.type === 'long_turn'); + expect(long).toBeDefined(); + expect(long?.severity).toBe('high'); + expect(long?.turnIndex).toBe(1); + expect(long?.tokensWasted).toBe(0); // observation, not waste — unlike other findings + expect(ledger.turns[0].activeMinutes).toBe(65); // 13 gaps × 5 min, под капом + expect(ledger.totals.longestTurn?.activeMinutes).toBe(65); + }); + + it('idle-heavy turns stay under the flag (anti-noise regression)', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go', timestamp: at(0) }), + makeMsg({ type: 'assistant', model: 'm1', timestamp: at(0), usage: usage(10, 0, 0, 1) }), + makeMsg({ type: 'assistant', model: 'm1', timestamp: at(35), usage: usage(10, 0, 0, 1) }), + makeMsg({ type: 'assistant', model: 'm1', timestamp: at(40), usage: usage(10, 0, 0, 1) }), + ]; + const ledger = buildLedger(messages); + expect(computeFindings(messages, ledger).some((f) => f.type === 'long_turn')).toBe(false); + expect(ledger.turns[0].activeMinutes).toBe(15); // 10 (кап) + 5 + }); + + it('filterLedgerByDate recomputes activeMinutes in the window', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go', timestamp: at(0) }), + makeMsg({ type: 'assistant', model: 'm1', timestamp: at(0), usage: usage(10, 0, 0, 1) }), + makeMsg({ type: 'assistant', model: 'm1', timestamp: at(35), usage: usage(10, 0, 0, 1) }), + makeMsg({ type: 'assistant', model: 'm1', timestamp: at(40), usage: usage(10, 0, 0, 1) }), + ]; + const ledger = filterLedgerByDate(buildLedger(messages), at(35)); + expect(ledger.turns[0].activeMinutes).toBe(5); + }); + + it('flags a turn of quiet expensive rounds as wait_loop', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go', timestamp: at(0) }), + ...Array.from({ length: 5 }, (_, i) => + makeMsg({ + type: 'assistant', + model: 'm1', + timestamp: at(i + 1), + // distinct outputs: identical counters would read as router-retry copies + usage: usage(60_000, 0, 0, 100 + i), // tick: 60k billed, ~100 out + }) + ), + ]; + const findings = computeFindings(messages, buildLedger(messages)); + const wait = findings.find((f) => f.type === 'wait_loop'); + expect(wait).toBeDefined(); + expect(wait?.severity).toBe('medium'); // 5 ticks — not yet a night watch + expect(wait?.tokensWasted).toBe(5 * 60_000); + }); + + it('needs 5 ticks: loud rounds and retry copies do not count', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + // 4 real ticks, distinct outputs so they don't match each other as copies + makeMsg({ type: 'assistant', model: 'm1', usage: usage(60_000, 0, 0, 100) }), + // a router-retry copy of tick 1 — identical counters, no requestId → skipped + makeMsg({ type: 'assistant', model: 'm1', usage: usage(60_000, 0, 0, 100) }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(60_000, 0, 0, 101) }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(60_000, 0, 0, 102) }), + makeMsg({ type: 'assistant', model: 'm1', usage: usage(60_000, 0, 0, 103) }), + // a loud round: 400 tok of output — not a tick + makeMsg({ type: 'assistant', model: 'm1', usage: usage(60_000, 0, 0, 400) }), + ]; + const findings = computeFindings(messages, buildLedger(messages)); + expect(findings.some((f) => f.type === 'wait_loop')).toBe(false); + }); + + // live regression: session 0779a2bc — a reviewer emitted Bash(echo w), Bash(echo v), + // Bash(echo u)... every round re-read ~134k context and grew it by the ~24-tok + // tool result only; distinct args mean the repeat-key walk sees no streak + it('flags a stall streak: tool rounds with no context growth (echo-marker loop)', () => { + const base = 134_200; + // context grows by exactly the tool result (+24) each round; inputs/outputs + // differ so the router retry-copy filter does not swallow rounds + const echoUsage = (i: number) => usage(100 + i, base + 24 * (i + 1) - (100 + i), 0, 19 + i); + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + // baseline round: real work — context jumps to `base`, loud output + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', + usage: usage(200, 134_000, 0, 2_000), + toolCalls: [call('e0', 'cat plan.md')], + }), + makeMsg({ type: 'user', isMeta: true, toolResults: [ok('e0')] }), + ...Array.from({ length: 4 }, (_, i) => + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', + usage: echoUsage(i), + toolCalls: [call(`e${i + 1}`, `echo ${String.fromCharCode(119 + i)}`)], + }) + ), + ]; + const findings = computeFindings(messages, buildLedger(messages)); + const stall = findings.find((f) => f.type === 'stall_streak'); + expect(stall).toBeDefined(); + expect(stall?.severity).toBe('medium'); // x4 — not yet a hang + expect(stall?.turnIndex).toBe(1); + // each stalled round re-read its full context — that is the burn + const stalledContexts = [1, 2, 3, 4].map((k) => base + 24 * k); + expect(stall?.tokensWasted).toBe(stalledContexts.reduce((s, c) => s + c, 0)); + }); + + it('working rounds with real context growth are not a stall streak', () => { + const messages = [ + makeMsg({ type: 'user', content: 'go' }), + ...Array.from({ length: 5 }, (_, i) => + makeMsg({ + type: 'assistant', + model: 'glm-5.3-flash', + usage: usage(500, 100_000 + 2_000 * i, 0, 150 + i), // +2k context per round + toolCalls: [call(`w${i}`, `edit file${i}.ts`)], + }) + ), + ]; + const findings = computeFindings(messages, buildLedger(messages)); + expect(findings.some((f) => f.type === 'stall_streak')).toBe(false); + }); +}); diff --git a/test/main/cli/importLoops.test.ts b/test/main/cli/importLoops.test.ts new file mode 100644 index 00000000..490671b1 --- /dev/null +++ b/test/main/cli/importLoops.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from 'vitest'; + +import { buildImportEntries, mergeImport } from '../../../src/cli/importLoops'; + +const fixture = { + sessionId: 's1', + projectId: '-p1', + filePath: '/x/s1.jsonl', + cycles: [ + { key: 'Read|/a', count: 25, startTs: '2026-09-01T10:00:00Z', toolUseId: 'c25' }, + { key: 'Bash|true', count: 19, startTs: '2026-09-02T10:00:20Z', toolUseId: 'c19' }, + ], +}; + +describe('importLoops pure core', () => { + it('filters cycles below the threshold and keeps deep links', () => { + const out = buildImportEntries([fixture], 20); + expect(out).toHaveLength(1); + expect(out[0].sessionId).toBe('s1'); + expect(out[0].toolUseId).toBe('c25'); + expect(out[0].triggerId).toBe('historical-loop'); + }); + + it('replaces previous imports and caps the bell', () => { + const existing: { id: string; timestamp: number; triggerId: string }[] = [ + { id: 'live-1', timestamp: 500, triggerId: 'trigger' }, + { id: 'old-import', timestamp: 900, triggerId: 'historical-loop' }, + ]; + const incoming = buildImportEntries([fixture], 20); + const merged = mergeImport(existing, incoming, 5); + expect(merged.map((n) => n.id)).toEqual([ + 'historical-loop-s1-25x-2026-09-01T10:00:00Z', + 'live-1', + ]); + }); +}); diff --git a/test/main/cli/sessionInventory.test.ts b/test/main/cli/sessionInventory.test.ts new file mode 100644 index 00000000..0be442d1 --- /dev/null +++ b/test/main/cli/sessionInventory.test.ts @@ -0,0 +1,410 @@ +/** + * Tests for the sessions inventory CLI flags (src/cli/sessionInventory.ts): + * unified flag grammar (--since/--until/--last N/--breakdown/--no-cost) and + * per-model token accumulation feeding --breakdown. + */ +import { mkdtemp, rm, writeFile } from 'fs/promises'; +import { tmpdir } from 'os'; +import * as path from 'path'; + +import { describe, expect, it } from 'vitest'; + +import { inDateRange, parseDayBound } from '../../../src/cli/args'; +import { parseInventoryArgs, scanSessionFile } from '../../../src/cli/sessionInventory'; + +describe('parseInventoryArgs (unified grammar)', () => { + it('has ccusage-compatible defaults', () => { + const o = parseInventoryArgs([]); + expect(o.minMinutes).toBe(0); + expect(o.sort).toBe('duration'); + expect(o.limit).toBe(Number.POSITIVE_INFINITY); + expect(o.json).toBe(false); + expect(o.breakdown).toBe(false); + expect(o.since).toBeUndefined(); + expect(o.until).toBeUndefined(); + expect(o.error).toBeUndefined(); + }); + + it('parses the full unified flag set', () => { + const o = parseInventoryArgs([ + '--project', + 'p', + '--min-minutes', + '30', + '--sort', + 'tokens', + '--limit', + '5', + '--breakdown', + '--no-cost', + '--json', + '--since', + '2026-09-01', + '--until', + '20260920', + ]); + expect(o.projectArg).toBe('p'); + expect(o.minMinutes).toBe(30); + expect(o.sort).toBe('tokens'); + expect(o.limit).toBe(5); + expect(o.breakdown).toBe(true); + expect(o.json).toBe(true); + expect(o.since).toEqual(new Date(2026, 8, 1)); + expect(o.until).toEqual(new Date(2026, 8, 20, 23, 59, 59, 999)); + expect(o.error).toBeUndefined(); + }); + + it('rejects malformed dates and --last without a positive number', () => { + expect(parseInventoryArgs(['--since', 'nah']).error).toContain('--since'); + expect(parseInventoryArgs(['--until', '2026-13-01']).error).toContain('--until'); + expect(parseInventoryArgs(['--last']).error).toContain('--last'); + expect(parseInventoryArgs(['--last', '--json']).error).toContain('--last'); + expect(parseInventoryArgs(['--last', '0']).error).toContain('--last'); + }); + + it('snaps --last N to local midnight N-1 days back', () => { + const midnight = (back: number): Date => { + const d = new Date(); + d.setHours(0, 0, 0, 0); + d.setDate(d.getDate() - back); + return d; + }; + expect(parseInventoryArgs(['--last', '1']).since).toEqual(midnight(0)); // today + expect(parseInventoryArgs(['--last', '7']).since).toEqual(midnight(6)); + }); +}); + +describe('parseDayBound / inDateRange', () => { + it('parses dash and compact forms with inclusive end of day', () => { + expect(parseDayBound('2026-09-01', false)).toEqual(new Date(2026, 8, 1)); + expect(parseDayBound('20260901', true)).toEqual(new Date(2026, 8, 1, 23, 59, 59, 999)); + expect(parseDayBound('nah', false)).toBeNull(); + expect(parseDayBound('2026-02-30', false)).toBeNull(); + expect(parseDayBound('', false)).toBeNull(); + }); + + it('checks the range inclusively on both ends', () => { + const since = new Date(2026, 8, 1); + const until = new Date(2026, 8, 20, 23, 59, 59, 999); + expect(inDateRange(new Date(2026, 8, 1), since, until)).toBe(true); + expect(inDateRange(new Date(2026, 8, 20, 12, 0), since, until)).toBe(true); + expect(inDateRange(new Date(2026, 7, 31), since, until)).toBe(false); + expect(inDateRange(new Date(2024, 9, 1), since, until)).toBe(false); + expect(inDateRange(new Date(2026, 8, 5))).toBe(true); + }); +}); + +describe('scanSessionFile tokensByModel', () => { + it('splits tokens per model, keeping synthetic and sidechain out', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-tbm-')); + try { + const file = path.join(dir, 'session1.jsonl'); + const lines = [ + JSON.stringify({ + type: 'user', + uuid: '1', + timestamp: '2026-09-20T10:00:00Z', + cwd: '/Users/x/tg-content-factory', + }), + JSON.stringify({ + type: 'assistant', + uuid: '2', + timestamp: '2026-09-20T10:05:00Z', + requestId: 'r1', + message: { + model: 'claude-sonnet-5', + usage: { input_tokens: 10, output_tokens: 1, cache_read_input_tokens: 100 }, + }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '3', + timestamp: '2026-09-20T10:06:00Z', + requestId: 'r1', + message: { + model: 'claude-sonnet-5', + usage: { input_tokens: 10, output_tokens: 5, cache_read_input_tokens: 100 }, + }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '4', + timestamp: '2026-09-20T10:07:00Z', + message: { model: 'claude-haiku-4-5', usage: { input_tokens: 7 } }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '5', + timestamp: '2026-09-20T10:08:00Z', + message: { model: '', usage: { input_tokens: 9 } }, + }), + JSON.stringify({ + type: 'assistant', + uuid: '6', + timestamp: '2026-09-20T10:09:00Z', + isSidechain: true, + message: { + model: 'claude-haiku-4-5', + usage: { input_tokens: 500, output_tokens: 50 }, + }, + }), + ]; + await writeFile(file, lines.join('\n')); + + const entry = await scanSessionFile(file); + expect(entry).not.toBeNull(); + // r1 dedup keeps the last entry (out 5); haiku arrives without requestId + expect(entry?.tokensByModel).toEqual({ + 'claude-sonnet-5': 115, + 'claude-haiku-4-5': 7, + }); + expect(entry?.models).toEqual(['claude-sonnet-5', 'claude-haiku-4-5']); + expect(entry?.totalTokens).toBe(122); + expect(entry?.messageCount).toBe(6); + // API time: r1's two streaming snapshots (10:05, 10:06) anchor once — the + // only gap is r1-last (10:06) → haiku (10:07); the intra-request minute + // does not count (parity with analyze:session's one round per request) + expect(entry?.activeMs).toBe(60000); + // real path from the session's cwd, not the lossy dash-decode of the dir name + expect(entry?.cwd).toBe('/Users/x/tg-content-factory'); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('parses --sort turn and --min-turn-minutes', () => { + expect(parseInventoryArgs(['--sort', 'turn']).sort).toBe('turn'); + expect(parseInventoryArgs(['--min-turn-minutes', '30']).minTurnMinutes).toBe(30); + }); + + it('parses --sort streak and --min-streak', () => { + expect(parseInventoryArgs(['--sort', 'streak']).sort).toBe('streak'); + expect(parseInventoryArgs(['--min-streak', '25']).minStreak).toBe(25); + }); + + it('tracks topRepeat across assistant lines with streaming dedup', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-tr-')); + try { + const file = path.join(dir, 'session-tr.jsonl'); + const usageLine = ( + uuid: string, + ts: string, + requestId?: string, + toolUse?: { + id: string; + command: string; + } + ): string => + JSON.stringify({ + type: 'assistant', + uuid, + timestamp: ts, + ...(requestId ? { requestId } : {}), + message: { + model: 'claude-sonnet-5', + usage: { input_tokens: 5 }, + content: toolUse + ? [ + { + type: 'tool_use', + id: toolUse.id, + name: 'Bash', + input: { command: toolUse.command }, + }, + ] + : [], + }, + }); + const lines = [ + // one stem, three calls: two exact + one piped variant (merged by stem) + usageLine('a1', '2026-09-20T10:00:00Z', 'r1', { id: 't1', command: 'git show abc' }), + usageLine('a2', '2026-09-20T10:01:00Z', 'r1', { id: 't2', command: 'git show abc' }), + usageLine('a3', '2026-09-20T10:02:00Z', undefined, { + id: 't3', + command: 'git show abc | wc -l', + }), + // streaming snapshot: same requestId + same toolUseId → counted once + usageLine('a4', '2026-09-20T10:03:00Z', 'r9', { id: 'd1', command: 'ls -la' }), + usageLine('a5', '2026-09-20T10:04:00Z', 'r9', { id: 'd1', command: 'ls -la' }), + ]; + await writeFile(file, lines.join('\n')); + const entry = await scanSessionFile(file); + // a1,a2,a3 share one stem back-to-back → one cycle of 3 + expect(entry?.cycles).toEqual([ + { key: 'Bash|git show abc', count: 3, startTs: '2026-09-20T10:00:00Z', toolUseId: 't3' }, + ]); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('caps idle gaps in activeMs', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-am-')); + try { + const file = path.join(dir, 'session-am.jsonl'); + const lines = [ + JSON.stringify({ + type: 'assistant', + uuid: 'a1', + timestamp: '2026-09-20T10:00:00Z', + message: { model: 'claude-sonnet-5', usage: { input_tokens: 5 } }, + }), + JSON.stringify({ + type: 'assistant', + uuid: 'a2', + timestamp: '2026-09-20T10:35:00Z', + message: { model: 'claude-sonnet-5', usage: { input_tokens: 5 } }, + }), + JSON.stringify({ + type: 'assistant', + uuid: 'a3', + timestamp: '2026-09-20T10:37:00Z', + message: { model: 'claude-sonnet-5', usage: { input_tokens: 5 } }, + }), + ]; + await writeFile(file, lines.join('\n')); + const entry = await scanSessionFile(file); + // 35 min gap capped at 10, plus 2 min — not 37 + expect(entry?.activeMs).toBe(720000); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('longestTurnMs resets at user-turn boundaries', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-lt-')); + try { + const file = path.join(dir, 'session-lt.jsonl'); + const userLine = (ts: string): string => + JSON.stringify({ + type: 'user', + uuid: 'u', + timestamp: ts, + message: { role: 'user', content: 'go' }, + }); + const usageLine = (uuid: string, ts: string): string => + JSON.stringify({ + type: 'assistant', + uuid, + timestamp: ts, + message: { model: 'claude-sonnet-5', usage: { input_tokens: 5 } }, + }); + const lines = [ + usageLine('a1', '2026-09-20T10:00:00Z'), // implicit turn 1: 0 active + userLine('2026-09-20T10:35:00Z'), // boundary + usageLine('a2', '2026-09-20T10:37:00Z'), + usageLine('a3', '2026-09-20T10:39:00Z'), // turn 2: 2 min active + userLine('2026-09-20T10:41:00Z'), // boundary + usageLine('a4', '2026-09-20T10:42:00Z'), // turn 3: 0 active + ]; + await writeFile(file, lines.join('\n')); + const entry = await scanSessionFile(file); + expect(entry?.activeMs).toBe(120000); // only turn 2's gaps count + expect(entry?.longestTurnMs).toBe(120000); // turn 2, not the sum across turns + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('array-content user messages split turns (faithful isParsedUserChunkMessage guard)', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-ac-')); + try { + const file = path.join(dir, 'session-ac.jsonl'); + const usageLine = (uuid: string, ts: string): string => + JSON.stringify({ + type: 'assistant', + uuid, + timestamp: ts, + message: { model: 'claude-sonnet-5', usage: { input_tokens: 5 } }, + }); + const lines = [ + usageLine('a1', '2026-09-20T10:00:00Z'), + usageLine('a2', '2026-09-20T10:01:00Z'), // turn 1: 1 min active + // newer-format user turn: array content with a text block + JSON.stringify({ + type: 'user', + uuid: 'u2', + timestamp: '2026-09-20T10:02:00Z', + message: { role: 'user', content: [{ type: 'text', text: 'go again' }] }, + }), + usageLine('a3', '2026-09-20T10:10:00Z'), // turn 2: 0 active + ]; + await writeFile(file, lines.join('\n')); + const entry = await scanSessionFile(file); + expect(entry?.activeMs).toBe(60000); // turn 1 only; turn 2 has a single round + expect(entry?.longestTurnMs).toBe(60000); // not the uncapped merge (10 min) + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('local-command-stdout lines do not split turns', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-so-')); + try { + const file = path.join(dir, 'session-so.jsonl'); + const usageLine = (uuid: string, ts: string): string => + JSON.stringify({ + type: 'assistant', + uuid, + timestamp: ts, + message: { model: 'claude-sonnet-5', usage: { input_tokens: 5 } }, + }); + const lines = [ + usageLine('a1', '2026-09-20T10:00:00Z'), + // bash-mode command output: type user, string content, falsy isMeta — + // must NOT be a boundary (buildLedger does not split here either) + JSON.stringify({ + type: 'user', + uuid: 'u1', + timestamp: '2026-09-20T10:01:00Z', + message: { + role: 'user', + content: 'done', + }, + }), + usageLine('a2', '2026-09-20T10:02:00Z'), // same turn: 2 min total + ]; + await writeFile(file, lines.join('\n')); + const entry = await scanSessionFile(file); + expect(entry?.activeMs).toBe(120000); // 10:00→10:02, the stdout line did not reset + expect(entry?.longestTurnMs).toBe(120000); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('distinguishes cycles from scattered repeats', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'devtools-sc-')); + try { + const file = path.join(dir, 'session-sc.jsonl'); + const line = (uuid: string, n: number, command: string): string => + JSON.stringify({ + type: 'assistant', + uuid, + timestamp: `2026-09-20T10:0${n}:00Z`, + message: { + model: 'claude-sonnet-5', + usage: { input_tokens: 5 }, + content: [{ type: 'tool_use', id: uuid, name: 'Bash', input: { command } }], + }, + }); + // A,A,B,A,A,A — repeat(A)=5 but the longest back-to-back run is 3 + const lines = [ + line('a1', 0, 'probe one'), + line('a2', 1, 'probe one'), + line('a3', 2, 'git show xyz'), + line('a4', 3, 'probe one'), + line('a5', 4, 'probe one'), + line('a6', 5, 'probe one'), + ]; + await writeFile(file, lines.join('\n')); + const entry = await scanSessionFile(file); + // A appears 5 times total, but only its 3-run qualifies as a cycle + expect(entry?.cycles).toEqual([ + { key: 'Bash|probe one', count: 3, startTs: '2026-09-20T10:03:00Z', toolUseId: 'a6' }, + ]); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/test/main/ipc/configValidation.test.ts b/test/main/ipc/configValidation.test.ts index 4dcb9714..509601c1 100644 --- a/test/main/ipc/configValidation.test.ts +++ b/test/main/ipc/configValidation.test.ts @@ -117,6 +117,23 @@ describe('configValidation', () => { } }); + it('accepts valid notifications.loopDetection payload', () => { + const result = validateConfigUpdatePayload('notifications', { + loopDetection: { enabled: true, cycleThreshold: 3 }, + }); + expect(result.valid).toBe(true); + }); + + it('rejects loopDetection with non-integer threshold', () => { + const result = validateConfigUpdatePayload('notifications', { + loopDetection: { enabled: true, cycleThreshold: 2.5 }, + }); + expect(result.valid).toBe(false); + if (!result.valid) { + expect(result.error).toContain('integer >= 1'); + } + }); + it('accepts valid display updates', () => { const result = validateConfigUpdatePayload('display', { compactMode: true, diff --git a/test/main/services/discovery/SessionSearcher.test.ts b/test/main/services/discovery/SessionSearcher.test.ts index 385fa83c..5c998984 100644 --- a/test/main/services/discovery/SessionSearcher.test.ts +++ b/test/main/services/discovery/SessionSearcher.test.ts @@ -117,4 +117,39 @@ describe('SessionSearcher', () => { expect(userResults).toHaveLength(1); expect(aiResults).toHaveLength(1); }); + + it('finds a session by its /name and shows the name as the result title', async () => { + const projectsDir = fs.mkdtempSync(path.join(os.tmpdir(), 'session-searcher-name-')); + tempDirs.push(projectsDir); + + const projectId = 'project-3'; + const sessionId = 'session-3'; + const projectPath = path.join(projectsDir, projectId); + fs.mkdirSync(projectPath, { recursive: true }); + + const sessionPath = path.join(projectPath, `${sessionId}.jsonl`); + const lines = [ + JSON.stringify({ + uuid: 'user-3', + type: 'user', + timestamp: '2026-01-01T00:00:00.000Z', + message: { role: 'user', content: 'completely unrelated text' }, + isMeta: false, + }), + JSON.stringify({ + type: 'agent-name', + agentName: 'profile-tiles-env-leak', + sessionId, + }), + ]; + fs.writeFileSync(sessionPath, `${lines.join('\n')}\n`, 'utf8'); + + const searcher = new SessionSearcher(projectsDir); + const result = await searcher.searchSessions(projectId, 'profile-tiles-env-leak', 50); + + expect(result.totalMatches).toBe(1); + expect(result.results[0].sessionId).toBe(sessionId); + expect(result.results[0].sessionTitle).toBe('profile-tiles-env-leak'); + expect(result.results[0].context).toContain('profile-tiles-env-leak'); + }); }); diff --git a/test/main/services/infrastructure/FileWatcher.test.ts b/test/main/services/infrastructure/FileWatcher.test.ts index 3be1d487..97a8a9bd 100644 --- a/test/main/services/infrastructure/FileWatcher.test.ts +++ b/test/main/services/infrastructure/FileWatcher.test.ts @@ -30,12 +30,19 @@ vi.mock('../../../../src/main/services/error/ErrorDetector', () => ({ }, })); +// Mutable so loop-detection tests can flip notifications.loopDetection.enabled +const mockConfig = vi.hoisted(() => ({ + notifications: { + includeSubagentErrors: true, + triggers: [] as never[], + loopDetection: { enabled: false, cycleThreshold: 3 }, + }, +})); + vi.mock('../../../../src/main/services/infrastructure/ConfigManager', () => ({ ConfigManager: { getInstance: () => ({ - getConfig: () => ({ - notifications: { includeSubagentErrors: true, triggers: [] }, - }), + getConfig: () => mockConfig, }), }, })); @@ -87,6 +94,23 @@ function jsonlLine(uuid: string, text: string): string { ); } +/** Assistant JSONL line carrying one Read tool_use — identical calls key as one loop */ +function toolUseLine(uuid: string, toolUseId: string): string { + return ( + JSON.stringify({ + type: 'assistant', + uuid, + timestamp: '2026-01-01T00:00:00.000Z', + cwd: '/tmp/loop-project', + message: { + role: 'assistant', + model: 'claude-sonnet-5', + content: [{ type: 'tool_use', id: toolUseId, name: 'Read', input: { file_path: '/x/f' } }], + }, + }) + '\n' + ); +} + describe('FileWatcher', () => { beforeEach(() => { vi.useFakeTimers(); @@ -95,6 +119,7 @@ describe('FileWatcher', () => { afterEach(() => { vi.useRealTimers(); vi.restoreAllMocks(); + mockConfig.notifications.loopDetection.enabled = false; }); it('retries and starts watchers when directories appear later', () => { @@ -266,6 +291,93 @@ describe('FileWatcher', () => { fs.rmSync(tempDir, { recursive: true, force: true }); }); + it('baselines a discovered file silently: old loops never ring', async () => { + vi.useRealTimers(); + useRealExistsSync(); + vi.mocked(errorDetector.detectErrors).mockResolvedValue([]); + + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'filewatcher-discover-')); + const projectsDir = path.join(tempDir, 'projects'); + const projectDir = path.join(projectsDir, 'new-project'); + fs.mkdirSync(projectDir, { recursive: true }); + + // history already contains a full loop (3 identical calls) + const filePath = path.join(projectDir, 'session-1.jsonl'); + fs.writeFileSync( + filePath, + toolUseLine('a1', 't1') + toolUseLine('a2', 't2') + toolUseLine('a3', 't3'), + 'utf8' + ); + const sizeAtDiscovery = fs.statSync(filePath).size; + + const dataCache = new DataCache(50, 10, false); + const notificationManager = createMockNotificationManager(); + const watcher = new FileWatcher(dataCache, projectsDir, path.join(tempDir, 'todos')); + watcher.setNotificationManager(notificationManager); + + const watcherAny = watcher as unknown as { + lastProcessedLineCount: Map; + lastProcessedSize: Map; + runCatchUpScan: () => Promise; + }; + + // fs.watch never fired for this file (mocked watch is silent) — catch-up + // must discover it, but its history predates the watcher: silent baseline + await watcherAny.runCatchUpScan(); + + expect(errorDetector.detectErrors).not.toHaveBeenCalled(); + expect(notificationManager.addError).not.toHaveBeenCalled(); + expect(watcherAny.lastProcessedSize.get(filePath)).toBe(sizeAtDiscovery); + expect(watcherAny.activeSessionFiles.has(filePath)).toBe(true); + + watcher.stop(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + + it('rings for a loop that develops after discovery', async () => { + vi.useRealTimers(); + useRealExistsSync(); + vi.mocked(errorDetector.detectErrors).mockResolvedValue([]); + + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'filewatcher-grow-')); + const projectsDir = path.join(tempDir, 'projects'); + const projectDir = path.join(projectsDir, 'new-project'); + fs.mkdirSync(projectDir, { recursive: true }); + + const filePath = path.join(projectDir, 'session-1.jsonl'); + fs.writeFileSync(filePath, jsonlLine('u1', 'hello'), 'utf8'); + + const dataCache = new DataCache(50, 10, false); + const notificationManager = createMockNotificationManager(); + const watcher = new FileWatcher(dataCache, projectsDir, path.join(tempDir, 'todos')); + watcher.setNotificationManager(notificationManager); + + const watcherAny = watcher as unknown as { + runCatchUpScan: () => Promise; + }; + + // discovery baselines the file silently + mockConfig.notifications.loopDetection.enabled = true; + await watcherAny.runCatchUpScan(); + expect(errorDetector.detectErrors).not.toHaveBeenCalled(); + + // now a NEW loop develops (3 identical calls appended post-discovery) + fs.appendFileSync( + filePath, + toolUseLine('a4', 't4') + toolUseLine('a5', 't5') + toolUseLine('a6', 't6'), + 'utf8' + ); + await watcherAny.runCatchUpScan(); + + expect(errorDetector.detectErrors).toHaveBeenCalled(); + expect(notificationManager.addError).toHaveBeenCalledWith( + expect.objectContaining({ source: 'loop' }) + ); + + watcher.stop(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + it('skips files with no size change', async () => { vi.useRealTimers(); useRealExistsSync(); @@ -511,6 +623,193 @@ describe('FileWatcher', () => { }); }); + // =========================================================================== + // Loop Detection Wiring + // =========================================================================== + + describe('loop detection wiring', () => { + it('emits exactly one synthetic DetectedError at threshold from an incremental append', async () => { + vi.useRealTimers(); + useRealExistsSync(); + mockConfig.notifications.loopDetection.enabled = true; + vi.mocked(errorDetector.detectErrors).mockResolvedValue([]); + + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'filewatcher-loop-')); + const projectsDir = path.join(tempDir, 'projects'); + const projectDir = path.join(projectsDir, 'test-project'); + fs.mkdirSync(projectDir, { recursive: true }); + + const filePath = path.join(projectDir, 'session-1.jsonl'); + fs.writeFileSync(filePath, jsonlLine('u1', 'hello'), 'utf8'); + + const dataCache = new DataCache(50, 10, false); + const notificationManager = createMockNotificationManager(); + const watcher = new FileWatcher(dataCache, projectsDir, path.join(tempDir, 'todos')); + watcher.setNotificationManager(notificationManager); + + const run = (): Promise => + ( + watcher as unknown as { + detectErrorsInSessionFile: (p: string, s: string, f: string) => Promise; + } + ).detectErrorsInSessionFile('test-project', 'session-1', filePath); + + // First read establishes the baseline — whole-file replay must not feed the detector + await run(); + expect(notificationManager.addError).not.toHaveBeenCalled(); + + // Incremental append of 3 identical Read calls (threshold 3) -> one incident + fs.appendFileSync( + filePath, + toolUseLine('a1', 't1') + toolUseLine('a2', 't2') + toolUseLine('a3', 't3'), + 'utf8' + ); + await run(); + + expect(notificationManager.addError).toHaveBeenCalledTimes(1); + const loopError = vi.mocked(notificationManager.addError).mock.calls[0][0]; + expect(loopError.source).toBe('loop'); + expect(loopError.triggerName).toBe('Loop detected'); + expect(loopError.toolUseId).toBe('t3'); + expect(loopError.message).toContain('Read|/x/f ×3'); + // pre-batch base (1 seed line) + batchIndex 2 + 1 + expect(loopError.lineNumber).toBe(4); + expect(loopError.sessionId).toBe('session-1'); + expect(loopError.projectId).toBe('test-project'); + + // Counts 4 and 5 stay below the doubling bar — quiet + fs.appendFileSync(filePath, toolUseLine('a4', 't4') + toolUseLine('a5', 't5'), 'utf8'); + await run(); + expect(notificationManager.addError).toHaveBeenCalledTimes(1); + + watcher.stop(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + + it('never fires for agent files or subagent calls', async () => { + vi.useRealTimers(); + useRealExistsSync(); + mockConfig.notifications.loopDetection.enabled = true; + vi.mocked(errorDetector.detectErrors).mockResolvedValue([]); + + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'filewatcher-loop-agent-')); + const projectsDir = path.join(tempDir, 'projects'); + const projectDir = path.join(projectsDir, 'test-project'); + fs.mkdirSync(projectDir, { recursive: true }); + + const dataCache = new DataCache(50, 10, false); + const notificationManager = createMockNotificationManager(); + const watcher = new FileWatcher(dataCache, projectsDir, path.join(tempDir, 'todos')); + watcher.setNotificationManager(notificationManager); + + const runWith = (file: string, subagentId?: string): Promise => + ( + watcher as unknown as { + detectErrorsInSessionFile: ( + p: string, + s: string, + f: string, + sub?: string + ) => Promise; + } + ).detectErrorsInSessionFile('test-project', 'session-1', file, subagentId); + + const seedAndLoop = (file: string): void => { + fs.writeFileSync(file, jsonlLine('u1', 'hello'), 'utf8'); + fs.appendFileSync( + file, + toolUseLine('a1', 't1') + toolUseLine('a2', 't2') + toolUseLine('a3', 't3'), + 'utf8' + ); + }; + + // Subagent-annotated file: baseline first, then an incremental append + // of 3 identical calls that would fire if the subagentId gate leaked + const agentPath = path.join(projectDir, 'session-1', 'subagents', 'agent-abc.jsonl'); + fs.mkdirSync(path.dirname(agentPath), { recursive: true }); + seedAndLoop(agentPath); + await runWith(agentPath, 'abc'); + fs.appendFileSync( + agentPath, + toolUseLine('a4', 't4') + toolUseLine('a5', 't5') + toolUseLine('a6', 't6'), + 'utf8' + ); + await runWith(agentPath, 'abc'); + expect(notificationManager.addError).not.toHaveBeenCalled(); + + // agent- named file arriving without subagentId: basename guard + const agentNamedPath = path.join(projectDir, 'agent-xyz.jsonl'); + seedAndLoop(agentNamedPath); + await runWith(agentNamedPath); + fs.appendFileSync( + agentNamedPath, + toolUseLine('a4', 't4') + toolUseLine('a5', 't5') + toolUseLine('a6', 't6'), + 'utf8' + ); + await runWith(agentNamedPath); + expect(notificationManager.addError).not.toHaveBeenCalled(); + + watcher.stop(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + + it('resets on truncation/rewrite: no phantom count, a fresh run notifies again', async () => { + vi.useRealTimers(); + useRealExistsSync(); + mockConfig.notifications.loopDetection.enabled = true; + vi.mocked(errorDetector.detectErrors).mockResolvedValue([]); + + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'filewatcher-loop-reset-')); + const projectsDir = path.join(tempDir, 'projects'); + const projectDir = path.join(projectsDir, 'test-project'); + fs.mkdirSync(projectDir, { recursive: true }); + + const filePath = path.join(projectDir, 'session-1.jsonl'); + const seed = jsonlLine('u1', 'hello'); + fs.writeFileSync(filePath, seed, 'utf8'); + + const dataCache = new DataCache(50, 10, false); + const notificationManager = createMockNotificationManager(); + const watcher = new FileWatcher(dataCache, projectsDir, path.join(tempDir, 'todos')); + watcher.setNotificationManager(notificationManager); + + const run = (): Promise => + ( + watcher as unknown as { + detectErrorsInSessionFile: (p: string, s: string, f: string) => Promise; + } + ).detectErrorsInSessionFile('test-project', 'session-1', filePath); + + await run(); // baseline + + fs.appendFileSync( + filePath, + toolUseLine('a1', 't1') + toolUseLine('a2', 't2') + toolUseLine('a3', 't3'), + 'utf8' + ); + await run(); + expect(notificationManager.addError).toHaveBeenCalledTimes(1); + + // Truncate back to the seed: fallback path resets detector state, no new incident + fs.writeFileSync(filePath, seed, 'utf8'); + await run(); + expect(notificationManager.addError).toHaveBeenCalledTimes(1); + + // Re-append the same loop with fresh ids: a fresh run notifies again, not ×6 + fs.appendFileSync( + filePath, + toolUseLine('b1', 'u1') + toolUseLine('b2', 'u2') + toolUseLine('b3', 'u3'), + 'utf8' + ); + await run(); + expect(notificationManager.addError).toHaveBeenCalledTimes(2); + expect(vi.mocked(notificationManager.addError).mock.calls[1][0].message).toContain('×3'); + + watcher.stop(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + }); + // =========================================================================== // Timer Lifecycle Tests // =========================================================================== diff --git a/test/main/utils/jsonl.test.ts b/test/main/utils/jsonl.test.ts index c7a1cf05..347e0221 100644 --- a/test/main/utils/jsonl.test.ts +++ b/test/main/utils/jsonl.test.ts @@ -3,7 +3,15 @@ import * as os from 'os'; import * as path from 'path'; import { describe, expect, it } from 'vitest'; -import { analyzeSessionFileMetadata, calculateMetrics } from '../../../src/main/utils/jsonl'; +import { + analyzeSessionFileMetadata, + calculateMetrics, + mergeAssistantFragments, + parseJsonlFile, + readSessionName, +} from '../../../src/main/utils/jsonl'; +import { ChunkBuilder } from '../../../src/main/services/analysis/ChunkBuilder'; +import { isAIChunk } from '../../../src/main/types'; import type { ParsedMessage } from '../../../src/main/types'; // Helper to create a minimal ParsedMessage @@ -136,6 +144,208 @@ describe('jsonl', () => { }); }); + describe('streaming fragments (one request split across JSONL lines)', () => { + const usageA = { input_tokens: 1000, output_tokens: 100 }; + + it('mergeAssistantFragments joins lines sharing a message.id into one request', () => { + const messages = [ + createMessage({ + uuid: 'a1', + messageId: 'msg_1', + usage: usageA, + content: [{ type: 'text', text: 'hi' }], + }), + createMessage({ + uuid: 'a2', + messageId: 'msg_1', + usage: usageA, + content: [{ type: 'tool_use', id: 't1', name: 'Read', input: {} }], + toolCalls: [{ id: 't1', name: 'Read', input: {}, isTask: false }], + }), + ]; + + const merged = mergeAssistantFragments(messages); + expect(merged).toHaveLength(1); + expect(merged[0].uuid).toBe('a1'); // first fragment anchors the round + expect(merged[0].usage).toEqual(usageA); + expect(merged[0].content).toHaveLength(2); // text + tool_use + expect(merged[0].toolCalls).toHaveLength(1); + }); + + it('mergeAssistantFragments joins string-content fragments without dropping text', () => { + const messages = [ + createMessage({ uuid: 'a1', messageId: 'msg_1', usage: usageA, content: 'hel' }), + createMessage({ uuid: 'a2', messageId: 'msg_1', usage: usageA, content: 'lo' }), + ]; + + const merged = mergeAssistantFragments(messages); + expect(merged).toHaveLength(1); + expect(merged[0].content).toBe('hello'); + }); + + it('keeps requestId-bearing snapshot lines untouched (dedupe path handles them)', () => { + const messages = [ + createMessage({ + uuid: 'a1', + requestId: 'req_1', + messageId: 'msg_1', + content: [{ type: 'text', text: 'partial' }], + }), + createMessage({ + uuid: 'a2', + requestId: 'req_1', + messageId: 'msg_1', + content: [{ type: 'text', text: 'full' }], + }), + ]; + + expect(mergeAssistantFragments(messages)).toHaveLength(2); + }); + + it('calculateMetrics bills a fragmented request once', () => { + const messages = [ + createMessage({ + uuid: 'a1', + messageId: 'msg_1', + usage: usageA, + content: [{ type: 'text', text: 'x' }], + }), + createMessage({ + uuid: 'a2', + messageId: 'msg_1', + usage: usageA, + content: [{ type: 'tool_use', id: 't1', name: 'Read', input: {} }], + }), + ]; + + // one request = one usage count (1100), not two (2200) + expect(calculateMetrics(messages).totalTokens).toBe(1100); + }); + + it('parseJsonlFile merges fragment lines before any consumer sees them', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'devtools-frag-')); + const file = path.join(dir, 'session.jsonl'); + const entry = (uuid: string, parentUuid: string | null, content: unknown[]) => ({ + type: 'assistant' as const, + uuid, + parentUuid, + timestamp: '2024-01-01T10:00:00Z', + isSidechain: false, + isMeta: false, + message: { + role: 'assistant' as const, + id: 'msg_1', + model: 'glm-4.6', + content, + usage: { input_tokens: 1000, output_tokens: 100 }, + }, + }); + fs.writeFileSync( + file, + [ + JSON.stringify({ + type: 'user', + uuid: 'u1', + parentUuid: null, + timestamp: '2024-01-01T10:00:00Z', + isSidechain: false, + isMeta: false, + message: { role: 'user', content: 'go' }, + }), + JSON.stringify(entry('a1', 'u1', [{ type: 'text', text: 'hi' }])), + JSON.stringify( + entry('a2', 'a1', [{ type: 'tool_use', id: 't1', name: 'Read', input: {} }]) + ), + JSON.stringify({ + type: 'user', + uuid: 'u2', + parentUuid: 'a2', + timestamp: '2024-01-01T10:00:03Z', + isSidechain: false, + isMeta: false, + message: { role: 'user', content: 'ok' }, + }), + ].join('\n') + ); + + const messages = await parseJsonlFile(file); + // two fragment lines with the same message.id arrive as ONE message + expect(messages.filter((m) => m.type === 'assistant')).toHaveLength(1); + + const chunks = new ChunkBuilder().buildChunks(messages); + const ai = chunks.filter(isAIChunk); + expect(ai).toHaveLength(1); + expect(ai[0].responses).toHaveLength(1); + expect(ai[0].responses[0].content).toHaveLength(2); + fs.rmSync(dir, { recursive: true, force: true }); + }); + }); + + describe('session name (agent-name / ai-title)', () => { + const USER = { + type: 'user', + uuid: 'u1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { role: 'user', content: 'hello' }, + isMeta: false, + }; + + function writeSession(lines: unknown[]): { filePath: string; cleanup: () => void } { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'jsonl-name-')); + const filePath = path.join(tempDir, 'session.jsonl'); + fs.writeFileSync(filePath, `${lines.map((l) => JSON.stringify(l)).join('\n')}\n`, 'utf8'); + return { + filePath, + cleanup: () => { + try { + fs.rmSync(tempDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 200 }); + } catch { + // best-effort + } + }, + }; + } + + it('extracts last agent-name, falling back to ai-title', async () => { + const { filePath, cleanup } = writeSession([ + USER, + { type: 'agent-name', agentName: 'first-name', sessionId: 's1' }, + { type: 'ai-title', aiTitle: 'auto title', sessionId: 's1' }, + { type: 'agent-name', agentName: 'renamed', sessionId: 's1' }, + ]); + try { + const meta = await analyzeSessionFileMetadata(filePath); + expect(meta.name).toBe('renamed'); + await expect(readSessionName(filePath)).resolves.toBe('renamed'); + } finally { + cleanup(); + } + }); + + it('falls back to ai-title when no agent-name, null when unnamed', async () => { + const withTitle = writeSession([ + USER, + { type: 'ai-title', aiTitle: 'auto title', sessionId: 's1' }, + ]); + try { + await expect(readSessionName(withTitle.filePath)).resolves.toBe('auto title'); + const meta = await analyzeSessionFileMetadata(withTitle.filePath); + expect(meta.name).toBe('auto title'); + } finally { + withTitle.cleanup(); + } + + const unnamed = writeSession([USER]); + try { + await expect(readSessionName(unnamed.filePath)).resolves.toBeNull(); + const meta = await analyzeSessionFileMetadata(unnamed.filePath); + expect(meta.name).toBeNull(); + } finally { + unnamed.cleanup(); + } + }); + }); + describe('analyzeSessionFileMetadata', () => { it('should extract first message, count, ongoing state, and git branch in one pass', async () => { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'jsonl-meta-')); @@ -182,5 +392,235 @@ describe('jsonl', () => { } } }); + + it('sums total spend across all assistant usage in one pass', async () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'jsonl-spend-')); + try { + const filePath = path.join(tempDir, 'session.jsonl'); + const lines = [ + JSON.stringify({ + type: 'user', + uuid: 'u1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { role: 'user', content: 'go' }, + isMeta: false, + }), + JSON.stringify({ + type: 'assistant', + uuid: 'a1', + timestamp: '2026-01-01T00:00:01.000Z', + message: { + role: 'assistant', + model: 'claude-fable-5-1', + content: [{ type: 'text', text: 'ok' }], + usage: { + input_tokens: 100, + cache_read_input_tokens: 5000, + cache_creation_input_tokens: 200, + output_tokens: 50, + }, + }, + }), + JSON.stringify({ + type: 'assistant', + uuid: 'a2', + timestamp: '2026-01-01T00:00:02.000Z', + message: { + role: 'assistant', + model: 'claude-fable-5-1', + content: [{ type: 'text', text: 'done' }], + usage: { input_tokens: 10, output_tokens: 5 }, + }, + }), + // sidechain counts too — this file's transcript cost + JSON.stringify({ + type: 'assistant', + uuid: 'a3', + isSidechain: true, + timestamp: '2026-01-01T00:00:03.000Z', + message: { + role: 'assistant', + model: 'claude-fable-5-1', + content: [], + usage: { input_tokens: 7, output_tokens: 3 }, + }, + }), + // synthetic / no-usage lines contribute nothing + JSON.stringify({ + type: 'assistant', + uuid: 'a4', + timestamp: '2026-01-01T00:00:04.000Z', + message: { role: 'assistant', model: '', content: [] }, + }), + ]; + fs.writeFileSync(filePath, `${lines.join('\n')}\n`, 'utf8'); + + const result = await analyzeSessionFileMetadata(filePath); + + // 100+5000+200+50 + 10+5 + 7+3 = 5375 + expect(result.totalTokens).toBe(5375); + } finally { + try { + fs.rmSync(tempDir, { + recursive: true, + force: true, + maxRetries: 5, + retryDelay: 200, + }); + } catch { + // Best-effort cleanup; ignore ENOTEMPTY on Windows when dir is in use + } + } + }); + it('counts turns — AI response groups, same rule as the chunk pipeline', async () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'jsonl-turns-')); + try { + const filePath = path.join(tempDir, 'session.jsonl'); + const user = ( + uuid: string, + parentUuid: string | null, + content: string, + isSidechain = false + ): string => + JSON.stringify({ + type: 'user', + uuid, + parentUuid: parentUuid ?? undefined, + timestamp: '2026-01-01T00:00:00.000Z', + isMeta: false, + isSidechain, + message: { role: 'user', content }, + }); + const assistant = (uuid: string, parentUuid: string, model: string): string => + JSON.stringify({ + type: 'assistant', + uuid, + parentUuid, + timestamp: '2026-01-01T00:00:01.000Z', + message: { + role: 'assistant', + model, + content: [{ type: 'text', text: 'ok' }], + usage: { input_tokens: 10, output_tokens: 2 }, + }, + }); + // root (parentUuid null) is hard noise everywhere; an assistant run + // closes on user/system/compact and counts exactly one turn — + // continuations, synthetic replies and sidechains never break a group + const lines = [ + user('u1', null, 'go'), + assistant('a1', 'u1', 'claude-fable-5-1'), + assistant('a1b', 'a1', 'claude-fable-5-1'), // continuation — same group + user('u2', 'a1b', 'again'), + assistant('a2-synthetic', 'u2', ''), // hard noise — no break + assistant('a2', 'a2-synthetic', 'claude-fable-5-1'), + user('sys', 'a2', 'ok'), // system break + assistant('a3', 'sys', 'claude-fable-5-1'), + user('side-u', 'a3', 'sidechat', true), // sidechain — skipped + assistant('side-a', 'side-u', 'claude-fable-5-1'), + user('u3', 'a3', 'more'), + assistant('a4', 'u3', 'claude-fable-5-1'), // still open — closed at EOF + ]; + fs.writeFileSync(filePath, `${lines.join('\n')}\n`, 'utf8'); + + const result = await analyzeSessionFileMetadata(filePath); + + // groups: [a1,a1b] [a2] [a3] [a4] = 4 + expect(result.turnCount).toBe(4); + } finally { + try { + fs.rmSync(tempDir, { + recursive: true, + force: true, + maxRetries: 5, + retryDelay: 200, + }); + } catch { + // Best-effort cleanup; ignore ENOTEMPTY on Windows when dir is in use + } + } + }); + + it('parity — scan turnCount equals the chunk pipeline AIChunk count', async () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'jsonl-parity-')); + try { + const msg = (over: Partial): ParsedMessage => ({ + uuid: over.uuid ?? 'x', + parentUuid: over.parentUuid ?? null, + type: over.type ?? 'assistant', + timestamp: new Date('2026-01-01T00:00:00.000Z'), + content: over.content ?? '', + isSidechain: over.isSidechain ?? false, + isMeta: over.isMeta ?? false, + isCompactSummary: over.isCompactSummary ?? false, + toolCalls: [], + toolResults: [], + }); + const ai = (uuid: string, parentUuid: string): ParsedMessage => + msg({ uuid, parentUuid, content: [{ type: 'text', text: 'ok' }] }); + const messages: ParsedMessage[] = [ + msg({ uuid: 'u1', type: 'user', content: 'go' }), // root → hard noise + ai('a1', 'u1'), + ai('a2', 'a1'), + msg({ uuid: 'u2', parentUuid: 'a2', type: 'user', content: 'again' }), // break + msg({ + uuid: 's1', + parentUuid: 'u2', + content: [{ type: 'text', text: 'side' }], + isSidechain: true, + }), + msg({ + uuid: 'sys', + parentUuid: 'u2', + type: 'user', + content: 'x', + }), // break + ai('a3', 'sys'), + msg({ uuid: 'c1', parentUuid: 'a3', type: 'user', isCompactSummary: true }), // break + ai('a4', 'c1'), // closed at EOF + ]; + // Serialize the same objects to JSONL the way the scanner reads them + const toEntry = (m: ParsedMessage): string => + JSON.stringify({ + uuid: m.uuid, + parentUuid: m.parentUuid ?? undefined, + type: m.type, + timestamp: '2026-01-01T00:00:00.000Z', + isMeta: m.isMeta || undefined, + isSidechain: m.isSidechain || undefined, + isCompactSummary: m.isCompactSummary || undefined, + message: + m.type === 'user' + ? { role: 'user', content: m.content } + : { + role: 'assistant', + model: 'claude-fable-5-1', + content: m.content, + usage: { input_tokens: 10, output_tokens: 2 }, + }, + }); + const filePath = path.join(tempDir, 'session.jsonl'); + fs.writeFileSync(filePath, `${messages.map(toEntry).join('\n')}\n`, 'utf8'); + + const scan = await analyzeSessionFileMetadata(filePath); + const chunks = new ChunkBuilder().buildChunks(messages); + const aiChunks = chunks.filter(isAIChunk).length; + + // groups: [a1,a2] [a3] [a4] = 3 on both paths + expect(aiChunks).toBe(3); + expect(scan.turnCount).toBe(aiChunks); + } finally { + try { + fs.rmSync(tempDir, { + recursive: true, + force: true, + maxRetries: 5, + retryDelay: 200, + }); + } catch { + // Best-effort cleanup; ignore ENOTEMPTY on Windows when dir is in use + } + } + }); }); }); diff --git a/test/main/utils/loopDetection.test.ts b/test/main/utils/loopDetection.test.ts new file mode 100644 index 00000000..4d041c01 --- /dev/null +++ b/test/main/utils/loopDetection.test.ts @@ -0,0 +1,244 @@ +import { describe, expect, it } from 'vitest'; + +import type { ParsedMessage } from '../../../src/main/types'; +import { LoopDetector, StallDetector } from '../../../src/main/utils/loopDetection'; + +/** Minimal ParsedMessage fixture: one assistant line carrying tool calls. */ +const assistant = ( + uuid: string, + calls: { id: string; name: string; input?: Record }[], + opts: { sidechain?: boolean; model?: string } = {} +): ParsedMessage => + ({ + uuid, + parentUuid: null, + type: 'assistant', + timestamp: new Date(), + content: [], + model: opts.model ?? 'claude-sonnet-5', + isSidechain: opts.sidechain ?? false, + isMeta: false, + toolCalls: calls.map((c) => ({ id: c.id, name: c.name, input: c.input ?? {}, isTask: false })), + toolResults: [], + }) as unknown as ParsedMessage; + +const read = ( + id: string, + file: string +): { id: string; name: string; input: Record } => ({ + id, + name: 'Read', + input: { file_path: file }, +}); + +describe('LoopDetector', () => { + it('fires once at threshold, stays quiet until doubling', () => { + const det = new LoopDetector(); + const msgs = [1, 2, 3, 4, 5, 6].map((n) => assistant(`m${n}`, [read(`t${n}`, '/x/f')])); + const first = det.feed('s1', msgs.slice(0, 3), 3); + expect(first).toEqual({ + key: 'Read|/x/f', + count: 3, + toolUseId: 't3', + cwd: undefined, + batchIndex: 2, + }); + // counts 4 and 5 are below the 2x doubling bar -> quiet + expect(det.feed('s1', msgs.slice(3, 5), 3)).toBeNull(); + // 6th call doubles the notified length 3 -> re-notify + const second = det.feed('s1', [msgs[5]], 3); + expect(second?.count).toBe(6); + expect(second?.toolUseId).toBe('t6'); + }); + + it('keeps counting after an incident to the end of the batch', () => { + const det = new LoopDetector(); + const msgs = [1, 2, 3, 4, 5, 6].map((n) => assistant(`m${n}`, [read(`t${n}`, '/x/f')])); + // first incident at 3; the rest of the batch still counts (no second incident) + expect(det.feed('s', msgs, 3)?.count).toBe(3); + // streak reached 6 in-batch, so the doubling bar (2×3) is crossed at 7, + // and the streaming snapshot dup of t6 is skipped (true last call id) + const next = det.feed( + 's', + [assistant('d', [read('t6', '/x/f')]), assistant('m7', [read('t7', '/x/f')])], + 3 + ); + expect(next?.count).toBe(7); + expect(next?.toolUseId).toBe('t7'); + }); + + it('dedupes streaming snapshots by consecutive toolUseId', () => { + const det = new LoopDetector(); + const msgs = [ + assistant('m1', [read('t1', '/x/f')]), + assistant('m2', [read('t1', '/x/f')]), + assistant('m3', [read('t2', '/x/f')]), + ]; + expect(det.feed('s', msgs, 2)).toEqual({ + key: 'Read|/x/f', + count: 2, + toolUseId: 't2', + cwd: undefined, + batchIndex: 2, + }); + }); + + it('resets the run when the key changes; a fresh run notifies again', () => { + const det = new LoopDetector(); + const msgs = [ + ...[1, 2].map((n) => assistant(`a${n}`, [read(`t${n}`, '/x/f')])), // 2x Read + assistant('b1', [read('t3', '/x/other')]), // key change -> run resets + ...[4, 5, 6].map((n) => assistant(`a${n}`, [read(`t${n}`, '/x/f')])), // fresh run + ]; + det.feed('s', msgs.slice(0, 3), 3); // 2x Read + break + const incident = det.feed('s', msgs.slice(3), 3); // 3x Read -> fire + expect(incident?.count).toBe(3); + }); + + it('files are independent', () => { + const det = new LoopDetector(); + const a = [1, 2].map((n) => assistant(`p${n}`, [read(`p${n}`, '/x/f')])); + const b = [1, 2, 3].map((n) => assistant(`q${n}`, [read(`q${n}`, '/x/f')])); + det.feed('s1', a, 3); + det.feed('s2', b, 3); + const s1Incident = det.feed('s1', [assistant('p3', [read('p3', '/x/f')])], 3); + expect(s1Incident?.count).toBe(3); + }); + + it('ignores sidechain and synthetic messages', () => { + const det = new LoopDetector(); + const msgs = [ + assistant('n1', [read('t1', '/x/f')]), + assistant('n2', [read('t2', '/x/f')], { sidechain: true }), + assistant('n3', [read('t3', '/x/f')], { model: '' }), + assistant('n4', [read('t4', '/x/f')]), + ]; + // only 2 counted, below threshold + expect(det.feed('s', msgs, 3)).toBeNull(); + expect(det.feed('s', [assistant('n5', [read('t5', '/x/f')])], 3)?.count).toBe(3); + }); + + it('reset() drops state: after reset the run restarts from scratch', () => { + const det = new LoopDetector(); + const msgs = [1, 2, 3].map((n) => assistant(`m${n}`, [read(`t${n}`, '/x/f')])); + expect(det.feed('s', msgs, 3)?.count).toBe(3); + det.reset('s'); + expect(det.feed('s', [], 3)).toBeNull(); + expect(det.feed('s', msgs, 3)?.count).toBe(3); + }); +}); + +let seq = 0; + +// Minimal main-chain assistant round with usage and optional Bash tool calls +function assistantMsg(overrides: { + input?: number; + cacheRead?: number; + output?: number; + commands?: string[]; + messageId?: string; +}): ParsedMessage { + seq += 1; + const { input = 0, cacheRead = 0, output = 0, commands = [], messageId } = overrides; + return { + uuid: `a${seq}`, + parentUuid: null, + type: 'assistant', + timestamp: new Date('2026-09-25T11:00:00Z'), + content: commands.map((command, i) => ({ + type: 'tool_use', + id: `t${seq}-${i}`, + name: 'Bash', + input: { command }, + })), + toolCalls: commands.map((command, i) => ({ + id: `t${seq}-${i}`, + name: 'Bash', + input: { command }, + isTask: false, + })), + toolResults: [], + isSidechain: false, + isMeta: false, + isCompactSummary: false, + model: 'glm-5.3-flash', + usage: { + input_tokens: input, + cache_read_input_tokens: cacheRead, + cache_creation_input_tokens: 0, + output_tokens: output, + }, + messageId, + } as unknown as ParsedMessage; +} + +// Live 0779a2bc shape: one echo-marker round re-reading ~134k, +24 delta +const echoRound = (n: number, messageId?: string) => + assistantMsg({ + input: 100 + n, + cacheRead: 134_100 + 23 * n, // context = 134_200 + 24n — grows by the tool result only + output: 19 + n, + commands: [`echo ${String.fromCharCode(119 + n)}`], + messageId, + }); + +describe('StallDetector', () => { + const threshold = 4; // notifications.loopDetection.cycleThreshold default + + it('notifies once the echo-marker streak reaches the threshold', () => { + const detector = new StallDetector(); + const batch = [ + // baseline work round: loud output, context jumps + assistantMsg({ input: 200, cacheRead: 134_000, output: 2_000, commands: ['cat plan.md'] }), + echoRound(1), + echoRound(2), + echoRound(3), + ]; + expect(detector.feed('/s.jsonl', batch, threshold)).toBeNull(); // streak 3 + + const incident = detector.feed('/s.jsonl', [echoRound(4)], threshold); + expect(incident).not.toBeNull(); + expect(incident?.count).toBe(4); + expect(incident?.toolUseId).toBe('t5-0'); + }); + + it('counts GLM-proxy fragments of one request once (messageId dedup)', () => { + const detector = new StallDetector(); + // request A streamed as two JSONL lines, EACH carrying the full usage — + // counting both would inflate the streak (the 66c45cf lesson) + const fragA = echoRound(1, 'msg_a'); + const fragA2 = { + ...assistantMsg({ input: 101, cacheRead: 134_123, output: 20 }), + messageId: 'msg_a', + toolCalls: [], + content: [{ type: 'text', text: 'text block of the same request' }], + } as unknown as ParsedMessage; + const batch = [ + assistantMsg({ input: 200, cacheRead: 134_000, output: 2_000, commands: ['cat plan.md'] }), + fragA, + fragA2, + echoRound(2, 'msg_b'), + echoRound(3, 'msg_c'), + ]; + // fragments bill once → streak 3 < threshold → silent + expect(detector.feed('/s.jsonl', batch, threshold)).toBeNull(); + + const incident = detector.feed('/s.jsonl', [echoRound(4, 'msg_d')], threshold); + expect(incident?.count).toBe(4); + }); + + it('a real work round resets the streak', () => { + const detector = new StallDetector(); + const batch = [ + echoRound(1), + echoRound(2), + // work round: context jumps +2k — progress + assistantMsg({ input: 500, cacheRead: 134_300, output: 2_000, commands: ['edit file.ts'] }), + echoRound(3), + echoRound(4), + echoRound(5), + ]; + // without the reset the streak would be 5 → incident; it must stay 3 + expect(detector.feed('/s.jsonl', batch, threshold)).toBeNull(); + }); +}); diff --git a/test/main/utils/toolExtraction.test.ts b/test/main/utils/toolExtraction.test.ts new file mode 100644 index 00000000..e111086f --- /dev/null +++ b/test/main/utils/toolExtraction.test.ts @@ -0,0 +1,43 @@ +/** + * Tests for tool-call extraction (src/main/utils/toolExtraction.ts), + * covering the Task → Agent tool rename in Claude Code 2.1.63. + */ +import { describe, expect, it } from 'vitest'; + +import { extractToolCalls } from '../../../src/main/utils/toolExtraction'; +import type { ContentBlock } from '../../../src/main/types'; + +describe('extractToolCalls', () => { + it('treats Agent-named blocks as subagent spawns (Task renamed in 2.1.63)', () => { + const blocks: ContentBlock[] = [ + { + type: 'tool_use', + id: 'a1', + name: 'Agent', + input: { description: 'Explore X', prompt: 'go', subagent_type: 'Explore' }, + } as ContentBlock, + { + type: 'tool_use', + id: 'a2', + name: 'Task', + input: { description: 'Legacy spawn' }, + } as ContentBlock, + ]; + + const calls = extractToolCalls(blocks); + expect(calls).toHaveLength(2); + expect(calls[0].isTask).toBe(true); + expect(calls[0].taskDescription).toBe('Explore X'); + expect(calls[0].taskSubagentType).toBe('Explore'); + expect(calls[1].isTask).toBe(true); + expect(calls[1].taskDescription).toBe('Legacy spawn'); + }); + + it('keeps regular tools non-task', () => { + const calls = extractToolCalls([ + { type: 'tool_use', id: 'b1', name: 'Bash', input: { command: 'ls' } } as ContentBlock, + ]); + expect(calls[0].isTask).toBe(false); + expect(calls[0].taskDescription).toBeUndefined(); + }); +}); diff --git a/test/renderer/components/burnHeaderNav.test.ts b/test/renderer/components/burnHeaderNav.test.ts new file mode 100644 index 00000000..24ff9ea9 --- /dev/null +++ b/test/renderer/components/burnHeaderNav.test.ts @@ -0,0 +1,137 @@ +/** + * Burn-category navigation from the Visible Context panel must target the + * turn header — where the aggregate burn pills ("Wait 4.3M · 58 rd") live — + * not an individual item in the middle of a long turn. + */ + +import React, { act } from 'react'; +import { createRoot } from 'react-dom/client'; +import { afterEach, describe, expect, it, vi } from 'vitest'; + +(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true; + +import { LoopSection } from '../../../src/renderer/components/chat/SessionContextPanel/components/LoopSection'; +import { WaitLoopSection } from '../../../src/renderer/components/chat/SessionContextPanel/components/WaitLoopSection'; + +import type { LoopInjection, WaitLoopInjection } from '@renderer/types/contextInjection'; + +const waitInjection = { + id: 'wait-loop-ai-0', + category: 'wait-loop', + turnIndex: 0, + aiGroupId: 'ai-0', + estimatedTokens: 4_284_616, + roundCount: 58, + rounds: [{ uuid: 'r1', index: 7, outputTokens: 118, billed: 4_284_616 }], +} as WaitLoopInjection; + +const loopInjection = { + id: 'loop-ai-0', + category: 'loop', + turnIndex: 0, + aiGroupId: 'ai-0', + estimatedTokens: 2300, + breakdown: [{ key: 'Edit|/Users/x/prompt_test.go', count: 3, tokenCount: 938 }], + rounds: [{ uuid: 'r2', index: 9, billed: 2300, keys: ['Edit|/Users/x/prompt_test.go'] }], +} as LoopInjection; + +const loopInjectionWithTool: LoopInjection = { + ...loopInjection, + breakdown: [ + { key: 'Edit|/Users/x/prompt_test.go', count: 3, tokenCount: 938, toolUseId: 'toolu-1' }, + ], +}; + +async function mount(ui: React.ReactElement): Promise<{ host: HTMLElement; unmount: () => void }> { + const host = document.createElement('div'); + document.body.appendChild(host); + const root = createRoot(host); + await act(async () => { + root.render(ui); + await Promise.resolve(); + }); + return { + host, + unmount: () => { + act(() => { + root.unmount(); + }); + }, + }; +} + +function clickEntryByTitle(host: HTMLElement, title: string): void { + const button = Array.from(host.querySelectorAll('button')).find((b) => + b.textContent?.includes(title) + ); + if (!button) throw new Error(`entry button not found: ${title}`); + act(() => { + button.click(); + }); +} + +describe('burn navigation targets the aggregate', () => { + afterEach(() => { + document.body.innerHTML = ''; + }); + + it('Wait-loop entry requests turn navigation with header flash', async () => { + const onNavigateToTurn = vi.fn(); + const { host, unmount } = await mount( + React.createElement(WaitLoopSection, { + injections: [waitInjection], + tokenCount: 4_284_616, + isExpanded: true, + onToggle: () => undefined, + onNavigateToTurn, + }) + ); + + clickEntryByTitle(host, 'Turn 1'); + + expect(onNavigateToTurn).toHaveBeenCalledWith(0, { flashHeader: true }); + unmount(); + }); + + it('Loop entry without toolUseId requests turn navigation with header flash', async () => { + const onNavigateToTurn = vi.fn(); + const onNavigateToTool = vi.fn(); + const { host, unmount } = await mount( + React.createElement(LoopSection, { + injections: [loopInjection], + tokenCount: 2300, + isExpanded: true, + onToggle: () => undefined, + onNavigateToTool, + onNavigateToTurn, + }) + ); + + clickEntryByTitle(host, 'Edit|/Users/x/prompt_test.go'); + + expect(onNavigateToTurn).toHaveBeenCalledWith(0, { flashHeader: true }); + expect(onNavigateToTool).not.toHaveBeenCalled(); + unmount(); + }); + + it('Loop entry with toolUseId still deep-links the specific call', async () => { + const onNavigateToTurn = vi.fn(); + const onNavigateToTool = vi.fn(); + const { host, unmount } = await mount( + React.createElement(LoopSection, { + injections: [loopInjectionWithTool], + tokenCount: 2300, + isExpanded: true, + onToggle: () => undefined, + onNavigateToTool, + onNavigateToTurn, + }) + ); + + clickEntryByTitle(host, 'Edit|/Users/x/prompt_test.go'); + + expect(onNavigateToTool).toHaveBeenCalledWith(0, 'toolu-1'); + expect(onNavigateToTurn).not.toHaveBeenCalled(); + unmount(); + }); +}); diff --git a/test/renderer/hooks/navigationUtils.test.ts b/test/renderer/hooks/navigationUtils.test.ts index c01505e4..8c4ded8f 100644 --- a/test/renderer/hooks/navigationUtils.test.ts +++ b/test/renderer/hooks/navigationUtils.test.ts @@ -4,9 +4,10 @@ import { describe, expect, it } from 'vitest'; -import { findAIGroupBySubagentId } from '@renderer/hooks/navigation/utils'; +import { findAIGroupBySubagentId, isGroupHeaderAlarm } from '@renderer/hooks/navigation/utils'; import type { ChatItem } from '@renderer/types/groups'; +import type { TriggerColor } from '@shared/constants/triggerColors'; import type { Process } from '@main/types'; /** Minimal AI chat item factory for testing. */ @@ -66,3 +67,28 @@ describe('findAIGroupBySubagentId', () => { expect(findAIGroupBySubagentId(items, 'agent-target')).toBe('ai-1'); }); }); + +describe('isGroupHeaderAlarm', () => { + const base = { + highlightedGroupId: 'ai-1' as string | null, + highlightColor: 'red' as TriggerColor | null, + highlightToolUseId: null as string | null, + }; + + it('red group-level error navigation lights the turn header — loop deep-links', () => { + expect(isGroupHeaderAlarm('ai-1', base)).toBe(true); + }); + + it('a tool-targeted navigation keeps the alarm on the tool card, not the header', () => { + expect(isGroupHeaderAlarm('ai-1', { ...base, highlightToolUseId: 'toolu-9' })).toBe(false); + }); + + it('non-red highlights are not header alarms', () => { + expect(isGroupHeaderAlarm('ai-1', { ...base, highlightColor: 'blue' })).toBe(false); + }); + + it('other groups and cleared navigation are not header alarms', () => { + expect(isGroupHeaderAlarm('ai-2', base)).toBe(false); + expect(isGroupHeaderAlarm('ai-1', { ...base, highlightedGroupId: null })).toBe(false); + }); +}); diff --git a/test/renderer/hooks/tabNavigationHighlight.test.ts b/test/renderer/hooks/tabNavigationHighlight.test.ts new file mode 100644 index 00000000..53ca8c3d --- /dev/null +++ b/test/renderer/hooks/tabNavigationHighlight.test.ts @@ -0,0 +1,14 @@ +import { describe, expect, it } from 'vitest'; + +import { isPersistentHighlight } from '../../../src/renderer/hooks/useTabNavigationController'; + +describe('isPersistentHighlight', () => { + it('error navigation is alarm state: highlight never auto-clears', () => { + expect(isPersistentHighlight('error')).toBe(true); + }); + + it('search and non-target kinds keep the flash semantics', () => { + expect(isPersistentHighlight('search')).toBe(false); + expect(isPersistentHighlight('autoBottom')).toBe(false); + }); +}); diff --git a/test/renderer/store/notificationSlice.test.ts b/test/renderer/store/notificationSlice.test.ts index 5ffb106d..411c74be 100644 --- a/test/renderer/store/notificationSlice.test.ts +++ b/test/renderer/store/notificationSlice.test.ts @@ -347,6 +347,25 @@ describe('notificationSlice', () => { expect(store.getState().openTabs[0].pendingNavigation?.kind).toBe('error'); }); + it('deep-links loop notifications to the turn, not one tool card', () => { + const error = createMockError({ source: 'loop' }); + + store.getState().navigateToError(error); + + const nav = store.getState().openTabs[0].pendingNavigation; + expect(nav?.kind).toBe('error'); + expect(nav?.payload.toolUseId).toBeUndefined(); + }); + + it('keeps toolUseId deep-linking for non-loop errors — regression guard', () => { + const error = createMockError(); + + store.getState().navigateToError(error); + + const nav = store.getState().openTabs[0].pendingNavigation; + expect(nav?.payload.toolUseId).toBe('tool-1'); + }); + it('should set selectedSessionId even when switching from different project', () => { // Start with a different project selected store.setState({ @@ -378,6 +397,37 @@ describe('notificationSlice', () => { expect(store.getState().selectedSessionId).not.toBe('session-old'); expect(store.getState().selectedSessionId).toBe('session-target'); }); + + it('re-fetches session detail when navigating to an already-open tab', () => { + // First click: opens the tab; the mount fetch can have failed earlier + // (e.g. the session file did not exist yet) leaving a stale empty view + store.getState().navigateToError(createMockError()); + const tabId = store.getState().openTabs[0]?.id; + expect(tabId).toBeTruthy(); + mockAPI.getSessionDetail.mockClear(); + + // Second click: explicit navigation to an existing tab must refresh + // the session detail instead of reusing the stale empty state + store.getState().navigateToError(createMockError({ id: 'error-2' })); + + expect(mockAPI.getSessionDetail).toHaveBeenCalled(); + const last = mockAPI.getSessionDetail.mock.calls.at(-1); + expect(last?.[0]).toBe('project-1'); + expect(last?.[1]).toBe('session-target'); + }); + + it('fetches session detail when opening a brand-new tab', () => { + mockAPI.getSessionDetail.mockClear(); + + store.getState().navigateToError(createMockError()); + + // A notification-opened tab has no other fetch trigger (openTab does + // not load, SessionTabContent only refetches on the Retry button) + expect(mockAPI.getSessionDetail).toHaveBeenCalled(); + const last = mockAPI.getSessionDetail.mock.calls.at(-1); + expect(last?.[0]).toBe('project-1'); + expect(last?.[1]).toBe('session-target'); + }); }); describe('grouped mode (viewMode === grouped)', () => { diff --git a/test/renderer/utils/contextTracker.test.ts b/test/renderer/utils/contextTracker.test.ts new file mode 100644 index 00000000..07649408 --- /dev/null +++ b/test/renderer/utils/contextTracker.test.ts @@ -0,0 +1,274 @@ +/** + * Tests for the Loop and Wait-loop categories in contextTracker. + * + * Loop: repeat calls (2..N of a back-to-back identical series, keyed via + * bashStem(normalizeCallKey)) are bucketed into the loop category; the first + * call stays in tool-output. The streak threads across AI groups. + * + * Wait-loop: quiet rounds (contextSize >= 50k, output <= 300) contribute their + * full billed usage (input + cache + output) — same criterion as the CLI's + * wait_loop findings. + * + * Loop: tokens are the billed usage of rounds carrying repeat calls (each + * round once), not content estimates — rounds without usage contribute 0. + */ + +import { describe, expect, it } from 'vitest'; + +import { processSessionContextWithPhases, classifyRounds } from '@renderer/utils/contextTracker'; + +import type { AIGroup, UserGroup } from '@renderer/types/groups'; +import type { ChatItem } from '@renderer/types/groups'; +import type { ParsedMessage } from '@renderer/types/data'; +import type { SemanticStep } from '@main/types/chunks'; +import type { ContextStats } from '@renderer/types/contextInjection'; + +// Minimal assistant message with usage for wait-loop accounting +function assistantMsg( + usage: { + input?: number; + cacheRead?: number; + output?: number; + }, + toolUseIds: string[] = [] +): ParsedMessage { + return { + type: 'assistant', + usage: { + input_tokens: usage.input ?? 0, + cache_read_input_tokens: usage.cacheRead ?? 0, + cache_creation_input_tokens: 0, + output_tokens: usage.output ?? 0, + }, + content: toolUseIds.map((id) => ({ type: 'tool_use', id, name: 'Read', input: {} })), + } as unknown as ParsedMessage; +} + +// tool_call + tool_result step pair with the given result token count +function toolCall(id: string, filePath: string, resultTokens: number): SemanticStep[] { + const call: SemanticStep = { + id, + type: 'tool_call', + startTime: new Date('2026-09-23T10:00:00Z'), + durationMs: 10, + content: { toolName: 'Read', toolInput: { file_path: filePath } }, + context: 'main', + }; + const result: SemanticStep = { + id, + type: 'tool_result', + startTime: new Date('2026-09-23T10:00:01Z'), + durationMs: 5, + content: { toolName: 'Read', tokenCount: resultTokens }, + context: 'main', + }; + return [call, result]; +} + +// Minimal AI group carrying the given steps and response messages +function aiGroup( + id: string, + turnIndex: number, + steps: SemanticStep[], + responses: ParsedMessage[] +): ChatItem { + return { + type: 'ai', + group: { + id, + turnIndex, + startTime: new Date(), + endTime: new Date(), + durationMs: 100, + steps, + responses, + processes: [], + } as unknown as AIGroup, + }; +} + +function userGroup(): ChatItem { + return { type: 'user', group: { content: { text: '' } } as unknown as UserGroup }; +} + +function lastStats(items: ChatItem[]): Map { + const { statsMap } = processSessionContextWithPhases(items, '/proj'); + return statsMap; +} + +describe('contextTracker loop category', () => { + it('buckets the 2nd identical call into loop, first stays in tool-output', () => { + const items = [ + userGroup(), + aiGroup( + 'ai-0', + 0, + [...toolCall('t1', '/src/a.ts', 5000), ...toolCall('t2', '/src/a.ts', 5000)], + // one round carrying both calls bills the loop exactly once + [assistantMsg({ input: 10000, cacheRead: 50000, output: 200 }, ['t1', 't2'])] + ), + ]; + + const stats = lastStats(items).get('ai-0'); + expect(stats).toBeDefined(); + expect(stats!.tokensByCategory.loop).toBe(60200); + + const loopInj = stats!.newInjections.find((inj) => inj.category === 'loop'); + expect(loopInj).toBeDefined(); + if (loopInj?.category === 'loop') { + expect(loopInj.breakdown).toHaveLength(1); + expect(loopInj.breakdown[0].key).toBe('Read|/src/a.ts'); + expect(loopInj.breakdown[0].count).toBe(1); + expect(loopInj.breakdown[0].toolUseId).toBe('t2'); + expect(loopInj.breakdown[0].tokenCount).toBe(60200); + } + }); + + it('resets the streak when the key changes', () => { + const items = [ + userGroup(), + aiGroup( + 'ai-0', + 0, + [ + ...toolCall('t1', '/src/a.ts', 1000), + ...toolCall('t2', '/src/b.ts', 1000), + ...toolCall('t3', '/src/a.ts', 1000), + ...toolCall('t4', '/src/a.ts', 1000), + ], + [assistantMsg({ input: 10000, cacheRead: 50000, output: 200 }, ['t4'])] + ), + ]; + + const stats = lastStats(items).get('ai-0'); + expect(stats).toBeDefined(); + // only t4 is a repeat (t1->t2->t3 breaks the series) + expect(stats!.accumulatedCounts.loop).toBe(1); + expect(stats!.tokensByCategory.loop).toBe(60200); + }); + + it('threads the streak across AI groups', () => { + const items = [ + userGroup(), + aiGroup('ai-0', 0, [...toolCall('t1', '/src/a.ts', 1000)], []), + aiGroup( + 'ai-1', + 1, + [...toolCall('t2', '/src/a.ts', 1000)], + [assistantMsg({ input: 10000, cacheRead: 50000, output: 200 }, ['t2'])] + ), + ]; + + const statsMap = lastStats(items); + const first = statsMap.get('ai-0'); + const second = statsMap.get('ai-1'); + expect(first!.tokensByCategory.loop).toBe(0); + expect(second!.tokensByCategory.loop).toBeGreaterThan(0); + }); +}); + +describe('classifyRounds', () => { + it('marks quiet, repeat and normal rounds with 4-component billing', () => { + const responses = [ + // quiet idle tick: big context, nothing produced, NO tool call + assistantMsg({ input: 10000, cacheRead: 50000, output: 200 }), + // working round (carries a repeat call) — same usage, but NOT quiet + assistantMsg({ input: 10000, cacheRead: 50000, output: 200 }, ['t1']), + // active round — big output, not quiet + assistantMsg({ input: 10000, cacheRead: 50000, output: 5000 }), + // normal small round + assistantMsg({ input: 500, cacheRead: 500, output: 100 }), + ]; + const keyByToolId = new Map([['t1', 'Read|/src/a.ts']]); + + const rounds = classifyRounds(responses, keyByToolId); + + expect(rounds).toHaveLength(4); + expect(rounds[0]).toMatchObject({ + uuid: 'round-1', + index: 1, + quiet: true, + repeat: false, + billed: 60200, + }); + // a round with a tool call is work, never a quiet tick — even with tiny output + expect(rounds[1]).toMatchObject({ index: 2, quiet: false, repeat: true, billed: 60200 }); + expect(rounds[1].keys).toEqual(['Read|/src/a.ts']); + expect(rounds[2].quiet).toBe(false); + expect(rounds[3]).toMatchObject({ index: 4, quiet: false, repeat: false, billed: 1100 }); + }); + + it('marks echo-marker rounds (tool call, no context growth) as stalled', () => { + const responses = [ + // baseline work round: context jumps to ~134k, loud output — not stalled + assistantMsg({ input: 200, cacheRead: 134_000, output: 2_000 }, ['t0']), + // the live 0779a2bc shape: tool call each round, context grows by the + // tiny tool result only (+24), output ~19 tok — `echo w/v/u` markers + assistantMsg({ input: 100, cacheRead: 134_124, output: 19 }, ['t1']), + assistantMsg({ input: 101, cacheRead: 134_148, output: 20 }, ['t2']), + assistantMsg({ input: 102, cacheRead: 134_172, output: 21 }, ['t3']), + ]; + + const rounds = classifyRounds(responses); + + expect(rounds[0].stalled).toBe(false); // loud output — real work + expect(rounds[1].stalled).toBe(true); + expect(rounds[2].stalled).toBe(true); + expect(rounds[3].stalled).toBe(true); + }); + + it('a shrinking window (compaction) is not stalled; the first round never is', () => { + const responses = [ + assistantMsg({ input: 100, cacheRead: 134_000, output: 100 }, ['t1']), + // context collapse — compaction, progress of a kind, not a stall + assistantMsg({ input: 100, cacheRead: 50_000, output: 100 }, ['t2']), + ]; + + const rounds = classifyRounds(responses); + + expect(rounds[0].stalled).toBe(false); // first round: no baseline yet + expect(rounds[1].stalled).toBe(false); // negative delta + }); +}); + +describe('contextTracker wait-loop category', () => { + it('counts quiet rounds (>=50k context, <=300 out) and skips active rounds', () => { + const items = [ + userGroup(), + aiGroup( + 'ai-0', + 0, + [], + [ + assistantMsg({ input: 10000, cacheRead: 50000, output: 200 }), // quiet: 60k + 200 out + assistantMsg({ input: 10000, cacheRead: 50000, output: 5000 }), // active + ] + ), + ]; + + const stats = lastStats(items).get('ai-0'); + expect(stats).toBeDefined(); + expect(stats!.tokensByCategory.waitLoop).toBe(60200); + expect(stats!.accumulatedCounts.waitLoop).toBe(1); + // total also carries the real global CLAUDE.md injections picked up on the + // first group — assert the wait-loop burn is included, not the exact total + expect(stats!.totalEstimatedTokens).toBeGreaterThanOrEqual(60000); + }); + + it('ignores rounds below the context threshold', () => { + const items = [ + userGroup(), + aiGroup( + 'ai-0', + 0, + [], + [ + assistantMsg({ input: 5000, cacheRead: 5000, output: 100 }), // 10k < 50k + ] + ), + ]; + + const stats = lastStats(items).get('ai-0'); + expect(stats!.tokensByCategory.waitLoop).toBe(0); + }); +}); diff --git a/test/scripts/turnBudgetHook.test.ts b/test/scripts/turnBudgetHook.test.ts new file mode 100644 index 00000000..c35f9cef --- /dev/null +++ b/test/scripts/turnBudgetHook.test.ts @@ -0,0 +1,77 @@ +import { describe, expect, it } from 'vitest'; + +import { feedLine, percentile } from '../../src/cli/turnSpendStats'; + +function userLine(content: unknown, isMeta?: boolean): string { + return JSON.stringify({ + type: 'user', + isMeta: isMeta ?? false, + message: { role: 'user', content }, + }); +} + +function assistantLine(input: number, cacheRead: number, output: number): string { + return JSON.stringify({ + type: 'assistant', + message: { + role: 'assistant', + content: [], + model: 'claude', + usage: { + input_tokens: input, + cache_read_input_tokens: cacheRead, + cache_creation_input_tokens: 0, + output_tokens: output, + }, + }, + }); +} + +function mkState() { + return { + current: null as null | { inputSide: number; rounds: number; file: string; turnIndex: number }, + spends: [] as { inputSide: number; rounds: number; file: string; turnIndex: number }[], + file: 'f.jsonl', + }; +} + +describe('turnSpendStats', () => { + it('bounds turns by real user messages and sums assistant input side', () => { + const s = mkState(); + for (const line of [ + userLine('first prompt'), + assistantLine(1000, 50_000, 200), + assistantLine(1000, 50_000, 10_000), + userLine('second prompt'), + assistantLine(2000, 60_000, 500), + ]) { + feedLine(line, s); + } + expect(s.spends).toHaveLength(1); + expect(s.spends[0].inputSide).toBe(102_000); + expect(s.spends[0].rounds).toBe(2); + }); + + it('ignores tool results and internal isMeta lines as boundaries', () => { + const s = mkState(); + for (const line of [ + userLine('prompt'), + assistantLine(500, 10_000, 100), + userLine([{ type: 'tool_result', content: 'x' }], true), + assistantLine(500, 10_000, 100), + userLine('next'), + ]) { + feedLine(line, s); + } + expect(s.spends).toHaveLength(1); + expect(s.spends[0].inputSide).toBe(21_000); + expect(s.spends[0].rounds).toBe(2); + }); + + it('computes nearest-rank percentile', () => { + const sorted = [10, 20, 30, 40, 100]; + expect(percentile(sorted, 50)).toBe(30); + expect(percentile(sorted, 99)).toBe(100); + expect(percentile([], 99)).toBe(0); + }); +}); diff --git a/test/scripts/turnBudgetHookScript.test.ts b/test/scripts/turnBudgetHookScript.test.ts new file mode 100644 index 00000000..d1954a36 --- /dev/null +++ b/test/scripts/turnBudgetHookScript.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from 'vitest'; + +// untyped plain-node hook script, imported directly +// @ts-expect-error .mjs without type declarations +import { analyzeTurn, isRealUserLine, readConfig } from '../../scripts/turn-budget-hook.mjs'; + +const USER_RAW = JSON.stringify({ type: 'user', message: { role: 'user', content: 'go' } }); +const COMPACT_RAW = JSON.stringify({ + type: 'user', + isCompactSummary: true, + message: { role: 'user', content: 'summary of previous context' }, +}); + +function assistantRaw(input: number): string { + return JSON.stringify({ + type: 'assistant', + message: { role: 'assistant', usage: { input_tokens: input } }, + }); +} + +describe('hook isRealUserLine (raw JSONL shapes)', () => { + it('recognizes a message-wrapped real user line — regression on the 872M bug', () => { + expect(isRealUserLine(JSON.parse(USER_RAW))).toBe(true); + }); + + it('accepts legacy flat content and text/image arrays, rejects the rest', () => { + expect(isRealUserLine({ type: 'user', content: 'legacy flat' })).toBe(true); + expect( + isRealUserLine({ type: 'user', message: { content: [{ type: 'text', text: 'hi' }] } }) + ).toBe(true); + expect( + isRealUserLine({ type: 'user', isMeta: true, message: { content: 'tool result' } }) + ).toBe(false); + expect(isRealUserLine({ type: 'user', message: { content: '' } })).toBe(false); + expect( + isRealUserLine({ type: 'user', message: { content: '[Request interrupted by user]' } }) + ).toBe(false); + expect( + isRealUserLine({ type: 'user', message: { content: 'yo' } }) + ).toBe(false); + expect(isRealUserLine({ type: 'assistant', message: { content: 'hi' } })).toBe(false); + }); +}); + +describe('hook analyzeTurn', () => { + it('sums assistants newest-first and stops at the user boundary', () => { + const r = analyzeTurn([assistantRaw(300), assistantRaw(200), USER_RAW, assistantRaw(9000)]); + expect(r).toEqual({ spent: 500, boundaryFound: true }); + }); + + it('stops at a compaction marker — pre-compact rounds do not count', () => { + const r = analyzeTurn([assistantRaw(100), COMPACT_RAW, assistantRaw(777)]); + expect(r).toEqual({ spent: 100, boundaryFound: true }); + }); + + it('reports boundaryFound=false when no boundary exists', () => { + const r = analyzeTurn([assistantRaw(50), assistantRaw(60)]); + expect(r).toEqual({ spent: 110, boundaryFound: false }); + }); +}); + +describe('hook readConfig', () => { + it('reads the section, defaults when missing, survives garbage', () => { + expect( + readConfig('{"notifications":{"turnBudget":{"enabled":false,"maxInputTokensPerTurn":42}}}') + ).toEqual({ + enabled: false, + budget: 42, + }); + expect(readConfig('{}').enabled).toBe(true); + expect(readConfig('not json').budget).toBeGreaterThan(0); + expect(readConfig(undefined).enabled).toBe(true); + }); +}); + +describe('hook deny output shape', async () => { + it('emits permissionDecision deny via stdout', async () => { + const { deny } = await import('../../scripts/turn-budget-hook.mjs'); + const write = vi.spyOn(process.stdout, 'write').mockReturnValue(true); + deny(16_000_000, 15_000_000); + const payload = JSON.parse(String(write.mock.calls[0][0])); + expect(payload.hookSpecificOutput.permissionDecision).toBe('deny'); + expect(payload.hookSpecificOutput.permissionDecisionReason).toContain('15M'); + write.mockRestore(); + }); +}); diff --git a/test/shared/utils/modelParser.test.ts b/test/shared/utils/modelParser.test.ts index fae2532c..d8ca2fa4 100644 --- a/test/shared/utils/modelParser.test.ts +++ b/test/shared/utils/modelParser.test.ts @@ -102,6 +102,24 @@ describe('modelParser', () => { }); }); + it('should parse short format without minor or date: claude-sonnet-5', () => { + const result = parseModelString('claude-sonnet-5'); + expect(result).toEqual({ + name: 'sonnet5', + family: 'sonnet', + majorVersion: 5, + minorVersion: null, + }); + }); + + it('should return null for invalid short format with non-numeric version', () => { + expect(parseModelString('claude-sonnet-x')).toBeNull(); + }); + + it('should return null for date-shaped major version: claude-sonnet-20250929', () => { + expect(parseModelString('claude-sonnet-20250929')).toBeNull(); + }); + it('should return null for invalid format with only two parts', () => { expect(parseModelString('claude-sonnet')).toBeNull(); });