From 712a98fd7e73e735e54e7d4f1911b5cea945f5e3 Mon Sep 17 00:00:00 2001 From: Mate Remias Date: Wed, 23 Sep 2026 17:44:48 +0200 Subject: [PATCH] fix(parser): index current OMP sessions (title slot, non-message tree nodes) Current Oh My Pi transcripts open with a fixed-width {type:"title"} slot before the {type:"session"} header, so detectConversationHarness fell through to the Claude parser and indexed zero exchanges. The active-path walk also only followed message entries, but OMP parents messages on model_change / thinking_level_change / custom / compaction entries, so the leaf->root walk stopped at the first non-message parent and kept only the tail of each session. Walk every id-bearing entry and keep the messages on the path; same fix for the show/read renderer. --- dist/mcp-server.js | 12 ++++++-- dist/parser.js | 27 +++++++++++++---- dist/show.js | 15 +++++++--- src/parser.ts | 30 ++++++++++++++----- src/show.ts | 15 +++++++--- test/omp-show.test.ts | 18 ++++++++++++ test/omp-transcripts.test.ts | 57 ++++++++++++++++++++++++++++++++++++ 7 files changed, 150 insertions(+), 24 deletions(-) diff --git a/dist/mcp-server.js b/dist/mcp-server.js index e3a9adac..1ee995d9 100644 --- a/dist/mcp-server.js +++ b/dist/mcp-server.js @@ -31467,6 +31467,7 @@ function extractOmpText(content) { function formatOmpConversationAsMarkdown(lines) { const metadata = {}; const nodesById = /* @__PURE__ */ new Map(); + const parentById = /* @__PURE__ */ new Map(); let leafId; for (const line of lines) { let entry; @@ -31480,6 +31481,9 @@ function formatOmpConversationAsMarkdown(lines) { metadata.cwd = entry.cwd || metadata.cwd; continue; } + if (entry.id) { + parentById.set(entry.id, entry.parentId ?? void 0); + } if (entry.type !== "message" || !entry.message || !entry.message.role || !entry.id) { continue; } @@ -31489,11 +31493,13 @@ function formatOmpConversationAsMarkdown(lines) { const chain = []; const seen = /* @__PURE__ */ new Set(); let currentId = leafId; - while (currentId && nodesById.has(currentId) && !seen.has(currentId)) { + while (currentId && parentById.has(currentId) && !seen.has(currentId)) { seen.add(currentId); const node2 = nodesById.get(currentId); - chain.push(node2); - currentId = node2.parentId ?? void 0; + if (node2) { + chain.push(node2); + } + currentId = parentById.get(currentId); } chain.reverse(); let output2 = "# Conversation\n\n"; diff --git a/dist/parser.js b/dist/parser.js index dcc1c270..cbe03a9f 100644 --- a/dist/parser.js +++ b/dist/parser.js @@ -28,7 +28,12 @@ async function detectConversationHarness(filePath) { // Oh My Pi (OMP) pi-lineage transcripts open with a bare session header // ({type:"session", id, cwd}) with no payload/session sub-object (which // would be Codex/opencode), and their turns are {type:"message", - // message:{role, content}} — a shape no other harness uses. + // message:{role, content}} — a shape no other harness uses. Current OMP + // writes a fixed-width {type:"title"} slot before the header; skip it so + // the header decides instead of the Claude fallback. + if (parsed.type === 'title') { + continue; + } const maybeOmp = parsed; if (parsed.type === 'session' && !parsed.payload && !maybeOmp.session) { return 'omp'; @@ -515,7 +520,10 @@ function extractOmpText(content) { * `type:"message"` entry in file order. Transcripts are append-only, so the * most recently written message is the current tip; abandoned/regenerated * branches remain earlier in the file but are not on the leaf's parentId chain. - * We follow parentId from that leaf up to the root (a message whose parentId is + * Non-message entries (model_change, thinking_level_change, custom, compaction, + * ...) are tree nodes too: a message's parent is often one of them, so the walk + * follows every id-bearing entry and keeps only the messages it passes. We + * follow parentId from that leaf up to the root (an entry whose parentId is * null/absent or not present in the file), then reverse to root->leaf order and * linearize. Because parents are always written before their children, line * numbers stay monotonic along the chain, keeping the #152 high-water mark and @@ -533,6 +541,7 @@ async function parseOmpConversation(filePath, projectName, archivePath) { let cwd; let headerTimestamp; const nodesById = new Map(); + const parentById = new Map(); let leafId; let lineNumber = 0; for await (const line of rl) { @@ -553,6 +562,9 @@ async function parseOmpConversation(filePath, projectName, archivePath) { headerTimestamp = parsed.timestamp ?? headerTimestamp; continue; } + if (parsed.id) { + parentById.set(parsed.id, parsed.parentId ?? undefined); + } // Skip title/session_init/custom line types and malformed messages. if (parsed.type !== 'message' || !parsed.message || !parsed.message.role || !parsed.id) { continue; @@ -568,15 +580,18 @@ async function parseOmpConversation(filePath, projectName, archivePath) { nodesById.set(node.id, node); leafId = node.id; // the last message wins as the active leaf } - // Walk from the active leaf back to the root, guarding against cycles. + // Walk from the active leaf back to the root through every entry, guarding + // against cycles; only message entries join the chain. const chain = []; const seen = new Set(); let currentId = leafId; - while (currentId && nodesById.has(currentId) && !seen.has(currentId)) { + while (currentId && parentById.has(currentId) && !seen.has(currentId)) { seen.add(currentId); const node = nodesById.get(currentId); - chain.push(node); - currentId = node.parentId; + if (node) { + chain.push(node); + } + currentId = parentById.get(currentId); } chain.reverse(); // root -> leaf const project = projectFromCwd(cwd) || projectName; diff --git a/dist/show.js b/dist/show.js index 2f297196..39f5ddf1 100644 --- a/dist/show.js +++ b/dist/show.js @@ -826,10 +826,12 @@ function extractOmpText(content) { .join('\n'); } // Render only the active path (leaf -> root via parentId, reversed), matching -// parseOmpConversation: abandoned/regenerated branches stay out of the output. +// parseOmpConversation: abandoned/regenerated branches stay out of the output, +// and the walk passes through non-message entries that parent messages. function formatOmpConversationAsMarkdown(lines) { const metadata = {}; const nodesById = new Map(); + const parentById = new Map(); let leafId; for (const line of lines) { let entry; @@ -844,6 +846,9 @@ function formatOmpConversationAsMarkdown(lines) { metadata.cwd = entry.cwd || metadata.cwd; continue; } + if (entry.id) { + parentById.set(entry.id, entry.parentId ?? undefined); + } if (entry.type !== 'message' || !entry.message || !entry.message.role || !entry.id) { continue; } @@ -853,11 +858,13 @@ function formatOmpConversationAsMarkdown(lines) { const chain = []; const seen = new Set(); let currentId = leafId; - while (currentId && nodesById.has(currentId) && !seen.has(currentId)) { + while (currentId && parentById.has(currentId) && !seen.has(currentId)) { seen.add(currentId); const node = nodesById.get(currentId); - chain.push(node); - currentId = node.parentId ?? undefined; + if (node) { + chain.push(node); + } + currentId = parentById.get(currentId); } chain.reverse(); let output = '# Conversation\n\n'; diff --git a/src/parser.ts b/src/parser.ts index a416a113..3e098796 100644 --- a/src/parser.ts +++ b/src/parser.ts @@ -120,7 +120,12 @@ async function detectConversationHarness(filePath: string): Promiseleaf order and * linearize. Because parents are always written before their children, line * numbers stay monotonic along the chain, keeping the #152 high-water mark and @@ -697,6 +705,7 @@ async function parseOmpConversation( let cwd: string | undefined; let headerTimestamp: string | undefined; const nodesById = new Map(); + const parentById = new Map(); let leafId: string | undefined; let lineNumber = 0; @@ -720,6 +729,10 @@ async function parseOmpConversation( continue; } + if (parsed.id) { + parentById.set(parsed.id, parsed.parentId ?? undefined); + } + // Skip title/session_init/custom line types and malformed messages. if (parsed.type !== 'message' || !parsed.message || !parsed.message.role || !parsed.id) { continue; @@ -737,15 +750,18 @@ async function parseOmpConversation( leafId = node.id; // the last message wins as the active leaf } - // Walk from the active leaf back to the root, guarding against cycles. + // Walk from the active leaf back to the root through every entry, guarding + // against cycles; only message entries join the chain. const chain: OmpNode[] = []; const seen = new Set(); let currentId: string | undefined = leafId; - while (currentId && nodesById.has(currentId) && !seen.has(currentId)) { + while (currentId && parentById.has(currentId) && !seen.has(currentId)) { seen.add(currentId); - const node = nodesById.get(currentId)!; - chain.push(node); - currentId = node.parentId; + const node = nodesById.get(currentId); + if (node) { + chain.push(node); + } + currentId = parentById.get(currentId); } chain.reverse(); // root -> leaf diff --git a/src/show.ts b/src/show.ts index 06a34b45..abf09008 100644 --- a/src/show.ts +++ b/src/show.ts @@ -864,10 +864,12 @@ function extractOmpText(content: unknown): string { } // Render only the active path (leaf -> root via parentId, reversed), matching -// parseOmpConversation: abandoned/regenerated branches stay out of the output. +// parseOmpConversation: abandoned/regenerated branches stay out of the output, +// and the walk passes through non-message entries that parent messages. function formatOmpConversationAsMarkdown(lines: string[]): string { const metadata: { sessionId?: string; cwd?: string } = {}; const nodesById = new Map(); + const parentById = new Map(); let leafId: string | undefined; for (const line of lines) { @@ -882,6 +884,9 @@ function formatOmpConversationAsMarkdown(lines: string[]): string { metadata.cwd = entry.cwd || metadata.cwd; continue; } + if (entry.id) { + parentById.set(entry.id, entry.parentId ?? undefined); + } if (entry.type !== 'message' || !entry.message || !entry.message.role || !entry.id) { continue; } @@ -892,11 +897,13 @@ function formatOmpConversationAsMarkdown(lines: string[]): string { const chain: any[] = []; const seen = new Set(); let currentId: string | undefined = leafId; - while (currentId && nodesById.has(currentId) && !seen.has(currentId)) { + while (currentId && parentById.has(currentId) && !seen.has(currentId)) { seen.add(currentId); const node = nodesById.get(currentId); - chain.push(node); - currentId = node.parentId ?? undefined; + if (node) { + chain.push(node); + } + currentId = parentById.get(currentId); } chain.reverse(); diff --git a/test/omp-show.test.ts b/test/omp-show.test.ts index a1939317..26bf536c 100644 --- a/test/omp-show.test.ts +++ b/test/omp-show.test.ts @@ -38,4 +38,22 @@ describe('show command - OMP markdown formatting', () => { expect(markdown).toContain('OMP transcripts render as markdown.'); expect(markdown).not.toContain('HIDDEN chain of thought'); }); + + it('renders turns whose parent is a non-message entry in a title-first file', () => { + const lines = [ + { type: 'title', v: 1, title: 'Show layout', pad: ' ' }, + { type: 'session', id: 'omp_sess_layout', timestamp: '2026-09-01T00:00:00Z', cwd: '/work/layout' }, + { type: 'model_change', id: 'm0', parentId: null, model: 'x/y' }, + { type: 'message', id: 'u1', parentId: 'm0', message: { role: 'user', content: 'Early question' } }, + { type: 'message', id: 'a1', parentId: 'u1', message: { role: 'assistant', content: 'Early answer' } }, + { type: 'custom', id: 'c1', parentId: 'a1' }, + { type: 'message', id: 'u2', parentId: 'c1', message: { role: 'user', content: 'Late question' } }, + { type: 'message', id: 'a2', parentId: 'u2', message: { role: 'assistant', content: 'Late answer' } }, + ]; + const markdown = formatConversationAsMarkdown(lines.map(line => JSON.stringify(line)).join('\n')); + + const order = ['Early question', 'Early answer', 'Late question', 'Late answer'].map(t => markdown.indexOf(t)); + expect(order.every(i => i >= 0)).toBe(true); + expect(order).toEqual([...order].sort((a, b) => a - b)); + }); }); diff --git a/test/omp-transcripts.test.ts b/test/omp-transcripts.test.ts index 0d1ae3a2..625ea12d 100644 --- a/test/omp-transcripts.test.ts +++ b/test/omp-transcripts.test.ts @@ -201,6 +201,63 @@ describe('OMP (Oh My Pi) transcript parser', () => { lineEnd: 5, }); }); + + it('parses current OMP files: title slot first, messages parented by non-message entries', async () => { + const transcriptPath = join(testDir, 'omp-real-layout.jsonl'); + const lines = [ + // L1 — fixed-width title slot precedes the session header + { type: 'title', v: 1, title: 'Real layout', source: 'auto', pad: ' ' }, + // L2 + { type: 'session', version: 3, id: 'omp_sess_real', timestamp: '2026-09-01T00:00:00Z', cwd: '/work/real' }, + // L3 — non-message tree node at the root + { type: 'model_change', id: 'm0', parentId: null, timestamp: '2026-09-01T00:00:00Z', model: 'x/y' }, + // L4 — user parented by the model_change entry + { + type: 'message', + id: 'u1', + parentId: 'm0', + timestamp: '2026-09-01T00:00:01Z', + message: { role: 'user', content: [{ type: 'text', text: 'First question' }] }, + }, + // L5 + { + type: 'message', + id: 'a1', + parentId: 'u1', + timestamp: '2026-09-01T00:00:02Z', + message: { role: 'assistant', content: [{ type: 'text', text: 'First answer' }] }, + }, + // L6 — mid-session non-message nodes between turns + { type: 'thinking_level_change', id: 't1', parentId: 'a1', timestamp: '2026-09-01T00:00:03Z' }, + // L7 + { type: 'custom', id: 'c1', parentId: 't1', timestamp: '2026-09-01T00:00:03Z' }, + // L8 — second user parented by the custom entry + { + type: 'message', + id: 'u2', + parentId: 'c1', + timestamp: '2026-09-01T00:00:04Z', + message: { role: 'user', content: [{ type: 'text', text: 'Second question' }] }, + }, + // L9 — leaf + { + type: 'message', + id: 'a2', + parentId: 'u2', + timestamp: '2026-09-01T00:00:05Z', + message: { role: 'assistant', content: [{ type: 'text', text: 'Second answer' }] }, + }, + ]; + writeLines(transcriptPath, lines); + + const exchanges = await parseConversation(transcriptPath, 'fallback-project', transcriptPath); + + expect(exchanges.map(e => [e.harness, e.userMessage, e.assistantMessage, e.lineStart, e.lineEnd])).toEqual([ + ['omp', 'First question', 'First answer', 4, 5], + ['omp', 'Second question', 'Second answer', 8, 9], + ]); + expect(exchanges[0]).toMatchObject({ sessionId: 'omp_sess_real', cwd: '/work/real', project: 'real' }); + }); }); describe('OMP source directory discovery', () => {