From 4357c6f747ef7ba9fe090b7339917fb5e14fed57 Mon Sep 17 00:00:00 2001 From: mrsibe Date: Mon, 28 Sep 2026 18:08:12 +0800 Subject: [PATCH] feat(search): a full-text index over chunks and a global search palette MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Chat retrieval and the search box were both vector-only. Embeddings are measurably bad at exact tokens — model names, ids, symbols, numbers — so there was no way to find a literal string, and a note could not be searched at all. Build the sparse index #77 also needs, and give it a user-facing surface: - `chunks_fts` (FTS5) is created and backfilled by `ensureChunksFts()`. The backfill runs on every ensure, so a library indexed before FTS existed, or a crash between the chunk write and the FTS write, heals itself. - The index is written with the chunks in the ingestion pipeline and deleted with the document, so it tracks re-index and delete. - `KnowledgeService.searchText()` returns the same `SearchResult` shape as the vector path, so provenance and the jump-to-source chain are not duplicated. - User text is quoted term by term before it reaches `MATCH`: an operator-looking query (`-`, `NEAR(`, `col:value`, an unbalanced quote) returns no rows instead of a syntax error, which would read as "no results rather than "bad query". - `SearchPalette` (Ctrl/Cmd+K) shows **matching text** and **related passages** as two labelled groups and never fuses them. Chat retrieval (#77) does fuse them, because the model needs one ordered list; the two must not share ranking semantics, which is why the shared part is the index, not the ranking. Verified: npm run typecheck; npm test (391 pass, incl. FTS5 integration against the bundled SQLite); npm run check:design; npm run build; `npm run eval` unchanged and byte-identical; `electron . --smoke-test` PASS — 27 checks, including a new one that a literal search finds the term after indexing and still finds exactly one row after a re-index. Not verified: the palette was not clicked in a running app (markup is covered by typecheck and the design guard). --- docs/eval/baseline-v1.5.md | 4 +- src/main/ipc/knowledgeHandlers.ts | 20 ++ src/main/ipc/validation.ts | 12 ++ src/main/services/KnowledgeService.ts | 42 ++++- src/main/services/fts.ts | 95 ++++++++++ src/main/services/ftsSql.ts | 89 +++++++++ src/main/smokeTest.ts | 22 +++ src/preload/index.d.ts | 6 + src/preload/index.ts | 6 + .../components/notebook/NotebookLayout.tsx | 4 + .../notebook/chat/SearchPalette.tsx | 177 ++++++++++++++++++ src/renderer/src/locales/en-US/ui.json | 8 +- src/renderer/src/locales/zh-CN/ui.json | 8 +- src/shared/types/knowledge.ts | 17 ++ test/fts.test.ts | 149 +++++++++++++++ 15 files changed, 654 insertions(+), 5 deletions(-) create mode 100644 src/main/services/fts.ts create mode 100644 src/main/services/ftsSql.ts create mode 100644 src/renderer/src/components/notebook/chat/SearchPalette.tsx create mode 100644 test/fts.test.ts diff --git a/docs/eval/baseline-v1.5.md b/docs/eval/baseline-v1.5.md index 797a698..ecf8ce9 100644 --- a/docs/eval/baseline-v1.5.md +++ b/docs/eval/baseline-v1.5.md @@ -25,8 +25,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | nDCG@10 | 0.9437 | | Evidence precision@5 | 0.2133 | -Timing is informational only and is **not** frozen: indexing 1941 ms, query -p50 14.39 ms, p95 27.71 ms on the +Timing is informational only and is **not** frozen: indexing 2515 ms, query +p50 13.99 ms, p95 23.62 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/src/main/ipc/knowledgeHandlers.ts b/src/main/ipc/knowledgeHandlers.ts index cebf8c2..9d492ce 100644 --- a/src/main/ipc/knowledgeHandlers.ts +++ b/src/main/ipc/knowledgeHandlers.ts @@ -150,6 +150,26 @@ export function registerKnowledgeHandlers(knowledgeService: KnowledgeService) { }) ) + // 字面搜索(#96):BM25,与语义搜索分开返回,UI 不融合两者。 + ipcMain.handle( + 'knowledge:search-text', + validate(KnowledgeSchemas.searchText, async (params) => { + Logger.debug('KnowledgeHandlers', 'search-text:', params) + + try { + const results = await knowledgeService.searchText( + params.notebookId, + params.query, + params.options ?? {} + ) + return { success: true, results } + } catch (error) { + Logger.error('KnowledgeHandlers', 'Error searching text:', error) + return { success: false, error: (error as Error).message, results: [] } + } + }) + ) + // 获取文档列表 ipcMain.handle( 'knowledge:get-documents', diff --git a/src/main/ipc/validation.ts b/src/main/ipc/validation.ts index 84ff353..f745d9b 100644 --- a/src/main/ipc/validation.ts +++ b/src/main/ipc/validation.ts @@ -202,6 +202,18 @@ export const KnowledgeSchemas = { includeContent: z.boolean().optional() }) .optional() + }), + + // 字面检索(#96):BM25 over chunks_fts,与语义检索分开呈现,不融合。 + searchText: z.object({ + notebookId: z.string().min(1, '笔记本 ID 不能为空'), + query: z.string().min(1, '搜索查询不能为空').max(1000, '搜索查询不能超过1000个字符'), + options: z + .object({ + limit: z.number().int().min(1).max(100).optional(), + documentIds: z.array(z.string().min(1)).optional() + }) + .optional() }) } diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 8f4e085..3da76a6 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -43,7 +43,7 @@ import { type IdentifiedBlockDraft } from './blocks/documentBlocks' import { resolveChunkProvenance, type ChunkProvenance } from './chunkProvenance' -import { DenseRetriever } from './retrieval' +import { DenseRetriever, hydrateEvidence } from './retrieval' import { advanceRun, completeRun, @@ -60,6 +60,12 @@ import type { Retriever } from './retrieval' import { WebFetchService } from './WebFetchService' +import { + deleteDocumentChunksFts, + ensureChunksFts, + indexChunksFts, + searchChunksFts +} from './fts' import { vectorStoreManager } from '../vectorstore' import Logger from '../../shared/utils/logger' @@ -467,6 +473,7 @@ export class KnowledgeService { // 4. 到此为止没有破坏任何东西。现在才替换旧的派生索引。 advanceRun(runId, 'finalizing', 85) onProgress?.('saving_chunks', 85) + ensureChunksFts() await this.clearDerivedIndex(documentId) this.persistDocumentBlocks(blocks) @@ -474,6 +481,17 @@ export class KnowledgeService { // 保存分块与 chunk↔block 映射(同一事务,不会出现没有映射的 chunk) const { chunkIds } = this.saveChunks(documentId, notebookId, chunkResults, now) + // 字面检索索引(#96)。与 chunk 写入在同一个 pipeline 里,所以两者不会各自 + // 漂移;就算漂移,`ensureChunksFts()` 的补齐也会在下次索引时自愈。 + indexChunksFts( + chunkResults.map((chunk, index) => ({ + chunkId: chunkIds[index], + notebookId, + documentId, + content: chunk.content + })) + ) + // 5. 保存嵌入元数据并添加到向量存储 onProgress?.('saving_embeddings', 90) const vectorStore = await vectorStoreManager.getStore( @@ -561,6 +579,10 @@ export class KnowledgeService { const doc = db.select().from(documents).where(eq(documents.id, documentId)).get() if (!doc) return + // 字面索引与 chunk 一起清掉;表不存在时先建出来,免得删除本身成了第一个错误。 + ensureChunksFts() + deleteDocumentChunksFts(documentId) + const oldChunks = db .select({ id: chunks.id }) .from(chunks) @@ -905,6 +927,24 @@ export class KnowledgeService { return toSearchResults(evidence, { includeContent }) } + /** + * 字面检索(#96):BM25 over `chunks_fts`,可限定来源(#94 的 scope)。 + * + * 与向量检索共用 `SearchResult` 形状,所以引用/定位链路不需要第二套。两个信号在 + * UI 里分开呈现,**不**做融合 —— 融合是 chat 检索(#77)的事。 + */ + async searchText( + notebookId: string, + query: string, + options: { limit?: number; documentIds?: string[] } = {} + ): Promise { + ensureChunksFts() + const hits = searchChunksFts(notebookId, query, options) + if (hits.length === 0) return [] + + return toSearchResults(hydrateEvidence(getDatabase(), hits)) + } + /** * 获取 notebook 的所有文档 */ diff --git a/src/main/services/fts.ts b/src/main/services/fts.ts new file mode 100644 index 0000000..0e40cf4 --- /dev/null +++ b/src/main/services/fts.ts @@ -0,0 +1,95 @@ +import { getSqlite } from '../db' +import { + backfillChunksFtsSql, + buildFtsMatchQuery, + createChunksFtsSql, + deleteChunksFtsSql, + insertChunksFtsSql, + searchChunksFtsSql +} from './ftsSql' + +/** + * Full-text index over chunks (#96) — the database operations. + * + * The SQL lives in `ftsSql.ts` (pure, testable without Electron); this file runs it + * against the real connection. + */ + +export * from './ftsSql' + +/** + * 建表 + 补齐缺失的行。 + * + * 每次调用都补齐而不是只建一次表:在「chunk 已写入、FTS 行还没写」之间崩溃,或一份 + * 旧库在引入 FTS 之前就已经索引过,都会留下没有 FTS 行的 chunk。补齐让这两种情况 + * 自愈,代价只是一条 INSERT..SELECT。 + */ +export function ensureChunksFts(): void { + const sqlite = getSqlite() + if (!sqlite) return + sqlite.exec(createChunksFtsSql()) + sqlite.exec(backfillChunksFtsSql()) +} + +export interface ChunksFtsRow { + chunkId: string + notebookId: string + documentId: string + content: string +} + +/** 写入刚索引好的 chunk。 */ +export function indexChunksFts(rows: readonly ChunksFtsRow[]): void { + if (rows.length === 0) return + const sqlite = getSqlite() + if (!sqlite) return + + const statement = sqlite.prepare(insertChunksFtsSql()) + const insert = sqlite.transaction((items: readonly ChunksFtsRow[]) => { + for (const item of items) { + statement.run(item.content, item.chunkId, item.notebookId, item.documentId) + } + }) + insert(rows) +} + +/** 删除一份文档的全部 FTS 行(重新索引、删除文档时)。 */ +export function deleteDocumentChunksFts(documentId: string): void { + const sqlite = getSqlite() + if (!sqlite) return + sqlite.prepare(deleteChunksFtsSql('document_id', 1)).run(documentId) +} + +export interface ChunksFtsHit { + chunkId: string + /** BM25 转成正数,越大越相关。 */ + score: number +} + +/** + * 字面检索:BM25 排序,可限定来源(#94 的 scope)。 + * + * 返回的顺序就是 BM25 顺序;命中已被删除的 chunk 由调用方补齐证据时丢弃。 + */ +export function searchChunksFts( + notebookId: string, + query: string, + options: { limit?: number; documentIds?: string[] } = {} +): ChunksFtsHit[] { + const match = buildFtsMatchQuery(query) + if (!match) return [] + + const sqlite = getSqlite() + if (!sqlite) return [] + + const documentIds = options.documentIds ?? [] + const limit = options.limit ?? 20 + const statement = sqlite.prepare(searchChunksFtsSql({ documentCount: documentIds.length })) + const rows = ( + documentIds.length > 0 + ? statement.all(match, notebookId, ...documentIds, limit) + : statement.all(match, notebookId, limit) + ) as Array<{ chunk_id: string; score: number }> + + return rows.map((row) => ({ chunkId: row.chunk_id, score: -row.score })) +} diff --git a/src/main/services/ftsSql.ts b/src/main/services/ftsSql.ts new file mode 100644 index 0000000..2fa9e02 --- /dev/null +++ b/src/main/services/ftsSql.ts @@ -0,0 +1,89 @@ +/** + * Full-text index SQL (#96) — pure builders, no database handle. + * + * Kept separate from `fts.ts` for the same reason `vectorTableSql.ts` is separate + * from `SQLiteVectorStore.ts`: the statements and the query sanitiser must be + * testable against the real FTS5 engine without loading Electron (the db module + * imports `electron`, which plain Node cannot). + * + * Two consumers share this index and must **not** share ranking semantics: + * + * - the global search surface presents literal matches as their own signal; + * - chat retrieval (#77) fuses BM25 with dense into one ranking. + */ + +const CHUNKS_FTS_TABLE = 'chunks_fts' + +export function createChunksFtsSql(): string { + // `content` is indexed; the ids are stored but not tokenized, so a result can be + // joined back to `chunks` without ever parsing the chunk text out of a snippet. + return ( + `CREATE VIRTUAL TABLE IF NOT EXISTS ${CHUNKS_FTS_TABLE} USING fts5(` + + 'content, chunk_id UNINDEXED, notebook_id UNINDEXED, document_id UNINDEXED)' + ) +} + +export function insertChunksFtsSql(): string { + return `INSERT INTO ${CHUNKS_FTS_TABLE} (content, chunk_id, notebook_id, document_id) VALUES (?, ?, ?, ?)` +} + +/** Backfill rows for chunks indexed before the FTS table existed. */ +export function backfillChunksFtsSql(): string { + return ( + `INSERT INTO ${CHUNKS_FTS_TABLE} (content, chunk_id, notebook_id, document_id) ` + + 'SELECT content, id, notebook_id, document_id FROM chunks ' + + `WHERE id NOT IN (SELECT chunk_id FROM ${CHUNKS_FTS_TABLE})` + ) +} + +export function deleteChunksFtsSql(column: 'chunk_id' | 'document_id', count: number): string { + if (count <= 0) throw new Error('deleteChunksFtsSql needs at least one id') + const placeholders = Array.from({ length: count }, () => '?').join(', ') + return `DELETE FROM ${CHUNKS_FTS_TABLE} WHERE ${column} IN (${placeholders})` +} + +/** + * Turn free text from a search box into an FTS5 MATCH expression. + * + * Every whitespace-separated term is quoted, so the user's text can never be + * parsed as FTS operators (`-`, `*`, `:`, `NEAR`, unbalanced quotes) — a query + * that is a syntax error is a query that returns nothing, which reads as "no + * results" rather than "your search is malformed". Terms are ANDed, which is + * FTS5's default for space-separated tokens. + * + * Returns null when there is nothing to search for. + */ +export function buildFtsMatchQuery(raw: string): string | null { + const terms = raw + .trim() + .split(/\s+/) + .filter((term) => term.length > 0) + if (terms.length === 0) return null + return terms.map((term) => `"${term.replace(/"/g, '""')}"`).join(' ') +} + +export interface FtsSearchSqlOptions { + /** Filter to these documents, e.g. the chat scope from #94. Empty = no filter. */ + documentCount?: number +} + +/** + * BM25-ranked search within one notebook. + * + * `bm25()` returns a negative number where more negative is a better match, so the + * caller orders ascending and negates it for display. + */ +export function searchChunksFtsSql(options: FtsSearchSqlOptions = {}): string { + const documentCount = options.documentCount ?? 0 + const documentFilter = + documentCount > 0 + ? ` AND document_id IN (${Array.from({ length: documentCount }, () => '?').join(', ')})` + : '' + + return ( + `SELECT chunk_id, bm25(${CHUNKS_FTS_TABLE}) AS score ` + + `FROM ${CHUNKS_FTS_TABLE} ` + + `WHERE ${CHUNKS_FTS_TABLE} MATCH ? AND notebook_id = ?${documentFilter} ` + + 'ORDER BY score ASC LIMIT ?' + ) +} diff --git a/src/main/smokeTest.ts b/src/main/smokeTest.ts index 2b95690..c06cab6 100644 --- a/src/main/smokeTest.ts +++ b/src/main/smokeTest.ts @@ -713,6 +713,28 @@ async function runChecks(): Promise { ) pass('DenseRetriever returns retrieved evidence with a page/block locator') + // --- the literal signal has its own index (#96) ---------------------------- + // FTS is written with the chunks and deleted with them, so a re-index must leave + // exactly one row: zero means the search silently went blind, two means a stale + // row survived. + const ftsDocId = await knowledge.addDocument(reindexNotebook, { + title: 'FTS smoke', + type: 'text', + content: 'the quick brown zephyrine jumps over the lazy dog' + }) + const literal = await knowledge.searchText(reindexNotebook, 'zephyrine') + assert( + literal.length === 1 && literal[0].documentId === ftsDocId, + 'full-text search did not find a literal term in the document it was indexed from' + ) + await knowledge.reindexDocument(ftsDocId) + const literalAfterReindex = await knowledge.searchText(reindexNotebook, 'zephyrine') + assert( + literalAfterReindex.length === 1 && literalAfterReindex[0].documentId === ftsDocId, + 'a re-index left the full-text index broken or duplicated' + ) + pass('full-text search finds literal terms and survives a re-index') + // --- a failed re-index leaves the previous index usable (#95) --------------- // The failure this guards: `reindexDocument()` used to clear the derived index // *before* embedding, so an embedding failure at 70% left the document with no diff --git a/src/preload/index.d.ts b/src/preload/index.d.ts index e731a82..8c066ba 100644 --- a/src/preload/index.d.ts +++ b/src/preload/index.d.ts @@ -274,6 +274,12 @@ declare global { query: string, options?: SearchOptions ) => Promise<{ success: boolean; results: KnowledgeSearchResult[]; error?: string }> + /** 字面搜索(#96):BM25,与语义搜索分开返回。 */ + searchText: ( + notebookId: string, + query: string, + options?: { limit?: number; documentIds?: string[] } + ) => Promise<{ success: boolean; results: KnowledgeSearchResult[]; error?: string }> // 文档管理 getDocuments: (notebookId: string) => Promise diff --git a/src/preload/index.ts b/src/preload/index.ts index bc0146b..8b8bc00 100644 --- a/src/preload/index.ts +++ b/src/preload/index.ts @@ -176,6 +176,12 @@ const api = { // 搜索 search: (notebookId: string, query: string, options?: any) => ipcRenderer.invoke('knowledge:search', { notebookId, query, options }), + // 字面搜索(#96) + searchText: ( + notebookId: string, + query: string, + options?: { limit?: number; documentIds?: string[] } + ) => ipcRenderer.invoke('knowledge:search-text', { notebookId, query, options }), // 文档管理 getDocuments: (notebookId: string) => diff --git a/src/renderer/src/components/notebook/NotebookLayout.tsx b/src/renderer/src/components/notebook/NotebookLayout.tsx index e561960..235382a 100644 --- a/src/renderer/src/components/notebook/NotebookLayout.tsx +++ b/src/renderer/src/components/notebook/NotebookLayout.tsx @@ -6,6 +6,7 @@ import ResizableLayout from '../layouts/ResizableLayout' import SourcePanel from './SourcePanel' import ProcessPanel from './ProcessPanel' import NotePanel from './NotePanel' +import SearchPalette from './chat/SearchPalette' import { useNotebookStore } from '../../store/notebookStore' import { useChatStore } from '../../store/chatStore' import { useUIStore } from '../../store/uiStore' @@ -113,6 +114,9 @@ export default function NotebookLayout(): ReactElement { onClose={handleDiscardClose} onConfirm={handleDiscardConfirm} /> + + {/* Global search (#96): Ctrl/Cmd+K from anywhere in the notebook. */} + ) } diff --git a/src/renderer/src/components/notebook/chat/SearchPalette.tsx b/src/renderer/src/components/notebook/chat/SearchPalette.tsx new file mode 100644 index 0000000..3a78cc9 --- /dev/null +++ b/src/renderer/src/components/notebook/chat/SearchPalette.tsx @@ -0,0 +1,177 @@ +import { ReactElement, useCallback, useEffect, useRef, useState } from 'react' +import { FileText, Search } from 'lucide-react' +import { useTranslation } from 'react-i18next' +import type { KnowledgeSearchResult } from '../../../../../shared/types/knowledge' +import { useNotebookStore } from '../../../store/notebookStore' +import { useSourceAnchorNavigation } from '../../../hooks/useSourceAnchorNavigation' +import { Button } from '../../ui/button' +import { Input } from '../../ui/input' + +/** + * Global search (#96). + * + * Two signals, presented separately and never fused: **matching text** is FTS/BM25 + * over the chunks — the only thing that finds an exact model name, id or number — + * and **related passages** is vector search, which finds a paraphrase with no shared + * words. Blending them into one score would hide which one answered the question. + * + * Chat retrieval (#77) *does* fuse them, because the model needs one ordered list. + * The two must not share ranking semantics. + */ +export default function SearchPalette(): ReactElement | null { + const { t } = useTranslation('ui') + const currentNotebook = useNotebookStore((state) => state.currentNotebook) + const { openSourceAnchor } = useSourceAnchorNavigation() + + const [isOpen, setIsOpen] = useState(false) + const [query, setQuery] = useState('') + const [literal, setLiteral] = useState([]) + const [semantic, setSemantic] = useState([]) + const [isSearching, setIsSearching] = useState(false) + const inputRef = useRef(null) + const notebookId = currentNotebook?.id + + const close = useCallback((): void => { + setIsOpen(false) + setQuery('') + setLiteral([]) + setSemantic([]) + }, []) + + // Ctrl/Cmd+K opens from anywhere in the notebook; Escape closes. + useEffect(() => { + const onKeyDown = (event: KeyboardEvent): void => { + if ((event.metaKey || event.ctrlKey) && event.key.toLowerCase() === 'k') { + event.preventDefault() + setIsOpen((open) => !open) + return + } + if (event.key === 'Escape') setIsOpen(false) + } + window.addEventListener('keydown', onKeyDown) + return () => window.removeEventListener('keydown', onKeyDown) + }, []) + + useEffect(() => { + if (isOpen) inputRef.current?.focus() + }, [isOpen]) + + // Debounced search. Both signals run for the same query; neither blocks the other + // from showing what it found. State is only written from the async continuation — + // an empty query simply stops the search and the results are hidden by `hasQuery`. + useEffect(() => { + if (!isOpen || !notebookId || query.trim().length === 0) return + + let cancelled = false + const timer = setTimeout(async () => { + setIsSearching(true) + try { + const [textResult, semanticResult] = await Promise.all([ + window.api.knowledge.searchText(notebookId, query, { limit: 10 }), + window.api.knowledge.search(notebookId, query, { topK: 10, includeContent: true }) + ]) + if (cancelled) return + if (textResult.success) setLiteral(textResult.results as KnowledgeSearchResult[]) + if (semanticResult.success) setSemantic(semanticResult.results as KnowledgeSearchResult[]) + } finally { + if (!cancelled) setIsSearching(false) + } + }, 200) + + return () => { + cancelled = true + clearTimeout(timer) + } + }, [isOpen, notebookId, query]) + + const openHit = (hit: KnowledgeSearchResult): void => { + const block = hit.locator?.blocks[0] + openSourceAnchor( + { + documentId: hit.documentId, + location: { + documentId: hit.documentId, + page: hit.locator?.pageStart ?? block?.page ?? undefined, + blockId: block?.blockId, + startOffset: block?.startOffset, + endOffset: block?.endOffset + } + }, + // The palette is unmounting as the reader opens; there is no origin element + // to return focus to, so the reader owns the focus handoff. + document.body + ) + close() + } + + if (!isOpen) return null + + const hasQuery = query.trim().length > 0 + // Stale results may still sit in state after the box is cleared; they are never + // shown, so there is nothing to reset from inside an effect. + const literalHits = hasQuery ? literal : [] + const semanticHits = hasQuery ? semantic : [] + const noResults = + hasQuery && !isSearching && literalHits.length === 0 && semanticHits.length === 0 + + const renderGroup = (label: string, hits: KnowledgeSearchResult[]): ReactElement | null => { + if (hits.length === 0) return null + return ( +
+

{label}

+
    + {hits.map((hit) => ( +
  • + +
  • + ))} +
+
+ ) + } + + return ( + <> +
+
+
+ + setQuery(event.target.value)} + placeholder={t('searchPlaceholder')} + className="h-8 border-0 bg-transparent p-0 text-sm focus-visible:ring-0 focus-visible:ring-offset-0" + /> +
+ +
+ {!hasQuery && ( +

{t('searchHint')}

+ )} + {noResults && ( +

{t('searchNoResults')}

+ )} + {renderGroup(t('searchMatchingText'), literalHits)} + {renderGroup(t('searchRelatedPassages'), semanticHits)} +
+
+ + ) +} diff --git a/src/renderer/src/locales/en-US/ui.json b/src/renderer/src/locales/en-US/ui.json index a686049..beb846a 100644 --- a/src/renderer/src/locales/en-US/ui.json +++ b/src/renderer/src/locales/en-US/ui.json @@ -84,5 +84,11 @@ "scopeClear": "Use all sources", "scopeNoDocuments": "No sources to choose from", "retryDocument": "Try again", - "reindexDocument": "Re-index" + "reindexDocument": "Re-index", + "searchPlaceholder": "Search sources", + "searchHint": "Find exact text or a related passage. Ctrl/Cmd+K toggles this palette.", + "searchNoResults": "No matches.", + "searchMatchingText": "Matching text", + "searchRelatedPassages": "Related passages", + "searchPage": "p. {{page}}" } diff --git a/src/renderer/src/locales/zh-CN/ui.json b/src/renderer/src/locales/zh-CN/ui.json index 9a8367e..e735b63 100644 --- a/src/renderer/src/locales/zh-CN/ui.json +++ b/src/renderer/src/locales/zh-CN/ui.json @@ -84,5 +84,11 @@ "scopeClear": "改用所有来源", "scopeNoDocuments": "没有可供选择的来源", "retryDocument": "重试", - "reindexDocument": "重建索引" + "reindexDocument": "重建索引", + "searchPlaceholder": "搜索来源", + "searchHint": "查找精确文本或相关段落。Ctrl/Cmd+K 开关此面板。", + "searchNoResults": "没有匹配结果。", + "searchMatchingText": "匹配的文本", + "searchRelatedPassages": "相关段落", + "searchPage": "第 {{page}} 页" } diff --git a/src/shared/types/knowledge.ts b/src/shared/types/knowledge.ts index 9673693..82f3ab8 100644 --- a/src/shared/types/knowledge.ts +++ b/src/shared/types/knowledge.ts @@ -37,6 +37,23 @@ export interface KnowledgeSearchResult { score: number chunkIndex: number metadata?: Record + /** + * 命中在来源里的位置(#96):页码区间与块区间。运行时会随结果一起送达 + * (`KnowledgeService.search` / `searchText` 共用 `SearchResult`),搜索面板用它 + * 跳到命中的段落,而不是只打开文档。 + */ + locator?: { + pageStart: number | null + pageEnd: number | null + blocks: Array<{ + blockId: string + page: number | null + startOffset: number + endOffset: number + startInBlock: number + endInBlock: number + }> + } } /** diff --git a/test/fts.test.ts b/test/fts.test.ts new file mode 100644 index 0000000..5392c17 --- /dev/null +++ b/test/fts.test.ts @@ -0,0 +1,149 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { DatabaseSync } from 'node:sqlite' +import { + buildFtsMatchQuery, + createChunksFtsSql, + deleteChunksFtsSql, + insertChunksFtsSql, + searchChunksFtsSql +} from '../src/main/services/ftsSql.ts' + +/** + * The full-text index (#96) is shared by the search surface and, later, BM25 chat + * retrieval (#77). Two things must hold: user text can never become FTS syntax, + * and BM25 ordering is real. The second needs the actual FTS5 engine, so this runs + * against the bundled SQLite rather than asserting on a SQL string. + */ + +test('a search query is quoted term by term, never parsed as FTS syntax', () => { + assert.equal(buildFtsMatchQuery('hello world'), '"hello" "world"') + assert.equal(buildFtsMatchQuery(' spaced out '), '"spaced" "out"') + + // Operators and unbalanced quotes would be a syntax error if passed through, and + // a syntax error reads to the user as "no results" rather than "bad query". + assert.equal(buildFtsMatchQuery('a -b'), '"a" "-b"') + assert.equal(buildFtsMatchQuery('NEAR(a b)'), '"NEAR(a" "b)"') + assert.equal(buildFtsMatchQuery('say "hi"'), '"say" """hi"""') + assert.equal(buildFtsMatchQuery('c*'), '"c*"') + + assert.equal(buildFtsMatchQuery(' '), null) + assert.equal(buildFtsMatchQuery(''), null) +}) + +test('delete is parameterised by count and refuses an empty list', () => { + assert.equal(deleteChunksFtsSql('chunk_id', 2), 'DELETE FROM chunks_fts WHERE chunk_id IN (?, ?)') + assert.match(deleteChunksFtsSql('document_id', 1), /WHERE document_id IN \(\?\)/) + assert.throws(() => deleteChunksFtsSql('chunk_id', 0)) +}) + +test('the search statement orders by BM25 and filters by notebook', () => { + const sql = searchChunksFtsSql() + assert.match(sql, /bm25\(chunks_fts\)/) + assert.match(sql, /chunks_fts MATCH \? AND notebook_id = \?/) + assert.match(sql, /ORDER BY score ASC LIMIT \?$/) + assert.doesNotMatch(sql, /document_id IN/) +}) + +test('a document filter adds placeholders before the limit', () => { + const sql = searchChunksFtsSql({ documentCount: 2 }) + assert.match(sql, /notebook_id = \? AND document_id IN \(\?, \?\) ORDER BY score ASC LIMIT \?$/) +}) + +/** + * Integration: the real FTS5 engine, the exact SQL the service runs. + */ +function fixture(): DatabaseSync { + const db = new DatabaseSync(':memory:') + db.exec(createChunksFtsSql()) + return db +} + +function index( + db: DatabaseSync, + chunkId: string, + notebookId: string, + documentId: string, + content: string +): void { + db.prepare(insertChunksFtsSql()).run(content, chunkId, notebookId, documentId) +} + +const search = ( + db: DatabaseSync, + notebookId: string, + query: string, + limit = 10, + documentIds: string[] = [] +) => { + const statement = db.prepare(searchChunksFtsSql({ documentCount: documentIds.length })) + const rows = + documentIds.length > 0 + ? statement.all(query, notebookId, ...documentIds, limit) + : statement.all(query, notebookId, limit) + return (rows as Array<{ chunk_id: string; score: number }>).map((row) => row.chunk_id) +} + +test('BM25 ranks the passage with more occurrences first', () => { + const db = fixture() + index(db, 'c1', 'nb1', 'doc1', 'photosynthesis converts light into chemical energy') + index( + db, + 'c2', + 'nb1', + 'doc1', + 'photosynthesis photosynthesis is the process by which plants convert light' + ) + index(db, 'c3', 'nb1', 'doc1', 'an unrelated passage about geology') + + const match = buildFtsMatchQuery('photosynthesis') + assert.ok(match) + assert.deepEqual(search(db, 'nb1', match), ['c2', 'c1']) +}) + +test('a notebook only sees its own chunks', () => { + const db = fixture() + index(db, 'c1', 'nb1', 'doc1', 'photosynthesis converts light') + index(db, 'c2', 'nb2', 'doc2', 'photosynthesis converts light') + + const match = buildFtsMatchQuery('photosynthesis') + assert.ok(match) + assert.deepEqual(search(db, 'nb1', match), ['c1']) + assert.deepEqual(search(db, 'nb2', match), ['c2']) +}) + +test('a document filter restricts results to the scoped sources', () => { + const db = fixture() + index(db, 'c1', 'nb1', 'docA', 'photosynthesis in plants') + index(db, 'c2', 'nb1', 'docB', 'photosynthesis in algae') + + const match = buildFtsMatchQuery('photosynthesis') + assert.ok(match) + assert.deepEqual(search(db, 'nb1', match, 10, ['docB']), ['c2']) +}) + +test('an operator-looking query returns no rows instead of throwing', () => { + const db = fixture() + index(db, 'c1', 'nb1', 'doc1', 'photosynthesis converts light') + + for (const raw of ['-', 'NEAR(', '"', 'a AND b', 'col:value', '^']) { + const match = buildFtsMatchQuery(raw) + assert.ok(match, `expected a match expression for ${JSON.stringify(raw)}`) + assert.doesNotThrow(() => search(db, 'nb1', match), `threw for ${JSON.stringify(raw)}`) + } +}) + +test('a snippet is available for the matching content column', () => { + const db = fixture() + index(db, 'c1', 'nb1', 'doc1', 'photosynthesis converts light into chemical energy') + + const match = buildFtsMatchQuery('chemical') + assert.ok(match) + const row = db + .prepare( + "SELECT snippet(chunks_fts, 0, '<', '>', '…', 6) AS s FROM chunks_fts WHERE chunks_fts MATCH ?" + ) + .get(match) as { s: string } + + assert.match(row.s, //) +})