From 8f2799c4714183a2081f914ddaebbfe123c2dbbd Mon Sep 17 00:00:00 2001 From: mrsibe Date: Fri, 2 Oct 2026 21:53:52 +0800 Subject: [PATCH 1/5] docs: record library source reuse design --- docs/adr/0002-library-source-reuse.md | 55 +++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) create mode 100644 docs/adr/0002-library-source-reuse.md diff --git a/docs/adr/0002-library-source-reuse.md b/docs/adr/0002-library-source-reuse.md new file mode 100644 index 0000000..88830f3 --- /dev/null +++ b/docs/adr/0002-library-source-reuse.md @@ -0,0 +1,55 @@ +# ADR 0002: Library sources and notebook memberships (#99) + +Status: accepted for implementation, recorded before code changes. + +## Schema and identity + +Add `library_sources` as the durable library identity and snapshot (canonical text, +parser structure, original URI and local file). Add nullable `documents.sourceId` +referencing it, unique within a notebook. Existing `documents` rows remain notebook +memberships and retain their IDs: historical citations and retrieval scopes must +not change during migration. Existing imports are backfilled one-to-one (do not +merge unrelated sources merely because titles or paths match). + +The existing document shape is a compatibility projection of the snapshot plus +membership-local indexing state. Keep the current text/structure fields as cached +projections for the existing retrieval/readers; reuse performs no parsing or file +copy. A library snapshot is immutable while shared. Explicit file refresh creates +a new library snapshot for that membership (copy-on-write); it never changes the +text or page offsets underneath another notebook's citations. Reindex uses the +persisted snapshot, not the original mutable file. + +## Embedding-space rule + +Chunks, chunk/block mappings, ingestion attempts, indexing status, and vectors +remain per membership/notebook. Citation document IDs identify the membership; +its `sourceId` identifies the shared library snapshot. This preserves every +existing citation's notebook context and page/span contract. + +Reuse is an explicit operation. If a donor membership is indexed and its stored +space identity and vector width match the target notebook (and the configured +embedding space), copy its chunks, provenance and vectors with new membership- +local IDs. Never call the embedding provider in the reuse operation. An empty +target may adopt the matching donor space. If there is no compatible indexed +donor, add a pending membership and show that indexing is required; only the +user's explicit reindex action may produce embeddings. Same dimensions alone +are not proof of compatibility. Copying never invalidates other target sources. + +## Removal and lifecycle + +Removing a source removes only that notebook's membership and derived index. +Keep the library snapshot and file, including after the last membership is +removed; it remains available for later reuse. Permanent library deletion is a +separate, explicitly confirmed operation and is refused while memberships exist. +Notebook deletion follows the same detach-only semantics. A notebook must never +unlink a file still used by a library source. + +## UI and validation + +Add “From library” to the existing source-import UI, list snapshots not already +attached, and distinguish ready-to-reuse sources from those needing explicit +indexing. Add confirmed permanent deletion for unused library sources. Provide +both English and Chinese strings. Regression tests cover legacy migration, +two-notebook reuse, no embedding calls on attach, embedding-space mismatch, +page/span preservation, independent removal, notebook deletion and confirmed +last-copy deletion. No automatic deduplication of new imports is introduced. From acc6a96b97f3306fa35bb92a111e6ffae91e7eca Mon Sep 17 00:00:00 2001 From: mrsibe Date: Fri, 2 Oct 2026 22:07:54 +0800 Subject: [PATCH 2/5] feat: persist library snapshots and reuse notebook indexes --- .../db/migrations/0023_sleepy_luckman.sql | 25 + .../db/migrations/meta/0023_snapshot.json | 1979 +++++++++++++++++ src/main/db/migrations/meta/_journal.json | 7 + src/main/db/queries.ts | 23 +- src/main/db/schema.ts | 38 +- src/main/services/KnowledgeService.ts | 317 ++- src/main/services/librarySources.ts | 628 ++++++ src/shared/types/knowledge.ts | 16 + test/librarySourceBackfill.test.ts | 141 ++ test/librarySourceGuards.test.ts | 119 + test/librarySourceReuse.test.ts | 867 ++++++++ 11 files changed, 4107 insertions(+), 53 deletions(-) create mode 100644 src/main/db/migrations/0023_sleepy_luckman.sql create mode 100644 src/main/db/migrations/meta/0023_snapshot.json create mode 100644 src/main/services/librarySources.ts create mode 100644 test/librarySourceBackfill.test.ts create mode 100644 test/librarySourceGuards.test.ts create mode 100644 test/librarySourceReuse.test.ts diff --git a/src/main/db/migrations/0023_sleepy_luckman.sql b/src/main/db/migrations/0023_sleepy_luckman.sql new file mode 100644 index 0000000..5eb05a8 --- /dev/null +++ b/src/main/db/migrations/0023_sleepy_luckman.sql @@ -0,0 +1,25 @@ +CREATE TABLE `library_sources` ( + `id` text PRIMARY KEY NOT NULL, + `title` text NOT NULL, + `type` text NOT NULL, + `source_uri` text, + `local_file_path` text, + `content` text, + `structure` text, + `content_hash` text, + `mime_type` text, + `file_size` integer, + `metadata` text, + `created_at` integer NOT NULL, + `updated_at` integer NOT NULL +); +--> statement-breakpoint +ALTER TABLE `documents` ADD `source_id` text REFERENCES library_sources(id);--> statement-breakpoint +CREATE UNIQUE INDEX `idx_documents_notebook_source` ON `documents` (`notebook_id`,`source_id`);--> statement-breakpoint +-- 数据迁移(#99):把现有 documents 一次性回填成一比一的 library_sources。 +-- 不按标题/路径合并不同来源;id 取 'lib_' || documents.id,保证一一对应、结果确定。 +INSERT INTO `library_sources` (`id`, `title`, `type`, `source_uri`, `local_file_path`, `content`, `structure`, `content_hash`, `mime_type`, `file_size`, `metadata`, `created_at`, `updated_at`) +SELECT 'lib_' || `id`, `title`, `type`, `source_uri`, `local_file_path`, `content`, `structure`, `content_hash`, `mime_type`, `file_size`, `metadata`, `created_at`, `updated_at` +FROM `documents`;--> statement-breakpoint +-- 每个 membership 指向它自己的 snapshot;此后重新索引读的就是这份快照。 +UPDATE `documents` SET `source_id` = 'lib_' || `id` WHERE `source_id` IS NULL; \ No newline at end of file diff --git a/src/main/db/migrations/meta/0023_snapshot.json b/src/main/db/migrations/meta/0023_snapshot.json new file mode 100644 index 0000000..87373a4 --- /dev/null +++ b/src/main/db/migrations/meta/0023_snapshot.json @@ -0,0 +1,1979 @@ +{ + "version": "6", + "dialect": "sqlite", + "id": "d8356a5b-c398-4294-b849-fa0c125f5baf", + "prevId": "a22e0598-116b-43ee-9ff7-3a9cc08b8a11", + "tables": { + "anki_cards": { + "name": "anki_cards", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "version": { + "name": "version", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": 1 + }, + "cards_data": { + "name": "cards_data", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "chunk_mapping": { + "name": "chunk_mapping", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "'generating'" + }, + "error_message": { + "name": "error_message", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_ankicards_notebook": { + "name": "idx_ankicards_notebook", + "columns": [ + "notebook_id", + "updated_at" + ], + "isUnique": false + }, + "idx_ankicards_version": { + "name": "idx_ankicards_version", + "columns": [ + "notebook_id", + "version" + ], + "isUnique": false + } + }, + "foreignKeys": { + "anki_cards_notebook_id_notebooks_id_fk": { + "name": "anki_cards_notebook_id_notebooks_id_fk", + "tableFrom": "anki_cards", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "chat_messages": { + "name": "chat_messages", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "session_id": { + "name": "session_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "role": { + "name": "role", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "reasoning_content": { + "name": "reasoning_content", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "finish_reason": { + "name": "finish_reason", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "error": { + "name": "error", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "usage": { + "name": "usage", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "finished_at": { + "name": "finished_at", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "attempt_of": { + "name": "attempt_of", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_messages_session": { + "name": "idx_messages_session", + "columns": [ + "session_id", + "created_at" + ], + "isUnique": false + } + }, + "foreignKeys": { + "chat_messages_session_id_chat_sessions_id_fk": { + "name": "chat_messages_session_id_chat_sessions_id_fk", + "tableFrom": "chat_messages", + "tableTo": "chat_sessions", + "columnsFrom": [ + "session_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "chat_sessions": { + "name": "chat_sessions", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title_is_auto": { + "name": "title_is_auto", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": false + }, + "summary": { + "name": "summary", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "total_tokens": { + "name": "total_tokens", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": 0 + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "'active'" + }, + "parent_session_id": { + "name": "parent_session_id", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "retrieval_scope": { + "name": "retrieval_scope", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "last_opened_at": { + "name": "last_opened_at", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_sessions_notebook": { + "name": "idx_sessions_notebook", + "columns": [ + "notebook_id", + "updated_at" + ], + "isUnique": false + }, + "idx_sessions_notebook_opened": { + "name": "idx_sessions_notebook_opened", + "columns": [ + "notebook_id", + "last_opened_at" + ], + "isUnique": false + } + }, + "foreignKeys": { + "chat_sessions_notebook_id_notebooks_id_fk": { + "name": "chat_sessions_notebook_id_notebooks_id_fk", + "tableFrom": "chat_sessions", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "chat_sessions_parent_session_id_chat_sessions_id_fk": { + "name": "chat_sessions_parent_session_id_chat_sessions_id_fk", + "tableFrom": "chat_sessions", + "tableTo": "chat_sessions", + "columnsFrom": [ + "parent_session_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "chunk_blocks": { + "name": "chunk_blocks", + "columns": { + "chunk_id": { + "name": "chunk_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "block_id": { + "name": "block_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "start_in_block": { + "name": "start_in_block", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "end_in_block": { + "name": "end_in_block", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_chunk_blocks_block": { + "name": "idx_chunk_blocks_block", + "columns": [ + "block_id" + ], + "isUnique": false + } + }, + "foreignKeys": { + "chunk_blocks_chunk_id_chunks_id_fk": { + "name": "chunk_blocks_chunk_id_chunks_id_fk", + "tableFrom": "chunk_blocks", + "tableTo": "chunks", + "columnsFrom": [ + "chunk_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "chunk_blocks_block_id_document_blocks_id_fk": { + "name": "chunk_blocks_block_id_document_blocks_id_fk", + "tableFrom": "chunk_blocks", + "tableTo": "document_blocks", + "columnsFrom": [ + "block_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": { + "chunk_blocks_chunk_id_block_id_pk": { + "columns": [ + "chunk_id", + "block_id" + ], + "name": "chunk_blocks_chunk_id_block_id_pk" + } + }, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "chunks": { + "name": "chunks", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "document_id": { + "name": "document_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "chunk_index": { + "name": "chunk_index", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "start_offset": { + "name": "start_offset", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "end_offset": { + "name": "end_offset", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "page_start": { + "name": "page_start", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "page_end": { + "name": "page_end", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "token_count": { + "name": "token_count", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_chunks_document": { + "name": "idx_chunks_document", + "columns": [ + "document_id" + ], + "isUnique": false + }, + "idx_chunks_notebook": { + "name": "idx_chunks_notebook", + "columns": [ + "notebook_id" + ], + "isUnique": false + } + }, + "foreignKeys": { + "chunks_document_id_documents_id_fk": { + "name": "chunks_document_id_documents_id_fk", + "tableFrom": "chunks", + "tableTo": "documents", + "columnsFrom": [ + "document_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "chunks_notebook_id_notebooks_id_fk": { + "name": "chunks_notebook_id_notebooks_id_fk", + "tableFrom": "chunks", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "document_blocks": { + "name": "document_blocks", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "document_id": { + "name": "document_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "kind": { + "name": "kind", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "order": { + "name": "order", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "page": { + "name": "page", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "level": { + "name": "level", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "text": { + "name": "text", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "start_offset": { + "name": "start_offset", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "end_offset": { + "name": "end_offset", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "bbox": { + "name": "bbox", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + } + }, + "indexes": { + "idx_blocks_document_order": { + "name": "idx_blocks_document_order", + "columns": [ + "document_id", + "order" + ], + "isUnique": false + }, + "idx_blocks_document_page": { + "name": "idx_blocks_document_page", + "columns": [ + "document_id", + "page" + ], + "isUnique": false + } + }, + "foreignKeys": { + "document_blocks_document_id_documents_id_fk": { + "name": "document_blocks_document_id_documents_id_fk", + "tableFrom": "document_blocks", + "tableTo": "documents", + "columnsFrom": [ + "document_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "documents": { + "name": "documents", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "type": { + "name": "type", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "source_uri": { + "name": "source_uri", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "local_file_path": { + "name": "local_file_path", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "source_note_id": { + "name": "source_note_id", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "source_id": { + "name": "source_id", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "structure": { + "name": "structure", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "content_hash": { + "name": "content_hash", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "mime_type": { + "name": "mime_type", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "file_size": { + "name": "file_size", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "'pending'" + }, + "source_state": { + "name": "source_state", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "'available'" + }, + "source_mtime_ms": { + "name": "source_mtime_ms", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "error_message": { + "name": "error_message", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "chunk_count": { + "name": "chunk_count", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false, + "default": 0 + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_documents_notebook": { + "name": "idx_documents_notebook", + "columns": [ + "notebook_id", + "updated_at" + ], + "isUnique": false + }, + "idx_documents_status": { + "name": "idx_documents_status", + "columns": [ + "status" + ], + "isUnique": false + }, + "idx_documents_notebook_source": { + "name": "idx_documents_notebook_source", + "columns": [ + "notebook_id", + "source_id" + ], + "isUnique": true + } + }, + "foreignKeys": { + "documents_notebook_id_notebooks_id_fk": { + "name": "documents_notebook_id_notebooks_id_fk", + "tableFrom": "documents", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "documents_source_note_id_notes_id_fk": { + "name": "documents_source_note_id_notes_id_fk", + "tableFrom": "documents", + "tableTo": "notes", + "columnsFrom": [ + "source_note_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + }, + "documents_source_id_library_sources_id_fk": { + "name": "documents_source_id_library_sources_id_fk", + "tableFrom": "documents", + "tableTo": "library_sources", + "columnsFrom": [ + "source_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "set null", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "embeddings": { + "name": "embeddings", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "chunk_id": { + "name": "chunk_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "model": { + "name": "model", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "dimensions": { + "name": "dimensions", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_embeddings_chunk": { + "name": "idx_embeddings_chunk", + "columns": [ + "chunk_id" + ], + "isUnique": false + }, + "idx_embeddings_notebook": { + "name": "idx_embeddings_notebook", + "columns": [ + "notebook_id" + ], + "isUnique": false + }, + "idx_embeddings_model": { + "name": "idx_embeddings_model", + "columns": [ + "model" + ], + "isUnique": false + } + }, + "foreignKeys": { + "embeddings_chunk_id_chunks_id_fk": { + "name": "embeddings_chunk_id_chunks_id_fk", + "tableFrom": "embeddings", + "tableTo": "chunks", + "columnsFrom": [ + "chunk_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "embeddings_notebook_id_notebooks_id_fk": { + "name": "embeddings_notebook_id_notebooks_id_fk", + "tableFrom": "embeddings", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "folder_watches": { + "name": "folder_watches", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "path": { + "name": "path", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_folder_watches_notebook_path": { + "name": "idx_folder_watches_notebook_path", + "columns": [ + "notebook_id", + "path" + ], + "isUnique": true + } + }, + "foreignKeys": { + "folder_watches_notebook_id_notebooks_id_fk": { + "name": "folder_watches_notebook_id_notebooks_id_fk", + "tableFrom": "folder_watches", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "ingestion_runs": { + "name": "ingestion_runs", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "document_id": { + "name": "document_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "kind": { + "name": "kind", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "stage": { + "name": "stage", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "progress": { + "name": "progress", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": 0 + }, + "error_message": { + "name": "error_message", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "started_at": { + "name": "started_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "finished_at": { + "name": "finished_at", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + } + }, + "indexes": { + "idx_ingestion_runs_document": { + "name": "idx_ingestion_runs_document", + "columns": [ + "document_id", + "started_at" + ], + "isUnique": false + } + }, + "foreignKeys": { + "ingestion_runs_document_id_documents_id_fk": { + "name": "ingestion_runs_document_id_documents_id_fk", + "tableFrom": "ingestion_runs", + "tableTo": "documents", + "columnsFrom": [ + "document_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "items": { + "name": "items", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "type": { + "name": "type", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "resource_id": { + "name": "resource_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "order": { + "name": "order", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": 0 + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_items_notebook_order": { + "name": "idx_items_notebook_order", + "columns": [ + "notebook_id", + "order" + ], + "isUnique": false + }, + "idx_items_type": { + "name": "idx_items_type", + "columns": [ + "type" + ], + "isUnique": false + }, + "idx_items_resource": { + "name": "idx_items_resource", + "columns": [ + "type", + "resource_id" + ], + "isUnique": false + } + }, + "foreignKeys": { + "items_notebook_id_notebooks_id_fk": { + "name": "items_notebook_id_notebooks_id_fk", + "tableFrom": "items", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "library_sources": { + "name": "library_sources", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "type": { + "name": "type", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "source_uri": { + "name": "source_uri", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "local_file_path": { + "name": "local_file_path", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "structure": { + "name": "structure", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "content_hash": { + "name": "content_hash", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "mime_type": { + "name": "mime_type", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "file_size": { + "name": "file_size", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": {}, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "mind_maps": { + "name": "mind_maps", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "version": { + "name": "version", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": 1 + }, + "tree_data": { + "name": "tree_data", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "chunk_mapping": { + "name": "chunk_mapping", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "'generating'" + }, + "error_message": { + "name": "error_message", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_mindmaps_notebook": { + "name": "idx_mindmaps_notebook", + "columns": [ + "notebook_id", + "updated_at" + ], + "isUnique": false + }, + "idx_mindmaps_version": { + "name": "idx_mindmaps_version", + "columns": [ + "notebook_id", + "version" + ], + "isUnique": false + } + }, + "foreignKeys": { + "mind_maps_notebook_id_notebooks_id_fk": { + "name": "mind_maps_notebook_id_notebooks_id_fk", + "tableFrom": "mind_maps", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "notebook_embedding_spaces": { + "name": "notebook_embedding_spaces", + "columns": { + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "space_id": { + "name": "space_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "backend": { + "name": "backend", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "model": { + "name": "model", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "revision": { + "name": "revision", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "''" + }, + "dimensions": { + "name": "dimensions", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": {}, + "foreignKeys": { + "notebook_embedding_spaces_notebook_id_notebooks_id_fk": { + "name": "notebook_embedding_spaces_notebook_id_notebooks_id_fk", + "tableFrom": "notebook_embedding_spaces", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "notebooks": { + "name": "notebooks", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "description": { + "name": "description", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_notebooks_updated": { + "name": "idx_notebooks_updated", + "columns": [ + "updated_at" + ], + "isUnique": false + } + }, + "foreignKeys": {}, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "notes": { + "name": "notes", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "content": { + "name": "content", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_notes_notebook": { + "name": "idx_notes_notebook", + "columns": [ + "notebook_id", + "updated_at" + ], + "isUnique": false + } + }, + "foreignKeys": { + "notes_notebook_id_notebooks_id_fk": { + "name": "notes_notebook_id_notebooks_id_fk", + "tableFrom": "notes", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "quiz_sessions": { + "name": "quiz_sessions", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "quiz_id": { + "name": "quiz_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "answers": { + "name": "answers", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "score": { + "name": "score", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "total_questions": { + "name": "total_questions", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "correct_count": { + "name": "correct_count", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "completed_at": { + "name": "completed_at", + "type": "integer", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_quiz_sessions_quiz": { + "name": "idx_quiz_sessions_quiz", + "columns": [ + "quiz_id", + "created_at" + ], + "isUnique": false + }, + "idx_quiz_sessions_notebook": { + "name": "idx_quiz_sessions_notebook", + "columns": [ + "notebook_id", + "created_at" + ], + "isUnique": false + } + }, + "foreignKeys": { + "quiz_sessions_quiz_id_quizzes_id_fk": { + "name": "quiz_sessions_quiz_id_quizzes_id_fk", + "tableFrom": "quiz_sessions", + "tableTo": "quizzes", + "columnsFrom": [ + "quiz_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + }, + "quiz_sessions_notebook_id_notebooks_id_fk": { + "name": "quiz_sessions_notebook_id_notebooks_id_fk", + "tableFrom": "quiz_sessions", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + }, + "quizzes": { + "name": "quizzes", + "columns": { + "id": { + "name": "id", + "type": "text", + "primaryKey": true, + "notNull": true, + "autoincrement": false + }, + "notebook_id": { + "name": "notebook_id", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "title": { + "name": "title", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "version": { + "name": "version", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": 1 + }, + "questions_data": { + "name": "questions_data", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "chunk_mapping": { + "name": "chunk_mapping", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "metadata": { + "name": "metadata", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "status": { + "name": "status", + "type": "text", + "primaryKey": false, + "notNull": true, + "autoincrement": false, + "default": "'generating'" + }, + "error_message": { + "name": "error_message", + "type": "text", + "primaryKey": false, + "notNull": false, + "autoincrement": false + }, + "created_at": { + "name": "created_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + }, + "updated_at": { + "name": "updated_at", + "type": "integer", + "primaryKey": false, + "notNull": true, + "autoincrement": false + } + }, + "indexes": { + "idx_quizzes_notebook": { + "name": "idx_quizzes_notebook", + "columns": [ + "notebook_id", + "updated_at" + ], + "isUnique": false + }, + "idx_quizzes_version": { + "name": "idx_quizzes_version", + "columns": [ + "notebook_id", + "version" + ], + "isUnique": false + } + }, + "foreignKeys": { + "quizzes_notebook_id_notebooks_id_fk": { + "name": "quizzes_notebook_id_notebooks_id_fk", + "tableFrom": "quizzes", + "tableTo": "notebooks", + "columnsFrom": [ + "notebook_id" + ], + "columnsTo": [ + "id" + ], + "onDelete": "cascade", + "onUpdate": "no action" + } + }, + "compositePrimaryKeys": {}, + "uniqueConstraints": {}, + "checkConstraints": {} + } + }, + "views": {}, + "enums": {}, + "_meta": { + "schemas": {}, + "tables": {}, + "columns": {} + }, + "internal": { + "indexes": {} + } +} \ No newline at end of file diff --git a/src/main/db/migrations/meta/_journal.json b/src/main/db/migrations/meta/_journal.json index be10baa..d8fdc92 100644 --- a/src/main/db/migrations/meta/_journal.json +++ b/src/main/db/migrations/meta/_journal.json @@ -162,6 +162,13 @@ "when": 1790753407323, "tag": "0022_concerned_spyke", "breakpoints": true + }, + { + "idx": 23, + "version": "6", + "when": 1790949516598, + "tag": "0023_sleepy_luckman", + "breakpoints": true } ] } \ No newline at end of file diff --git a/src/main/db/queries.ts b/src/main/db/queries.ts index 55435ce..dba23fa 100644 --- a/src/main/db/queries.ts +++ b/src/main/db/queries.ts @@ -455,13 +455,6 @@ export async function deleteNotebook(id: string) { const db = getDatabase() try { - // 先获取该笔记本下所有带本地文件的文档 - const docsWithLocalFiles = db - .select({ localFilePath: documents.localFilePath }) - .from(documents) - .where(eq(documents.notebookId, id)) - .all() - // 删除笔记本(外键级联会自动删除所有关联的 sessions、messages 和 documents) db.delete(notebooks).where(eq(notebooks.id, id)).run() @@ -473,20 +466,8 @@ export async function deleteNotebook(id: string) { console.error(`[Database] Failed to drop vector table of notebook ${id}:`, error) } - // 删除本地文件(异步执行,不阻塞数据库操作) - if (docsWithLocalFiles.length > 0) { - const { unlink } = await import('fs/promises') - for (const doc of docsWithLocalFiles) { - if (doc.localFilePath) { - try { - await unlink(doc.localFilePath) - console.log(`[Database] Deleted local file: ${doc.localFilePath}`) - } catch (error) { - console.error(`[Database] Failed to delete local file: ${doc.localFilePath}`, error) - } - } - } - } + // 本地文件归库 snapshot(library_sources)所有,可能还被别的 notebook 复用(#99)。 + // 删除 notebook 只解除挂载,绝不 unlink;库文件的生命周期由显式删除 snapshot 管理。 // 执行 checkpoint 确保数据持久化 executeCheckpoint('PASSIVE') diff --git a/src/main/db/schema.ts b/src/main/db/schema.ts index a59782d..a0f211b 100644 --- a/src/main/db/schema.ts +++ b/src/main/db/schema.ts @@ -145,6 +145,36 @@ export type NewNote = typeof notes.$inferInsert // ==================== RAG 相关表 ==================== +/** + * 库来源表(#99) + * + * 一个来源在库层面只存在一次:规范文本、解析结构、原始 URI 与本地文件都属于 + * snapshot,与任何 notebook 无关。`documents` 行是某个 notebook 对这个 snapshot 的 + * 一次挂载(membership),保留自己的 ID,所以历史 citation 不受迁移影响。 + * + * snapshot 在共享期间不可变:刷新走 copy-on-write(新开一个 snapshot),重新索引 + * 只读已持久化的 snapshot,不会碰到别的 notebook。文件生命周期也归 snapshot —— + * 删除 notebook / 解除挂载都不 unlink,只有显式删除未使用的 snapshot 才会。 + */ +export const librarySources = sqliteTable('library_sources', { + id: text('id').primaryKey(), + title: text('title').notNull(), + type: text('type', { enum: ['file', 'note', 'url', 'text'] }).notNull(), + sourceUri: text('source_uri'), // 原始文件路径或 URL + localFilePath: text('local_file_path'), // 库持有的本地拷贝文件(可共享) + content: text('content'), // 规范文本 + structure: text('structure', { mode: 'json' }).$type(), + contentHash: text('content_hash'), + mimeType: text('mime_type'), + fileSize: integer('file_size'), + metadata: text('metadata', { mode: 'json' }).$type>(), + createdAt: integer('created_at', { mode: 'timestamp' }).notNull(), + updatedAt: integer('updated_at', { mode: 'timestamp' }).notNull() +}) + +export type LibrarySource = typeof librarySources.$inferSelect +export type NewLibrarySource = typeof librarySources.$inferInsert + /** * 知识库文档表 * 存储上传的文档元信息(知识来源) @@ -161,6 +191,9 @@ export const documents = sqliteTable( sourceUri: text('source_uri'), // 原始文件路径或 URL localFilePath: text('local_file_path'), // 本地拷贝文件路径 sourceNoteId: text('source_note_id').references(() => notes.id, { onDelete: 'set null' }), + // 这个 membership 指向的库 snapshot(#99)。NULL = 迁移前/尚未解析的来源, + // 它还不能被别的 notebook 复用。 + sourceId: text('source_id').references(() => librarySources.id, { onDelete: 'set null' }), content: text('content'), // 原始内容(可选存储) // 解析器给出的结构(页/章节)。这是 source record 的一部分,不是派生索引: // 重新索引必须能重建出与首次导入一致的 document_blocks,所以不能靠重新解析 @@ -189,7 +222,10 @@ export const documents = sqliteTable( }, (table) => ({ notebookIdx: index('idx_documents_notebook').on(table.notebookId, table.updatedAt), - statusIdx: index('idx_documents_status').on(table.status) + statusIdx: index('idx_documents_status').on(table.status), + // 同一个 notebook 不会重复挂载同一个 snapshot(#99)。NULL 在 SQLite 的唯一索引里 + // 互不相同,所以尚未解析的 membership 不受约束。 + sourceIdx: uniqueIndex('idx_documents_notebook_source').on(table.notebookId, table.sourceId) }) ) diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 11272b1..3b79ae4 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -7,11 +7,14 @@ import { createHash } from 'crypto' import { app } from 'electron' import { join, basename, sep } from 'path' import { mkdir, copyFile, unlink, stat } from 'fs/promises' +import type Database from 'better-sqlite3' import { getDatabase, executeCheckpoint, getNotebookVectorTable, - rebuildNotebookVectorTable + createNotebookVectorTable, + rebuildNotebookVectorTable, + getSqlite } from '../db' import { documents, @@ -34,6 +37,18 @@ import type { import { eq, desc, inArray, sql } from 'drizzle-orm' import { EmbeddingService } from './EmbeddingService' import type { EmbeddingSpace } from '../../shared/types' +import type { LibrarySourceSummary, DocumentType } from '../../shared/types/knowledge' +import { + attachLibrarySource as attachLibrarySourceMembership, + countMemberships, + deleteLibrarySource as deleteLibrarySourceRecord, + insertLibrarySource as insertLibrarySourceRow, + listLibrarySources as listLibrarySourceSummaries, + planSnapshotWrite, + updateLibrarySource as updateLibrarySourceRow, + type EmbeddingSpaceIdentity, + type VectorTableAccess +} from './librarySources' import { ChunkingService, type ChunkOptions, type ChunkResult } from './ChunkingService' import { FileParserService } from './FileParserService' import type { DocumentStructure } from './loaders/types' @@ -168,6 +183,23 @@ export interface BatchImportResult { failed: BatchImportFailure[] } +/** + * 一个库 snapshot 的可选字段(#99)。导入/刷新在解析或内容确定之后用它建 snapshot, + * membership 再缓存同一批字段,检索/阅读器继续读 `documents` 上的兼容投影。 + */ +interface LibrarySnapshotFields { + title: string + type: DocumentType + sourceUri?: string + localFilePath?: string + content?: string + structure?: DocumentStructure + contentHash?: string + mimeType?: string + fileSize?: number + metadata?: Record +} + /** * `RetrievedEvidence` → 兼容的 `SearchResult` 形状。 * @@ -239,20 +271,21 @@ export class KnowledgeService { } /** - * 拷贝文件到知识库目录 + * 拷贝文件到知识库目录。 + * + * 文件按 `ownerId`(库 snapshot 的 id)命名,而不是 document id:文件归 snapshot 所有, + * 复制式刷新会为新 snapshot 生成新文件名,不会覆盖别的 notebook 正在用的物理文件。 + * * @param sourceFilePath 源文件路径 - * @param documentId 文档 ID + * @param ownerId 拥有这份拷贝的库 snapshot id * @returns 本地文件路径 */ - private async copyFileToKnowledgeDir( - sourceFilePath: string, - documentId: string - ): Promise { + private async copyFileToKnowledgeDir(sourceFilePath: string, ownerId: string): Promise { await this.ensureKnowledgeFilesDir() // 提取文件扩展名 const extension = sourceFilePath.split('.').pop() || 'bin' - const localFileName = `${documentId}.${extension}` + const localFileName = `${ownerId}.${extension}` const localFilePath = join(this.knowledgeFilesDir, localFileName) // 拷贝文件 @@ -299,6 +332,19 @@ export class KnowledgeService { // 这一行,不会再生成新的 documentId。 onProgress?.('creating_document', 0) + // 库 snapshot 在解析/内容确定之后、索引之前建立(#99):这份内容从此可以被别的 + // notebook 复用,且复用不依赖原始可变文件。 + const sourceId = this.createLibrarySnapshot({ + title: options.title, + type: options.type, + sourceUri: options.sourceUri, + content: options.content, + contentHash, + mimeType: options.mimeType, + fileSize: options.fileSize, + metadata: options.metadata + }) + const newDoc: NewDocument = { id: documentId, notebookId, @@ -306,6 +352,7 @@ export class KnowledgeService { type: options.type, sourceUri: options.sourceUri, sourceNoteId: options.sourceNoteId, + sourceId, content: options.content, contentHash, mimeType: options.mimeType, @@ -494,26 +541,63 @@ export class KnowledgeService { } /** - * 一条 ingestion pipeline:copy(可选)→ parse → chunk → embed → 写入派生索引。 + * 一条 ingestion pipeline:copy(可选)→ parse → snapshot → chunk → embed → 写入派生索引。 * * 导入与「解析阶段失败后的重试」共用它:后者已经有本地副本,直接解析副本,不再拷贝。 * run 的生命周期也在这里维护(#95)—— 每条路径都必须留下一次完整的尝试记录。 + * + * 库 snapshot 在**解析之后、索引之前**建立(#99):内容一旦确定就能被复用,而索引 + * 失败也不会留下一个空 snapshot。`refresh` 为真时走 copy-on-write,新开一个 + * snapshot,旧 snapshot 与它正在服务的别的 notebook 完全不受影响。 */ private async ingestFile( documentId: string, filePath: string, - options: { copyFrom?: string; chunkOptions?: ChunkOptions }, + options: { copyFrom?: string; chunkOptions?: ChunkOptions; refresh?: boolean }, kind: IngestionRunKind, onProgress?: IndexProgressCallback ): Promise { const db = getDatabase() + const existing = this.getDocument(documentId) + if (!existing) throw new Error(`Document ${documentId} not found`) + const runId = startRun(documentId, kind) + // snapshot 计划:刷新/首次/被共享 -> 新开;一份从未解析成功的独占空快照 -> 原地补齐; + // 已解析且独占 -> 保持不动。共享期间快照不可变,任何路径都不会原地改写别人的快照。 + const plan = planSnapshotWrite({ + refresh: options.refresh === true, + hasSourceId: Boolean(existing.sourceId), + hasContent: Boolean(existing.content), + membershipCount: existing.sourceId + ? countMemberships(this.requireRawSqlite(), existing.sourceId) + : 0 + }) + const isNewSnapshot = plan === 'new' + const sourceId = isNewSnapshot ? this.newLibrarySourceId() : existing.sourceId! + try { + if (plan === 'keep') { + // 当前调用图不会走到这里(retryDocument 对已解析来源直接转 reindexDocument),但保留 + // 这条路径:已解析的独占快照按 ADR 从持久化快照重建,绝不因重试而漂移。 + await this.indexDocument( + documentId, + runId, + existing.content!, + existing.structure ?? undefined, + { chunkOptions: options.chunkOptions }, + onProgress + ) + completeRun(runId) + return + } + + let localFilePath = existing.localFilePath if (options.copyFrom) { advanceRun(runId, 'copying', 0) onProgress?.('copying', 0) - const localFilePath = await this.copyFileToKnowledgeDir(options.copyFrom, documentId) + // 文件归 snapshot 所有:新 snapshot 得到新文件名,不会覆盖共享的物理文件。 + localFilePath = await this.copyFileToKnowledgeDir(options.copyFrom, sourceId) db.update(documents) .set({ localFilePath, updatedAt: new Date() }) .where(eq(documents.id, documentId)) @@ -526,15 +610,39 @@ export class KnowledgeService { // 计算内容哈希 const contentHash = createHash('md5').update(parseResult.content).digest('hex') - const docType = + const docType: DocumentType = parseResult.mimeType === 'text/plain' || parseResult.mimeType === 'text/markdown' ? 'text' : 'file' + const title = parseResult.title || basename(filePath) || 'Untitled' + const sourceMtimeMs = (await stat(filePath).catch(() => null))?.mtimeMs + + // 解析完成之后、索引之前:先落 snapshot,再让 membership 指向它。 + const snapshotFields: LibrarySnapshotFields = { + title, + type: docType, + sourceUri: existing.sourceUri ?? filePath, + localFilePath: localFilePath ?? undefined, + content: parseResult.content, + structure: parseResult.structure ?? undefined, + contentHash, + mimeType: parseResult.mimeType, + fileSize: parseResult.metadata?.fileSize as number | undefined, + metadata: parseResult.metadata + } + if (isNewSnapshot) { + this.insertLibrarySnapshot(sourceId, snapshotFields) + } else { + // plan === 'fill':升级前解析失败、迁移只给它一个空的独占 snapshot(content = NULL)。 + // 重试成功后原地补齐,这份快照才真正可复用;空快照没有任何 citation 引用。 + this.updateLibrarySnapshot(sourceId, snapshotFields) + } db.update(documents) .set({ - title: parseResult.title || basename(filePath) || 'Untitled', + title, type: docType, + sourceId, content: parseResult.content, // 结构随 source 一起持久化,重新索引才能不加解析地重建同一批块 structure: parseResult.structure ?? undefined, @@ -545,7 +653,7 @@ export class KnowledgeService { errorMessage: null, // 文件回到 available,并记下这次看到的 mtime:watch(#158)靠它判断是否被改过。 sourceState: 'available', - sourceMtimeMs: (await stat(filePath).catch(() => null))?.mtimeMs, + sourceMtimeMs, updatedAt: new Date() }) .where(eq(documents.id, documentId)) @@ -953,6 +1061,18 @@ export class KnowledgeService { const now = new Date() const contentHash = createHash('md5').update(options.content).digest('hex') + // 内容已经可用,snapshot 立刻建立(#99):列表里这一行就是可复用的库来源。 + const sourceId = this.createLibrarySnapshot({ + title: options.title, + type: options.type, + sourceUri: options.sourceUri, + content: options.content, + contentHash, + mimeType: options.mimeType, + fileSize: options.fileSize, + metadata: options.metadata + }) + const newDoc: NewDocument = { id: documentId, notebookId, @@ -960,6 +1080,7 @@ export class KnowledgeService { type: options.type, sourceUri: options.sourceUri, sourceNoteId: options.sourceNoteId, + sourceId, content: options.content, contentHash, mimeType: options.mimeType, @@ -1061,17 +1182,31 @@ export class KnowledgeService { jobProgress('fetching_url', 0) const fetchResult = await this.webFetchService.fetchUrl(url) const content = fetchResult.content + const title = fetchResult.title || url + const metadata = { + ...fetchResult.metadata, + description: fetchResult.description + } + + // 抓取完成之后、索引之前建立 snapshot(#99)。 + const sourceId = this.createLibrarySnapshot({ + title, + type: 'url', + sourceUri: url, + content, + contentHash: createHash('md5').update(content).digest('hex'), + mimeType: fetchResult.mimeType, + metadata + }) db.update(documents) .set({ - title: fetchResult.title || url, + title, + sourceId, content, contentHash: createHash('md5').update(content).digest('hex'), mimeType: fetchResult.mimeType, - metadata: { - ...fetchResult.metadata, - description: fetchResult.description - }, + metadata, updatedAt: new Date() }) .where(eq(documents.id, documentId)) @@ -1321,27 +1456,145 @@ export class KnowledgeService { } /** - * 删除文档 + * 从一个 notebook 解除挂载(#99)。 + * + * 只删这个 membership 及其派生索引:库 snapshot 与它的物理文件保留(可能还有别的 + * notebook 在用,也可能只是留着以后复用)。永久删除库来源是另一个显式、已确认的 + * 操作(`deleteLibrarySource`)。notebook 删除走同样的 detach-only 语义。 */ async deleteDocument(documentId: string): Promise { const db = getDatabase() - // 获取文档信息 const doc = db.select().from(documents).where(eq(documents.id, documentId)).get() if (!doc) return - // 删除全部派生索引(向量、映射、chunks、embeddings、blocks) + // 删除全部派生索引(向量、映射、chunks、embeddings、blocks)。 await this.clearDerivedIndex(documentId) - // 删除本地拷贝的文件(如果存在) - if (doc.localFilePath) { - await this.deleteLocalFile(doc.localFilePath) - } - + // 只删 membership 行:文件名归库 snapshot 所有,这里绝不能 unlink。 db.delete(documents).where(eq(documents.id, documentId)).run() executeCheckpoint('PASSIVE') - Logger.info('KnowledgeService', `Document deleted: ${documentId}`) + Logger.info('KnowledgeService', `Document detached: ${documentId}`) + } + + /** + * 列出这个 notebook 还没挂载的库 snapshot(#99)。 + * + * `canReuseIndex` 需要「当前配置的 space」,所以它是异步的:只有 donor 持久化的 + * space 与当前模型一致、且目标 notebook 能接住时,才可能不调用 embedding 直接复用。 + */ + async listLibrarySources(notebookId: string): Promise { + const space = await this.embeddingService.getSpace() + return listLibrarySourceSummaries( + this.requireRawSqlite(), + notebookId, + this.embeddingSpaceIdentity(space), + this.vectorTableAccess() + ) + } + + /** + * 把一个库 snapshot 挂载到 notebook(#99)。 + * + * 可复用时只复制 donor 的派生索引与向量,**绝不调用 embedding**;不可复用(模型/ + * 维度不匹配,或没有 donor)时落一个 pending membership,等用户显式重新索引(这条 + * 路径不碰任何已有向量)。重复挂载幂等。所有 DB 操作在 `getSpace()` 之后同步完成, + * 所以挂载与删除不会交错。 + */ + async attachLibrarySource( + notebookId: string, + sourceId: string + ): Promise<{ documentId: string; indexed: boolean }> { + const space = await this.embeddingService.getSpace() + const result = attachLibrarySourceMembership(this.requireRawSqlite(), { + notebookId, + sourceId, + currentSpace: this.embeddingSpaceIdentity(space), + tables: this.vectorTableAccess() + }) + // 空目标会在 attach 里新建向量表,而 vectorStoreManager 可能已经缓存了一个 + // 无向量表的 store(初始化时读过 vec_metadata 为空)。丢掉缓存,下一次检索/写入 + // 会从新的元数据重新初始化,否则会以为目标还没有向量表。 + await vectorStoreManager.closeStore(notebookId) + Logger.info( + 'KnowledgeService', + `Library source attached: ${sourceId} -> ${notebookId} (indexed: ${result.indexed})` + ) + return { documentId: result.documentId, indexed: result.indexed } + } + + /** + * 永久删除一个库 snapshot(#99)。未确认或仍被挂载时拒绝;删成功后 unlink 它的 + * 本地文件。这是唯一会删库文件的路径。 + */ + async deleteLibrarySource(sourceId: string, confirmed: boolean): Promise { + const { localFilePath, deleted } = deleteLibrarySourceRecord( + this.requireRawSqlite(), + sourceId, + confirmed + ) + if (deleted && localFilePath) { + await this.deleteLocalFile(localFilePath) + } + Logger.info('KnowledgeService', `Library source deleted: ${sourceId}`) + } + + /** 当前 embedding 空间的身份,库复用用它判断向量可比性。 */ + private embeddingSpaceIdentity(space: EmbeddingSpace): EmbeddingSpaceIdentity { + return { id: space.id, dimensions: space.dimensions } + } + + /** 库复用边界需要的向量表读写,从 db 层注入。 */ + private vectorTableAccess(): VectorTableAccess { + return { + read: (notebookId) => getNotebookVectorTable(notebookId), + ensure: (notebookId, dimensions) => createNotebookVectorTable(notebookId, dimensions) + } + } + + private requireRawSqlite(): Database.Database { + const sqlite = getSqlite() + if (!sqlite) throw new Error('Database not initialized') + return sqlite + } + + private newLibrarySourceId(): string { + return `lib_${Date.now()}_${Math.random().toString(36).slice(2, 9)}` + } + + private createLibrarySnapshot(fields: LibrarySnapshotFields): string { + const id = this.newLibrarySourceId() + this.insertLibrarySnapshot(id, fields) + return id + } + + /** 插入一个库 snapshot 行。调用方负责在解析/内容确定之后、索引之前调用。 */ + private insertLibrarySnapshot(id: string, fields: LibrarySnapshotFields): void { + insertLibrarySourceRow(this.requireRawSqlite(), id, this.snapshotRowFields(fields)) + } + + /** 原地补齐一份从没有过内容的 snapshot(迁移回填的解析失败来源)。 */ + private updateLibrarySnapshot(id: string, fields: LibrarySnapshotFields): void { + updateLibrarySourceRow(this.requireRawSqlite(), id, this.snapshotRowFields(fields)) + } + + /** `LibrarySnapshotFields` → 落库字段;structure/metadata 与 drizzle 的 json 模式一致。 */ + private snapshotRowFields( + fields: LibrarySnapshotFields + ): Parameters[2] { + return { + title: fields.title, + type: fields.type, + sourceUri: fields.sourceUri ?? null, + localFilePath: fields.localFilePath ?? null, + content: fields.content ?? null, + structure: fields.structure ? JSON.stringify(fields.structure) : null, + contentHash: fields.contentHash ?? null, + mimeType: fields.mimeType ?? null, + fileSize: fields.fileSize ?? null, + metadata: fields.metadata ? JSON.stringify(fields.metadata) : null + } } /** @@ -1448,13 +1701,15 @@ export class KnowledgeService { } /** - * 来源文件变了:重新拷贝、解析并索引**同一个** documentId(#158)。 + * 来源文件变了:重新拷贝、解析并索引**同一个** membership(#158)。 * * 与 reindex 的区别:reindex 用的是已持久化的 content,而这里文件本身被改过,必须 - * 重新解析。来源身份不变,所以历史 citation 与摘录仍然指向同一个来源。 + * 重新解析。刷新是 copy-on-write(#99):新开一个库 snapshot(新文件、新 id),旧 + * snapshot 与它正在服务的别的 notebook 的文本/页偏移完全不变。membership 的 ID + * 不变,所以历史 citation 与摘录仍然指向同一个来源行。 */ async refreshDocumentFromFile(documentId: string, filePath: string): Promise { - await this.ingestFile(documentId, filePath, { copyFrom: filePath }, 'reindex') + await this.ingestFile(documentId, filePath, { copyFrom: filePath, refresh: true }, 'reindex') } /** diff --git a/src/main/services/librarySources.ts b/src/main/services/librarySources.ts new file mode 100644 index 0000000..988d349 --- /dev/null +++ b/src/main/services/librarySources.ts @@ -0,0 +1,628 @@ +/** + * 库来源复用边界(#99)。 + * + * 一个来源在库层面只存在一次(`library_sources` 的一行 snapshot);`documents` 行是 + * 某个 notebook 对它的一次挂载(membership),保留自己的 ID 与派生索引。本模块实现 + * 复用这条边界: + * + * - 列出某个 notebook 还没挂载的 snapshot; + * - 挂载一个 snapshot(能复用就复制已索引 donor 的 blocks/chunks/mappings/向量, + * 绝不调用 embedding); + * - 显式、已确认地删除一个未被挂载的 snapshot。 + * + * 刻意不 import Electron:入参是 raw better-sqlite3 连接,所以这条边界能在测试里用 + * 真实的 SQLite + sqlite-vec 跑,而不必启动桌面应用。向量表信息通过 `VectorTableAccess` + * 注入,测试可以自己建真的 vec0 表。 + * + * 同步性:所有数据库操作都是同步的(better-sqlite3 / node 单线程),函数内部没有 + * await,所以「检查重复 → 写入」之间不会有另一个挂载/删除插进来,挂载与删除天然不会 + * 互相踩到对方的中间状态。 + */ + +import type Database from 'better-sqlite3' +import type { DocumentType, LibrarySourceSummary } from '../../shared/types/knowledge' +import { safeIdentifier } from '../vectorstore/vectorTableSql' + +/** 决定向量可比性的最小 space 身份:模型/维度一致才允许复制向量。 */ +export interface EmbeddingSpaceIdentity { + id: string + dimensions: number +} + +export interface VectorTable { + tableName: string + dimensions: number +} + +/** 注入向量表读写,避免把 Electron 的 db/index 拉进本模块。 */ +export interface VectorTableAccess { + read(notebookId: string): VectorTable | undefined + ensure(notebookId: string, dimensions: number): VectorTable +} + +export interface AttachResult { + documentId: string + indexed: boolean + chunkCount: number +} + +/** + * 一个库 snapshot 的字段。`structure` / `metadata` 已经是 JSON 文本(与 drizzle 的 + * json 模式一致),由调用方序列化。 + */ +export interface LibrarySnapshotFields { + title: string + type: string + sourceUri: string | null + localFilePath: string | null + content: string | null + structure: string | null + contentHash: string | null + mimeType: string | null + fileSize: number | null + metadata: string | null +} + +interface LibrarySourceRow { + id: string + title: string + type: string + source_uri: string | null + local_file_path: string | null + content: string | null + structure: string | null + content_hash: string | null + mime_type: string | null + file_size: number | null + metadata: string | null + created_at: number + updated_at: number +} + +interface DonorRow { + document_id: string + notebook_id: string + space_id: string + dimensions: number +} + +const newId = (prefix: string): string => + `${prefix}_${Date.now()}_${Math.random().toString(36).slice(2, 9)}` + +const epochSeconds = (date: Date): number => Math.floor(date.getTime() / 1000) + +/** + * 写入一份新的库 snapshot。导入与 copy-on-write 刷新共用它:刷新就是新开一行,旧 + * snapshot 一行都不改。 + */ +export function insertLibrarySource( + sqlite: Database.Database, + id: string, + fields: LibrarySnapshotFields, + now = new Date() +): void { + const nowSec = epochSeconds(now) + sqlite + .prepare( + `INSERT INTO library_sources + (id, title, type, source_uri, local_file_path, content, structure, content_hash, + mime_type, file_size, metadata, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + ) + .run( + id, + fields.title, + fields.type, + fields.sourceUri, + fields.localFilePath, + fields.content, + fields.structure, + fields.contentHash, + fields.mimeType, + fields.fileSize, + fields.metadata, + nowSec, + nowSec + ) +} + +/** + * 补齐一份从没有过内容的 snapshot(升级前解析失败、迁移回填时 content 为 NULL)。 + * 共享期间不可变只保护真正有效的快照;一份空快照原地补齐不会被任何 citation 引用。 + */ +export function updateLibrarySource( + sqlite: Database.Database, + id: string, + fields: LibrarySnapshotFields, + now = new Date() +): void { + sqlite + .prepare( + `UPDATE library_sources + SET title = ?, type = ?, source_uri = ?, local_file_path = ?, content = ?, + structure = ?, content_hash = ?, mime_type = ?, file_size = ?, metadata = ?, + updated_at = ? + WHERE id = ?` + ) + .run( + fields.title, + fields.type, + fields.sourceUri, + fields.localFilePath, + fields.content, + fields.structure, + fields.contentHash, + fields.mimeType, + fields.fileSize, + fields.metadata, + epochSeconds(now), + id + ) +} + +/** + * 重试/刷新一份来源时 snapshot 该怎么处理。ADR 要求共享期间快照不可变,所以: + * - refresh:永远新开(copy-on-write),绝不动共享的文件。 + * - 没有 snapshot:新开。 + * - 被多个 membership 共享:新开 —— 绝不为一个 membership 原地改写别人的快照。 + * - 独占、且从未解析成功(content 为空):原地补齐(空快照没有有效内容可漂移)。 + * - 独占、已解析:保持不动,重试只重建派生索引。 + */ +export type SnapshotWritePlan = 'new' | 'fill' | 'keep' + +export function planSnapshotWrite(params: { + refresh: boolean + hasSourceId: boolean + hasContent: boolean + membershipCount: number +}): SnapshotWritePlan { + if (params.refresh || !params.hasSourceId) return 'new' + if (params.membershipCount > 1) return 'new' + if (!params.hasContent) return 'fill' + return 'keep' +} + +/** 一个 snapshot 被多少个 notebook membership 挂载。 */ +export function countMemberships(sqlite: Database.Database, sourceId: string): number { + return ( + sqlite.prepare('SELECT COUNT(*) AS count FROM documents WHERE source_id = ?').get(sourceId) as { + count: number + } + ).count +} + +/** + * 找一个可复用的 donor:同 snapshot、别的 notebook、已索引、且持久化的 space 与当前 + * 配置的 space 完全一致。只用维度不足以证明可比(换模型但维度相同时旧向量不可比)。 + */ +function findIndexedDonor( + sqlite: Database.Database, + sourceId: string, + targetNotebookId: string, + currentSpace: EmbeddingSpaceIdentity +): DonorRow | undefined { + const rows = sqlite + .prepare( + `SELECT d.id AS document_id, d.notebook_id, s.space_id, s.dimensions + FROM documents d + JOIN notebook_embedding_spaces s ON s.notebook_id = d.notebook_id + WHERE d.source_id = ? + AND d.notebook_id <> ? + AND d.status = 'indexed' + AND d.chunk_count > 0 + ORDER BY d.updated_at DESC` + ) + .all(sourceId, targetNotebookId) as DonorRow[] + + return rows.find( + (row) => row.space_id === currentSpace.id && row.dimensions === currentSpace.dimensions + ) +} + +/** + * 目标 notebook 能不能接住 donor 的向量:没有向量表(空目标)就采用 donor 的 space; + * 已有向量表则维度必须一致,且记录的 space 身份也必须一致。不一致就不可能复用。 + */ +function targetAcceptsDonor( + sqlite: Database.Database, + targetNotebookId: string, + donor: DonorRow, + tables: VectorTableAccess +): boolean { + const targetTable = tables.read(targetNotebookId) + if (!targetTable) return true + if (targetTable.dimensions !== donor.dimensions) return false + + const targetSpace = sqlite + .prepare('SELECT space_id FROM notebook_embedding_spaces WHERE notebook_id = ?') + .get(targetNotebookId) as { space_id: string } | undefined + return targetSpace?.space_id === donor.space_id +} + +/** + * 复用判定只有这一处:`listLibrarySources` 的 `canReuseIndex` 与 `attachLibrarySource` + * 实际会不会复制向量必须得出同一个答案,否则 UI 会承诺一个 attach 兑现不了的复用。 + */ +function resolveReuse( + sqlite: Database.Database, + sourceId: string, + targetNotebookId: string, + currentSpace: EmbeddingSpaceIdentity, + tables: VectorTableAccess +): { donor: DonorRow | undefined; reusable: boolean } { + const donor = findIndexedDonor(sqlite, sourceId, targetNotebookId, currentSpace) + return { + donor, + reusable: donor ? targetAcceptsDonor(sqlite, targetNotebookId, donor, tables) : false + } +} + +/** + * 某个 notebook 可以挂载的库 snapshot:排除已经挂载到它的那些。`canReuseIndex` 表示 + * 存在一个与当前 space 一致、且目标能接住的已索引 donor。 + */ +export function listLibrarySources( + sqlite: Database.Database, + notebookId: string, + currentSpace: EmbeddingSpaceIdentity, + tables: VectorTableAccess +): LibrarySourceSummary[] { + const rows = sqlite + .prepare( + `SELECT ls.*, + (SELECT COUNT(*) FROM documents d WHERE d.source_id = ls.id) AS membership_count + FROM library_sources ls + WHERE ls.content IS NOT NULL + AND NOT EXISTS ( + SELECT 1 FROM documents d2 + WHERE d2.source_id = ls.id AND d2.notebook_id = ? + ) + ORDER BY ls.updated_at DESC` + ) + .all(notebookId) as Array + + return rows.map((row) => ({ + id: row.id, + title: row.title, + type: row.type as DocumentType, + mimeType: row.mime_type, + membershipCount: row.membership_count, + canReuseIndex: resolveReuse(sqlite, row.id, notebookId, currentSpace, tables).reusable + })) +} + +const MEMBERSHIP_COLUMNS = `(id, notebook_id, title, type, source_uri, local_file_path, + content, structure, content_hash, mime_type, file_size, metadata, source_id, status, + source_state, chunk_count, created_at, updated_at)` + +function insertMembership( + sqlite: Database.Database, + source: LibrarySourceRow, + notebookId: string, + documentId: string, + status: 'pending' | 'indexed', + chunkCount: number, + nowSec: number +): void { + sqlite + .prepare( + `INSERT INTO documents ${MEMBERSHIP_COLUMNS} + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'available', ?, ?, ?)` + ) + .run( + documentId, + notebookId, + source.title, + source.type, + source.source_uri, + source.local_file_path, + source.content, + source.structure, + source.content_hash, + source.mime_type, + source.file_size, + source.metadata, + source.id, + status, + chunkCount, + nowSec, + nowSec + ) +} + +/** + * 把一个库 snapshot 挂载到 notebook。 + * + * 重复挂载是幂等的:已经挂载过就返回现有 membership,不新建、不改索引。没有可复用 + * donor(或目标不兼容)时落一个 `pending` membership,等用户显式重新索引 —— 这条路径 + * 不碰任何已有向量。可复用时不调用 embedding,只按新 ID 复制派生索引与向量。 + */ +export function attachLibrarySource( + sqlite: Database.Database, + params: { + notebookId: string + sourceId: string + currentSpace: EmbeddingSpaceIdentity + tables: VectorTableAccess + now?: Date + } +): AttachResult { + const now = params.now ?? new Date() + const nowSec = epochSeconds(now) + + const notebook = sqlite.prepare('SELECT id FROM notebooks WHERE id = ?').get(params.notebookId) + if (!notebook) throw new Error(`Notebook ${params.notebookId} not found`) + + const source = sqlite + .prepare('SELECT * FROM library_sources WHERE id = ?') + .get(params.sourceId) as LibrarySourceRow | undefined + if (!source) throw new Error(`Library source ${params.sourceId} not found`) + if (source.content === null) { + throw new Error( + `Library source ${params.sourceId} has no parsed content and cannot be reused; re-import the file instead` + ) + } + + const existing = sqlite + .prepare( + 'SELECT id, status, chunk_count FROM documents WHERE notebook_id = ? AND source_id = ?' + ) + .get(params.notebookId, params.sourceId) as + { id: string; status: string; chunk_count: number | null } | undefined + if (existing) { + const chunkCount = existing.chunk_count ?? 0 + return { + documentId: existing.id, + indexed: existing.status === 'indexed' && chunkCount > 0, + chunkCount + } + } + + const { donor, reusable } = resolveReuse( + sqlite, + params.sourceId, + params.notebookId, + params.currentSpace, + params.tables + ) + + const documentId = newId('doc') + + if (!reusable) { + insertMembership(sqlite, source, params.notebookId, documentId, 'pending', 0, nowSec) + return { documentId, indexed: false, chunkCount: 0 } + } + + // 空目标采用 donor 的 space 身份:建一张同宽度的向量表并记录 space,之后复制向量。 + let targetTable = params.tables.read(params.notebookId) + if (!targetTable) { + targetTable = params.tables.ensure(params.notebookId, donor!.dimensions) + const donorSpace = sqlite + .prepare( + `SELECT space_id, backend, model, revision, dimensions + FROM notebook_embedding_spaces WHERE notebook_id = ?` + ) + .get(donor!.notebook_id) as + | { + space_id: string + backend: string + model: string + revision: string + dimensions: number + } + | undefined + if (donorSpace) { + sqlite + .prepare( + `INSERT OR REPLACE INTO notebook_embedding_spaces + (notebook_id, space_id, backend, model, revision, dimensions, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?)` + ) + .run( + params.notebookId, + donorSpace.space_id, + donorSpace.backend, + donorSpace.model, + donorSpace.revision, + donorSpace.dimensions, + nowSec + ) + } + } + + const donorTable = params.tables.read(donor!.notebook_id) + if (!donorTable) { + throw new Error( + `Donor notebook ${donor!.notebook_id} is indexed but has no vector table to copy from` + ) + } + + const chunkCount = sqlite.transaction(() => { + insertMembership(sqlite, source, params.notebookId, documentId, 'indexed', 0, nowSec) + return cloneIndexedMembership(sqlite, { + donorDocumentId: donor!.document_id, + targetDocumentId: documentId, + targetNotebookId: params.notebookId, + donorVectorTable: donorTable.tableName, + targetVectorTable: targetTable!.tableName, + now + }) + })() + + sqlite.prepare('UPDATE documents SET chunk_count = ? WHERE id = ?').run(chunkCount, documentId) + + return { documentId, indexed: true, chunkCount } +} + +/** + * 复制一份已经索引的 membership 的派生索引:blocks、chunks、chunk↔block 映射、 + * embeddings 元数据,以及 vec0 表里的向量。全部用新 ID,因此 donor 与目标互不影响; + * page/offset/span 原样复制,历史 citation 的分页与字符区间不变。 + * + * 纯同步 SQL:向量用 `INSERT ... SELECT embedding FROM donor` 在同一连接里搬,所以 + * 整个过程可以在一个事务里完成,不需要在事务中间 await。 + */ +export function cloneIndexedMembership( + sqlite: Database.Database, + params: { + donorDocumentId: string + targetDocumentId: string + targetNotebookId: string + donorVectorTable: string + targetVectorTable: string + now: Date + } +): number { + const nowSec = epochSeconds(params.now) + + const blocks = sqlite + .prepare('SELECT * FROM document_blocks WHERE document_id = ? ORDER BY "order"') + .all(params.donorDocumentId) as Array & { id: string }> + const chunks = sqlite + .prepare('SELECT * FROM chunks WHERE document_id = ? ORDER BY chunk_index') + .all(params.donorDocumentId) as Array & { id: string }> + const mappings = sqlite + .prepare( + `SELECT cb.chunk_id, cb.block_id, cb.start_in_block, cb.end_in_block + FROM chunk_blocks cb + JOIN chunks c ON c.id = cb.chunk_id + WHERE c.document_id = ?` + ) + .all(params.donorDocumentId) as Array<{ + chunk_id: string + block_id: string + start_in_block: number + end_in_block: number + }> + const embeddings = sqlite + .prepare( + `SELECT e.id, e.chunk_id, e.model, e.dimensions + FROM embeddings e + JOIN chunks c ON c.id = e.chunk_id + WHERE c.document_id = ?` + ) + .all(params.donorDocumentId) as Array<{ + id: string + chunk_id: string + model: string + dimensions: number + }> + + const blockIdMap = new Map() + const chunkIdMap = new Map() + + const insertBlock = sqlite.prepare( + `INSERT INTO document_blocks + (id, document_id, kind, "order", page, level, text, start_offset, end_offset, bbox, metadata) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + ) + for (const block of blocks) { + const id = newId('block') + blockIdMap.set(block.id, id) + insertBlock.run( + id, + params.targetDocumentId, + block.kind, + block.order, + block.page, + block.level, + block.text, + block.start_offset, + block.end_offset, + block.bbox, + block.metadata + ) + } + + const insertChunk = sqlite.prepare( + `INSERT INTO chunks + (id, document_id, notebook_id, content, chunk_index, start_offset, end_offset, + page_start, page_end, metadata, token_count, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + ) + for (const chunk of chunks) { + const id = newId('chunk') + chunkIdMap.set(chunk.id, id) + insertChunk.run( + id, + params.targetDocumentId, + params.targetNotebookId, + chunk.content, + chunk.chunk_index, + chunk.start_offset, + chunk.end_offset, + chunk.page_start, + chunk.page_end, + chunk.metadata, + chunk.token_count, + nowSec + ) + } + + const insertMapping = sqlite.prepare( + 'INSERT INTO chunk_blocks (chunk_id, block_id, start_in_block, end_in_block) VALUES (?, ?, ?, ?)' + ) + for (const mapping of mappings) { + const chunkId = chunkIdMap.get(mapping.chunk_id) + const blockId = blockIdMap.get(mapping.block_id) + if (!chunkId || !blockId) continue + insertMapping.run(chunkId, blockId, mapping.start_in_block, mapping.end_in_block) + } + + const insertEmbedding = sqlite.prepare( + 'INSERT INTO embeddings (id, chunk_id, notebook_id, model, dimensions, created_at) VALUES (?, ?, ?, ?, ?, ?)' + ) + const copyVector = sqlite.prepare( + `INSERT INTO ${safeIdentifier(params.targetVectorTable)} + (embedding_id, chunk_id, embedding) + SELECT ?, ?, embedding FROM ${safeIdentifier(params.donorVectorTable)} + WHERE embedding_id = ?` + ) + for (const embedding of embeddings) { + const chunkId = chunkIdMap.get(embedding.chunk_id) + if (!chunkId) continue + const embeddingId = newId('emb') + insertEmbedding.run( + embeddingId, + chunkId, + params.targetNotebookId, + embedding.model, + embedding.dimensions, + nowSec + ) + copyVector.run(embeddingId, chunkId, embedding.id) + } + + return chunks.length +} + +/** + * 永久删除一个库 snapshot。未确认或仍被挂载时拒绝;成功时返回本地文件路径,由调用方 + * (持有文件系统权限的 KnowledgeService)决定 unlink。 + */ +export function deleteLibrarySource( + sqlite: Database.Database, + sourceId: string, + confirmed: boolean +): { localFilePath: string | null; deleted: boolean } { + if (!confirmed) { + throw new Error('Permanent deletion of a library source requires explicit confirmation') + } + + const source = sqlite + .prepare('SELECT id, local_file_path FROM library_sources WHERE id = ?') + .get(sourceId) as { id: string; local_file_path: string | null } | undefined + if (!source) return { localFilePath: null, deleted: false } + + const attached = sqlite + .prepare('SELECT COUNT(*) AS count FROM documents WHERE source_id = ?') + .get(sourceId) as { count: number } + if (attached.count > 0) { + throw new Error( + `Library source ${sourceId} is still attached to ${attached.count} notebook membership(s)` + ) + } + + sqlite.prepare('DELETE FROM library_sources WHERE id = ?').run(sourceId) + return { localFilePath: source.local_file_path, deleted: true } +} diff --git a/src/shared/types/knowledge.ts b/src/shared/types/knowledge.ts index d3241d4..616655f 100644 --- a/src/shared/types/knowledge.ts +++ b/src/shared/types/knowledge.ts @@ -20,6 +20,22 @@ export type KnowledgeChunk = Chunk */ export type DocumentType = 'file' | 'note' | 'url' | 'text' +/** + * 库来源摘要(#99)。 + * + * "从库中添加"只看得见这些字段:snapshot 的身份、能否复用已有向量。`membershipCount` + * 是已经挂载它的 notebook 数;`canReuseIndex` 表示目标 notebook 能直接复制一份已索引 + * 的 donor(不调用 embedding),否则新成员先落成 pending,等用户显式重新索引。 + */ +export interface LibrarySourceSummary { + id: string + title: string + type: DocumentType + mimeType: string | null + membershipCount: number + canReuseIndex: boolean +} + /** * 文档状态枚举 */ diff --git a/test/librarySourceBackfill.test.ts b/test/librarySourceBackfill.test.ts new file mode 100644 index 0000000..9e92b87 --- /dev/null +++ b/test/librarySourceBackfill.test.ts @@ -0,0 +1,141 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { DatabaseSync } from 'node:sqlite' +import { readFileSync } from 'node:fs' + +/** + * 迁移 0023 的一比一回填(#99)。 + * + * 现有 documents 要在升级时各自得到自己的 library_sources 行,**不合并**标题或路径 + * 相同的来源;membership 保留原 id,历史 citation 不变。这里执行真实迁移文件的语句, + * 而不是复述它的 SQL。 + */ + +const MIGRATION = 'src/main/db/migrations/0023_sleepy_luckman.sql' + +function applyMigration(db: DatabaseSync): void { + const statements = readFileSync(MIGRATION, 'utf8') + .split('--> statement-breakpoint') + .map((statement) => statement.trim()) + .filter((statement) => statement.length > 0) + for (const statement of statements) db.exec(statement) +} + +function seedLegacyDocuments(db: DatabaseSync): void { + db.exec(`CREATE TABLE documents ( + id TEXT PRIMARY KEY, + notebook_id TEXT NOT NULL, + title TEXT NOT NULL, + type TEXT NOT NULL, + source_uri TEXT, + local_file_path TEXT, + content TEXT, + structure TEXT, + content_hash TEXT, + mime_type TEXT, + file_size INTEGER, + metadata TEXT, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL + )`) + const insert = db.prepare( + 'INSERT INTO documents VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)' + ) + // 同样的标题、不同的内容:必须回填成两行,而不是被合并成一个来源。 + insert.run( + 'doc_1', + 'nb_1', + 'Shared title', + 'file', + '/src/a.pdf', + '/files/doc_1.pdf', + 'content one', + JSON.stringify({ sections: ['one'] }), + 'hash1', + 'application/pdf', + 10, + JSON.stringify({ pages: 3 }), + 100, + 200 + ) + insert.run( + 'doc_2', + 'nb_2', + 'Shared title', + 'text', + '/src/b.md', + '/files/doc_2.md', + 'content two', + null, + 'hash2', + 'text/markdown', + 20, + null, + 300, + 400 + ) +} + +test('migration backfills one library source per legacy document without merging', () => { + const db = new DatabaseSync(':memory:') + try { + seedLegacyDocuments(db) + applyMigration(db) + + const rows = db.prepare('SELECT * FROM library_sources ORDER BY id').all() as Array< + Record + > + assert.equal(rows.length, 2) + assert.deepEqual( + rows.map((row) => row.id), + ['lib_doc_1', 'lib_doc_2'] + ) + + // 规范文本、解析结构、原始 URI、本地文件都跟着 snapshot 走。 + assert.equal(rows[0].title, 'Shared title') + assert.equal(rows[0].content, 'content one') + assert.equal(rows[0].structure, JSON.stringify({ sections: ['one'] })) + assert.equal(rows[0].source_uri, '/src/a.pdf') + assert.equal(rows[0].local_file_path, '/files/doc_1.pdf') + assert.equal(rows[0].content_hash, 'hash1') + assert.equal(rows[0].mime_type, 'application/pdf') + assert.equal(rows[0].file_size, 10) + assert.equal(rows[0].metadata, JSON.stringify({ pages: 3 })) + assert.equal(rows[1].type, 'text') + + const documents = ( + db.prepare('SELECT id, source_id FROM documents ORDER BY id').all() as Array<{ + id: string + source_id: string + }> + ).map((row) => ({ id: row.id, source_id: row.source_id })) + assert.deepEqual(documents, [ + { id: 'doc_1', source_id: 'lib_doc_1' }, + { id: 'doc_2', source_id: 'lib_doc_2' } + ]) + } finally { + db.close() + } +}) + +test('the backfill is idempotent on re-run because it only fills NULL source ids', () => { + const db = new DatabaseSync(':memory:') + try { + seedLegacyDocuments(db) + applyMigration(db) + // 再跑一次 UPDATE(迁移只跑一次,但 guard 让重跑安全)。 + db.exec("UPDATE documents SET source_id = 'lib_' || id WHERE source_id IS NULL") + const documents = ( + db.prepare('SELECT id, source_id FROM documents ORDER BY id').all() as Array<{ + id: string + source_id: string + }> + ).map((row) => ({ id: row.id, source_id: row.source_id })) + assert.deepEqual(documents, [ + { id: 'doc_1', source_id: 'lib_doc_1' }, + { id: 'doc_2', source_id: 'lib_doc_2' } + ]) + } finally { + db.close() + } +}) diff --git a/test/librarySourceGuards.test.ts b/test/librarySourceGuards.test.ts new file mode 100644 index 0000000..cd481a5 --- /dev/null +++ b/test/librarySourceGuards.test.ts @@ -0,0 +1,119 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' + +/** + * 库来源复用里「不调用 embedding」「只解除挂载」「不 unlink」这些约束。 + * + * 这些行为落在 Electron 的 KnowledgeService / db/queries 上,node test 里没法实例化 + * 它们(会 import electron),所以按仓库里 architectureBoundary.test.ts 的既有做法, + * 扫描真实源码把方向固定下来。真正能跑的行为在 librarySourceReuse.test.ts 里。 + */ + +const KNOWLEDGE = 'src/main/services/KnowledgeService.ts' +const LIBRARY_MODULE = 'src/main/services/librarySources.ts' +const QUERIES = 'src/main/db/queries.ts' + +/** 从 `signature` 处开始,按花括号配对取出整个方法体(跳过参数表里的对象类型)。 */ +function methodBody(source: string, signature: string): string { + const start = source.indexOf(signature) + assert.ok(start >= 0, `signature not found: ${signature}`) + + // 先跳过参数表(对象类型字面量的 `{}` 不是方法体)。 + let cursor = source.indexOf('(', start) + let paren = 0 + for (; cursor < source.length; cursor++) { + if (source[cursor] === '(') paren++ + else if (source[cursor] === ')') { + paren-- + if (paren === 0) { + cursor++ + break + } + } + } + + const open = source.indexOf('{', cursor) + assert.ok(open >= 0, `opening brace not found after: ${signature}`) + + let depth = 0 + for (let i = open; i < source.length; i++) { + if (source[i] === '{') depth++ + else if (source[i] === '}') { + depth-- + if (depth === 0) return source.slice(open, i + 1) + } + } + throw new Error(`unbalanced braces for: ${signature}`) +} + +test('attaching a library source never reaches for the embedding provider', () => { + const knowledge = readFileSync(KNOWLEDGE, 'utf8') + const attach = methodBody(knowledge, 'async attachLibrarySource(') + for (const forbidden of ['embedBatch', 'embedDocumentChunks', 'indexDocument']) { + assert.equal( + attach.includes(forbidden), + false, + `attachLibrarySource must not call ${forbidden}; reuse copies vectors, it never embeds` + ) + } + + const module = readFileSync(LIBRARY_MODULE, 'utf8') + assert.equal(module.includes('embedBatch'), false) + assert.equal(module.includes('embedDocumentChunks'), false) + assert.equal( + module.includes('EmbeddingService'), + false, + 'the reuse boundary must not import an embedding provider at all' + ) +}) + +test('deleting a document is detach-only: it never unlinks the library file', () => { + const knowledge = readFileSync(KNOWLEDGE, 'utf8') + const detach = methodBody(knowledge, 'async deleteDocument(') + assert.equal(detach.includes('deleteLocalFile('), false, 'deleteDocument unlinked a library file') + assert.equal(detach.includes('unlink('), false) + assert.match(detach, /db\.delete\(documents\)/) +}) + +test('deleting a notebook never unlinks files owned by a library source', () => { + const queries = readFileSync(QUERIES, 'utf8') + const deleteNotebook = methodBody(queries, 'export async function deleteNotebook(') + assert.equal(deleteNotebook.includes('unlink('), false) + assert.equal(deleteNotebook.includes('localFilePath'), false) +}) + +test('only an explicit, confirmed library deletion unlinks the file', () => { + const knowledge = readFileSync(KNOWLEDGE, 'utf8') + const deleteSource = methodBody(knowledge, 'async deleteLibrarySource(') + assert.match(deleteSource, /this\.deleteLocalFile\(/) + assert.match(deleteSource, /deleteLibrarySourceRecord\(/) +}) + +test('file refresh is copy-on-write: new snapshot id and an owner-named file', () => { + const knowledge = readFileSync(KNOWLEDGE, 'utf8') + + const refresh = methodBody(knowledge, 'async refreshDocumentFromFile(') + assert.match(refresh, /refresh: true/) + + const ingest = methodBody(knowledge, 'private async ingestFile(') + assert.match(ingest, /isNewSnapshot/, 'ingestFile must allocate a fresh snapshot on refresh') + assert.match( + ingest, + /planSnapshotWrite\(/, + 'the snapshot write plan must come from the shared rule, so a shared/parsed snapshot is never mutated' + ) + assert.match(ingest, /countMemberships\(/) + // 解析完成之后(parseFile)、索引之前(indexDocument)落 snapshot。 + const insertAt = ingest.indexOf('insertLibrarySnapshot(') + const indexAt = ingest.lastIndexOf('await this.indexDocument(') + assert.ok(insertAt > ingest.indexOf('parseFile('), 'the snapshot must be created after parsing') + assert.ok(insertAt < indexAt, 'the snapshot must exist before a successful index can be reused') + + const copy = methodBody(knowledge, 'private async copyFileToKnowledgeDir(') + assert.match( + copy, + /\$\{ownerId\}\.\$\{extension\}/, + 'the local file is named by the snapshot id, so a refresh cannot overwrite a shared file' + ) +}) diff --git a/test/librarySourceReuse.test.ts b/test/librarySourceReuse.test.ts new file mode 100644 index 0000000..5842111 --- /dev/null +++ b/test/librarySourceReuse.test.ts @@ -0,0 +1,867 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import Database from 'better-sqlite3' +import * as sqliteVec from 'sqlite-vec' +import { + createVectorTableSql, + knnQuerySql, + upsertVectorsSql +} from '../src/main/vectorstore/vectorTableSql.ts' +import { + attachLibrarySource, + countMemberships, + deleteLibrarySource, + insertLibrarySource, + listLibrarySources, + planSnapshotWrite, + updateLibrarySource, + type EmbeddingSpaceIdentity, + type VectorTableAccess +} from '../src/main/services/librarySources.ts' + +/** + * 库来源复用边界(#99)的行为,跑在真实 SQLite + sqlite-vec 上。 + * + * 这一层不 import Electron:`librarySources.ts` 只拿一个 raw better-sqlite3 连接,所以 + * 可以真的建 vec0 表、真的复制向量,而不是只断言 SQL 字符串。这里盯住 ADR 的硬要求: + * 挂载不调用 embedding、只在 space/维度匹配时复制向量、page/offset 原样保留、donor 与 + * 目标互不影响、未确认或仍被挂载的库来源不能删。 + */ + +const DIMENSIONS = 4 +const SPACE_ID = 'space_test' +const NOW = new Date('2026-01-01T00:00:00Z') +const NOW_SECONDS = Math.floor(NOW.getTime() / 1000) + +function createSchema(db: Database.Database): void { + db.exec(` + CREATE TABLE notebooks (id TEXT PRIMARY KEY, title TEXT, description TEXT, created_at INTEGER, updated_at INTEGER); + CREATE TABLE library_sources ( + id TEXT PRIMARY KEY, + title TEXT NOT NULL, + type TEXT NOT NULL, + source_uri TEXT, + local_file_path TEXT, + content TEXT, + structure TEXT, + content_hash TEXT, + mime_type TEXT, + file_size INTEGER, + metadata TEXT, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL + ); + CREATE TABLE documents ( + id TEXT PRIMARY KEY, + notebook_id TEXT NOT NULL, + title TEXT NOT NULL, + type TEXT NOT NULL, + source_uri TEXT, + local_file_path TEXT, + source_note_id TEXT, + source_id TEXT, + content TEXT, + structure TEXT, + content_hash TEXT, + mime_type TEXT, + file_size INTEGER, + metadata TEXT, + status TEXT NOT NULL, + source_state TEXT NOT NULL, + source_mtime_ms INTEGER, + error_message TEXT, + chunk_count INTEGER, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL + ); + CREATE TABLE notebook_embedding_spaces ( + notebook_id TEXT PRIMARY KEY, + space_id TEXT NOT NULL, + backend TEXT NOT NULL, + model TEXT NOT NULL, + revision TEXT NOT NULL DEFAULT '', + dimensions INTEGER NOT NULL, + updated_at INTEGER NOT NULL + ); + CREATE TABLE chunks ( + id TEXT PRIMARY KEY, + document_id TEXT NOT NULL, + notebook_id TEXT NOT NULL, + content TEXT NOT NULL, + chunk_index INTEGER NOT NULL, + start_offset INTEGER, + end_offset INTEGER, + page_start INTEGER, + page_end INTEGER, + metadata TEXT, + token_count INTEGER, + created_at INTEGER NOT NULL + ); + CREATE TABLE document_blocks ( + id TEXT PRIMARY KEY, + document_id TEXT NOT NULL, + kind TEXT NOT NULL, + "order" INTEGER NOT NULL, + page INTEGER, + level INTEGER, + text TEXT NOT NULL, + start_offset INTEGER NOT NULL, + end_offset INTEGER NOT NULL, + bbox TEXT, + metadata TEXT + ); + CREATE TABLE chunk_blocks ( + chunk_id TEXT NOT NULL, + block_id TEXT NOT NULL, + start_in_block INTEGER NOT NULL, + end_in_block INTEGER NOT NULL, + PRIMARY KEY (chunk_id, block_id) + ); + CREATE TABLE embeddings ( + id TEXT PRIMARY KEY, + chunk_id TEXT NOT NULL, + notebook_id TEXT NOT NULL, + model TEXT NOT NULL, + dimensions INTEGER NOT NULL, + created_at INTEGER NOT NULL + ); + CREATE TABLE vec_metadata ( + notebook_id TEXT PRIMARY KEY, + table_name TEXT NOT NULL, + dimensions INTEGER NOT NULL + ); + `) +} + +interface Fixture { + db: Database.Database + tables: VectorTableAccess +} + +function fixture(): Fixture { + const db = new Database(':memory:') + sqliteVec.load(db) + createSchema(db) + + const tables: VectorTableAccess = { + read: (notebookId) => { + const row = db + .prepare('SELECT table_name, dimensions FROM vec_metadata WHERE notebook_id = ?') + .get(notebookId) as { table_name: string; dimensions: number } | undefined + return row ? { tableName: row.table_name, dimensions: row.dimensions } : undefined + }, + ensure: (notebookId, dimensions) => { + const tableName = `vec_${notebookId}` + db.exec(createVectorTableSql(tableName, dimensions)) + db.prepare( + 'INSERT OR REPLACE INTO vec_metadata (notebook_id, table_name, dimensions) VALUES (?, ?, ?)' + ).run(notebookId, tableName, dimensions) + return { tableName, dimensions } + } + } + + return { db, tables } +} + +function addNotebook(db: Database.Database, id: string): void { + db.prepare('INSERT INTO notebooks VALUES (?, ?, ?, ?, ?)').run(id, id, null, 0, 0) +} + +function addSnapshot( + db: Database.Database, + id: string, + overrides: Record = {} +): void { + insertLibrarySource( + db, + id, + { + title: 'Guide', + type: 'file', + sourceUri: '/src/guide.pdf', + localFilePath: `/files/${id}.pdf`, + content: 'canonical text', + structure: JSON.stringify({ sections: [] }), + contentHash: 'hash', + mimeType: 'application/pdf', + fileSize: 123, + metadata: null, + ...overrides + } as Parameters[2], + NOW + ) +} + +/** 一个已索引的 membership:2 块、2 chunk、2 向量,页码分别是 3 和 4。 */ +function seedIndexedMembership( + db: Database.Database, + params: { + documentId: string + notebookId: string + sourceId: string + spaceId?: string + dimensions?: number + } +): void { + const { documentId, notebookId, sourceId } = params + const spaceId = params.spaceId ?? SPACE_ID + const dimensions = params.dimensions ?? DIMENSIONS + + db.prepare( + `INSERT INTO documents + (id, notebook_id, title, type, source_uri, local_file_path, source_id, content, + structure, content_hash, mime_type, file_size, metadata, status, source_state, + chunk_count, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + ).run( + documentId, + notebookId, + 'Guide', + 'file', + '/src/guide.pdf', + `/files/${sourceId}.pdf`, + sourceId, + 'canonical text', + JSON.stringify({ sections: [] }), + 'hash', + 'application/pdf', + 123, + null, + 'indexed', + 'available', + 2, + NOW_SECONDS, + NOW_SECONDS + ) + + db.prepare( + `INSERT OR REPLACE INTO notebook_embedding_spaces + (notebook_id, space_id, backend, model, revision, dimensions, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?)` + ).run(notebookId, spaceId, 'local', 'model-x', 'rev', dimensions, NOW_SECONDS) + + const block = db.prepare('INSERT INTO document_blocks VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)') + block.run( + `${documentId}_b1`, + documentId, + 'paragraph', + 0, + 3, + null, + 'first paragraph', + 0, + 15, + null, + null + ) + block.run( + `${documentId}_b2`, + documentId, + 'paragraph', + 1, + 4, + null, + 'second paragraph', + 15, + 32, + null, + null + ) + + const chunk = db.prepare('INSERT INTO chunks VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)') + chunk.run( + `${documentId}_c1`, + documentId, + notebookId, + 'first paragraph', + 0, + 0, + 15, + 3, + 3, + null, + 3, + NOW_SECONDS + ) + chunk.run( + `${documentId}_c2`, + documentId, + notebookId, + 'second paragraph', + 1, + 15, + 32, + 4, + 4, + null, + 3, + NOW_SECONDS + ) + + const mapping = db.prepare('INSERT INTO chunk_blocks VALUES (?, ?, ?, ?)') + mapping.run(`${documentId}_c1`, `${documentId}_b1`, 0, 15) + mapping.run(`${documentId}_c2`, `${documentId}_b2`, 0, 16) + + const tableName = `vec_${notebookId}` + db.exec(createVectorTableSql(tableName, dimensions)) + db.prepare( + 'INSERT OR REPLACE INTO vec_metadata (notebook_id, table_name, dimensions) VALUES (?, ?, ?)' + ).run(notebookId, tableName, dimensions) + + db.prepare('INSERT INTO embeddings VALUES (?, ?, ?, ?, ?, ?)').run( + `${documentId}_e1`, + `${documentId}_c1`, + notebookId, + 'model-x', + dimensions, + NOW_SECONDS + ) + db.prepare('INSERT INTO embeddings VALUES (?, ?, ?, ?, ?, ?)').run( + `${documentId}_e2`, + `${documentId}_c2`, + notebookId, + 'model-x', + dimensions, + NOW_SECONDS + ) + db.prepare(upsertVectorsSql(tableName)).run( + `${documentId}_e1`, + `${documentId}_c1`, + new Float32Array([1, 0, 0, 0]) + ) + db.prepare(upsertVectorsSql(tableName)).run( + `${documentId}_e2`, + `${documentId}_c2`, + new Float32Array([0, 1, 0, 0]) + ) +} + +const count = (db: Database.Database, sql: string, ...params: unknown[]): number => + (db.prepare(sql).get(...params) as { c: number }).c + +test('reuse copies the indexed donor into the target notebook without touching the donor', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + + const result = attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + assert.equal(result.indexed, true) + assert.notEqual(result.documentId, 'docA') + assert.equal(result.chunkCount, 2) + + const membership = db + .prepare('SELECT * FROM documents WHERE id = ?') + .get(result.documentId) as Record + assert.equal(membership.source_id, 'lib_1') + assert.equal(membership.notebook_id, 'nbB') + assert.equal(membership.status, 'indexed') + assert.equal(membership.chunk_count, 2) + assert.equal(membership.content, 'canonical text') + + // page/span/offset 原样保留,只是换了 membership-local 的 id。 + const donorChunks = db + .prepare('SELECT * FROM chunks WHERE document_id = ? ORDER BY chunk_index') + .all('docA') as Array> + const targetChunks = db + .prepare('SELECT * FROM chunks WHERE document_id = ? ORDER BY chunk_index') + .all(result.documentId) as Array> + assert.equal(targetChunks.length, 2) + for (let i = 0; i < donorChunks.length; i++) { + assert.notEqual(targetChunks[i].id, donorChunks[i].id) + assert.equal(targetChunks[i].chunk_index, donorChunks[i].chunk_index) + assert.equal(targetChunks[i].page_start, donorChunks[i].page_start) + assert.equal(targetChunks[i].page_end, donorChunks[i].page_end) + assert.equal(targetChunks[i].start_offset, donorChunks[i].start_offset) + assert.equal(targetChunks[i].end_offset, donorChunks[i].end_offset) + assert.equal(targetChunks[i].notebook_id, 'nbB') + } + + // 原 derive 的 chunk→block 页面映射也复制过来了。 + const provenance = db + .prepare( + `SELECT b.page, cb.start_in_block, cb.end_in_block + FROM chunk_blocks cb + JOIN document_blocks b ON b.id = cb.block_id + JOIN chunks c ON c.id = cb.chunk_id + WHERE c.document_id = ? + ORDER BY c.chunk_index` + ) + .all(result.documentId) as Array<{ page: number; start_in_block: number; end_in_block: number }> + assert.deepEqual( + provenance.map((row) => [row.page, row.start_in_block, row.end_in_block]), + [ + [3, 0, 15], + [4, 0, 16] + ] + ) + + // 目标向量真的可查,且与 donor 的向量一致。 + const targetTable = tables.read('nbB')!.tableName + const hits = db + .prepare(knnQuerySql(targetTable)) + .all(new Float32Array([1, 0, 0, 0]), 5) as Array<{ + chunk_id: string + distance: number + }> + assert.equal(hits.length, 2) + assert.equal(hits[0].distance, 0) + + // donor 一行未动。 + assert.equal(count(db, 'SELECT COUNT(*) c FROM chunks WHERE document_id = ?', 'docA'), 2) + const donorTable = tables.read('nbA')!.tableName + assert.equal(count(db, `SELECT COUNT(*) c FROM ${donorTable}`), 2) +}) + +test('an empty target adopts the donor embedding space', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + + attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + const adopted = db + .prepare('SELECT * FROM notebook_embedding_spaces WHERE notebook_id = ?') + .get('nbB') as { space_id: string; dimensions: number } + assert.equal(adopted.space_id, SPACE_ID) + assert.equal(adopted.dimensions, DIMENSIONS) + assert.equal(tables.read('nbB')?.dimensions, DIMENSIONS) +}) + +test('a mismatched current space leaves a pending membership and touches no vectors', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + + const result = attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: 'space_other', dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + assert.equal(result.indexed, false) + const membership = db.prepare('SELECT * FROM documents WHERE id = ?').get(result.documentId) as { + status: string + chunk_count: number + } + assert.equal(membership.status, 'pending') + assert.equal(membership.chunk_count, 0) + assert.equal( + count(db, 'SELECT COUNT(*) c FROM chunks WHERE document_id = ?', result.documentId), + 0 + ) + assert.equal(tables.read('nbB'), undefined) + + // donor 的向量没被清掉。 + const donorTable = tables.read('nbA')!.tableName + assert.equal(count(db, `SELECT COUNT(*) c FROM ${donorTable}`), 2) +}) + +test('a target with an incompatible space does not clear its existing vectors', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + + // nbB 已经用另一个 space 索引过:宽度相同但 space 身份不同,不能混用。 + seedIndexedMembership(db, { + documentId: 'docB', + notebookId: 'nbB', + sourceId: 'lib_other', + spaceId: 'space_other' + }) + + const result = attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + assert.equal(result.indexed, false) + const bTable = tables.read('nbB')!.tableName + assert.equal( + count(db, `SELECT COUNT(*) c FROM ${bTable}`), + 2, + 'existing target vectors were cleared' + ) +}) + +test('attaching the same source twice is idempotent', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + + const first = attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + const second = attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + assert.equal(second.documentId, first.documentId) + assert.equal(second.indexed, true) + assert.equal( + count( + db, + 'SELECT COUNT(*) c FROM documents WHERE notebook_id = ? AND source_id = ?', + 'nbB', + 'lib_1' + ), + 1 + ) +}) + +test('attach validates that the notebook and the source exist', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addSnapshot(db, 'lib_1') + + assert.throws( + () => + attachLibrarySource(db, { + notebookId: 'missing', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }), + /Notebook missing not found/ + ) + assert.throws( + () => + attachLibrarySource(db, { + notebookId: 'nbA', + sourceId: 'missing', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }), + /Library source missing not found/ + ) +}) + +test('listLibrarySources hides attached snapshots and reports reuse readiness', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + + // nbB 还没挂载 lib_1:可见、membershipCount 1、可复用。 + const forB = listLibrarySources(db, 'nbB', { id: SPACE_ID, dimensions: DIMENSIONS }, tables) + assert.deepEqual( + forB.map((s) => s.id), + ['lib_1'] + ) + assert.equal(forB[0].membershipCount, 1) + assert.equal(forB[0].canReuseIndex, true) + assert.equal(forB[0].type, 'file') + assert.equal(forB[0].mimeType, 'application/pdf') + + // 已挂载的 notebook 看不到它。 + const forA = listLibrarySources(db, 'nbA', { id: SPACE_ID, dimensions: DIMENSIONS }, tables) + assert.deepEqual(forA, []) + + // 当前模型不同 -> 不可直接复用。 + const mismatched = listLibrarySources( + db, + 'nbB', + { id: 'space_other', dimensions: DIMENSIONS }, + tables + ) + assert.equal(mismatched[0].canReuseIndex, false) +}) + +test('deleteLibrarySource refuses unconfirmed or attached sources and reports the file', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + assert.throws(() => deleteLibrarySource(db, 'lib_1', false), /confirmation/) + assert.throws(() => deleteLibrarySource(db, 'lib_1', true), /still attached/) + + // 解除两个 membership 后,永久删除才被允许,并交出待 unlink 的文件路径。 + db.prepare('DELETE FROM documents WHERE source_id = ?').run('lib_1') + const result = deleteLibrarySource(db, 'lib_1', true) + assert.deepEqual(result, { localFilePath: '/files/lib_1.pdf', deleted: true }) + assert.equal(count(db, 'SELECT COUNT(*) c FROM library_sources WHERE id = ?', 'lib_1'), 0) + + // 已经不存在时是幂等的 no-op。 + assert.deepEqual(deleteLibrarySource(db, 'lib_1', true), { localFilePath: null, deleted: false }) +}) + +test('a snapshot with no parsed content is hidden and cannot be reused', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + // 迁移回填一个解析前就失败的来源:content 为 NULL。 + insertLibrarySource( + db, + 'lib_empty', + { + title: 'empty', + type: 'file', + sourceUri: '/src/empty.pdf', + localFilePath: '/files/lib_empty.pdf', + content: null, + structure: null, + contentHash: null, + mimeType: null, + fileSize: null, + metadata: null + }, + NOW + ) + + assert.deepEqual( + listLibrarySources(db, 'nbA', { id: SPACE_ID, dimensions: DIMENSIONS }, tables), + [] + ) + assert.throws( + () => + attachLibrarySource(db, { + notebookId: 'nbA', + sourceId: 'lib_empty', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }), + /no parsed content/ + ) + + // 重试成功后原地补齐,这份来源才重新出现在库里。 + updateLibrarySource( + db, + 'lib_empty', + { + title: 'empty', + type: 'file', + sourceUri: '/src/empty.pdf', + localFilePath: '/files/lib_empty.pdf', + content: 'now parsed', + structure: null, + contentHash: 'h', + mimeType: 'application/pdf', + fileSize: 1, + metadata: null + }, + NOW + ) + const listed = listLibrarySources(db, 'nbA', { id: SPACE_ID, dimensions: DIMENSIONS }, tables) + assert.deepEqual( + listed.map((s) => s.id), + ['lib_empty'] + ) + assert.equal(listed[0].canReuseIndex, false) +}) + +test('listLibrarySources and attachLibrarySource make the same reuse decision', () => { + // UI 从 canReuseIndex 承诺的复用,attach 必须真的能兑现 —— 两边共用同一个判定。 + const check = ( + setup: (db: Database.Database) => { + notebookId: string + sourceId: string + space: EmbeddingSpaceIdentity + } + ): void => { + const { db, tables } = fixture() + const { notebookId, sourceId, space } = setup(db) + + const listed = listLibrarySources(db, notebookId, space, tables).find((s) => s.id === sourceId) + assert.ok(listed, `source ${sourceId} should be listed for ${notebookId}`) + + const attached = attachLibrarySource(db, { + notebookId, + sourceId, + currentSpace: space, + tables, + now: NOW + }) + assert.equal( + attached.indexed, + listed.canReuseIndex, + `list canReuseIndex=${listed.canReuseIndex} but attach indexed=${attached.indexed}` + ) + } + + check((db) => { + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + return { notebookId: 'nbB', sourceId: 'lib_1', space: { id: SPACE_ID, dimensions: DIMENSIONS } } + }) + + check((db) => { + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + // 当前配置的 space 不同:两边都必须是 false。 + return { + notebookId: 'nbB', + sourceId: 'lib_1', + space: { id: 'space_other', dimensions: DIMENSIONS } + } + }) + + check((db) => { + addNotebook(db, 'nbA') + addSnapshot(db, 'lib_1') + // 没有 donor:两边都必须是 false。 + return { notebookId: 'nbA', sourceId: 'lib_1', space: { id: SPACE_ID, dimensions: DIMENSIONS } } + }) + + check((db) => { + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + // 目标已用不兼容的 space 索引:两边都必须是 false。 + seedIndexedMembership(db, { + documentId: 'docB', + notebookId: 'nbB', + sourceId: 'lib_other', + spaceId: 'space_b' + }) + return { notebookId: 'nbB', sourceId: 'lib_1', space: { id: SPACE_ID, dimensions: DIMENSIONS } } + }) +}) + +test('planSnapshotWrite never mutates a shared or parsed snapshot on retry', () => { + // 共享(>1 membership)永远新开,绝不为一个 membership 改写别人的快照。 + assert.equal( + planSnapshotWrite({ refresh: false, hasSourceId: true, hasContent: true, membershipCount: 2 }), + 'new' + ) + // 刷新永远新开。 + assert.equal( + planSnapshotWrite({ refresh: true, hasSourceId: true, hasContent: true, membershipCount: 1 }), + 'new' + ) + // 首次导入新开。 + assert.equal( + planSnapshotWrite({ + refresh: false, + hasSourceId: false, + hasContent: false, + membershipCount: 0 + }), + 'new' + ) + // 独占的已解析快照保持不动。 + assert.equal( + planSnapshotWrite({ refresh: false, hasSourceId: true, hasContent: true, membershipCount: 1 }), + 'keep' + ) + // 独占但从未解析成功(迁移回填的空快照)才允许原地补齐。 + assert.equal( + planSnapshotWrite({ refresh: false, hasSourceId: true, hasContent: false, membershipCount: 1 }), + 'fill' + ) +}) + +test('countMemberships counts every notebook that mounted the snapshot', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_1') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_1' }) + assert.equal(countMemberships(db, 'lib_1'), 1) + + attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_1', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + assert.equal(countMemberships(db, 'lib_1'), 2) + assert.equal(countMemberships(db, 'lib_unknown'), 0) +}) + +test('refresh copy-on-write leaves the shared snapshot and its peers untouched', () => { + const { db, tables } = fixture() + addNotebook(db, 'nbA') + addNotebook(db, 'nbB') + addSnapshot(db, 'lib_shared') + seedIndexedMembership(db, { documentId: 'docA', notebookId: 'nbA', sourceId: 'lib_shared' }) + // B 也挂载同一 snapshot,所以刷新 A 必须新开 snapshot,不能覆盖共享文件。 + const forB = attachLibrarySource(db, { + notebookId: 'nbB', + sourceId: 'lib_shared', + currentSpace: { id: SPACE_ID, dimensions: DIMENSIONS }, + tables, + now: NOW + }) + + // 刷新 A:新 snapshot 一行 + 把 A 的 membership 指过去(ingestFile 的 copy-on-write 结果)。 + addSnapshot(db, 'lib_refreshed', { + content: 'refreshed text', + localFilePath: '/files/lib_refreshed.pdf' + }) + db.prepare('UPDATE documents SET source_id = ?, content = ? WHERE id = ?').run( + 'lib_refreshed', + 'refreshed text', + 'docA' + ) + + const peer = db + .prepare('SELECT source_id, content FROM documents WHERE id = ?') + .get(forB.documentId) as { source_id: string; content: string } + assert.equal(peer.source_id, 'lib_shared', 'peers must stay on the old snapshot') + assert.equal(peer.content, 'canonical text') + + const original = db + .prepare('SELECT content, local_file_path FROM library_sources WHERE id = ?') + .get('lib_shared') as { content: string; local_file_path: string } + assert.equal(original.content, 'canonical text') + assert.equal(original.local_file_path, '/files/lib_shared.pdf') + + const refreshed = db + .prepare('SELECT content, local_file_path FROM library_sources WHERE id = ?') + .get('lib_refreshed') as { content: string; local_file_path: string } + assert.equal(refreshed.content, 'refreshed text') + assert.notEqual(refreshed.local_file_path, original.local_file_path) +}) From 4063f92736586bf0d2c960f4bb112c1f9a3ced56 Mon Sep 17 00:00:00 2001 From: mrsibe Date: Fri, 2 Oct 2026 21:59:50 +0800 Subject: [PATCH 3/5] feat: add library source reuse and removal controls --- src/main/ipc/knowledgeHandlers.ts | 33 +++ src/main/ipc/validation.ts | 17 +- src/preload/index.d.ts | 12 +- src/preload/index.ts | 6 + .../src/components/notebook/SourcePanel.tsx | 39 +++- .../notebook/source/DocumentList.tsx | 9 +- .../notebook/source/LibrarySourceDialog.tsx | 195 ++++++++++++++++++ src/renderer/src/locales/en-US/ui.json | 16 +- src/renderer/src/locales/zh-CN/ui.json | 17 +- test/librarySourceUi.test.ts | 58 ++++++ 10 files changed, 390 insertions(+), 12 deletions(-) create mode 100644 src/renderer/src/components/notebook/source/LibrarySourceDialog.tsx create mode 100644 test/librarySourceUi.test.ts diff --git a/src/main/ipc/knowledgeHandlers.ts b/src/main/ipc/knowledgeHandlers.ts index 088a171..5c70ada 100644 --- a/src/main/ipc/knowledgeHandlers.ts +++ b/src/main/ipc/knowledgeHandlers.ts @@ -208,6 +208,39 @@ export function registerKnowledgeHandlers(knowledgeService: KnowledgeService) { }) ) + ipcMain.handle( + 'knowledge:list-library-sources', + validate(KnowledgeSchemas.listLibrarySources, async ({ notebookId }) => { + return knowledgeService.listLibrarySources(notebookId) + }) + ) + + ipcMain.handle( + 'knowledge:attach-library-source', + validate(KnowledgeSchemas.attachLibrarySource, async ({ notebookId, sourceId }) => { + try { + return { + success: true, + ...(await knowledgeService.attachLibrarySource(notebookId, sourceId)) + } + } catch (error) { + return { success: false, error: (error as Error).message } + } + }) + ) + + ipcMain.handle( + 'knowledge:delete-library-source', + validate(KnowledgeSchemas.deleteLibrarySource, async ({ sourceId, confirmed }) => { + try { + await knowledgeService.deleteLibrarySource(sourceId, confirmed) + return { success: true } + } catch (error) { + return { success: false, error: (error as Error).message } + } + }) + ) + // 获取单个文档 ipcMain.handle( 'knowledge:get-document', diff --git a/src/main/ipc/validation.ts b/src/main/ipc/validation.ts index cf959fa..9012693 100644 --- a/src/main/ipc/validation.ts +++ b/src/main/ipc/validation.ts @@ -5,7 +5,8 @@ import { z } from 'zod' import { API_PROTOCOLS, MODEL_CAPABILITIES, REASONING_EFFORTS } from '../../shared/types' -import { Result, Err, Ok } from '../../shared/types/result' +import { Err, Ok } from '../../shared/types/result' +import type { Result } from '../../shared/types/result' import Logger from '../../shared/utils/logger' /** @@ -187,6 +188,20 @@ export const KnowledgeSchemas = { notebookId: z.string().min(1, '笔记本 ID 不能为空') }), + listLibrarySources: z.object({ + notebookId: z.string().min(1) + }), + + attachLibrarySource: z.object({ + notebookId: z.string().min(1), + sourceId: z.string().min(1) + }), + + deleteLibrarySource: z.object({ + sourceId: z.string().min(1), + confirmed: z.literal(true) + }), + getDocument: z.object({ documentId: z.string().min(1, '文档 ID 不能为空') }), diff --git a/src/preload/index.d.ts b/src/preload/index.d.ts index 0ce1852..0435b5e 100644 --- a/src/preload/index.d.ts +++ b/src/preload/index.d.ts @@ -28,7 +28,8 @@ import type { AddDocumentOptions, SearchOptions, IndexProgress, - BatchImportOutcome + BatchImportOutcome, + LibrarySourceSummary } from '../shared/types/knowledge' import type { UpdateState, UpdateCheckResult, UpdateOperationResult } from '../shared/types/update' import type { MindMap, Quiz, QuizSession, AnkiCard } from '../main/db/schema' @@ -319,6 +320,15 @@ declare global { // 文档管理 getDocuments: (notebookId: string) => Promise + listLibrarySources: (notebookId: string) => Promise + attachLibrarySource: ( + notebookId: string, + sourceId: string + ) => Promise<{ success: boolean; documentId?: string; indexed?: boolean; error?: string }> + deleteLibrarySource: ( + sourceId: string, + confirmed: boolean + ) => Promise<{ success: boolean; error?: string }> getDocument: (documentId: string) => Promise getDocumentChunks: (documentId: string) => Promise getDocumentBlocks: (documentId: string) => Promise diff --git a/src/preload/index.ts b/src/preload/index.ts index 0273c11..56c5f37 100644 --- a/src/preload/index.ts +++ b/src/preload/index.ts @@ -211,6 +211,12 @@ const api = { // 文档管理 getDocuments: (notebookId: string) => ipcRenderer.invoke('knowledge:get-documents', { notebookId }), + listLibrarySources: (notebookId: string) => + ipcRenderer.invoke('knowledge:list-library-sources', { notebookId }), + attachLibrarySource: (notebookId: string, sourceId: string) => + ipcRenderer.invoke('knowledge:attach-library-source', { notebookId, sourceId }), + deleteLibrarySource: (sourceId: string, confirmed: boolean) => + ipcRenderer.invoke('knowledge:delete-library-source', { sourceId, confirmed }), getDocument: (documentId: string) => ipcRenderer.invoke('knowledge:get-document', { documentId }), getDocumentChunks: (documentId: string) => diff --git a/src/renderer/src/components/notebook/SourcePanel.tsx b/src/renderer/src/components/notebook/SourcePanel.tsx index 15a6c2e..b6bcfda 100644 --- a/src/renderer/src/components/notebook/SourcePanel.tsx +++ b/src/renderer/src/components/notebook/SourcePanel.tsx @@ -12,7 +12,8 @@ import { Quote, ExternalLink, FolderOpen, - FolderSync + FolderSync, + Library } from 'lucide-react' import { toast } from 'sonner' import { useKnowledgeStore, setupKnowledgeListeners } from '../../store/knowledgeStore' @@ -34,6 +35,7 @@ import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from '. import { Card } from '../ui/card' import { PanelHeader } from '../ui/panel-header' import DocumentList from './source/DocumentList' +import LibrarySourceDialog from './source/LibrarySourceDialog' import SourceReader from './source/reader/SourceReader' import type { KnowledgeDocument, BatchImportOutcome } from '../../../../shared/types/knowledge' import type { ReaderAnchor, ReaderSelection } from '../../../../shared/types/source' @@ -331,6 +333,7 @@ export default function SourcePanel(): ReactElement { const { t } = useTranslation('ui') const { id: notebookId } = useParams() const [showAddMenu, setShowAddMenu] = useState(false) + const [showLibraryDialog, setShowLibraryDialog] = useState(false) const [isDragging, setIsDragging] = useState(false) const [modalType, setModalType] = useState(null) const [hasEmbeddingModel, setHasEmbeddingModel] = useState(false) @@ -812,11 +815,10 @@ export default function SourcePanel(): ReactElement { > @@ -825,6 +827,18 @@ export default function SourcePanel(): ReactElement { {showAddMenu && (
+
)} + {showLibraryDialog && notebookId && ( + setShowLibraryDialog(false)} + onAdded={async () => { + if (useKnowledgeStore.getState().activeNotebookId === notebookId) { + await loadDocuments(notebookId) + await loadStats(notebookId) + } + }} + /> + )} + {/* 添加来源弹窗 - 使用 key 强制在 type 变化时重新挂载组件 */} {modalType && modalType !== 'file' && (