diff --git a/README.md b/README.md index f2f6c6b..a3c2218 100644 --- a/README.md +++ b/README.md @@ -54,6 +54,37 @@ AnythingLLM. sources, and turn a notebook into a mind map. - **No deployment.** Download, open, start reading. The embedder downloads once and then runs on your machine. +- **An MCP export surface.** Serve a notebook to a local agent over read-only + stdio [MCP](https://modelcontextprotocol.io) — the same retrieval path the app + uses, as a server, not a second client (see below). + +## Use it from an agent (MCP) + +KnowNote can serve your library over MCP on stdio, read-only, so an agent — +Claude Code, Codex, Cursor, Claude Desktop — can search the documents you already +have instead of asking you to paste them in. + +```jsonc +// your MCP client's config +{ + "mcpServers": { + "knownote": { + "command": "/path/to/knownote", + "args": ["--mcp"] + } + } +} +``` + +The tools are `list_notebooks`, `search_notebook`, `get_source`, `read_document` +and `search_notes`. `search_notebook` returns passages **with provenance** +(document id, page, character offsets) rather than bare text, so a claim an agent +makes can be checked against the source. There is no write tool, no tool that +calls a model, and no result that reports a filesystem path — that is what makes +this surface safe to grant to an agent reading untrusted documents. + +By default the server uses the same library as the desktop app. Point it at a +different profile with `KNOWNOTE_DATA_DIR=/path/to/profile`. ## What is not here yet diff --git a/README_CN.md b/README_CN.md index 10c617b..a92ed9f 100644 --- a/README_CN.md +++ b/README_CN.md @@ -39,6 +39,31 @@ KnowNote 把你自己的文档变成可以提问的知识库,并让每个回 - **自带模型。** 任何兼容 OpenAI、Anthropic 或 Google 协议的端点,或 Ollama 这类本地服务。没有内置对话模型,也没有账号。 - **笔记与思维导图。** 把你的结论作为结构化笔记保存在资料旁边,也可以把一本笔记本转成思维导图。 - **无需部署。** 下载、打开、开始阅读。嵌入模型只下载一次,之后都在你的机器上运行。 +- **MCP 导出面。** 通过只读的 stdio [MCP](https://modelcontextprotocol.io) 把笔记本提供给本地 agent —— 用的是 app 自己的检索路径,KnowNote 作服务端,不是再做一个客户端(见下)。 + +## 从 agent 调用(MCP) + +KnowNote 可以通过 stdio 以只读方式把知识库作为 MCP 服务提供给 agent —— +Claude Code、Codex、Cursor、Claude Desktop —— 让它直接检索你已有的文档,而不是让你把文档粘进去。 + +```jsonc +// 你的 MCP 客户端配置 +{ + "mcpServers": { + "knownote": { + "command": "/path/to/knownote", + "args": ["--mcp"] + } + } +} +``` + +工具是 `list_notebooks`、`search_notebook`、`get_source`、`read_document`、`search_notes`。 +`search_notebook` 返回带 **provenance**(文档 id、页码、字符偏移)的片段,而不是裸文本, +所以 agent 给出的结论可以回到原文核对。没有写入工具、没有调用模型的工具、也没有任何 +返回文件路径的结果 —— 正是这一点让这个面可以安全地授予一个正在读不可信文档的 agent。 + +默认使用与桌面应用相同的知识库;用 `KNOWNOTE_DATA_DIR=/path/to/profile` 可以指向别的 profile。 ## 还没有的部分 diff --git a/package-lock.json b/package-lock.json index f667709..e911b70 100644 --- a/package-lock.json +++ b/package-lock.json @@ -12,6 +12,7 @@ "dependencies": { "@huggingface/transformers": "^4.3.0", "@mixmark-io/domino": "^2.2.0", + "@modelcontextprotocol/server": "^2.1.0", "anki-apkg-export": "^4.0.0", "better-sqlite3": "^12.6.2", "jsdom": "^27.4.0", @@ -35,6 +36,7 @@ "@electron-toolkit/preload": "^3.0.2", "@electron-toolkit/tsconfig": "^2.0.0", "@electron-toolkit/utils": "^4.0.0", + "@modelcontextprotocol/client": "^2.1.0", "@mozilla/readability": "^0.6.0", "@radix-ui/react-alert-dialog": "^1.1.15", "@radix-ui/react-checkbox": "^1.3.3", @@ -3303,6 +3305,50 @@ "integrity": "sha512-Y28PR25bHXUg88kCV7nivXrP2Nj2RueZ3/l/jdx6J9f8J4nsEGcgX0Qe6lt7Pa+J79+kPiJU3LguR6O/6zrLOw==", "license": "BSD-2-Clause" }, + "node_modules/@modelcontextprotocol/client": { + "version": "2.1.0", + "resolved": "https://registry.npmmirror.com/@modelcontextprotocol/client/-/client-2.1.0.tgz", + "integrity": "sha512-mDVhoy5WjDb0U+4dQPzLcciC2erSex/GRVQqnoZdiuMoE3YiTZdiS7ezNcFwX4XEOWOaj7jZr0kLYnILnL8orA==", + "dev": true, + "license": "MIT", + "dependencies": { + "@modelcontextprotocol/core": "2.1.0", + "cross-spawn": "^7.0.5", + "eventsource": "^3.0.2", + "eventsource-parser": "^3.0.0", + "jose": "^6.1.3", + "pkce-challenge": "^5.0.0", + "zod": "^4.2.0" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/@modelcontextprotocol/core": { + "version": "2.1.0", + "resolved": "https://registry.npmmirror.com/@modelcontextprotocol/core/-/core-2.1.0.tgz", + "integrity": "sha512-YdFj5gMRHr0gYNAhTbQDgjLHiEJdIAUYxomhKseUNye7ZGlISQmUoLh2sveFFCiQ7j2YMffT1kgSND0FSfIFjQ==", + "license": "MIT", + "dependencies": { + "zod": "^4.2.0" + }, + "engines": { + "node": ">=20" + } + }, + "node_modules/@modelcontextprotocol/server": { + "version": "2.1.0", + "resolved": "https://registry.npmmirror.com/@modelcontextprotocol/server/-/server-2.1.0.tgz", + "integrity": "sha512-FdvwZU5N03O77Suh35JqjAqq5R1ZTEPFtwu6N4BgBq28qrQtrCnpbLL5U2pwYCuE194jAw0xnlOVmIqhfoYuUQ==", + "license": "MIT", + "dependencies": { + "@modelcontextprotocol/core": "2.1.0", + "zod": "^4.2.0" + }, + "engines": { + "node": ">=20" + } + }, "node_modules/@mozilla/readability": { "version": "0.6.0", "resolved": "https://registry.npmmirror.com/@mozilla/readability/-/readability-0.6.0.tgz", @@ -10074,6 +10120,19 @@ "node": ">=0.8.x" } }, + "node_modules/eventsource": { + "version": "3.0.7", + "resolved": "https://registry.npmmirror.com/eventsource/-/eventsource-3.0.7.tgz", + "integrity": "sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA==", + "dev": true, + "license": "MIT", + "dependencies": { + "eventsource-parser": "^3.0.1" + }, + "engines": { + "node": ">=18.0.0" + } + }, "node_modules/eventsource-parser": { "version": "3.1.1", "resolved": "https://registry.npmmirror.com/eventsource-parser/-/eventsource-parser-3.1.1.tgz", @@ -11942,6 +12001,16 @@ "jiti": "lib/jiti-cli.mjs" } }, + "node_modules/jose": { + "version": "6.2.12", + "resolved": "https://registry.npmmirror.com/jose/-/jose-6.2.12.tgz", + "integrity": "sha512-9NiFmJEex0sy2Dk58j2UGBSHgUs2ypF9eZSu4L6vjOX3Dp96Sw1F3uL+H+D1sx02jZZdzUT0HgvCy59CuvXcWw==", + "dev": true, + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/panva" + } + }, "node_modules/js-tokens": { "version": "4.0.0", "resolved": "https://registry.npmmirror.com/js-tokens/-/js-tokens-4.0.0.tgz", @@ -14946,6 +15015,16 @@ "url": "https://github.com/sponsors/jonschlinkert" } }, + "node_modules/pkce-challenge": { + "version": "5.0.1", + "resolved": "https://registry.npmmirror.com/pkce-challenge/-/pkce-challenge-5.0.1.tgz", + "integrity": "sha512-wQ0b/W4Fr01qtpHlqSqspcj3EhBvimsdh0KlHhH8HRZnMsEa0ea2fTULOXOS9ccQr3om+GcGRk4e+isrZWV8qQ==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=16.20.0" + } + }, "node_modules/pkijs": { "version": "3.4.1", "resolved": "https://registry.npmmirror.com/pkijs/-/pkijs-3.4.1.tgz", @@ -19588,7 +19667,6 @@ "version": "4.6.5", "resolved": "https://registry.npmmirror.com/zod/-/zod-4.6.5.tgz", "integrity": "sha512-v5l/aFXZQeai4awLbOpSoHecE9UiMrnfx75tEXLjNonXVARxQ5mOeipTjROUchszUNCqnE+hqAMujRsRHsut2Q==", - "dev": true, "license": "MIT", "funding": { "url": "https://github.com/sponsors/colinhacks" diff --git a/package.json b/package.json index 2bc9bec..5e0eea9 100644 --- a/package.json +++ b/package.json @@ -100,6 +100,7 @@ "dependencies": { "@huggingface/transformers": "^4.3.0", "@mixmark-io/domino": "^2.2.0", + "@modelcontextprotocol/server": "^2.1.0", "anki-apkg-export": "^4.0.0", "better-sqlite3": "^12.6.2", "jsdom": "^27.4.0", @@ -129,6 +130,7 @@ "@electron-toolkit/preload": "^3.0.2", "@electron-toolkit/tsconfig": "^2.0.0", "@electron-toolkit/utils": "^4.0.0", + "@modelcontextprotocol/client": "^2.1.0", "@mozilla/readability": "^0.6.0", "@radix-ui/react-alert-dialog": "^1.1.15", "@radix-ui/react-checkbox": "^1.3.3", diff --git a/src/main/db/index.ts b/src/main/db/index.ts index f3aa99f..6ad414e 100644 --- a/src/main/db/index.ts +++ b/src/main/db/index.ts @@ -1,5 +1,6 @@ import { app } from 'electron' import { join } from 'path' +import { mkdirSync } from 'fs' import { stat } from 'fs/promises' import Database from 'better-sqlite3' import { drizzle } from 'drizzle-orm/better-sqlite3' @@ -23,8 +24,16 @@ let isClosing = false // 防止重复关闭 * 在 Electron 主进程的 app.whenReady() 中调用 */ export function initDatabase() { - // 数据库文件存放在用户数据目录 - const dbPath = join(app.getPath('userData'), 'knownote.db') + // 数据库文件存放在用户数据目录。 + // + // `KNOWNOTE_DATA_DIR` 让非桌面入口(#80 的 MCP server)能指向同一个 profile,而不 + // 必经过 Electron 的 `app.getPath('userData')`。默认值仍然是 Electron 的 userData, + // 所以桌面应用的行为不变。 + const dataDir = process.env.KNOWNOTE_DATA_DIR?.trim() || app.getPath('userData') + // Electron creates its userData directory; a `KNOWNOTE_DATA_DIR` override (#80) is + // a path we were handed, so it has to exist before sqlite can open a file in it. + mkdirSync(dataDir, { recursive: true }) + const dbPath = join(dataDir, 'knownote.db') console.log('[Database] Initializing database at:', dbPath) diff --git a/src/main/index.ts b/src/main/index.ts index 4b6df15..a6e24cf 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -21,6 +21,7 @@ import { getStore } from './config/store' import { migrateProvidersToConnections } from './config/connectionMigration' import { isSmokeTestRequested, runSmokeTest } from './smokeTest' import { isEvalRequested, runEvalCli } from './eval/run' +import { isMcpRequested, runMcpServer } from './mcp/entry' import { declareDocumentScheme, registerDocumentProtocolHandler } from './protocol/documentProtocol' import { declarePdfjsAssetScheme, @@ -70,6 +71,14 @@ app.whenReady().then(async () => { return } + // MCP stdio server (#80). Like the other headless entries it runs before any + // window exists; unlike them it stays alive until the client closes stdin. + if (isMcpRequested()) { + await runMcpServer() + app.exit(0) + return + } + // Set app user model id for windows electronApp.setAppUserModelId('com.knownote.app') diff --git a/src/main/mcp/entry.ts b/src/main/mcp/entry.ts new file mode 100644 index 0000000..d25372a --- /dev/null +++ b/src/main/mcp/entry.ts @@ -0,0 +1,61 @@ +import { join } from 'path' +import { app } from 'electron' +import { serveStdio } from '@modelcontextprotocol/server/stdio' +import { closeDatabase, initDatabase, initVectorStore, runMigrations } from '../db' +import { ConnectionManager } from '../models/ConnectionManager' +import { EmbeddingService } from '../services/EmbeddingService' +import { KnowledgeService } from '../services/KnowledgeService' +import { createKnowledgeRuntime } from './runtime' +import { buildMcpServer } from './server' + +export const MCP_FLAG = '--mcp' + +export function isMcpRequested(argv: readonly string[] = process.argv): boolean { + return argv.includes(MCP_FLAG) +} + +/** + * Serve the knowledge base over MCP on stdio (#80). + * + * The stdio transport **is** the protocol: the JSON-RPC frames and the process's + * standard output are the same bytes. Anything the app logs to stdout would be + * read by the client as a frame, and the database logs on init — so every console + * method is redirected to stderr *before* anything else runs. This patches the + * console methods, not `process.stdout`: the SDK writes to that stream directly. + * + * The runtime is built once here, outside the factory. `serveStdio` may call the + * factory more than once for a single connection (an optimistic modern probe, + * then a legacy instance when the probe falls back), so opening the database or + * the index inside the factory would do it twice. + * + * Resolves when the client closes stdin, which is how a stdio server is supposed + * to end. + */ +export async function runMcpServer(): Promise { + console.log = console.error + console.info = console.error + console.debug = console.error + + initDatabase() + runMigrations() + initVectorStore() + + const connectionManager = new ConnectionManager() + const embeddingService = new EmbeddingService(connectionManager, { + cacheDir: join(app.getPath('userData'), 'models') + }) + const knowledgeService = new KnowledgeService(embeddingService) + const runtime = createKnowledgeRuntime(knowledgeService) + + // Dual-era by default: `legacy` is left as the SDK's 'serve', so a 2025-era + // `initialize` and a 2026-07-28 per-request envelope are both served by the same + // factory — one tool surface, no second implementation. + serveStdio(() => buildMcpServer(runtime, app.getVersion())) + + await new Promise((resolve) => { + process.stdin.once('end', resolve) + process.stdin.once('close', resolve) + }) + + closeDatabase() +} diff --git a/src/main/mcp/runtime.ts b/src/main/mcp/runtime.ts new file mode 100644 index 0000000..04c9355 --- /dev/null +++ b/src/main/mcp/runtime.ts @@ -0,0 +1,250 @@ +import { desc, eq, sql } from 'drizzle-orm' +import { getDatabase } from '../db' +import { documents, notes, notebooks } from '../db/schema' +import type { KnowledgeService } from '../services/KnowledgeService' + +/** + * Knowledge query layer for MCP (#80). + * + * The tool handlers talk to this interface, never to `KnowledgeService`, + * Electron or the database directly. That keeps the protocol surface testable + * with a stub runtime and keeps the MCP server an *adapter* over the same + * retrieval path the chat does — the ADR's "do not fork a query path". + * + * Everything here is read-only. There is no tool that writes, spends money or + * calls a model, and no result carries a filesystem path. + */ + +export interface NotebookSummary { + id: string + title: string + description?: string + sourceCount: number + noteCount: number +} + +/** One retrieved passage with the provenance a citation needs. */ +export interface EvidenceView { + documentId: string + documentTitle: string + page?: number + blockId?: string + startOffset?: number + endOffset?: number + quote: string + score: number +} + +export interface SearchResultView { + strategy: string + evidence: EvidenceView[] + /** Set when the requested strategy could not run (e.g. no embedding backend). */ + degraded?: string +} + +export interface SourceOutlineEntry { + blockId: string + kind: string + page: number | null + level: number | null + text: string +} + +export interface SourceView { + documentId: string + title: string + type: string + status: string + mimeType?: string + chunkCount: number + outline: SourceOutlineEntry[] +} + +export interface DocumentTextView { + documentId: string + title: string + /** The page the text was scoped to, when `page` was given. */ + page?: number + text: string + /** True when the returned text is a bounded window rather than the whole source. */ + truncated: boolean +} + +export interface NoteView { + id: string + title: string + content: string +} + +export interface McpRuntime { + listNotebooks(): Promise + searchNotebook(notebookId: string, query: string, topK: number): Promise + getSource(documentId: string): Promise + readDocument(documentId: string, page?: number): Promise + searchNotes(notebookId: string, query: string): Promise +} + +/** How much of a source `read_document` returns when no page is given. */ +export const READ_DOCUMENT_WINDOW = 20_000 + +export function createKnowledgeRuntime(knowledgeService: KnowledgeService): McpRuntime { + const db = getDatabase() + + return { + async listNotebooks(): Promise { + const rows = await db.select().from(notebooks).orderBy(desc(notebooks.updatedAt)).all() + + const sourceCounts = new Map( + db + .select({ notebookId: documents.notebookId, count: sql`count(*)` }) + .from(documents) + .groupBy(documents.notebookId) + .all() + .map((row) => [row.notebookId, Number(row.count)]) + ) + const noteCounts = new Map( + db + .select({ notebookId: notes.notebookId, count: sql`count(*)` }) + .from(notes) + .groupBy(notes.notebookId) + .all() + .map((row) => [row.notebookId, Number(row.count)]) + ) + + return rows.map((row) => ({ + id: row.id, + title: row.title, + ...(row.description ? { description: row.description } : {}), + sourceCount: sourceCounts.get(row.id) ?? 0, + noteCount: noteCounts.get(row.id) ?? 0 + })) + }, + + async searchNotebook(notebookId, query, topK): Promise { + // The same Retriever the chat and the search panel use. When no embedding + // backend is configured the dense path throws; the literal index is still a + // valid answer, so it is used and the result says so rather than failing the + // whole tool call. + try { + const result = await knowledgeService.retrieve({ notebookId, query, topK }) + return { + strategy: result.trace.strategy, + evidence: result.evidence.map((item) => { + const block = item.locator.blocks[0] + return { + documentId: item.documentId, + documentTitle: item.source.title, + ...(item.locator.pageStart !== null ? { page: item.locator.pageStart } : {}), + ...(block + ? { + blockId: block.blockId, + startOffset: block.startOffset + block.startInBlock, + endOffset: block.startOffset + block.endInBlock + } + : {}), + quote: item.content, + score: item.score + } + }) + } + } catch (error) { + const literal = await knowledgeService.searchText(notebookId, query, { limit: topK }) + return { + strategy: 'sparse', + degraded: `Dense retrieval was unavailable (${(error as Error).message}); these are literal BM25 matches only.`, + evidence: literal.map((item) => { + const block = item.locator.blocks[0] + return { + documentId: item.documentId, + documentTitle: item.documentTitle, + ...(item.locator.pageStart !== null ? { page: item.locator.pageStart } : {}), + ...(block + ? { + blockId: block.blockId, + startOffset: block.startOffset + block.startInBlock, + endOffset: block.startOffset + block.endInBlock + } + : {}), + quote: item.content, + score: item.score + } + }) + } + } + }, + + async getSource(documentId): Promise { + const document = knowledgeService.getDocument(documentId) + if (!document) return null + + const outline = knowledgeService.getDocumentBlocks(documentId).map((block) => ({ + blockId: block.id, + kind: block.kind, + page: block.page, + level: block.level, + text: block.text + })) + + return { + documentId: document.id, + title: document.title, + type: document.type, + status: document.status, + ...(document.mimeType ? { mimeType: document.mimeType } : {}), + chunkCount: document.chunkCount ?? 0, + outline + } + }, + + async readDocument(documentId, page): Promise { + const document = knowledgeService.getDocument(documentId) + if (!document?.content) return null + + const content = document.content + + // Page-scoped read: the block table is what knows where a page starts and + // ends, so the slice is derived from it rather than guessed from a page count. + if (page !== undefined) { + const pageBlocks = knowledgeService + .getDocumentBlocks(documentId) + .filter((block) => block.page === page) + .sort((a, b) => a.order - b.order) + + if (pageBlocks.length === 0) { + return { documentId, title: document.title, page, text: '', truncated: false } + } + const start = pageBlocks[0].startOffset + const end = pageBlocks[pageBlocks.length - 1].endOffset + return { + documentId, + title: document.title, + page, + text: content.slice(start, end), + truncated: false + } + } + + return { + documentId, + title: document.title, + text: content.slice(0, READ_DOCUMENT_WINDOW), + truncated: content.length > READ_DOCUMENT_WINDOW + } + }, + + async searchNotes(notebookId, query): Promise { + const needle = query.trim().toLowerCase() + if (needle.length === 0) return [] + + return getDatabase() + .select({ id: notes.id, title: notes.title, content: notes.content }) + .from(notes) + .where(eq(notes.notebookId, notebookId)) + .all() + .filter( + (note) => + note.title.toLowerCase().includes(needle) || note.content.toLowerCase().includes(needle) + ) + } + } +} diff --git a/src/main/mcp/server.ts b/src/main/mcp/server.ts new file mode 100644 index 0000000..9836b16 --- /dev/null +++ b/src/main/mcp/server.ts @@ -0,0 +1,106 @@ +import { McpServer } from '@modelcontextprotocol/server' +import { z } from 'zod' +import type { McpRuntime } from './runtime' + +/** + * The MCP tool surface (#80). + * + * Read-only, and that is the point: this surface is granted to an agent that is + * itself acting on untrusted document text, so no tool writes, spends money, + * calls a model, or reports a filesystem path. + * + * `server/discover` is not here on purpose — it is a protocol RPC owned by the + * SDK's `serveStdio` entry, not a tool, and must not appear in `tools/list`. + */ + +/** The tools this server registers, in a stable order. */ +export const MCP_TOOL_NAMES = [ + 'list_notebooks', + 'search_notebook', + 'get_source', + 'read_document', + 'search_notes' +] as const + +const textResult = (value: unknown): { content: Array<{ type: 'text'; text: string }> } => ({ + content: [{ type: 'text', text: JSON.stringify(value, null, 2) }] +}) + +/** + * Build one server instance. + * + * `serveStdio` may call the factory twice for a single connection (an optimistic + * modern probe, then a legacy instance when the probe falls back), so this must + * stay cheap and side-effect-free. It closes over an already-constructed + * runtime and opens nothing. + */ +export function buildMcpServer(runtime: McpRuntime, version: string): McpServer { + const server = new McpServer({ name: 'knownote', version }) + + server.registerTool( + 'list_notebooks', + { + description: + 'List the notebooks in this KnowNote library, with source and note counts. Call this first to get a notebook_id.', + annotations: { readOnlyHint: true }, + inputSchema: z.object({}) + }, + async () => textResult(await runtime.listNotebooks()) + ) + + server.registerTool( + 'search_notebook', + { + description: + 'Search one notebook and return matching passages WITH provenance (document id, page, character offsets). Prefer this over guessing: the provenance is what lets a claim be checked.', + annotations: { readOnlyHint: true }, + inputSchema: z.object({ + notebook_id: z.string().min(1), + query: z.string().min(1).max(1000), + top_k: z.number().int().min(1).max(50).default(5) + }) + }, + async ({ notebook_id, query, top_k }) => + textResult(await runtime.searchNotebook(notebook_id, query, top_k)) + ) + + server.registerTool( + 'get_source', + { + description: + 'Get one source (document) by id: its metadata and a structure outline of blocks (page, heading level, text).', + annotations: { readOnlyHint: true }, + inputSchema: z.object({ document_id: z.string().min(1) }) + }, + async ({ document_id }) => textResult(await runtime.getSource(document_id)) + ) + + server.registerTool( + 'read_document', + { + description: + 'Read a source as canonical text. Pass page to read one page; otherwise a bounded window of the document is returned and `truncated` says so.', + annotations: { readOnlyHint: true }, + inputSchema: z.object({ + document_id: z.string().min(1), + page: z.number().int().min(1).optional() + }) + }, + async ({ document_id, page }) => textResult(await runtime.readDocument(document_id, page)) + ) + + server.registerTool( + 'search_notes', + { + description: 'Search the notes of one notebook by substring (case-insensitive).', + annotations: { readOnlyHint: true }, + inputSchema: z.object({ + notebook_id: z.string().min(1), + query: z.string().min(1).max(1000) + }) + }, + async ({ notebook_id, query }) => textResult(await runtime.searchNotes(notebook_id, query)) + ) + + return server +} diff --git a/test/mcp.test.ts b/test/mcp.test.ts new file mode 100644 index 0000000..658dbe8 --- /dev/null +++ b/test/mcp.test.ts @@ -0,0 +1,160 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { Client } from '@modelcontextprotocol/client' +import { InMemoryTransport } from '@modelcontextprotocol/server' +import { buildMcpServer, MCP_TOOL_NAMES } from '../src/main/mcp/server.ts' +import type { McpRuntime } from '../src/main/mcp/runtime.ts' + +/** + * The MCP surface (#80). + * + * These drive the real server over the SDK's in-memory transport, so what is + * verified is the protocol dispatch and the registered tool surface — not a + * mocked handler call. The runtime is a stub, which is the point of the + * `McpRuntime` seam. + */ + +const runtime: McpRuntime = { + listNotebooks: async () => [{ id: 'nb_1', title: 'Papers', sourceCount: 2, noteCount: 1 }], + searchNotebook: async () => ({ + strategy: 'dense', + evidence: [ + { + documentId: 'doc_1', + documentTitle: 'Attention Is All You Need', + page: 7, + blockId: 'blk_1', + startOffset: 100, + endOffset: 120, + quote: 'self-attention connects all positions', + score: 0.9 + } + ] + }), + getSource: async (documentId) => + documentId === 'doc_1' + ? { + documentId: 'doc_1', + title: 'Attention Is All You Need', + type: 'file', + status: 'indexed', + chunkCount: 3, + outline: [{ blockId: 'blk_1', kind: 'paragraph', page: 7, level: null, text: '…' }] + } + : null, + readDocument: async (documentId, page) => ({ + documentId, + title: 'Attention Is All You Need', + ...(page !== undefined ? { page } : {}), + text: 'the canonical text', + truncated: false + }), + searchNotes: async () => [{ id: 'note_1', title: 'Note', content: 'hello world' }] +} + +async function connected(): Promise<{ client: Client; close: () => Promise }> { + const server = buildMcpServer(runtime, '0.0.0-test') + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair() + const client = new Client({ name: 'knownote-test', version: '0.0.0' }) + await Promise.all([server.connect(serverTransport), client.connect(clientTransport)]) + return { + client, + close: async () => { + await client.close() + await server.close() + } + } +} + +const textOf = (result: unknown): unknown => { + const content = (result as { content: Array<{ type: string; text?: string }> }).content + const first = content[0] + return JSON.parse(first.text ?? '') +} + +test('the tool list is exactly the read-only surface, and nothing else', async () => { + const { client, close } = await connected() + + const listed = await client.listTools() + const names: string[] = listed.tools.map((tool) => tool.name).sort() + + // Exactly the five application tools. A write tool, or `server/discover` + // registered as a tool, would fail here — and the ADR requires both to be absent. + // The `includes` check runs first: `assert.deepEqual` is an assertion signature, so + // it narrows `names` to the expected tuple's element union. + assert.equal(names.includes('server/discover'), false) + assert.deepEqual(names, [...MCP_TOOL_NAMES].sort()) + + // Every tool declares itself read-only. + for (const tool of listed.tools) { + assert.equal(tool.annotations?.readOnlyHint, true, `${tool.name} is not marked read-only`) + } + + await close() +}) + +test('list_notebooks returns the runtime payload', async () => { + const { client, close } = await connected() + const result = await client.callTool({ name: 'list_notebooks', arguments: {} }) + assert.deepEqual(textOf(result), [{ id: 'nb_1', title: 'Papers', sourceCount: 2, noteCount: 1 }]) + await close() +}) + +test('search_notebook returns evidence with provenance, not bare text', async () => { + const { client, close } = await connected() + const result = await client.callTool({ + name: 'search_notebook', + arguments: { notebook_id: 'nb_1', query: 'attention', top_k: 3 } + }) + + const parsed = textOf(result) as { + strategy: string + evidence: Array<{ documentId: string; page?: number; blockId?: string; quote: string }> + } + + assert.equal(parsed.strategy, 'dense') + assert.equal(parsed.evidence[0].documentId, 'doc_1') + assert.equal(parsed.evidence[0].page, 7) + assert.equal(parsed.evidence[0].blockId, 'blk_1') + assert.match(parsed.evidence[0].quote, /self-attention/) + + await close() +}) + +test('read_document passes an optional page through', async () => { + const { client, close } = await connected() + + const whole = textOf( + await client.callTool({ name: 'read_document', arguments: { document_id: 'doc_1' } }) + ) as { + page?: number + } + assert.equal(whole.page, undefined) + + const scoped = textOf( + await client.callTool({ name: 'read_document', arguments: { document_id: 'doc_1', page: 7 } }) + ) as { page?: number } + assert.equal(scoped.page, 7) + + await close() +}) + +test('a missing source is returned as null, not as an error', async () => { + const { client, close } = await connected() + const result = await client.callTool({ + name: 'get_source', + arguments: { document_id: 'doc_missing' } + }) + assert.equal(textOf(result), null) + await close() +}) + +test('search_notebook rejects top_k above 50 at the schema boundary', async () => { + const { client, close } = await connected() + const result = await client.callTool({ + name: 'search_notebook', + arguments: { notebook_id: 'nb_1', query: 'x', top_k: 500 } + }) + assert.equal((result as { isError?: boolean }).isError, true) + await close() +})