diff --git a/src/main/ipc/knowledgeHandlers.ts b/src/main/ipc/knowledgeHandlers.ts index 9d492ce..f624cf9 100644 --- a/src/main/ipc/knowledgeHandlers.ts +++ b/src/main/ipc/knowledgeHandlers.ts @@ -327,6 +327,75 @@ export function registerKnowledgeHandlers(knowledgeService: KnowledgeService) { return result.canceled ? [] : result.filePaths }) + // 打开文件夹选择对话框(#98) + ipcMain.handle('knowledge:select-folder', async () => { + const result = await dialog.showOpenDialog({ + properties: ['openDirectory', 'multiSelections'] + }) + + return result.canceled ? [] : result.filePaths + }) + + // 文件夹导入:一次快照(#98) + ipcMain.handle( + 'knowledge:add-folder', + async (event: IpcMainInvokeEvent, args: unknown) => { + const validated = await validate(KnowledgeSchemas.addFolder, async (params) => { + Logger.debug('KnowledgeHandlers', 'add-folder:', params) + + try { + const result = await knowledgeService.addFolder( + params.notebookId, + params.folderPath, + (stage, progress) => { + event.sender.send('knowledge:index-progress', { + notebookId: params.notebookId, + stage, + progress + }) + } + ) + return { success: true, ...result } + } catch (error) { + Logger.error('KnowledgeHandlers', 'Error adding folder:', error) + return { success: false, added: [], skipped: [], failed: [], error: (error as Error).message } + } + })(event, args) + + return validated + } + ) + + // 批量导入一组文件(#98,拖放/多选共用) + ipcMain.handle( + 'knowledge:add-files', + async (event: IpcMainInvokeEvent, args: unknown) => { + const validated = await validate(KnowledgeSchemas.addFiles, async (params) => { + Logger.debug('KnowledgeHandlers', 'add-files:', params.paths.length) + + try { + const result = await knowledgeService.addDocumentsFromPaths( + params.notebookId, + params.paths, + (stage, progress) => { + event.sender.send('knowledge:index-progress', { + notebookId: params.notebookId, + stage, + progress + }) + } + ) + return { success: true, ...result } + } catch (error) { + Logger.error('KnowledgeHandlers', 'Error adding files:', error) + return { success: false, added: [], skipped: [], failed: [], error: (error as Error).message } + } + })(event, args) + + return validated + } + ) + // 打开文档源文件 ipcMain.handle( 'knowledge:open-source', diff --git a/src/main/ipc/validation.ts b/src/main/ipc/validation.ts index f745d9b..fee8194 100644 --- a/src/main/ipc/validation.ts +++ b/src/main/ipc/validation.ts @@ -146,6 +146,17 @@ export const KnowledgeSchemas = { filePath: z.string().min(1, '文件路径不能为空') }), + // 批量/文件夹导入(#98) + addFolder: z.object({ + notebookId: z.string().min(1, '笔记本 ID 不能为空'), + folderPath: z.string().min(1, '文件夹路径不能为空') + }), + + addFiles: z.object({ + notebookId: z.string().min(1, '笔记本 ID 不能为空'), + paths: z.array(z.string().min(1)).min(1, '至少需要一个文件') + }), + addDocumentFromUrl: z.object({ notebookId: z.string().min(1, '笔记本 ID 不能为空'), url: z.string().min(1, '无效的 URL') diff --git a/src/main/services/FileParserService.ts b/src/main/services/FileParserService.ts index e658823..8ea3f42 100644 --- a/src/main/services/FileParserService.ts +++ b/src/main/services/FileParserService.ts @@ -106,6 +106,22 @@ export class FileParserService { return this.loaders.has(fileType.toLowerCase()) } + /** + * 所有已注册 loader 认识的扩展名(小写,不含点)。 + * + * 文件夹导入用它决定「扫什么」和「跳过了什么」(#98):扩展名集合只有一个来源, + * 扫描器不会和 loader 各自维护一份而逐渐漂移。 + */ + supportedExtensions(): string[] { + const extensions = new Set() + for (const loader of this.allLoaders) { + for (const extension of loader.supportedExtensions) { + extensions.add(extension.toLowerCase()) + } + } + return [...extensions].sort((a, b) => a.localeCompare(b)) + } + /** * 根据 MIME 类型获取文件类型 */ diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 597a301..9a38a45 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -60,6 +60,7 @@ import type { Retriever } from './retrieval' import { WebFetchService } from './WebFetchService' +import { scanFolder } from './ingestion/folderScan' import { deleteDocumentChunksFts, ensureChunksFts, indexChunksFts, searchChunksFts } from './fts' import { vectorStoreManager } from '../vectorstore' import Logger from '../../shared/utils/logger' @@ -115,6 +116,31 @@ export interface SearchResult { */ export type IndexProgressCallback = (stage: string, progress: number) => void +/** 批量导入里被跳过的文件(#98):已经在同一个 notebook 里。 */ +export interface BatchImportSkip { + path: string + reason: string +} + +/** 批量导入里失败的文件。每个文件都有自己的 ingestion run,这里只汇总。 */ +export interface BatchImportFailure { + path: string + error: string +} + +/** + * 一次批量导入的结果。 + * + * 三组互斥:`added` 是新建的 documentId,`skipped` 是重复路径,`failed` 是解析/索引 + * 阶段失败的路径。调用方永远知道「42 个里成功多少、跳过多少、失败多少」,而不是只 + * 看到一个总数。 + */ +export interface BatchImportResult { + added: string[] + skipped: BatchImportSkip[] + failed: BatchImportFailure[] +} + /** * `RetrievedEvidence` → 兼容的 `SearchResult` 形状。 * @@ -322,6 +348,76 @@ export class KnowledgeService { return documentId } + /** + * 批量导入一组已经存在的文件路径(#98)。 + * + * 一个文件失败不会中止其余的:每个文件仍然走 `addDocumentFromFile`,所以每个都有自己 + * 的 ingestion run 与失败记录。已经在这个 notebook 里的路径被跳过并报告,而不是重复 + * 导入一份。 + */ + async addDocumentsFromPaths( + notebookId: string, + paths: readonly string[], + onProgress?: IndexProgressCallback + ): Promise { + const db = getDatabase() + const existing = new Set( + db + .select({ sourceUri: documents.sourceUri }) + .from(documents) + .where(eq(documents.notebookId, notebookId)) + .all() + .map((row) => row.sourceUri) + .filter((uri): uri is string => typeof uri === 'string' && uri.length > 0) + ) + + const added: string[] = [] + const skipped: BatchImportSkip[] = [] + const failed: BatchImportFailure[] = [] + + for (let index = 0; index < paths.length; index++) { + const filePath = paths[index] + + if (existing.has(filePath)) { + skipped.push({ path: filePath, reason: 'already imported into this notebook' }) + continue + } + + try { + added.push(await this.addDocumentFromFile(notebookId, filePath)) + existing.add(filePath) + } catch (error) { + failed.push({ path: filePath, error: (error as Error).message }) + } + + onProgress?.( + `importing ${index + 1}/${paths.length}`, + Math.round(((index + 1) / paths.length) * 100) + ) + } + + return { added, skipped, failed } + } + + /** + * 把一个文件夹导入为**一次快照**(#98)。 + * + * 扫描只挑解析器认识的扩展名,其余文件被跳过并计数,而不是静默忽略。这是一次快照 + * —— 之后新增到文件夹里的文件不会被自动带走,那是 #158 的监听。 + */ + async addFolder( + notebookId: string, + folderPath: string, + onProgress?: IndexProgressCallback + ): Promise { + const scanned = await scanFolder(folderPath, this.fileParserService.supportedExtensions()) + return this.addDocumentsFromPaths( + notebookId, + scanned.map((file) => file.path), + onProgress + ) + } + /** * 一条 ingestion pipeline:copy(可选)→ parse → chunk → embed → 写入派生索引。 * diff --git a/src/main/services/ingestion/folderScan.ts b/src/main/services/ingestion/folderScan.ts new file mode 100644 index 0000000..ff29c3c --- /dev/null +++ b/src/main/services/ingestion/folderScan.ts @@ -0,0 +1,56 @@ +import { readdir } from 'fs/promises' +import { extname, join, relative } from 'path' + +/** + * Folder scan for batch import (#98). + * + * A folder import is a **snapshot, not a watch** — this returns the files present + * now. Watching is #158 and lives elsewhere; keeping the two apart is what stops + * "one-shot import" and "the folder changed" from becoming the same code path + * with different triggers. + */ + +export interface ScannedFile { + /** Absolute path, as the importer will read it. */ + path: string + /** Path relative to the scanned root, for stable ordering and messages. */ + relativePath: string +} + +/** + * Every supported file under `dir`, recursively, in a deterministic order. + * + * Directories whose name starts with `.` are skipped (`.git`, `.obsidian`, …), and + * symlinks are not followed — `Dirent.isDirectory()` is false for a symlink to a + * directory, so a link that points back up the tree cannot make this loop. + */ +export async function scanFolder( + dir: string, + extensions: readonly string[] +): Promise { + const allowed = new Set(extensions.map((extension) => extension.toLowerCase())) + const files: ScannedFile[] = [] + + const walk = async (current: string): Promise => { + const entries = await readdir(current, { withFileTypes: true }) + + for (const entry of entries) { + if (entry.name.startsWith('.')) continue + + const fullPath = join(current, entry.name) + if (entry.isDirectory()) { + await walk(fullPath) + continue + } + if (!entry.isFile()) continue + + const extension = extname(entry.name).toLowerCase().slice(1) + if (allowed.has(extension)) { + files.push({ path: fullPath, relativePath: relative(dir, fullPath) }) + } + } + } + + await walk(dir) + return files.sort((a, b) => a.relativePath.localeCompare(b.relativePath)) +} diff --git a/src/main/smokeTest.ts b/src/main/smokeTest.ts index c06cab6..e07c777 100644 --- a/src/main/smokeTest.ts +++ b/src/main/smokeTest.ts @@ -28,7 +28,7 @@ import Logger from '../shared/utils/logger' import { readFile, readdir } from 'fs/promises' -import { mkdtempSync, rmSync, writeFileSync } from 'fs' +import { mkdtempSync, rmSync, writeFileSync, mkdirSync, copyFileSync } from 'fs' import { tmpdir } from 'os' import { join } from 'path' import { net, BrowserWindow } from 'electron' @@ -735,6 +735,28 @@ async function runChecks(): Promise { ) pass('full-text search finds literal terms and survives a re-index') + // --- folder import is a snapshot with honest counts (#98) ------------------ + const folderRoot = mkdtempSync(join(tmpdir(), 'knownote-folder-')) + mkdirSync(join(folderRoot, 'nested'), { recursive: true }) + copyFileSync(join(fixtures, 'sample.md'), join(folderRoot, 'kept.md')) + writeFileSync(join(folderRoot, 'ignored.xyz'), 'not a document') + writeFileSync(join(folderRoot, 'nested', 'deep.txt'), 'also unsupported') + + const firstFolderImport = await knowledge.addFolder(reindexNotebook, folderRoot) + assert( + firstFolderImport.added.length === 1, + `folder import added ${firstFolderImport.added.length}, expected exactly the one supported file` + ) + + // A folder import is a snapshot: running it twice must not duplicate anything. + const secondFolderImport = await knowledge.addFolder(reindexNotebook, folderRoot) + assert(secondFolderImport.added.length === 0, 're-importing the same folder added duplicates') + assert( + secondFolderImport.skipped.length === 1, + `re-import skipped ${secondFolderImport.skipped.length}, expected the already-imported file` + ) + pass('a folder import is a snapshot: unsupported files are skipped and re-import is a no-op') + // --- a failed re-index leaves the previous index usable (#95) --------------- // The failure this guards: `reindexDocument()` used to clear the derived index // *before* embedding, so an embedding failure at 70% left the document with no diff --git a/src/preload/index.d.ts b/src/preload/index.d.ts index 8c066ba..18bdb80 100644 --- a/src/preload/index.d.ts +++ b/src/preload/index.d.ts @@ -21,7 +21,8 @@ import type { KnowledgeStats, AddDocumentOptions, SearchOptions, - IndexProgress + IndexProgress, + BatchImportOutcome } from '../shared/types/knowledge' import type { UpdateState, UpdateCheckResult, UpdateOperationResult } from '../shared/types/update' import type { MindMap, Quiz, QuizSession, AnkiCard } from '../main/db/schema' @@ -259,6 +260,12 @@ declare global { notebookId: string, filePath: string ) => Promise<{ success: boolean; documentId?: string; error?: string }> + /** 文件夹导入:一次快照(#98)。 */ + addFolder: (notebookId: string, folderPath: string) => Promise + /** 批量导入一组文件(#98):拖放与多选共用。 */ + addFiles: (notebookId: string, paths: string[]) => Promise + /** 拖放的 `File` → 绝对路径(#98)。 */ + getPathForFile: (file: File) => string addDocumentFromUrl: ( notebookId: string, url: string @@ -295,6 +302,7 @@ declare global { // 文件选择对话框 selectFiles: () => Promise + selectFolder: () => Promise // 打开源文件 openSource: (documentId: string) => Promise<{ success: boolean; error?: string }> diff --git a/src/preload/index.ts b/src/preload/index.ts index 8b8bc00..191c69a 100644 --- a/src/preload/index.ts +++ b/src/preload/index.ts @@ -1,4 +1,4 @@ -import { contextBridge, ipcRenderer } from 'electron' +import { contextBridge, ipcRenderer, webUtils } from 'electron' import type { IpcRendererEvent } from 'electron' import { electronAPI } from '@electron-toolkit/preload' import type { EmbeddingDownloadProgress } from '../shared/types' @@ -168,6 +168,17 @@ const api = { ipcRenderer.invoke('knowledge:add-document', { notebookId, options }), addDocumentFromFile: (notebookId: string, filePath: string) => ipcRenderer.invoke('knowledge:add-document-from-file', { notebookId, filePath }), + addFolder: (notebookId: string, folderPath: string) => + ipcRenderer.invoke('knowledge:add-folder', { notebookId, folderPath }), + addFiles: (notebookId: string, paths: string[]) => + ipcRenderer.invoke('knowledge:add-files', { notebookId, paths }), + /** + * 拖放进来的 `File` 对应的绝对路径(#98)。 + * + * Electron 39 移除了 `File.path`,唯一受支持的取法是 `webUtils.getPathForFile`, + * 且必须在 preload 里对真实的 File 对象调用。 + */ + getPathForFile: (file: File) => webUtils.getPathForFile(file), addDocumentFromUrl: (notebookId: string, url: string) => ipcRenderer.invoke('knowledge:add-document-from-url', { notebookId, url }), addNote: (notebookId: string, noteId: string) => @@ -204,6 +215,7 @@ const api = { // 文件选择 selectFiles: () => ipcRenderer.invoke('knowledge:select-files'), + selectFolder: () => ipcRenderer.invoke('knowledge:select-folder'), // 打开源文件 openSource: (documentId: string) => ipcRenderer.invoke('knowledge:open-source', { documentId }), diff --git a/src/renderer/src/components/notebook/SourcePanel.tsx b/src/renderer/src/components/notebook/SourcePanel.tsx index 766a1a5..7268052 100644 --- a/src/renderer/src/components/notebook/SourcePanel.tsx +++ b/src/renderer/src/components/notebook/SourcePanel.tsx @@ -10,7 +10,8 @@ import { StickyNote, ArrowLeft, Quote, - ExternalLink + ExternalLink, + FolderOpen } from 'lucide-react' import { toast } from 'sonner' import { useKnowledgeStore, setupKnowledgeListeners } from '../../store/knowledgeStore' @@ -26,7 +27,10 @@ import { Card } from '../ui/card' import { PanelHeader } from '../ui/panel-header' import DocumentList from './source/DocumentList' import SourceReader from './source/reader/SourceReader' -import type { KnowledgeDocument } from '../../../../shared/types/knowledge' +import type { + KnowledgeDocument, + BatchImportOutcome +} from '../../../../shared/types/knowledge' import type { ReaderAnchor, ReaderSelection } from '../../../../shared/types/source' import { selectionToSourceAnchor, @@ -42,9 +46,6 @@ import { requestAppendExcerpt } from './note/appendExcerptCommand' // 添加来源类型 type AddSourceType = 'file' | 'url' | 'text' | 'note' -/** “/home/me/notes.pdf” → “notes.pdf”。渲染进程没有 node 的 `path`。 */ -const fileName = (filePath: string): string => filePath.split(/[\\/]/).pop() || filePath - // 添加来源弹窗组件 interface AddSourceModalProps { type: AddSourceType @@ -311,6 +312,7 @@ export default function SourcePanel(): ReactElement { const { t } = useTranslation('ui') const { id: notebookId } = useParams() const [showAddMenu, setShowAddMenu] = useState(false) + const [isDragging, setIsDragging] = useState(false) const [modalType, setModalType] = useState(null) const [hasEmbeddingModel, setHasEmbeddingModel] = useState(false) const [selectedDocument, setSelectedDocument] = useState(null) @@ -326,13 +328,15 @@ export default function SourcePanel(): ReactElement { loadDocuments, loadStats, addDocument, - addDocumentFromFile, + addFolder, + addFiles, addDocumentFromUrl, addNoteToKnowledge, deleteDocument, retryDocument, reindexDocument, - selectFiles + selectFiles, + selectFolder } = useKnowledgeStore() const { notes, loadNotes, currentNote, createNote } = useItemStore() @@ -472,7 +476,25 @@ export default function SourcePanel(): ReactElement { [t] ) - // 处理文件上传 + /** + * 批量导入的汇总(#98):成功/跳过/失败三组分开说,而不是只报一个总数。 + * 用户要能知道「42 个里跳过了 3 个、失败了 1 个」。 + */ + const reportBatch = useCallback( + (result: BatchImportOutcome): void => { + const parts: string[] = [] + if (result.added.length > 0) parts.push(t('importAdded', { count: result.added.length })) + if (result.skipped.length > 0) + parts.push(t('importSkipped', { count: result.skipped.length })) + if (result.failed.length > 0) + parts.push(t('importFailedCount', { count: result.failed.length })) + if (!result.success && result.error) parts.push(result.error) + if (parts.length > 0) toast(parts.join(' · ')) + }, + [t] + ) + + // 处理文件上传(多选 → 批量导入,#98) const handleFileUpload = useCallback(async () => { if (!notebookId) return @@ -483,14 +505,43 @@ export default function SourcePanel(): ReactElement { } const files = await selectFiles() - for (const filePath of files) { - // 一个文件失败不能影响其余文件,但也不能被吞掉: - // 吞掉它,用户看到的就是“什么都没发生” - const result = await addDocumentFromFile(notebookId, filePath) - reportImportFailure(fileName(filePath), result) + setShowAddMenu(false) + if (files.length === 0) return + // 一次批量导入:重复路径被跳过并报告,一个失败不中止其余。 + reportBatch(await addFiles(notebookId, files)) + }, [notebookId, hasEmbeddingModel, selectFiles, addFiles, reportBatch, t]) + + // 处理文件夹导入(#98):一次快照,不是监听(监听是 #158) + const handleFolderUpload = useCallback(async () => { + if (!notebookId) return + if (!hasEmbeddingModel) { + alert(t('noEmbeddingModelConfigured')) + return } + + const folders = await selectFolder() setShowAddMenu(false) - }, [notebookId, hasEmbeddingModel, selectFiles, addDocumentFromFile, reportImportFailure, t]) + for (const folder of folders) { + reportBatch(await addFolder(notebookId, folder)) + } + }, [notebookId, hasEmbeddingModel, selectFolder, addFolder, reportBatch, t]) + + // 拖放文件/文件夹(#98)。Electron 39 下路径必须由 preload 的 webUtils 给出。 + const handleDrop = useCallback( + async (event: React.DragEvent): Promise => { + event.preventDefault() + setIsDragging(false) + if (!notebookId || !hasEmbeddingModel) return + + const paths = Array.from(event.dataTransfer.files) + .map((file) => window.api.knowledge.getPathForFile(file)) + .filter((path) => path.length > 0) + + if (paths.length === 0) return + reportBatch(await addFiles(notebookId, paths)) + }, + [notebookId, hasEmbeddingModel, addFiles, reportBatch] + ) // 处理 URL 导入 const handleUrlImport = useCallback( @@ -645,7 +696,22 @@ export default function SourcePanel(): ReactElement { ) return ( - + { + event.preventDefault() + setIsDragging(true) + }} + onDragLeave={() => setIsDragging(false)} + onDrop={handleDrop} + > + {isDragging && ( +
+

+ {t('dropToImport')} +

+
+ )} {openDocument ? ( // 文档预览页面 {t('uploadFile')} +