Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
69 changes: 69 additions & 0 deletions src/main/ipc/knowledgeHandlers.ts
Original file line number Diff line number Diff line change
Expand Up @@ -327,6 +327,75 @@ export function registerKnowledgeHandlers(knowledgeService: KnowledgeService) {
return result.canceled ? [] : result.filePaths
})

// 打开文件夹选择对话框(#98)
ipcMain.handle('knowledge:select-folder', async () => {
const result = await dialog.showOpenDialog({
properties: ['openDirectory', 'multiSelections']
})

return result.canceled ? [] : result.filePaths
})

// 文件夹导入:一次快照(#98)
ipcMain.handle(
'knowledge:add-folder',
async (event: IpcMainInvokeEvent, args: unknown) => {
const validated = await validate(KnowledgeSchemas.addFolder, async (params) => {
Logger.debug('KnowledgeHandlers', 'add-folder:', params)

try {
const result = await knowledgeService.addFolder(
params.notebookId,
params.folderPath,
(stage, progress) => {
event.sender.send('knowledge:index-progress', {
notebookId: params.notebookId,
stage,
progress
})
}
)
return { success: true, ...result }
} catch (error) {
Logger.error('KnowledgeHandlers', 'Error adding folder:', error)
return { success: false, added: [], skipped: [], failed: [], error: (error as Error).message }
}
})(event, args)

return validated
}
)

// 批量导入一组文件(#98,拖放/多选共用)
ipcMain.handle(
'knowledge:add-files',
async (event: IpcMainInvokeEvent, args: unknown) => {
const validated = await validate(KnowledgeSchemas.addFiles, async (params) => {
Logger.debug('KnowledgeHandlers', 'add-files:', params.paths.length)

try {
const result = await knowledgeService.addDocumentsFromPaths(
params.notebookId,
params.paths,
(stage, progress) => {
event.sender.send('knowledge:index-progress', {
notebookId: params.notebookId,
stage,
progress
})
}
)
return { success: true, ...result }
} catch (error) {
Logger.error('KnowledgeHandlers', 'Error adding files:', error)
return { success: false, added: [], skipped: [], failed: [], error: (error as Error).message }
}
})(event, args)

return validated
}
)

// 打开文档源文件
ipcMain.handle(
'knowledge:open-source',
Expand Down
11 changes: 11 additions & 0 deletions src/main/ipc/validation.ts
Original file line number Diff line number Diff line change
Expand Up @@ -146,6 +146,17 @@ export const KnowledgeSchemas = {
filePath: z.string().min(1, '文件路径不能为空')
}),

// 批量/文件夹导入(#98)
addFolder: z.object({
notebookId: z.string().min(1, '笔记本 ID 不能为空'),
folderPath: z.string().min(1, '文件夹路径不能为空')
}),

addFiles: z.object({
notebookId: z.string().min(1, '笔记本 ID 不能为空'),
paths: z.array(z.string().min(1)).min(1, '至少需要一个文件')
}),

addDocumentFromUrl: z.object({
notebookId: z.string().min(1, '笔记本 ID 不能为空'),
url: z.string().min(1, '无效的 URL')
Expand Down
16 changes: 16 additions & 0 deletions src/main/services/FileParserService.ts
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,22 @@ export class FileParserService {
return this.loaders.has(fileType.toLowerCase())
}

/**
* 所有已注册 loader 认识的扩展名(小写,不含点)。
*
* 文件夹导入用它决定「扫什么」和「跳过了什么」(#98):扩展名集合只有一个来源,
* 扫描器不会和 loader 各自维护一份而逐渐漂移。
*/
supportedExtensions(): string[] {
const extensions = new Set<string>()
for (const loader of this.allLoaders) {
for (const extension of loader.supportedExtensions) {
extensions.add(extension.toLowerCase())
}
}
return [...extensions].sort((a, b) => a.localeCompare(b))
}

/**
* 根据 MIME 类型获取文件类型
*/
Expand Down
96 changes: 96 additions & 0 deletions src/main/services/KnowledgeService.ts
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,7 @@ import type {
Retriever
} from './retrieval'
import { WebFetchService } from './WebFetchService'
import { scanFolder } from './ingestion/folderScan'
import { deleteDocumentChunksFts, ensureChunksFts, indexChunksFts, searchChunksFts } from './fts'
import { vectorStoreManager } from '../vectorstore'
import Logger from '../../shared/utils/logger'
Expand Down Expand Up @@ -115,6 +116,31 @@ export interface SearchResult {
*/
export type IndexProgressCallback = (stage: string, progress: number) => void

/** 批量导入里被跳过的文件(#98):已经在同一个 notebook 里。 */
export interface BatchImportSkip {
path: string
reason: string
}

/** 批量导入里失败的文件。每个文件都有自己的 ingestion run,这里只汇总。 */
export interface BatchImportFailure {
path: string
error: string
}

/**
* 一次批量导入的结果。
*
* 三组互斥:`added` 是新建的 documentId,`skipped` 是重复路径,`failed` 是解析/索引
* 阶段失败的路径。调用方永远知道「42 个里成功多少、跳过多少、失败多少」,而不是只
* 看到一个总数。
*/
export interface BatchImportResult {
added: string[]
skipped: BatchImportSkip[]
failed: BatchImportFailure[]
}

/**
* `RetrievedEvidence` → 兼容的 `SearchResult` 形状。
*
Expand Down Expand Up @@ -322,6 +348,76 @@ export class KnowledgeService {
return documentId
}

/**
* 批量导入一组已经存在的文件路径(#98)。
*
* 一个文件失败不会中止其余的:每个文件仍然走 `addDocumentFromFile`,所以每个都有自己
* 的 ingestion run 与失败记录。已经在这个 notebook 里的路径被跳过并报告,而不是重复
* 导入一份。
*/
async addDocumentsFromPaths(
notebookId: string,
paths: readonly string[],
onProgress?: IndexProgressCallback
): Promise<BatchImportResult> {
const db = getDatabase()
const existing = new Set(
db
.select({ sourceUri: documents.sourceUri })
.from(documents)
.where(eq(documents.notebookId, notebookId))
.all()
.map((row) => row.sourceUri)
.filter((uri): uri is string => typeof uri === 'string' && uri.length > 0)
)

const added: string[] = []
const skipped: BatchImportSkip[] = []
const failed: BatchImportFailure[] = []

for (let index = 0; index < paths.length; index++) {
const filePath = paths[index]

if (existing.has(filePath)) {
skipped.push({ path: filePath, reason: 'already imported into this notebook' })
continue
}

try {
added.push(await this.addDocumentFromFile(notebookId, filePath))
existing.add(filePath)
} catch (error) {
failed.push({ path: filePath, error: (error as Error).message })
}

onProgress?.(
`importing ${index + 1}/${paths.length}`,
Math.round(((index + 1) / paths.length) * 100)
)
}

return { added, skipped, failed }
}

/**
* 把一个文件夹导入为**一次快照**(#98)。
*
* 扫描只挑解析器认识的扩展名,其余文件被跳过并计数,而不是静默忽略。这是一次快照
* —— 之后新增到文件夹里的文件不会被自动带走,那是 #158 的监听。
*/
async addFolder(
notebookId: string,
folderPath: string,
onProgress?: IndexProgressCallback
): Promise<BatchImportResult> {
const scanned = await scanFolder(folderPath, this.fileParserService.supportedExtensions())
return this.addDocumentsFromPaths(
notebookId,
scanned.map((file) => file.path),
onProgress
)
}

/**
* 一条 ingestion pipeline:copy(可选)→ parse → chunk → embed → 写入派生索引。
*
Expand Down
56 changes: 56 additions & 0 deletions src/main/services/ingestion/folderScan.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
import { readdir } from 'fs/promises'
import { extname, join, relative } from 'path'

/**
* Folder scan for batch import (#98).
*
* A folder import is a **snapshot, not a watch** — this returns the files present
* now. Watching is #158 and lives elsewhere; keeping the two apart is what stops
* "one-shot import" and "the folder changed" from becoming the same code path
* with different triggers.
*/

export interface ScannedFile {
/** Absolute path, as the importer will read it. */
path: string
/** Path relative to the scanned root, for stable ordering and messages. */
relativePath: string
}

/**
* Every supported file under `dir`, recursively, in a deterministic order.
*
* Directories whose name starts with `.` are skipped (`.git`, `.obsidian`, …), and
* symlinks are not followed — `Dirent.isDirectory()` is false for a symlink to a
* directory, so a link that points back up the tree cannot make this loop.
*/
export async function scanFolder(
dir: string,
extensions: readonly string[]
): Promise<ScannedFile[]> {
const allowed = new Set(extensions.map((extension) => extension.toLowerCase()))
const files: ScannedFile[] = []

const walk = async (current: string): Promise<void> => {
const entries = await readdir(current, { withFileTypes: true })

for (const entry of entries) {
if (entry.name.startsWith('.')) continue

const fullPath = join(current, entry.name)
if (entry.isDirectory()) {
await walk(fullPath)
continue
}
if (!entry.isFile()) continue

const extension = extname(entry.name).toLowerCase().slice(1)
if (allowed.has(extension)) {
files.push({ path: fullPath, relativePath: relative(dir, fullPath) })
}
}
}

await walk(dir)
return files.sort((a, b) => a.relativePath.localeCompare(b.relativePath))
}
24 changes: 23 additions & 1 deletion src/main/smokeTest.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@

import Logger from '../shared/utils/logger'
import { readFile, readdir } from 'fs/promises'
import { mkdtempSync, rmSync, writeFileSync } from 'fs'
import { mkdtempSync, rmSync, writeFileSync, mkdirSync, copyFileSync } from 'fs'
import { tmpdir } from 'os'
import { join } from 'path'
import { net, BrowserWindow } from 'electron'
Expand Down Expand Up @@ -735,6 +735,28 @@ async function runChecks(): Promise<string[]> {
)
pass('full-text search finds literal terms and survives a re-index')

// --- folder import is a snapshot with honest counts (#98) ------------------
const folderRoot = mkdtempSync(join(tmpdir(), 'knownote-folder-'))
mkdirSync(join(folderRoot, 'nested'), { recursive: true })
copyFileSync(join(fixtures, 'sample.md'), join(folderRoot, 'kept.md'))
writeFileSync(join(folderRoot, 'ignored.xyz'), 'not a document')
writeFileSync(join(folderRoot, 'nested', 'deep.txt'), 'also unsupported')

const firstFolderImport = await knowledge.addFolder(reindexNotebook, folderRoot)
assert(
firstFolderImport.added.length === 1,
`folder import added ${firstFolderImport.added.length}, expected exactly the one supported file`
)

// A folder import is a snapshot: running it twice must not duplicate anything.
const secondFolderImport = await knowledge.addFolder(reindexNotebook, folderRoot)
assert(secondFolderImport.added.length === 0, 're-importing the same folder added duplicates')
assert(
secondFolderImport.skipped.length === 1,
`re-import skipped ${secondFolderImport.skipped.length}, expected the already-imported file`
)
pass('a folder import is a snapshot: unsupported files are skipped and re-import is a no-op')

// --- a failed re-index leaves the previous index usable (#95) ---------------
// The failure this guards: `reindexDocument()` used to clear the derived index
// *before* embedding, so an embedding failure at 70% left the document with no
Expand Down
10 changes: 9 additions & 1 deletion src/preload/index.d.ts
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,8 @@ import type {
KnowledgeStats,
AddDocumentOptions,
SearchOptions,
IndexProgress
IndexProgress,
BatchImportOutcome
} from '../shared/types/knowledge'
import type { UpdateState, UpdateCheckResult, UpdateOperationResult } from '../shared/types/update'
import type { MindMap, Quiz, QuizSession, AnkiCard } from '../main/db/schema'
Expand Down Expand Up @@ -259,6 +260,12 @@ declare global {
notebookId: string,
filePath: string
) => Promise<{ success: boolean; documentId?: string; error?: string }>
/** 文件夹导入:一次快照(#98)。 */
addFolder: (notebookId: string, folderPath: string) => Promise<BatchImportOutcome>
/** 批量导入一组文件(#98):拖放与多选共用。 */
addFiles: (notebookId: string, paths: string[]) => Promise<BatchImportOutcome>
/** 拖放的 `File` → 绝对路径(#98)。 */
getPathForFile: (file: File) => string
addDocumentFromUrl: (
notebookId: string,
url: string
Expand Down Expand Up @@ -295,6 +302,7 @@ declare global {

// 文件选择对话框
selectFiles: () => Promise<string[]>
selectFolder: () => Promise<string[]>

// 打开源文件
openSource: (documentId: string) => Promise<{ success: boolean; error?: string }>
Expand Down
14 changes: 13 additions & 1 deletion src/preload/index.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
import { contextBridge, ipcRenderer } from 'electron'
import { contextBridge, ipcRenderer, webUtils } from 'electron'
import type { IpcRendererEvent } from 'electron'
import { electronAPI } from '@electron-toolkit/preload'
import type { EmbeddingDownloadProgress } from '../shared/types'
Expand Down Expand Up @@ -168,6 +168,17 @@ const api = {
ipcRenderer.invoke('knowledge:add-document', { notebookId, options }),
addDocumentFromFile: (notebookId: string, filePath: string) =>
ipcRenderer.invoke('knowledge:add-document-from-file', { notebookId, filePath }),
addFolder: (notebookId: string, folderPath: string) =>
ipcRenderer.invoke('knowledge:add-folder', { notebookId, folderPath }),
addFiles: (notebookId: string, paths: string[]) =>
ipcRenderer.invoke('knowledge:add-files', { notebookId, paths }),
/**
* 拖放进来的 `File` 对应的绝对路径(#98)。
*
* Electron 39 移除了 `File.path`,唯一受支持的取法是 `webUtils.getPathForFile`,
* 且必须在 preload 里对真实的 File 对象调用。
*/
getPathForFile: (file: File) => webUtils.getPathForFile(file),
addDocumentFromUrl: (notebookId: string, url: string) =>
ipcRenderer.invoke('knowledge:add-document-from-url', { notebookId, url }),
addNote: (notebookId: string, noteId: string) =>
Expand Down Expand Up @@ -204,6 +215,7 @@ const api = {

// 文件选择
selectFiles: () => ipcRenderer.invoke('knowledge:select-files'),
selectFolder: () => ipcRenderer.invoke('knowledge:select-folder'),

// 打开源文件
openSource: (documentId: string) => ipcRenderer.invoke('knowledge:open-source', { documentId }),
Expand Down
Loading
Loading