Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions skill-data/wiki/references/agents/kb-doc-generator.md
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,7 @@ Use the `Glob → Grep → Read` three-step method (**adapt to the language of t
TypeScript: app.ts / index.ts / main.ts / server.ts
Rust: main.rs / src/main.rs
Swift: main.swift / App.swift
Scala: Main.scala / App.scala

2. Grep: locate the core Handlers/Routers (choose the pattern by language + framework)
Go: grep -rn 'func.*Handler\|\.GET\|\.POST\|router\.\|@handler' <dir>
Expand Down
2 changes: 1 addition & 1 deletion skill-data/wiki/scripts/scan_repo.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@
# Language extension map
LANG_MAP = {
".py": "Python", ".go": "Go", ".js": "JavaScript", ".ts": "TypeScript",
".java": "Java", ".rs": "Rust", ".swift": "Swift", ".rb": "Ruby", ".php": "PHP",
".java": "Java", ".rs": "Rust", ".swift": "Swift", ".scala": "Scala", ".rb": "Ruby", ".php": "PHP",
".c": "C", ".cpp": "C++", ".h": "C/C++ Header",
".proto": "Protobuf", ".thrift": "Thrift", ".graphql": "GraphQL",
".sql": "SQL", ".sh": "Shell", ".bash": "Shell",
Expand Down
424 changes: 424 additions & 0 deletions src/__tests__/scala-extractor.test.ts

Large diffs are not rendered by default.

2 changes: 1 addition & 1 deletion src/ci/extract-mr.ts
Original file line number Diff line number Diff line change
Expand Up @@ -256,7 +256,7 @@ export async function ciExtractMr(opts: CiExtractMrOptions): Promise<void> {
{ cwd: businessRepo, encoding: 'utf-8', timeout: 10_000 },
);
changedFiles = diffOutput.trim().split('\n')
.filter(f => f && /\.(ts|tsx|js|jsx|py|go|rs|java)$/.test(f));
.filter(f => f && /\.(ts|tsx|js|jsx|py|go|rs|java|swift|scala)$/.test(f));
if (changedFiles.length > 0) break;
} catch {
continue;
Expand Down
65 changes: 59 additions & 6 deletions src/codebase-extract.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,10 @@ import {
formatAstStatsSummary,
} from './wiki-engine/adapters/index.js';
import type { CodeFact, InterfaceInventory, CallChain } from './wiki-engine/adapters/index.js';
import type { CodeCollectedFile } from './wiki-engine/code-knowledge/code-collector.js';
import type { ExtractorContext } from './wiki-engine/code-knowledge/extractors/index.js';
import { SCALA_DECL_PREFIX, SCALA_WILDCARD_PREFIX } from './wiki-engine/code-knowledge/extractors/index.js';
import { isMetadataRelation } from './wiki-engine/code-knowledge/code-extractors.js';
import {
loadFactsCache,
saveFactsCache,
Expand Down Expand Up @@ -96,7 +100,7 @@ function detectKnowledgeGaps(
}

// 1. 未解析的外部依赖:import target 不在扫描范围内
const relationFacts = facts.filter((f) => f.kind === 'relation');
const relationFacts = facts.filter((f) => f.kind === 'relation' && !isMetadataRelation(f.name));
const unresolvedImports = new Set<string>();
for (const rel of relationFacts) {
const target = rel.name;
Expand Down Expand Up @@ -216,7 +220,7 @@ function buildEvidencePages(
pages.set(`${kind}.md`, lines.join('\n'));
}

const relationFacts = facts.filter((f) => f.kind === 'relation');
const relationFacts = facts.filter((f) => f.kind === 'relation' && !isMetadataRelation(f.name));
if (relationFacts.length > 0) {
const byDir = new Map<string, CodeFact[]>();
for (const fact of relationFacts) {
Expand Down Expand Up @@ -586,6 +590,33 @@ export async function extractCodebase(opts: ExtractCodebaseOptions): Promise<voi
}
}

// 增量模式下,import 在提取时物化为具体文件(通配 → 各成员、具名 → 声明
// 文件);若本次只变更了被指向的文件而 importer 未变,物化结果已过期。
// 缓存的 scala-wildcard: 标记记录通配 importer 及其包;其余以 .scala/.java
// 结尾的 relation 名就是物化目标文件本身——两者任一受影响都重提取 importer
const indicesDir = path.join(wikiRoot, '.indices');
let cachedFacts: CodeFact[] | undefined;
if (changedFiles !== undefined) {
cachedFacts = await loadFactsCache(indicesDir);
const changedSet = new Set([...changedFiles, ...deletedFiles]);
const staleImporters = new Set<string>();
for (const fact of cachedFacts) {
if (fact.kind !== 'relation') continue;
if (fact.name.startsWith(SCALA_WILDCARD_PREFIX)) {
const wildcardPackage = fact.name.slice(SCALA_WILDCARD_PREFIX.length);
const touched = changedFiles.some((f) => f.includes(`/${wildcardPackage}/`) || f.startsWith(`${wildcardPackage}/`))
|| deletedFiles.some((f) => f.includes(`/${wildcardPackage}/`) || f.startsWith(`${wildcardPackage}/`));
if (touched) staleImporters.add(fact.file);
} else if (/\.(?:scala|java)$/.test(fact.name) && changedSet.has(fact.name)) {
// a materialized target changed — the importer must re-resolve it
staleImporters.add(fact.file);
}
}
if (staleImporters.size > 0) {
changedFiles = [...new Set([...changedFiles, ...staleImporters])];
}
}

const { files, manifest: collectionManifest } = await collectCode({ root, maxFiles, changedFiles });
if (files.length === 0 && !changedFiles) {
// 全量模式下无文件
Expand All @@ -597,17 +628,39 @@ export async function extractCodebase(opts: ExtractCodebaseOptions): Promise<voi
return;
}

// 提取变更文件的新 facts
const newFacts = files.length > 0 ? extractCodeFacts(files) : [];
// 增量模式下,跨文件解析(通配展开、符号定位到声明文件)需要未变更文件的
// 路径与声明;两者都能从上一轮 facts 缓存得到,未变更文件无需重新读取。
// 声明只从 scala-decl: 标记重建——component facts 分不出嵌套成员,
// 标记在提取时就只记包级名字
let extractionContext: ExtractorContext | undefined;
if (changedFiles !== undefined) {
const priorDeclarations = new Map<string, Set<string>>();
const stubs = new Map<string, CodeCollectedFile>();
const removed = new Set([...changedFiles, ...deletedFiles]);
for (const fact of cachedFacts ?? []) {
if (!removed.has(fact.file)) {
stubs.set(fact.file, { path: fact.file, relativePath: fact.file, language: 'text', sha256: '', content: '' });
}
if (fact.kind === 'relation' && fact.name.startsWith(SCALA_DECL_PREFIX)) {
const names = new Set(fact.name.slice(SCALA_DECL_PREFIX.length).split(','));
priorDeclarations.set(fact.file, names);
}
}
for (const file of files) {
stubs.delete(file.relativePath); // 变更文件以本批为准
}
extractionContext = { allFiles: [...files, ...stubs.values()], priorDeclarations };
}

const newFacts = files.length > 0 ? extractCodeFacts(files, extractionContext) : [];

// 增量模式:加载缓存 → 剪除 → 合并
let facts: CodeFact[];
let interfaceInventory: InterfaceInventory;
const indicesDir = path.join(wikiRoot, '.indices');

if (changedFiles !== undefined) {
// 增量模式(含 changedFiles=[] 即仅删除场景)
const oldFacts = await loadFactsCache(indicesDir);
const oldFacts = cachedFacts ?? (await loadFactsCache(indicesDir));
const oldInterfaces = await loadInterfacesCache(indicesDir);

// 剪除已变更/删除的旧数据
Expand Down
4 changes: 2 additions & 2 deletions src/import-repo.ts
Original file line number Diff line number Diff line change
Expand Up @@ -119,7 +119,7 @@ export function detectCrossRepoEdges(
for (const edge of overlay.edges) {
if (edge.relation !== 'imports') continue;
const segments = edge.to.split('/');
const fileName = segments[segments.length - 1]?.replace(/\.(ts|tsx|js|jsx|py|go|rs|java|swift)$/, '') ?? '';
const fileName = segments[segments.length - 1]?.replace(/\.(ts|tsx|js|jsx|py|go|rs|java|swift|scala)$/, '') ?? '';
const pascalName = fileName.split(/[-_]/).map(s => s.charAt(0).toUpperCase() + s.slice(1)).join('');

const match = existingIndex.get(pascalName.toLowerCase());
Expand All @@ -146,7 +146,7 @@ export function detectCrossRepoEdges(
for (const edge of existing.edges) {
if (edge.relation !== 'imports') continue;
const segments = edge.to.split('/');
const fileName = segments[segments.length - 1]?.replace(/\.(ts|tsx|js|jsx|py|go|rs|java|swift)$/, '') ?? '';
const fileName = segments[segments.length - 1]?.replace(/\.(ts|tsx|js|jsx|py|go|rs|java|swift|scala)$/, '') ?? '';
const pascalName = fileName.split(/[-_]/).map(s => s.charAt(0).toUpperCase() + s.slice(1)).join('');

const match = overlayIndex.get(pascalName.toLowerCase());
Expand Down
10 changes: 6 additions & 4 deletions src/wiki-engine/call-chain-tracer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,9 +24,11 @@ const ENTRY_PATTERNS = [
/route/i,
/controller/i,
/endpoint/i,
/main\.(ts|go|py|rs|java|swift)$/,
/server\.(ts|go|py|rs|java|swift)$/,
/app\.(ts|go|py|rs|java|swift)$/,
// Case-insensitive: Scala's key files are Main.scala / App.scala, and a
// capitalized App.ts or Main.go is just as much an entry point.
/main\.(ts|go|py|rs|java|swift|scala)$/i,
/server\.(ts|go|py|rs|java|swift|scala)$/i,
/app\.(ts|go|py|rs|java|swift|scala)$/i,
];

const ORCHESTRATION_PATTERNS = [
Expand Down Expand Up @@ -237,7 +239,7 @@ function resolveRelationTarget(importPath: string, filesByModule: Map<string, st
// Normalize import path
const normalized = importPath
.replace(/^\.\//, "")
.replace(/\.(ts|tsx|js|jsx|mjs|cjs|py|go|rs|java|swift)$/, "");
.replace(/\.(ts|tsx|js|jsx|mjs|cjs|py|go|rs|java|swift|scala)$/, "");

// Try exact match first
const exact = filesByModule.get(normalized);
Expand Down
7 changes: 4 additions & 3 deletions src/wiki-engine/code-knowledge/code-collector.ts
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,8 @@ export const KEY_FILE_PATTERNS: Record<string, RegExp[]> = {
java: [/Application\.java$/, /Controller\.java$/, /Service\.java$/],
typescript: [/index\.ts$/, /server\.ts$/, /app\.ts$/, /router\.ts$/],
rust: [/main\.rs$/, /lib\.rs$/, /mod\.rs$/],
swift: [/main\.swift$/, /App\.swift$/, /Package\.swift$/]
swift: [/main\.swift$/, /App\.swift$/, /Package\.swift$/],
scala: [/Main\.scala$/, /App\.scala$/]
};

export function isKeyFile(relativePath: string, language: string): boolean {
Expand Down Expand Up @@ -132,7 +133,7 @@ async function walk(directory: string, results: string[], includeTests: boolean)
}

function isCodeFile(filePath: string): boolean {
return [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".go", ".rs", ".java", ".swift", ".json", ".yaml", ".yml", ".toml", ".sql", ".conf", ".ini"].includes(
return [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".go", ".rs", ".java", ".swift", ".scala", ".json", ".yaml", ".yml", ".toml", ".sql", ".conf", ".ini"].includes(
path.extname(filePath).toLowerCase()
);
}
Expand All @@ -145,7 +146,7 @@ function languageFor(filePath: string): string {
const ext = path.extname(filePath).toLowerCase();
const map: Record<string, string> = {
".ts": "typescript", ".tsx": "typescript", ".js": "javascript", ".jsx": "javascript",
".py": "python", ".go": "go", ".rs": "rust", ".java": "java", ".swift": "swift",
".py": "python", ".go": "go", ".rs": "rust", ".java": "java", ".swift": "swift", ".scala": "scala",
".json": "json", ".yaml": "yaml", ".yml": "yaml",
".toml": "toml", ".sql": "sql", ".conf": "toml", ".ini": "toml",
};
Expand Down
19 changes: 16 additions & 3 deletions src/wiki-engine/code-knowledge/code-extractors.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import { type CodeCollectedFile } from "./code-collector.js";
import { extractForLanguage } from "./extractors/index.js";
import { extractForLanguage, type ExtractorContext } from "./extractors/index.js";

export type CodeFactKind = "component" | "interface" | "config" | "error" | "data" | "style" | "relation";

Expand Down Expand Up @@ -36,15 +36,28 @@ export interface CodeFact {
evidenceType?: CodeEvidenceType;
}

/**
* Relations that carry extraction metadata for the incremental layer — a
* Scala wildcard's package, a file's top-level declarations — rather than a
* dependency. Evidence pages and gap detection skip them, and the graph
* builder resolves none of them.
*/
export function isMetadataRelation(name: string): boolean {
return /^scala-(?:wildcard|decl):/u.test(name);
}

/**
* Extract code facts from collected files.
* Groups files by language, then dispatches to language-specific extractors.
* `context` — the run's full file list and the previous run's declarations —
* lets extractors resolve cross-file constructs (see ExtractorContext).
*/
export function extractCodeFacts(files: CodeCollectedFile[]): CodeFact[] {
export function extractCodeFacts(files: CodeCollectedFile[], context?: ExtractorContext): CodeFact[] {
const byLanguage = groupByLanguage(files);
const allFacts: CodeFact[] = [];
const effectiveContext: ExtractorContext = context ?? { allFiles: files, priorDeclarations: new Map() };
for (const [language, langFiles] of byLanguage) {
allFacts.push(...extractForLanguage(language, langFiles));
allFacts.push(...extractForLanguage(language, langFiles, effectiveContext));
}
// Deduplicate facts by kind:name:file (same symbol in same file only kept once)
const seen = new Set<string>();
Expand Down
4 changes: 2 additions & 2 deletions src/wiki-engine/code-knowledge/code-graph.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
import path from "node:path";

import { type CodeFact } from "./code-extractors.js";
import { type CodeFact, isMetadataRelation } from "./code-extractors.js";
import {
type GraphIndex,
type GraphNode,
Expand All @@ -27,7 +27,7 @@ export function buildCodeGraph(facts: CodeFact[]): GraphIndex {

const nodeFiles = new Set(facts.filter(f => f.kind !== "relation").map(f => f.file));
const edges: GraphEdge[] = facts
.filter((fact) => fact.kind === "relation")
.filter((fact) => fact.kind === "relation" && !isMetadataRelation(fact.name))
.flatMap((fact) => {
// AST-derived relation facts carry a resolved target file in fact.name and a
// "(code-ast)" marker in detail — trust them directly instead of fuzzy matching.
Expand Down
23 changes: 20 additions & 3 deletions src/wiki-engine/code-knowledge/extractors/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,10 +5,23 @@ import { extractGo } from "./go.js";
import { extractJava } from "./java.js";
import { extractPython } from "./python.js";
import { extractRust } from "./rust.js";
import { extractScala } from "./scala.js";
import { extractSwift } from "./swift.js";
import { extractTypescript } from "./typescript.js";

type LanguageExtractor = (files: CodeCollectedFile[]) => CodeFact[];
/** What an extractor may need beyond its own language batch. */
export interface ExtractorContext {
/** Every collected file of the run, all languages — a Scala wildcard imports Java files just as freely. */
allFiles: CodeCollectedFile[];
/**
* The symbol names each file declares, from the previous run's facts. An
* incremental run re-extracts only changed files, so unchanged files are
* known by their cached declarations alone.
*/
priorDeclarations: Map<string, Set<string>>;
}

type LanguageExtractor = (files: CodeCollectedFile[], context?: ExtractorContext) => CodeFact[];

/**
* Registry mapping language identifiers to their specialized extractors.
Expand All @@ -20,21 +33,24 @@ const EXTRACTOR_REGISTRY: Record<string, LanguageExtractor> = {
python: extractPython,
java: extractJava,
rust: extractRust,
scala: extractScala,
swift: extractSwift,
toml: extractToml,
sql: extractSql,
};

/**
* Dispatch extraction to the appropriate language-specific extractor.
* `context` — the run's full file list and the previous run's declarations —
* lets an extractor resolve cross-language and cross-file constructs.
* Falls back to an empty array for unsupported languages (json, yaml, text, etc.).
*/
export function extractForLanguage(language: string, files: CodeCollectedFile[]): CodeFact[] {
export function extractForLanguage(language: string, files: CodeCollectedFile[], context?: ExtractorContext): CodeFact[] {
const extractor = EXTRACTOR_REGISTRY[language];
if (!extractor) {
return [];
}
return extractor(files);
return extractor(files, context);
}

/**
Expand All @@ -48,5 +64,6 @@ export { extractGo } from "./go.js";
export { extractJava } from "./java.js";
export { extractPython } from "./python.js";
export { extractRust } from "./rust.js";
export { extractScala, SCALA_DECL_PREFIX, SCALA_WILDCARD_PREFIX } from "./scala.js";
export { extractSwift } from "./swift.js";
export { extractTypescript } from "./typescript.js";
Loading
Loading