Organisasi ulang layer-first (domain/application/infrastructure) menjadi vertikal per-feature, behavior tidak berubah: - src/shared/domain/ -> fondasi bersama (ex-domain/core): message, provider, tool_call, tool_result, store, error, usage, conversation - src/features/agent/ -> LLM turn loop: domain (TurnEvent, AgentTurnParams) + application (turn_service, ports) + infrastructure (llm LlmClient + tools) - src/features/cms/ -> settings/conversation/memory persistence: domain + application (services) + infrastructure (persistence repos) - src/features/subagent/ -> orchestration: domain (AccessTier) + infrastructure (engine runner, context, provider, delegate, spawn_tools) - src/features/workflow/ -> multi-phase + hive-mind: domain + infrastructure - src/interfaces/ -> cli + tui (outer ring) tetap Subagent-runner (engine.ts runAgent, division.ts toolsFor) dipindah ke features/subagent (bukan agent tools) utk memutus coupling agent<->subagent. Alias tsconfig: @zesdex/domain -> shared/domain, + @zesdex/agent, @zesdex/agent-infra, @zesdex/cms, @zesdex/subagent, @zesdex/workflow, @zesdex/shared. Bersihkan dead: application/ports PasswordService/TokenService/AuthService. compose.ts (composition root) & bootstrap repoint ke barrel feature. Gates: tsc --noEmit 0 error, bun test 70 pass/0 fail, bun build 147 modules, ./dist/zesdex --headless -> real turn OK.
819 lines
31 KiB
TypeScript
819 lines
31 KiB
TypeScript
/**
|
|
* Multi-language code symbol index — functions, classes, variables, structs,
|
|
* enums, interfaces, traits, modules. Mirrors `tools/semantic_search.rs`.
|
|
*
|
|
* Per-workspace in-memory cache keyed by workspace path; rebuilds run
|
|
* outside the global lock so concurrent searches never block on I/O.
|
|
*/
|
|
import * as fs from "node:fs";
|
|
import * as path from "node:path";
|
|
import type { JsonValue } from "@zesdex/domain";
|
|
import { type Tool, type ToolCtx } from "../mod.ts";
|
|
import { argStr, optBool, optInt, optStr } from "../util.ts";
|
|
|
|
export type Language = "rust" | "typescript" | "javascript" | "python" | "go" | "other";
|
|
|
|
export type SymbolKind =
|
|
| "fn"
|
|
| "struct"
|
|
| "enum"
|
|
| "trait"
|
|
| "mod"
|
|
| "impl"
|
|
| "type"
|
|
| "const"
|
|
| "macro"
|
|
| "class"
|
|
| "interface"
|
|
| "var"
|
|
| "symbol";
|
|
|
|
export interface CodeSymbol {
|
|
name: string;
|
|
kind: SymbolKind;
|
|
language: Language;
|
|
file: string;
|
|
line: number;
|
|
parent: string | null;
|
|
doc_comment: string | null;
|
|
context: string;
|
|
}
|
|
|
|
// ── Global per-workspace index cache ─────────────────────────────────────
|
|
const symbolIndexStore: Record<string, SymbolIndex> = {};
|
|
|
|
let pendingRebuilds: Record<string, Promise<number>> = {};
|
|
|
|
// ── Language regexes (precompiled) ───────────────────────────────────────
|
|
const RX = {
|
|
rust_fn: /^\s*(?:pub\s+)?(?:(?:unsafe\s+)?async\s+)?fn\s+(\w+)/m,
|
|
rust_struct: /^\s*(?:pub\s+)?struct\s+(\w+)/m,
|
|
rust_enum: /^\s*(?:pub\s+)?enum\s+(\w+)/m,
|
|
rust_trait: /^\s*(?:pub\s+)?(?:(?:unsafe\s+)?)?trait\s+(\w+)/m,
|
|
rust_mod: /^\s*(?:pub\s+)?mod\s+(\w+)/m,
|
|
rust_impl: /^\s*(?:pub\s+)?(?:unsafe\s+)?impl(?:\s*<[^>]*>)?\s+(?:for\s+)?(\w+)/m,
|
|
rust_type: /^\s*(?:pub\s+)?type\s+(\w+)/m,
|
|
rust_const: /^\s*(?:pub\s+)?const\s+(\w+)/m,
|
|
rust_macro: /^\s*(?:pub\s+)?macro_rules!\s*\(\s*(\w+)/m,
|
|
|
|
ts_fn: /^\s*(?:export\s+)?(?:(?:async\s+)?function\s+|(?:public|private|protected)\s+)?(\w+)\s*(?:\(|=\s*(?:async\s+)?\()/m,
|
|
ts_class: /^\s*(?:export\s+)?(?:abstract\s+)?class\s+(\w+)/m,
|
|
ts_interface: /^\s*(?:export\s+)?interface\s+(\w+)/m,
|
|
ts_type: /^\s*(?:export\s+)?type\s+(\w+)\s*=/m,
|
|
ts_enum: /^\s*(?:export\s+)?enum\s+(\w+)/m,
|
|
ts_var: /^\s*(?:export\s+)?(?:const|let|var)\s+(\w+)\s*(?::\s*\w+\s*)?=/m,
|
|
|
|
py_def: /^\s*def\s+(\w+)/m,
|
|
py_class: /^\s*class\s+(\w+)/m,
|
|
py_async_def: /^\s*async\s+def\s+(\w+)/m,
|
|
|
|
go_func: /^\s*func\s+(?:\([^)]*\)\s+)?(\w+)/m,
|
|
go_type: /^\s*type\s+(\w+)/m,
|
|
go_struct: /^\s*type\s+(\w+)\s+struct/m,
|
|
go_interface: /^\s*type\s+(\w+)\s+interface/m,
|
|
go_const: /^\s*const\s+(\w+)/m,
|
|
go_var: /^\s*var\s+(\w+)/m,
|
|
};
|
|
|
|
// ── SymbolIndex ──────────────────────────────────────────────────────────
|
|
export class SymbolIndex {
|
|
symbols: CodeSymbol[] = [];
|
|
workspacePath: string | null = null;
|
|
|
|
is_empty(): boolean {
|
|
return this.symbols.length === 0;
|
|
}
|
|
len(): number {
|
|
return this.symbols.length;
|
|
}
|
|
needs_rebuild(workspace: string): boolean {
|
|
return this.is_empty() || this.workspacePath !== workspace;
|
|
}
|
|
|
|
/** Walk the workspace and extract symbols from supported languages. */
|
|
async rebuild(workspace: string): Promise<number> {
|
|
if (!fs.existsSync(workspace)) {
|
|
throw new Error(`workspace path does not exist: ${workspace}`);
|
|
}
|
|
const dispatch: Record<string, (c: string, r: string) => CodeSymbol[]> = {
|
|
rs: extractRust,
|
|
ts: extractTypescript,
|
|
tsx: extractTypescript,
|
|
mts: extractTypescript,
|
|
js: extractJavascript,
|
|
jsx: extractJavascript,
|
|
mjs: extractJavascript,
|
|
py: extractPython,
|
|
go: extractGo,
|
|
};
|
|
|
|
const symbols: CodeSymbol[] = [];
|
|
const walk = (dir: string) => {
|
|
let entries: fs.Dirent[];
|
|
try {
|
|
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
} catch {
|
|
return;
|
|
}
|
|
for (const entry of entries) {
|
|
if (entry.name === ".git" || entry.name === "node_modules" || entry.name === "target") continue;
|
|
const full = path.join(dir, entry.name);
|
|
if (entry.isDirectory()) {
|
|
walk(full);
|
|
} else if (entry.isFile()) {
|
|
const ext = path.extname(full).slice(1).toLowerCase();
|
|
const extractor = dispatch[ext];
|
|
if (!extractor) continue;
|
|
const relPath = path.relative(workspace, full);
|
|
try {
|
|
const content = fs.readFileSync(full, "utf8");
|
|
symbols.push(...extractor(content, relPath));
|
|
} catch {
|
|
// skip unreadable files
|
|
}
|
|
}
|
|
}
|
|
};
|
|
// yield to event loop periodically to avoid blocking the runtime for long walks
|
|
await new Promise((r) => setTimeout(r, 0));
|
|
walk(workspace);
|
|
|
|
symbols.sort((a, b) => a.name.localeCompare(b.name));
|
|
this.symbols = symbols;
|
|
this.workspacePath = workspace;
|
|
return symbols.length;
|
|
}
|
|
|
|
search(query: string, maxResults: number): CodeSymbol[] {
|
|
if (this.symbols.length === 0) return [];
|
|
const queryLower = query.toLowerCase();
|
|
const queryWords = queryLower.split(/\s+/).filter((w) => w !== "");
|
|
const scored = this.symbols
|
|
.map((sym) => ({ score: scoreSymbol(sym, queryLower, queryWords), sym }))
|
|
.filter((s) => s.score > 0)
|
|
.sort((a, b) => b.score - a.score || a.sym.name.localeCompare(b.sym.name));
|
|
return scored.slice(0, maxResults).map((s) => s.sym);
|
|
}
|
|
|
|
list(
|
|
languageFilter: Language | null,
|
|
kindFilter: SymbolKind | null,
|
|
fileFilter: string | null,
|
|
maxResults: number,
|
|
): CodeSymbol[] {
|
|
return this.symbols
|
|
.filter((s) => {
|
|
if (languageFilter && s.language !== languageFilter) return false;
|
|
if (kindFilter && s.kind !== kindFilter) return false;
|
|
if (fileFilter && !s.file.includes(fileFilter)) return false;
|
|
return true;
|
|
})
|
|
.slice(0, maxResults);
|
|
}
|
|
|
|
count_by_language(): Array<[Language, number]> {
|
|
const counts: Record<string, number> = {};
|
|
for (const sym of this.symbols) {
|
|
counts[sym.language] = (counts[sym.language] ?? 0) + 1;
|
|
}
|
|
return Object.entries(counts)
|
|
.map(([lang, c]) => [lang as Language, c] as [Language, number])
|
|
.sort((a, b) => b[1] - a[1]);
|
|
}
|
|
|
|
count_by_kind(): Array<[SymbolKind, number]> {
|
|
const counts: Record<string, number> = {};
|
|
for (const sym of this.symbols) {
|
|
counts[sym.kind] = (counts[sym.kind] ?? 0) + 1;
|
|
}
|
|
return Object.entries(counts)
|
|
.map(([kind, c]) => [kind as SymbolKind, c] as [SymbolKind, number])
|
|
.sort((a, b) => b[1] - a[1]);
|
|
}
|
|
}
|
|
|
|
// ── Index management ─────────────────────────────────────────────────────
|
|
/** Ensure a per-workspace index is built. */
|
|
export async function ensureSymbolIndex(workspace: string, force: boolean): Promise<number> {
|
|
const existing = symbolIndexStore[workspace];
|
|
if (!force && existing && !existing.is_empty()) {
|
|
return existing.len();
|
|
}
|
|
// Rebuild outside the global lock; dedupe concurrent rebuilds for same workspace.
|
|
if (!pendingRebuilds[workspace]) {
|
|
pendingRebuilds[workspace] = (async () => {
|
|
const fresh = new SymbolIndex();
|
|
const count = await fresh.rebuild(workspace);
|
|
const cur = symbolIndexStore[workspace];
|
|
if (!cur || cur.is_empty()) {
|
|
symbolIndexStore[workspace] = fresh;
|
|
}
|
|
return count;
|
|
})();
|
|
}
|
|
const count = await pendingRebuilds[workspace];
|
|
delete pendingRebuilds[workspace];
|
|
return count;
|
|
}
|
|
|
|
function getIndex(workspace: string): SymbolIndex {
|
|
let idx = symbolIndexStore[workspace];
|
|
if (!idx) {
|
|
idx = new SymbolIndex();
|
|
symbolIndexStore[workspace] = idx;
|
|
}
|
|
return idx;
|
|
}
|
|
|
|
// ── Scoring ──────────────────────────────────────────────────────────────
|
|
function scoreSymbol(sym: CodeSymbol, queryLower: string, queryWords: string[]): number {
|
|
const nameLower = sym.name.toLowerCase();
|
|
let score = 0;
|
|
if (nameLower === queryLower) score += 1000;
|
|
if (nameLower.startsWith(queryLower)) score += 500;
|
|
if (nameLower.includes(queryLower)) score += 200;
|
|
for (const word of queryWords) {
|
|
if (nameLower.includes(word)) score += 50;
|
|
}
|
|
const doc = sym.doc_comment ? sym.doc_comment.toLowerCase() : "";
|
|
if (doc) {
|
|
if (doc.includes(queryLower)) score += 30;
|
|
for (const word of queryWords) {
|
|
if (doc.includes(word)) score += 10;
|
|
}
|
|
}
|
|
const contextLower = sym.context.toLowerCase();
|
|
if (contextLower.includes(queryLower)) score += 20;
|
|
return score;
|
|
}
|
|
|
|
// ── Doc comment extraction ───────────────────────────────────────────────
|
|
function findNextDeclarationLine(lines: string[], start: number): number | null {
|
|
for (let i = start; i < lines.length; i++) {
|
|
const trimmed = lines[i]!.trim();
|
|
if (
|
|
trimmed !== "" &&
|
|
!trimmed.startsWith("///") &&
|
|
!trimmed.startsWith("//!") &&
|
|
!trimmed.startsWith("#")
|
|
) {
|
|
return i;
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
function extractDocComments(lines: string[]): Map<number, string> {
|
|
const map = new Map<number, string>();
|
|
let i = 0;
|
|
while (i < lines.length) {
|
|
const line = lines[i]!.trim();
|
|
if (line.startsWith("///")) {
|
|
const parts: string[] = [];
|
|
while (i < lines.length) {
|
|
const l = lines[i]!.trim();
|
|
if (l.startsWith("///")) {
|
|
parts.push(l.replace(/^\/\/\//, "").trim());
|
|
i += 1;
|
|
} else {
|
|
break;
|
|
}
|
|
}
|
|
if (parts.length > 0) {
|
|
const target = findNextDeclarationLine(lines, i);
|
|
if (target !== null) map.set(target + 1, parts.join(" "));
|
|
}
|
|
} else {
|
|
i += 1;
|
|
}
|
|
}
|
|
return map;
|
|
}
|
|
|
|
// ── Rust extractor ───────────────────────────────────────────────────────
|
|
export function extractRust(content: string, relPath: string): CodeSymbol[] {
|
|
const symbols: CodeSymbol[] = [];
|
|
const lines = content.split("\n");
|
|
const docComments = extractDocComments(lines);
|
|
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const lineNum = i + 1;
|
|
const trimmed = lines[i]!.trim();
|
|
const entries: Array<[SymbolKind, RegExp]> = [
|
|
["fn", RX.rust_fn],
|
|
["struct", RX.rust_struct],
|
|
["enum", RX.rust_enum],
|
|
["trait", RX.rust_trait],
|
|
["mod", RX.rust_mod],
|
|
["type", RX.rust_type],
|
|
["const", RX.rust_const],
|
|
["macro", RX.rust_macro],
|
|
];
|
|
for (const [kind, re] of entries) {
|
|
const m = re.exec(trimmed);
|
|
if (m && m[1]) {
|
|
symbols.push({
|
|
name: m[1],
|
|
kind,
|
|
language: "rust",
|
|
file: relPath,
|
|
line: lineNum,
|
|
parent: null,
|
|
doc_comment: docComments.get(lineNum) ?? null,
|
|
context: trimmed,
|
|
});
|
|
}
|
|
}
|
|
|
|
// Parse impl blocks for methods
|
|
const implMatch = RX.rust_impl.exec(trimmed);
|
|
if (implMatch && implMatch[1]) {
|
|
const implFor = implMatch[1];
|
|
let braceDepth = 0;
|
|
let started = false;
|
|
for (let j = 0; i + j < lines.length; j++) {
|
|
const l = lines[i + j]!;
|
|
for (const ch of l) {
|
|
if (ch === "{") {
|
|
braceDepth += 1;
|
|
started = true;
|
|
} else if (ch === "}") {
|
|
braceDepth -= 1;
|
|
}
|
|
}
|
|
if (started && braceDepth <= 0 && j > 1) break;
|
|
if (j > 0) {
|
|
const inner = l.trim();
|
|
const fnMatch = RX.rust_fn.exec(inner);
|
|
if (fnMatch && fnMatch[1]) {
|
|
const absLine = i + j + 1;
|
|
symbols.push({
|
|
name: `${implFor}::${fnMatch[1]}`,
|
|
kind: "fn",
|
|
language: "rust",
|
|
file: relPath,
|
|
line: absLine,
|
|
parent: implFor,
|
|
doc_comment: docComments.get(absLine) ?? null,
|
|
context: inner,
|
|
});
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
return symbols;
|
|
}
|
|
|
|
// ── TypeScript extractor ─────────────────────────────────────────────────
|
|
function extractTsDoc(lines: string[]): Map<number, string> {
|
|
const map = new Map<number, string>();
|
|
let i = 0;
|
|
const collect = (lookahead: boolean): boolean => {
|
|
void lookahead;
|
|
return true;
|
|
};
|
|
void collect;
|
|
while (i < lines.length) {
|
|
const line = lines[i]!.trim();
|
|
if (line.startsWith("/**") || line.startsWith("///")) {
|
|
const parts: string[] = [];
|
|
if (line.startsWith("/**")) {
|
|
if (line.endsWith("*/") && line.length > 4) {
|
|
const content = line.replace(/^\/\*\*/, "").replace(/\*\/$/, "").trim();
|
|
if (content) parts.push(content);
|
|
} else {
|
|
while (i < lines.length) {
|
|
let l = lines[i]!.trim().replace(/^\*/, "").trim();
|
|
if (l.endsWith("*/")) {
|
|
parts.push(l.replace(/\*\/$/, "").trim());
|
|
break;
|
|
}
|
|
if (l) parts.push(l);
|
|
i += 1;
|
|
}
|
|
}
|
|
} else {
|
|
while (i < lines.length) {
|
|
const l = lines[i]!.trim();
|
|
if (l.startsWith("///")) {
|
|
parts.push(l.replace(/^\/\/\//, "").trim());
|
|
i += 1;
|
|
} else {
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
const target = findNextDeclarationLine(lines, i);
|
|
if (target !== null) map.set(target + 1, parts.join(" "));
|
|
} else {
|
|
i += 1;
|
|
}
|
|
}
|
|
return map;
|
|
}
|
|
|
|
export function extractTypescript(content: string, relPath: string): CodeSymbol[] {
|
|
const symbols: CodeSymbol[] = [];
|
|
const lines = content.split("\n");
|
|
const docComments = extractTsDoc(lines);
|
|
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const lineNum = i + 1;
|
|
const trimmed = lines[i]!.trim();
|
|
const doc = docComments.get(lineNum) ?? null;
|
|
|
|
const fnMatch = RX.ts_fn.exec(trimmed);
|
|
if (fnMatch && fnMatch[1]) {
|
|
const name = fnMatch[1];
|
|
if (!name.startsWith("(") && name !== "function" && name !== "async") {
|
|
symbols.push({ name, kind: "fn", language: "typescript", file: relPath, line: lineNum, parent: null, doc_comment: doc, context: trimmed });
|
|
}
|
|
continue;
|
|
}
|
|
const classMatch = RX.ts_class.exec(trimmed);
|
|
if (classMatch && classMatch[1]) {
|
|
symbols.push({ name: classMatch[1], kind: "class", language: "typescript", file: relPath, line: lineNum, parent: null, doc_comment: doc, context: trimmed });
|
|
continue;
|
|
}
|
|
const ifaceMatch = RX.ts_interface.exec(trimmed);
|
|
if (ifaceMatch && ifaceMatch[1]) {
|
|
symbols.push({ name: ifaceMatch[1], kind: "interface", language: "typescript", file: relPath, line: lineNum, parent: null, doc_comment: doc, context: trimmed });
|
|
continue;
|
|
}
|
|
const typeMatch = RX.ts_type.exec(trimmed);
|
|
if (typeMatch && typeMatch[1]) {
|
|
symbols.push({ name: typeMatch[1], kind: "type", language: "typescript", file: relPath, line: lineNum, parent: null, doc_comment: doc, context: trimmed });
|
|
continue;
|
|
}
|
|
const enumMatch = RX.ts_enum.exec(trimmed);
|
|
if (enumMatch && enumMatch[1]) {
|
|
symbols.push({ name: enumMatch[1], kind: "enum", language: "typescript", file: relPath, line: lineNum, parent: null, doc_comment: doc, context: trimmed });
|
|
continue;
|
|
}
|
|
const varMatch = RX.ts_var.exec(trimmed);
|
|
if (varMatch && varMatch[1]) {
|
|
const name = varMatch[1];
|
|
const isTopLevel = !/^\s/.test(lines[i]!) || trimmed.startsWith("export");
|
|
if (isTopLevel) {
|
|
const kind: SymbolKind = trimmed.includes("const ") ? "const" : "var";
|
|
symbols.push({ name, kind, language: "typescript", file: relPath, line: lineNum, parent: null, doc_comment: doc, context: trimmed });
|
|
}
|
|
}
|
|
}
|
|
return symbols;
|
|
}
|
|
|
|
// ── JavaScript extractor ─────────────────────────────────────────────────
|
|
export function extractJavascript(content: string, relPath: string): CodeSymbol[] {
|
|
const symbols: CodeSymbol[] = [];
|
|
const lines = content.split("\n");
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const lineNum = i + 1;
|
|
const trimmed = lines[i]!.trim();
|
|
|
|
const fnMatch = RX.ts_fn.exec(trimmed);
|
|
if (fnMatch && fnMatch[1]) {
|
|
const name = fnMatch[1];
|
|
if (!name.startsWith("(") && name !== "function" && name !== "async") {
|
|
symbols.push({ name, kind: "fn", language: "javascript", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
}
|
|
continue;
|
|
}
|
|
const classMatch = RX.ts_class.exec(trimmed);
|
|
if (classMatch && classMatch[1]) {
|
|
symbols.push({ name: classMatch[1], kind: "class", language: "javascript", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const varMatch = RX.ts_var.exec(trimmed);
|
|
if (varMatch && varMatch[1]) {
|
|
const name = varMatch[1];
|
|
if (!/^\s/.test(lines[i]!)) {
|
|
const kind: SymbolKind = trimmed.includes("const ") ? "const" : "var";
|
|
symbols.push({ name, kind, language: "javascript", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
}
|
|
}
|
|
}
|
|
return symbols;
|
|
}
|
|
|
|
// ── Python extractor ─────────────────────────────────────────────────────
|
|
export function extractPython(content: string, relPath: string): CodeSymbol[] {
|
|
const symbols: CodeSymbol[] = [];
|
|
const lines = content.split("\n");
|
|
let currentClass: string | null = null;
|
|
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const lineNum = i + 1;
|
|
const line = lines[i]!;
|
|
const trimmed = line.trim();
|
|
const indent = line.length - trimmed.length;
|
|
|
|
if (currentClass !== null && indent === 0 && trimmed !== "") {
|
|
currentClass = null;
|
|
}
|
|
const classMatch = RX.py_class.exec(trimmed);
|
|
if (classMatch && classMatch[1]) {
|
|
const name = classMatch[1];
|
|
currentClass = name;
|
|
symbols.push({ name, kind: "class", language: "python", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const asyncMatch = RX.py_async_def.exec(trimmed);
|
|
if (asyncMatch && asyncMatch[1]) {
|
|
const name = asyncMatch[1];
|
|
const full = currentClass ? `${currentClass}.${name}` : name;
|
|
symbols.push({ name: full, kind: "fn", language: "python", file: relPath, line: lineNum, parent: currentClass, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const defMatch = RX.py_def.exec(trimmed);
|
|
if (defMatch && defMatch[1]) {
|
|
const name = defMatch[1];
|
|
const full = currentClass ? `${currentClass}.${name}` : name;
|
|
symbols.push({ name: full, kind: "fn", language: "python", file: relPath, line: lineNum, parent: currentClass, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
|
|
// Module-level variable assignment
|
|
if (
|
|
!/^\s/.test(line) &&
|
|
!trimmed.startsWith("#") &&
|
|
!trimmed.startsWith("def ") &&
|
|
!trimmed.startsWith("class ") &&
|
|
!trimmed.startsWith("import ") &&
|
|
!trimmed.startsWith("from ") &&
|
|
!trimmed.startsWith("@") &&
|
|
!trimmed.startsWith("return") &&
|
|
!trimmed.startsWith("if ") &&
|
|
!trimmed.startsWith("elif ") &&
|
|
!trimmed.startsWith("else:") &&
|
|
!trimmed.startsWith("for ") &&
|
|
!trimmed.startsWith("while ") &&
|
|
!trimmed.startsWith("try:") &&
|
|
!trimmed.startsWith("except") &&
|
|
!trimmed.startsWith("with ") &&
|
|
!trimmed.startsWith("raise") &&
|
|
!trimmed.startsWith("pass") &&
|
|
!trimmed.startsWith("self.") &&
|
|
!trimmed.startsWith("cls.") &&
|
|
trimmed.includes(" = ") &&
|
|
!trimmed.includes("==")
|
|
) {
|
|
const name = trimmed.split("=")[0]!.trim();
|
|
if (name !== "" && !name.startsWith("_") && !name.includes(" ")) {
|
|
const kind: SymbolKind = /^[A-Z_]+$/.test(name) ? "const" : "var";
|
|
symbols.push({ name, kind, language: "python", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
}
|
|
}
|
|
}
|
|
return symbols;
|
|
}
|
|
|
|
// ── Go extractor ─────────────────────────────────────────────────────────
|
|
export function extractGo(content: string, relPath: string): CodeSymbol[] {
|
|
const symbols: CodeSymbol[] = [];
|
|
const lines = content.split("\n");
|
|
for (let i = 0; i < lines.length; i++) {
|
|
const lineNum = i + 1;
|
|
const trimmed = lines[i]!.trim();
|
|
|
|
const structMatch = RX.go_struct.exec(trimmed);
|
|
if (structMatch && structMatch[1]) {
|
|
symbols.push({ name: structMatch[1], kind: "struct", language: "go", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const ifaceMatch = RX.go_interface.exec(trimmed);
|
|
if (ifaceMatch && ifaceMatch[1]) {
|
|
symbols.push({ name: ifaceMatch[1], kind: "interface", language: "go", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const typeMatch = RX.go_type.exec(trimmed);
|
|
if (typeMatch && typeMatch[1]) {
|
|
const name = typeMatch[1];
|
|
if (!trimmed.includes(" struct") && !trimmed.includes(" interface")) {
|
|
symbols.push({ name, kind: "type", language: "go", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
}
|
|
continue;
|
|
}
|
|
const funcMatch = RX.go_func.exec(trimmed);
|
|
if (funcMatch && funcMatch[1]) {
|
|
const name = funcMatch[1];
|
|
symbols.push({ name, kind: "fn", language: "go", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const constMatch = RX.go_const.exec(trimmed);
|
|
if (constMatch && constMatch[1]) {
|
|
symbols.push({ name: constMatch[1], kind: "const", language: "go", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
continue;
|
|
}
|
|
const varMatch = RX.go_var.exec(trimmed);
|
|
if (varMatch && varMatch[1]) {
|
|
symbols.push({ name: varMatch[1], kind: "var", language: "go", file: relPath, line: lineNum, parent: null, doc_comment: null, context: trimmed });
|
|
}
|
|
}
|
|
return symbols;
|
|
}
|
|
|
|
// ── Format listing ───────────────────────────────────────────────────────
|
|
export function formatSymbolListing(index: SymbolIndex): string {
|
|
let out = `## Indexed Symbols (${index.len()} total)\n\n`;
|
|
const byLang = index.count_by_language();
|
|
if (byLang.length === 0) {
|
|
out += "_No symbols indexed. Rebuild the index first._\n";
|
|
return out;
|
|
}
|
|
const byLangFile: Record<string, Record<string, CodeSymbol[]>> = {};
|
|
for (const sym of index.symbols) {
|
|
(byLangFile[sym.language] ??= {})[sym.file] ??= [];
|
|
byLangFile[sym.language]![sym.file]!.push(sym);
|
|
}
|
|
for (const [lang, files] of Object.entries(byLangFile).sort()) {
|
|
const count = Object.values(files).reduce((n, v) => n + v.length, 0);
|
|
out += `### ${lang} (${count})\n`;
|
|
for (const [file, syms] of Object.entries(files).sort()) {
|
|
out += ` ${file}\n`;
|
|
const byKind: Record<string, string[]> = {};
|
|
for (const sym of syms) {
|
|
(byKind[sym.kind] ??= []).push(sym.name);
|
|
}
|
|
for (const [kind, names] of Object.entries(byKind).sort()) {
|
|
out += ` ${kind}: ${names.join(", ")}\n`;
|
|
}
|
|
}
|
|
out += "\n";
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// ── Language/kind mappers ────────────────────────────────────────────────
|
|
function mapLanguage(s: string): Language | null {
|
|
switch (s) {
|
|
case "rust": return "rust";
|
|
case "typescript": return "typescript";
|
|
case "javascript": return "javascript";
|
|
case "python": return "python";
|
|
case "go": return "go";
|
|
default: return null;
|
|
}
|
|
}
|
|
function mapKind(s: string): SymbolKind | null {
|
|
switch (s) {
|
|
case "fn": return "fn";
|
|
case "class": return "class";
|
|
case "struct": return "struct";
|
|
case "enum": return "enum";
|
|
case "interface": return "interface";
|
|
case "trait": return "trait";
|
|
case "const": return "const";
|
|
case "var": return "var";
|
|
case "mod": return "mod";
|
|
default: return null;
|
|
}
|
|
}
|
|
|
|
// ── Tools ────────────────────────────────────────────────────────────────
|
|
|
|
function workspaceOf(ctx: ToolCtx): string {
|
|
return ctx.workspaces[0] ?? ".";
|
|
}
|
|
|
|
/** Search for code symbols by name or concept. */
|
|
export class SemanticSearch implements Tool {
|
|
name = "semantic_search";
|
|
description = "Search for code symbols (functions, structs, classes, interfaces, variables) by name, concept, or meaning across Rust, TypeScript, JavaScript, Python, and Go";
|
|
|
|
parameters = {
|
|
type: "object",
|
|
properties: {
|
|
query: { type: "string", description: "Search query — symbol name, concept, or meaning" },
|
|
kind: { type: "string", enum: ["fn", "class", "struct", "enum", "interface", "trait", "const", "var", "mod", "all"], default: "all", description: "Filter by symbol kind" },
|
|
language: { type: "string", enum: ["rust", "typescript", "javascript", "python", "go", "all"], default: "all", description: "Filter by language" },
|
|
max_results: { type: "integer", default: 10, description: "Maximum results (default 10, max 30)" },
|
|
rebuild_index: { type: "boolean", default: false, description: "Force rebuild the symbol index before searching" },
|
|
},
|
|
required: ["query"],
|
|
};
|
|
|
|
async run(ctx: ToolCtx, args: JsonValue): Promise<string> {
|
|
const query = argStr(args, "query");
|
|
const kindFilter = optStr(args, "kind") ?? "all";
|
|
const langFilter = optStr(args, "language") ?? "all";
|
|
const maxResults = Math.min(optInt(args, "max_results", 10), 30);
|
|
const rebuild = optBool(args, "rebuild_index", false);
|
|
|
|
const workspace = workspaceOf(ctx);
|
|
await ensureSymbolIndex(workspace, rebuild);
|
|
const index = getIndex(workspace);
|
|
|
|
const targetKind = mapKind(kindFilter);
|
|
const targetLang = mapLanguage(langFilter);
|
|
|
|
const results = index.search(query, maxResults * 2);
|
|
const filtered = results
|
|
.filter((s) => (targetKind === null || s.kind === targetKind))
|
|
.filter((s) => (targetLang === null || s.language === targetLang))
|
|
.slice(0, maxResults);
|
|
|
|
if (filtered.length === 0) {
|
|
return `No symbols found matching '${query}'.\nTry a different query, or use \`rebuild_index: true\` to rebuild the index first.\nIndex has ${index.len()} symbols across ${index.count_by_language().length} languages.`;
|
|
}
|
|
|
|
const total = index.len();
|
|
const byFile: Record<string, CodeSymbol[]> = {};
|
|
for (const sym of filtered) {
|
|
(byFile[sym.file] ??= []).push(sym);
|
|
}
|
|
|
|
let output = `## Semantic Search Results\n\n**Query:** ${query}\n**Index size:** ${total} symbols\n**Matches:** ${filtered.length}\n\n`;
|
|
for (const [file, symbols] of Object.entries(byFile).sort()) {
|
|
output += `### \`${file}\`\n\n`;
|
|
for (const sym of symbols) {
|
|
const parentStr = sym.parent ? ` [${sym.parent}]` : "";
|
|
const docStr = sym.doc_comment ? ` — ${sym.doc_comment.slice(0, 100)}` : "";
|
|
output += `- \`${sym.kind}\` **${sym.name}**${parentStr} \`[${sym.language}]\` at line ${sym.line} \`${sym.context.trim()}\`${docStr}${sym.context.trim().length > 80 ? "…" : ""}\n`;
|
|
}
|
|
output += "\n";
|
|
}
|
|
output += `---\n*${total} symbols indexed across ${index.count_by_language().length} languages. Use \`rebuild_index: true\` to refresh.*\n`;
|
|
return output;
|
|
}
|
|
}
|
|
|
|
/** Rebuild the code symbol index. */
|
|
export class RebuildIndex implements Tool {
|
|
name = "rebuild_index";
|
|
description = "Rebuild the code symbol index for semantic search (supports Rust, TypeScript, JavaScript, Python, Go)";
|
|
parameters = { type: "object", properties: {} };
|
|
|
|
async run(ctx: ToolCtx, _args: JsonValue): Promise<string> {
|
|
const workspace = workspaceOf(ctx);
|
|
const count = await ensureSymbolIndex(workspace, true);
|
|
const index = getIndex(workspace);
|
|
const byLang = index.count_by_language();
|
|
let out = `Symbol index rebuilt successfully. ${count} symbols indexed.\n\nBy language:\n`;
|
|
for (const [lang, c] of byLang) out += ` ${lang}: ${c}\n`;
|
|
return out;
|
|
}
|
|
}
|
|
|
|
/** List all indexed symbols. */
|
|
export class ListSymbols implements Tool {
|
|
name = "list_symbols";
|
|
description = "List all indexed code symbols across Rust, TypeScript, JavaScript, Python, and Go. Optionally filter by language, kind, or file path.";
|
|
|
|
parameters = {
|
|
type: "object",
|
|
properties: {
|
|
language: { type: "string", enum: ["rust", "typescript", "javascript", "python", "go", "all"], default: "all" },
|
|
kind: { type: "string", enum: ["fn", "class", "struct", "enum", "interface", "trait", "const", "var", "mod", "all"], default: "all" },
|
|
file: { type: "string", description: "Filter by file path substring" },
|
|
max_results: { type: "integer", default: 50, description: "Maximum symbols to list (default 50, max 200)" },
|
|
rebuild_index: { type: "boolean", default: false },
|
|
},
|
|
};
|
|
|
|
async run(ctx: ToolCtx, args: JsonValue): Promise<string> {
|
|
const langFilter = optStr(args, "language") ?? "all";
|
|
const kindFilter = optStr(args, "kind") ?? "all";
|
|
const fileFilter = optStr(args, "file") ?? null;
|
|
const maxResults = Math.min(optInt(args, "max_results", 50), 200);
|
|
const rebuild = optBool(args, "rebuild_index", false);
|
|
|
|
const workspace = workspaceOf(ctx);
|
|
await ensureSymbolIndex(workspace, rebuild);
|
|
const index = getIndex(workspace);
|
|
|
|
const targetLang = mapLanguage(langFilter);
|
|
const targetKind = mapKind(kindFilter);
|
|
const symbols = index.list(targetLang, targetKind, fileFilter, maxResults);
|
|
|
|
const total = index.len();
|
|
const byLang = index.count_by_language();
|
|
|
|
if (symbols.length === 0) {
|
|
return `No symbols match the filters. Index has ${total} total symbols.\nLanguages: ${byLang.map(([l, c]) => `${l}: ${c}`).join(", ")}`;
|
|
}
|
|
|
|
let out = `## Indexed Symbols\n\n**Total:** ${total} | **Showing:** ${symbols.length} | **Filter:** lang=${langFilter}, kind=${kindFilter}\n\n`;
|
|
const byLangMap: Record<string, CodeSymbol[]> = {};
|
|
for (const sym of symbols) (byLangMap[sym.language] ??= []).push(sym);
|
|
|
|
for (const [lang, syms] of Object.entries(byLangMap).sort()) {
|
|
out += `### ${lang}\n\n`;
|
|
const byFile: Record<string, CodeSymbol[]> = {};
|
|
for (const sym of syms) (byFile[sym.file] ??= []).push(sym);
|
|
for (const [file, fileSyms] of Object.entries(byFile).sort()) {
|
|
out += `\`${file}\`:\n`;
|
|
for (const sym of fileSyms) {
|
|
out += ` \`${sym.kind}\` ${sym.name} L${sym.line}\n`;
|
|
}
|
|
}
|
|
out += "\n";
|
|
}
|
|
out += "---\n";
|
|
out += `By language: ${byLang.map(([l, c]) => `${l}: ${c}`).join(", ")}\n`;
|
|
out += `By kind: ${index.count_by_kind().map(([k, c]) => `${k}: ${c}`).join(", ")}\n`;
|
|
return out;
|
|
}
|
|
}
|