diff --git a/src/features/agent/application/turn_service.test.ts b/src/features/agent/application/turn_service.test.ts index 2aab64e..ac62714 100644 --- a/src/features/agent/application/turn_service.test.ts +++ b/src/features/agent/application/turn_service.test.ts @@ -4,8 +4,22 @@ import { conversationChars, ErrorTracker, truncateToolOutput, + bashExitCode, + isToolFailure, + deriveVerifyCommand, + AgentTurnServiceImpl, } from "./turn_service.ts"; -import { type ChatMessage, newConversation, systemMessage, userMessage, toolResultMessage } from "@zesdex/domain"; +import { + type ChatMessage, + newConversation, + systemMessage, + userMessage, + assistantMessage, + toolResultMessage, +} from "@zesdex/domain"; +import type { AgentTurnParams } from "@zesdex/agent"; +import type { ToolExecutor } from "./index.ts"; +import type { ProviderService } from "./ports.ts"; describe("truncateToolOutput", () => { it("short output is unchanged", () => { @@ -56,4 +70,197 @@ describe("conversationChars", () => { conv.messages.push(systemMessage("sys"), userMessage("hello world"), toolResultMessage("id", "output")); expect(conversationChars(conv.messages)).toBe(3 + 11 + 6); }); -}); \ No newline at end of file +}); +/* ── New flow helpers ─────────────────────────────────────────────── */ + +describe("bashExitCode / isToolFailure", () => { + it("extracts a non-zero exit code", () => { + expect(bashExitCode("boom\n\nExit code: 1 (1s)")).toBe(1); + expect(bashExitCode("done\n\nExit code: 0 (0.5s)")).toBe(0); + }); + it("returns null when no exit-code line exists", () => { + expect(bashExitCode("plain output")).toBeNull(); + }); + it("isToolFailure flags bash non-zero exits as failures", () => { + expect(isToolFailure("bash", "nope\n\nExit code: 2 (1s)")).toBe(true); + expect(isToolFailure("bash", "ok\n\nExit code: 0 (1s)")).toBe(false); + }); + it("isToolFailure still flags Error: prefixes for other tools", () => { + expect(isToolFailure("read", "Error: no such file")).toBe(true); + }); +}); + +describe("deriveVerifyCommand", () => { + it("defaults to bun check+test for a Bun repo", () => { + // No real manifests in a control dir we guarantee to not exist. + expect(deriveVerifyCommand("/nonexistent-zesdex-dir")).toContain("bun"); + }); +}); + +/* ── runTurn flow tests (fakes, no network) ───────────────────────── */ + +interface ScriptedStep { + content?: string | null; + tools?: Array<{ name: string; args?: string; id?: string }>; +} + +/** A ProviderService that replays scripted chatStream responses. */ +function makeScriptedProvider(script: ScriptedStep[]): { + provider: ProviderService; + seen: Array; +} { + const seen: Array = []; + let step = 0; + const chatResponses: string[] = []; + const provider: ProviderService = { + async chat(_messages, _tools, _maxTokens, _temperature) { + // Used by compaction and review. For review tests we script below. + const text = chatResponses.shift() ?? ""; + return { message: { role: "assistant", content: text || null }, usage: [10, 5] }; + }, + async chatStream(messages, _tools, _max, _temp, onEvent) { + // Snapshot a copy: runTurn freely mutates the live array (splice/compact). + seen.push([...messages]); + const s = script[Math.min(step, script.length - 1)] ?? { content: "done" }; + step += 1; + if (s.content !== undefined && s.content !== null) onEvent({ kind: "token", content: s.content }); + if (s.tools && s.tools.length > 0) { + const msg = assistantMessage(null) as ChatMessage; + msg.tool_calls = s.tools.map((t, i) => ({ + id: t.id ?? `call-${i}`, + type: "function", + function: { name: t.name, arguments: t.args ?? "{}" }, + })); + return { message: msg, usage: [10, 2] }; + } + return { + message: { role: "assistant", content: s.content ?? null }, + usage: [10, 2], + }; + }, + // expose a way for tests to script chat replies + } as ProviderService; + (provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => { + chatResponses.push(text); + }; + return { provider, seen }; +} + +/** A ToolExecutor that echoes fixed outputs per tool. */ +function fixedExecutor(outs: Record): ToolExecutor { + return { + async execute(name) { + return outs[name] ?? "ok"; + }, + isParallelSafe() { + return false; + }, + }; +} + +function buildParams(userText: string): { + params: AgentTurnParams; + events: Array; +} { + const events: Array = []; + const params: AgentTurnParams = { + messages: [userMessage(userText)], + session_dir: "/tmp/zesdex-flow-test", + workspace_roots: ["/nonexistent-zesdex-dir"], // no AGENTS.md/package.json side effects + turn_events: { push: (e) => events.push(e), drain: () => [] }, + in_flight: { value: false }, + abort: new AbortController(), + api_key: "k", + model: "m", + api_base: "https://x", + }; + return { params, events }; +} + +/** Extract the `[Flow]` steering system message from a captured batch. */ +function flowDirective(batch: ChatMessage[]): string { + const m = batch.find((x) => (x.content ?? "").includes("[Flow]")); + return m?.content ?? ""; +} + +describe("runTurn complexity steering", () => { + it("injects a complex-request directive for a complex prompt", async () => { + const { provider, seen } = makeScriptedProvider([ + { content: "final answer" }, + ]); + const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never); + const { params } = buildParams("Please refactor the architecture across multiple files"); + await svc.runTurn(params); + // The complexity directive system message must be present before the LLM. + const flow = flowDirective(seen[0] ?? []); + expect(flow).toContain("flagged as complex"); + }); + + it("injects a keep-it-simple directive instead", async () => { + const { provider, seen } = makeScriptedProvider([{ content: "hi" }]); + const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never); + const { params } = buildParams("what is 2+2"); + await svc.runTurn(params); + const flow = flowDirective(seen[0] ?? []); + expect(flow).toContain("looks simple"); + expect(flow).not.toContain("complex"); + }); +}); + +describe("runTurn verify-after-edit", () => { + it("injects a verify nudge after a write, cleared after a successful bash", async () => { + const { provider, seen } = makeScriptedProvider([ + { tools: [{ name: "write" }] }, + { tools: [{ name: "bash" }] }, + { content: "done" }, + ]); + const executor = fixedExecutor({ write: "wrote it", bash: "ok\n\nExit code: 0 (1s)" }); + const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never); + const { params } = buildParams("add a comment"); + await svc.runTurn(params); + // The LLM call after the write should carry the verify nudge. + const second = seen[1] ?? []; + expect(second.some((m) => (m.content ?? "").includes("Run the verify command"))).toBe(true); + // The successful bash exits 0, which clears pendingVerify — so the nudge is + // injected exactly once, not repeated on every later call. + const finalBatch = seen[seen.length - 1] ?? []; + const nudgeCount = finalBatch.filter((m) => (m.content ?? "").includes("Run the verify command")).length; + expect(nudgeCount).toBe(1); + }); +}); + +describe("runTurn convergence guard", () => { + it("stops after repeated identical read calls", async () => { + const script: ScriptedStep[] = []; + for (let i = 0; i < 6; i++) script.push({ tools: [{ name: "read" }] }); + const { provider, seen } = makeScriptedProvider(script); + const executor = fixedExecutor({ read: "same content" }); + const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never); + const { params, events } = buildParams("read something"); + await svc.runTurn(params); + const warned = events.some((e) => e.kind === "system_note" && /without any progress|repeated/.test(e.message ?? "")); + expect(warned).toBe(true); + // Not every call got issued — the guard broke early. + expect(seen.length).toBeLessThan(script.length); + }); +}); + +describe("runTurn self-review", () => { + it("feeds a reviewer critique back as a system message after a mutator", async () => { + const { provider, seen } = makeScriptedProvider([ + { tools: [{ name: "edit" }] }, + { content: "fixed" }, + ]); + const withChat = provider as unknown as { setChatReply(s: string): void }; + withChat.setChatReply("- [PRIORITY: high] handle empty input in parse()"); + const executor = fixedExecutor({ edit: "edited" }); + const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never); + const { params, events } = buildParams("fix parse()"); + await svc.runTurn(params); + // The reviewer critique must have reached the next LLM call's history. + const second = seen[1] ?? []; + expect(second.some((m) => (m.content ?? "").includes("[Reviewer]"))).toBe(true); + // review_usage emitted. + expect(events.some((e) => e.kind === "review_usage")).toBe(true); + }); +}); diff --git a/src/features/agent/application/turn_service.ts b/src/features/agent/application/turn_service.ts index 2d625e1..822e5c7 100644 --- a/src/features/agent/application/turn_service.ts +++ b/src/features/agent/application/turn_service.ts @@ -20,7 +20,9 @@ import { mainAgentPromptWithProjectContext, compactionPrompt, errorRecoveryNote, + reviewerPrompt, } from "@zesdex/agent"; +import { isComplexRequest } from "@zesdex/workflow"; import type { ProviderService } from "./ports.ts"; import type { ToolExecutor } from "./index.ts"; @@ -38,11 +40,80 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000; const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"]; const COMPACT_KEEP_TAIL = 6; +/** Tools that mutate the filesystem — after these, a verify run is expected. */ +const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]); + +/** Consecutive identical, non-progressing tool iterations before the loop stops. */ +const MAX_NO_PROGRESS_STREAK = 4; + +/** Self-review is bounded to this many passes per turn. */ +const MAX_REVIEW_PASSES = 1; + /** Whether the output string denotes a tool error. */ function isErrorOutput(output: string): boolean { return output.startsWith("Error:"); } +/** Extract the numeric exit code from a `bash` tool result, or null if not a failure (0). */ +export function bashExitCode(output: string): number | null { + const match = output.match(/Exit code:\s*(\d+)/); + if (!match) return null; + const code = Number(match[1]); + return Number.isInteger(code) && code >= 0 ? code : null; +} + +/** + * Whether a tool result is a real failure. `Error:` prefixes cover most tools; + * a `bash` command that exits non-zero returns an `Exit code: N` line instead. + */ +export function isToolFailure(toolName: string, output: string): boolean { + if (isErrorOutput(output)) return true; + if (toolName === "bash" || toolName === "bash_output") { + const code = bashExitCode(output); + if (code === null) return false; // no exit-code line → no signal + return code !== 0; + } + return false; +} + +/** + * Best-effort derivation of the repository's verify command from convention + * manifests. Defaults to a safe lint+test invocation for Bun. + */ +export function deriveVerifyCommand(root: string): string { + const fs = require("node:fs"); + // Bun project: prefer an explicit `check` (typecheck) script then test. + try { + const pkg = JSON.parse(fs.readFileSync(`${root}/package.json`, "utf8")) as { + scripts?: Record; + }; + const s = pkg?.scripts ?? {}; + const parts: string[] = []; + if (s["check"] && typeof s["check"] === "string") parts.push(`bun run check`); + else if (s["typecheck"] && typeof s["typecheck"] === "string") parts.push(`bun run typecheck`); + if (s["lint"] && typeof s["lint"] === "string") parts.push(`bun run lint`); + if (s["test"] && typeof s["test"] === "string") parts.push(`bun run test`); + if (parts.length > 0) return parts.join(" && "); + } catch { + /* no package.json — fall through */ + } + // Rust project. + try { + if (fs.existsSync(`${root}/Cargo.toml`)) return "cargo check && cargo test"; + } catch { + /* ignore */ + } + // Java/Maven. + try { + if (fs.existsSync(`${root}/pom.xml`) || fs.existsSync(`${root}/build.gradle`)) { + return "mvn test"; + } + } catch { + /* ignore */ + } + return "bun run check && bun test"; +} + /** Truncate a long tool output, preserving the head + truncation marker. */ export function truncateToolOutput(output: string): string { if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output; @@ -136,7 +207,7 @@ async function executeToolCall( output = `Error: ${(e as Error).message}`; } - const isError = isErrorOutput(output); + const isError = isToolFailure(name, output); const truncated = truncateToolOutput(output); sink.push({ @@ -268,18 +339,43 @@ export class AgentTurnServiceImpl { const abort = params.abort; const { in_flight } = params; - // Insert system prompt at index 0 with repo conventions loaded. - const projectContext = buildProjectContext(params.workspace_roots[0] ?? "."); - const systemPrompt = mainAgentPromptWithProjectContext(projectContext); + // Insert system prompt at index 0 with repo conventions + verify command. + const root = params.workspace_roots[0] ?? "."; + const projectContext = buildProjectContext(root); + const verifyCommand = deriveVerifyCommand(root); + const systemPrompt = mainAgentPromptWithProjectContext(projectContext, verifyCommand); params.messages.unshift(systemMessage(systemPrompt)); const originalCount = params.messages.length; // Estimate request complexity from the last user message. const last = params.messages[params.messages.length - 1]; const requestLen = last?.content?.length ?? 0; + const userText = last?.content ?? ""; + + // Complexity steering: a one-shot directive so the model picks the right + // depth instead of relying on prose memory. Pure and cheap. + const complex = isComplexRequest(userText); + this.push(sink, { + kind: "system_note", + systemKind: "info", + message: complex ? "Complex request detected — plan before executing." : "Simple request — keep tool use minimal.", + }); + params.messages.push( + systemMessage( + complex + ? "[Flow] This request is flagged as complex. Enter a short plan with `plan_enter` and track steps with `todowrite` before starting edits." + : "[Flow] This request looks simple. If you can answer directly without tools, do so — do not spawn agents or workflows for it.", + ), + ); const errors = new ErrorTracker(); let sawToolCalls = false; + let sawMutator = false; + let pendingVerify = false; + let verifyPrompted = false; + let noProgressStreak = 0; + let lastSignature: string | null = null; + let reviewPasses = 0; for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) { // Check abort flag. @@ -293,6 +389,26 @@ export class AgentTurnServiceImpl { break; } + // Convergence guard: bail out of a loop stuck re-issuing the same call. + if (noProgressStreak >= MAX_NO_PROGRESS_STREAK) { + this.push(sink, { + kind: "system_note", + systemKind: "warn", + message: "Stopping: the same tool call is being repeated without any progress.", + }); + break; + } + + // Verify-after-edit: nudge the model to run the check before concluding. + if (pendingVerify && !verifyPrompted) { + params.messages.push( + systemMessage( + `[Flow] You just modified files. Run the verify command via \`bash\` now (${verifyCommand}) and resolve any failures before concluding your turn.`, + ), + ); + verifyPrompted = true; + } + // Auto-compact oversized history before the LLM call. await this.autoCompactIfNeeded(params.messages); @@ -343,12 +459,44 @@ export class AgentTurnServiceImpl { return seq; })(); + let mutated = false; + let verified = false; + let batchSignature = ""; for (let i = 0; i < toolCalls.length; i++) { const tc = toolCalls[i]!; const output = outputs[i]!; - if (isErrorOutput(output)) errors.record(tc.function.name, output, params.messages); + if (isToolFailure(tc.function.name, output)) errors.record(tc.function.name, output, params.messages); + if (MUTATOR_TOOLS.has(tc.function.name)) { + mutated = true; + sawMutator = true; + } + if ((tc.function.name === "bash" || tc.function.name === "bash_output") && bashExitCode(output) === 0) { + verified = true; + } + // Batch signature: concat of tool+output for no-progress detection. + batchSignature += `${tc.function.name}${output.length}`; params.messages.push(toolResultMessage(tc.id, output)); } + + // No-progress: byte-identical batch (same tools, same output lengths) → streak. + if (toolCalls.length === 1 && batchSignature === lastSignature) noProgressStreak += 1; + else noProgressStreak = 0; + lastSignature = batchSignature; + + if (mutated) { + pendingVerify = true; + verifyPrompted = false; // allow a fresh nudge after the next edit batch + } + if (verified) { + pendingVerify = false; // a successful bash run satisfies the verify nudge + verifyPrompted = false; + } + + // Bounded self-review pass after file mutations (at most one per turn). + if (sawMutator && reviewPasses < MAX_REVIEW_PASSES && !abort.signal.aborted) { + reviewPasses += 1; + await this.reviewWork(params.messages, abort, sink); + } } // Remove the synthetic sys_msg before emitting to the transcript. @@ -357,4 +505,30 @@ export class AgentTurnServiceImpl { this.push(sink, { kind: "done" }); in_flight.value = false; } + + /** Run one bounded self-review of the work so far and feed critique back. */ + private async reviewWork( + messages: ChatMessage[], + abort: AbortController, + sink: TurnEventSink, + ): Promise { + if (abort.signal.aborted) return; + try { + const result = await this.provider.chat( + [systemMessage(reviewerPrompt())], + undefined, + 1024, + 0.3, + ); + const critique = result.message.content ?? ""; + if (result.usage) { + this.push(sink, { kind: "review_usage", tokens_in: result.usage[0], tokens_out: result.usage[1] }); + } + if (critique.trim() === "" || critique.trim().toUpperCase() === "NO ISSUES FOUND") return; + // Feed the critique back as a system message so the next iteration resolves it. + messages.push(systemMessage(`[Reviewer]\n${critique.trim()}`)); + } catch { + /* review is best-effort — non-fatal */ + } + } } \ No newline at end of file diff --git a/src/features/agent/domain/prompt.test.ts b/src/features/agent/domain/prompt.test.ts new file mode 100644 index 0000000..4561d12 --- /dev/null +++ b/src/features/agent/domain/prompt.test.ts @@ -0,0 +1,105 @@ +/** + * Tests pinning the rewritten prompts to be consistent with the REAL tool + * inventory: they must never reference the phantom `explore_codebase`, must + * list real tool names, and must carry the concrete verify command. + */ +import { describe, expect, test } from "bun:test"; +import { + mainAgentPrompt, + mainAgentPromptWithProjectContext, + subagentDirective, + compactionPrompt, + reviewerPrompt, + consensusSynthesizerPrompt, + errorRecoveryNote, +} from "./prompt.ts"; + +/** All real tool names (mirrors registry.allTools()). */ +const REAL_TOOLS = new Set([ + "read", "write", "edit", "delete", + "grep", "glob", "semantic_search", "rebuild_index", "list_symbols", + "bash", "bash_output", "bash_kill", "git_operator", "git_worktree", "git_cred", + "cd", "dir_list", "dir_cache_update", + "plan_enter", "plan_ready", "sequential_think", "todowrite", "todofinish", + "workflow_run", "hive_mind", "note_finding", "read_findings", + "spawn_agents", "spawn_pipeline", "parallel_delegate", + "remember", "forget", "recall", + "web_search", "best_practice", "commit_convention", "pong", +]); + +describe("prompt tool-consistency", () => { + test("main prompt never references non-existent explore_codebase", () => { + expect(mainAgentPrompt()).not.toContain("explore_codebase"); + }); + + test("every backticked tool token in the main prompt is a real tool", () => { + const s = mainAgentPrompt(); + const backticks = s.match(/`([a-z_]+)`/g) ?? []; + const tokens = backticks.map((t) => t.slice(1, -1)); + // `plan` and `verify` appear in backticks as prose (a command/word), not tools. + const proseWords = new Set(["plan", "verify", "check"]); + const ghosts = tokens.filter((t) => !REAL_TOOLS.has(t) && !proseWords.has(t)); + expect(ghosts).toEqual([]); + }); + + test("main prompt anchors the key real tools", () => { + const s = mainAgentPrompt(); + for (const tool of ["plan_enter", "todowrite", "workflow_run", "spawn_agents", "hive_mind", "recall", "remember"]) { + expect(s).toContain(tool); + } + }); + + test("verify command is injected as a concrete line when provided", () => { + const s = mainAgentPrompt("bun run check && bun test"); + expect(s).toContain("VERIFY COMMAND"); + expect(s).toContain("bun run check && bun test"); + }); + + test("without a verify command, no VERIFY COMMAND block is emitted", () => { + expect(mainAgentPrompt()).not.toContain("VERIFY COMMAND"); + }); + + test("project-context variant appends the context block", () => { + const s = mainAgentPromptWithProjectContext("## Rules\nno threads", "bun test"); + expect(s).toContain("## PROJECT CONTEXT"); + expect(s).toContain("no threads"); + expect(s).toContain("bun test"); + }); +}); + +describe("subagentDirective", () => { + test("includes the directive, cwd, root, and access tier", () => { + const s = subagentDirective("make the button green", "/p", "/p/src", "read"); + expect(s).toContain("make the button green"); + expect(s).toContain("/p"); + expect(s).toContain("READ-ONLY"); + }); + + test("no-tier call defaults to full access", () => { + const s = subagentDirective("do it", "/p", "/p"); + expect(s).toContain("full tool set"); + }); +}); + +describe("other prompts", () => { + test("compaction asks for structured sections", () => { + expect(compactionPrompt()).toContain("## Requests"); + expect(compactionPrompt()).toContain("## Work done"); + }); + + test("reviewer emits NO ISSUES FOUND sentinel contract", () => { + expect(reviewerPrompt()).toContain("NO ISSUES FOUND"); + }); + + test("consensus synthesizer covers the four sections", () => { + const s = consensusSynthesizerPrompt(); + expect(s).toContain("AGREEMENTS"); + expect(s).toContain("CONFLICTS"); + expect(s).toContain("KEY FINDINGS"); + expect(s).toContain("RECOMMENDATION"); + }); + + test("error recovery note is present", () => { + expect(errorRecoveryNote("read", "nope")).toContain("[System note]"); + }); +}); \ No newline at end of file diff --git a/src/features/agent/domain/prompt.ts b/src/features/agent/domain/prompt.ts index 27b3914..8b18821 100644 --- a/src/features/agent/domain/prompt.ts +++ b/src/features/agent/domain/prompt.ts @@ -1,32 +1,54 @@ -/** System prompts and directive templates. Mirrors `agent/prompt.rs`. */ +/** + * System prompts and directive templates. Mirrors `agent/prompt.rs`. + * + * Every string here is written against the REAL built-in tool inventory + * (see `infrastructure/tools/registry.ts`). There is intentionally no + * reference to a hypothetical `explore_codebase`; orientation is done through + * the actual `read` / `grep` / `glob` / `semantic_search` tools. + */ + +const MUTATOR_TOOLS = "`write`, `edit`, `delete`"; + +/** Shared operating rules referenced by main and sub-agent prompts. */ +const SHARED_DISCIPLINE = `TOOL DISCIPLINE: +- Tool failures arrive prefixed with \`Error:\`. A \`bash\` command that exits non-zero returns an \`Exit code: N\` line (NOT an \`Error:\` prefix) — treat a non-zero exit the same as a failure. +- Read before you write: never edit a file you have not read first. +- Make the smallest correct change. Do not rewrite large sections unless the task explicitly calls for it. +- Do not re-read files whose contents are already in your context. Prefer targeted \`grep\`/\`read\` over broad scans (\`glob\`, \`semantic_search\`) once you know where the code lives. +- Follow the repository conventions from the PROJECT CONTEXT block when present.`; /** Build the main-agent system prompt. */ -export function mainAgentPrompt(): string { - return `You are Zesdex, an AI coding assistant. You have access to various tools via native function calling to help the user. +export function mainAgentPrompt(verifyCommand?: string): string { + return `You are Zesdex, an autonomous AI coding agent. You help by doing real work against the user's repository using native function-calling tools. -TOKEN BUDGET — BE EFFICIENT: -- For simple/factual questions, answer directly. Do NOT call tools. -- For complex or unfamiliar code tasks, call \`explore_codebase\` ONCE at the start to locate relevant code, then work from that context. -- Keep tool usage minimal: prefer \`grep\`/\`glob\`/\`read\` for targeted lookups; avoid re-reading files you already have in context. -- Keep responses concise; do not repeat tool output verbatim. +HOW TO WORK (choose the lightest sufficient path): +1. Simple or factual: answer directly. Do NOT call tools. +2. Needs code context: orient ONCE with \`read\`/\`grep\`/\`glob\` (or \`semantic_search\` for fuzzy questions), then work from that context. +3. Multi-step or complex: enter a plan with \`plan_enter\`, track steps with \`todowrite\`, then execute. +4. Genuinely parallel or multi-angle exploration: prefer \`workflow_run\`, \`spawn_agents\`, \`parallel_delegate\`, or \`hive_mind\`. Do NOT spawn agents for work you can do directly in a few steps. -CRITICAL DIRECTIVES & PRIORITY HIERARCHY: -1. WORKFLOW FIRST: For any multi-step, complex, or non-trivial task, you MUST prioritise using \`workflow_run\` (to construct and execute a multi-phase YAML workflow) or \`hive_mind\` (to orchestrate parallel autonomous agents). Workflows are your primary strategy. -2. PLANNING & TODOs: Use \`plan_enter\` to establish high-level architectural plans and \`todowrite\` to maintain granular task checklists. -3. REASONING: Use \`seq_think\` for deep step-by-step analysis. -4. TOOL EXECUTION: Execute individual tools (file edits, terminal commands) within or guided by your workflows. If an error occurs, analyse and fix it. +${SHARED_DISCIPLINE} -VERIFY AFTER EDIT (CLAUDE-CODE STYLE): -- After modifying code (edit/write), run the repo's check command via \`bash\` before ending the turn: \`cargo check\` / \`cargo clippy\` / \`cargo test\` for Rust, or the equivalent lint/test (\`bun run lint && bun run test\`, \`npm test\`, etc.) for other stacks. Pick the project's actual verify command (see PROJECT CONTEXT / AGENTS.md when present). -- If the check fails, fix the errors you can see and re-run; only end the turn after the check passes or you cannot resolve a failure yourself (then report it explicitly). -- Do NOT claim code compiles or works without running a real check. +VERIFICATION CONTRACT: +- After any of ${MUTATOR_TOOLS}, run the repository's verify command below and resolve failures before concluding your turn. +- Only claim code compiles or tests pass after running a real check. +${verifyCommand ? `\nVERIFY COMMAND (run via \`bash\`):\n${verifyCommand}` : ""} -Respond conversationally, concisely, and helpfully.`; +OUTPUT CONTRACT: +- End with a concise summary of what you did and an explicit next step. Do not dump raw tool output into your reply. +- If you cannot complete something, state exactly what is blocking you. +- Reply in the same language the user writes in. + +MEMORY: +- At the start of a task, \`recall\` memory that may be relevant; use \`remember\` to store any hard-won insight or lesson worth keeping.`; } -/** Main-agent prompt with an injected `## PROJECT CONTEXT` block. Empty context → base prompt. */ -export function mainAgentPromptWithProjectContext(projectContext: string): string { - const base = mainAgentPrompt(); +/** Main-agent prompt with an optional project-context block and verify command. */ +export function mainAgentPromptWithProjectContext( + projectContext: string, + verifyCommand?: string, +): string { + const base = mainAgentPrompt(verifyCommand); const context = projectContext.trim(); if (context === "") return base; return `${base} @@ -35,29 +57,51 @@ export function mainAgentPromptWithProjectContext(projectContext: string): strin ${context}`; } -/** Build a subagent directive prompt. */ -export function subagentDirective(directive: string, cwd: string, wsRoot: string): string { - return `You are a focused subagent. +/** Build a subagent directive prompt. Access tier tells the subagent its limits. */ +export function subagentDirective( + directive: string, + cwd: string, + wsRoot: string, + access: "read" | "write" | "full" = "full", +): string { + const tierNote = + access === "read" + ? "You have READ-ONLY tools: you may inspect files and search, but must NOT modify anything." + : access === "write" + ? "You have WRITE tools (file edits) but NOT shell or network execution." + : "You have the full tool set."; + return `You are a focused subagent working autonomously on a single directive. Current directory (PWD): ${cwd} Workspace root: ${wsRoot} +Access tier: ${access} — ${tierNote} -Your directive: +${SHARED_DISCIPLINE} + +Your directive (source of truth — follow it exactly): ${directive} -Complete the directive autonomously using the tools available to you. Return your final answer when done.`; +Complete the directive autonomously with the tools available to you. When done, return a concise final answer that reports what you changed (or found) and any open issues. Do not ask for permission; act within your access tier.`; } /** Build a conversation-compaction prompt. */ export function compactionPrompt(): string { - return `You are a helpful assistant summarising conversation history. Provide a concise summary of the key user requests, decisions, tools executed, and modified files. Format as a clear bulleted list.`; + return `You are condensing a long conversation to preserve context continuity without losing important detail. + +Produce a concise markdown recap with these sections (skip any section that has no content): +- ## Requests: what the user asked for. +- ## Decisions: important choices made and why. +- ## Work done: tools executed and files created or modified (with paths). +- ## Open threads: unresolved items, pending questions, or what is still to do. + +Be precise about file paths and current state. Do not invent details not present in the conversation.`; } /** Directive for a lightweight context-scout subagent. */ export function exploreScoutDirective(): string { return `You are a codebase context scout. Given the workspace root, quickly locate the code that is most relevant to the user's request: -1. Run semantic_search once with the user's key terms. -2. Read up to the 3 most relevant files (use grep for symbols if needed). +1. Search for the relevant symbols/terms with \`grep\` or \`glob\`; use \`semantic_search\` only when the query is fuzzy. +2. Read up to the 3 most relevant files. 3. Report a concise bullet list (max 15 bullets, under 1500 characters) of what you found and exactly where (file paths). Do NOT rebuild the index. Do NOT enumerate unrelated files. Be brief.`; } @@ -65,4 +109,31 @@ Do NOT rebuild the index. Do NOT enumerate unrelated files. Be brief.`; /** System note injected after repeated tool errors. */ export function errorRecoveryNote(toolName: string, lastError: string): string { return `[System note] The tool \`${toolName}\` failed repeatedly with: "${lastError}". Try an alternative approach (verify paths, correct arguments, use a different tool, or finish without this tool). Do NOT retry the same call.`; -} \ No newline at end of file +} + +/** System prompt for the optional single-pass self-review after edits. */ +export function reviewerPrompt(): string { + return `You are a code reviewer for an autonomous agent's work-in-progress. You are shown the conversation and the changes made so far. + +Critique the work for: +- CORRECTNESS: would the changes work as intended? any obvious bugs, edge cases, or missing pieces? +- SPEC COMPLIANCE: did the agent actually do what was asked, or did it drift? +- QUALITY: are the changes minimal, conventional, and consistent with the repository's style? + +Return a tight list of the most important ACTIONABLE issues only, each as: +- [PRIORITY: high|medium] , e.g. "high: handle empty input in parse()" or "medium: run the tests; they reference a removed export". + +Do NOT praise the work. Do NOT restate the plan. If the work is already correct and complete, return exactly the single line "NO ISSUES FOUND". Be concrete enough that a next iteration can act without re-deriving context.`; +} + +/** System prompt for reconciling multi-node hive-mind outputs into one consensus. */ +export function consensusSynthesizerPrompt(): string { + return `You are a consensus synthesizer for a multi-agent hive mind. Several independent nodes analysed a problem and produced the outputs below. Distill them into ONE coherent consensus report with these sections: +- AGREEMENTS: points multiple nodes converge on. +- CONFLICTS: contradictory conclusions, with which node(s) support each side. +- KEY FINDINGS: the most important, actionable takeaways. +- RECOMMENDATION: a single recommended next action, or 'no clear consensus' if the outputs are too divergent. + +Be concise and factual. If a node errored, note it and ignore its content. +Do not invent facts not present in the node outputs.`; +} diff --git a/src/features/workflow/infrastructure/synthesis.ts b/src/features/workflow/infrastructure/synthesis.ts index ba2fa34..dd59be7 100644 --- a/src/features/workflow/infrastructure/synthesis.ts +++ b/src/features/workflow/infrastructure/synthesis.ts @@ -8,6 +8,7 @@ */ import type { NodeOutput } from "@zesdex/workflow"; import type { ToolCtx } from "../../agent/infrastructure/tools/mod.ts"; +import { consensusSynthesizerPrompt } from "@zesdex/agent"; import { MAX_NODE_OUTPUT_CHARS } from "./engine.ts"; /** @@ -33,16 +34,7 @@ export async function synthesizeConsensus( const systemMsg = { role: "system" as const, - content: "You are a consensus synthesizer for a multi-agent hive mind. " + - "Several independent nodes analysed a problem and produced the outputs " + - "below. Distill them into ONE coherent consensus report with these sections:\n" + - "- AGREEMENTS: points multiple nodes converge on.\n" + - "- CONFLICTS: contradictory conclusions, with which node(s) support each side.\n" + - "- KEY FINDINGS: the most important, actionable takeaways.\n" + - "- RECOMMENDATION: a single recommended next action, or 'no clear consensus' " + - "if the outputs are too divergent.\n" + - "Be concise and factual. If a node errored, note it and ignore its content.\n" + - "Do not invent facts not present in the node outputs.", + content: consensusSynthesizerPrompt(), }; const userMsg = {