diff --git a/src/features/agent/application/turn_service.test.ts b/src/features/agent/application/turn_service.test.ts index ac62714..2346d80 100644 --- a/src/features/agent/application/turn_service.test.ts +++ b/src/features/agent/application/turn_service.test.ts @@ -8,6 +8,9 @@ import { isToolFailure, deriveVerifyCommand, AgentTurnServiceImpl, + takeRecentTail, + reduceMessagesToDigest, + compactMessagesWithAi, } from "./turn_service.ts"; import { type ChatMessage, @@ -264,3 +267,118 @@ describe("runTurn self-review", () => { expect(events.some((e) => e.kind === "review_usage")).toBe(true); }); }); + +/* ── Compaction: efficient + accurate ─────────────────────────────── */ + +describe("takeRecentTail", () => { + it("keeps the most recent messages and evicts an oversized older blob", () => { + const messages = [ + toolResultMessage("t0", "y".repeat(5000)), // huge old tool output + userMessage("old request"), + assistantMessage("recent reply A"), + userMessage("recent reply B"), + ]; + const { tail, evicted } = takeRecentTail(messages, 25, 2); + expect(evicted.length).toBe(2); // the blob + the old request + expect(tail.map((m) => m.content)).toEqual(["recent reply A", "recent reply B"]); + }); + + it("gives the whole history as tail when it is small enough", () => { + const messages = [userMessage("a"), assistantMessage("b")]; + const { tail, evicted } = takeRecentTail(messages, 1000, 6); + expect(evicted).toEqual([]); + expect(tail.length).toBe(2); + }); + + it("does not let a huge tool output blow the tail past the min count", () => { + // minTail 2: newest two kept regardless; the huge tool body stays evicted. + const messages = [ + toolResultMessage("t0", "x".repeat(50000)), + userMessage("keep1"), + assistantMessage("keep2"), + ]; + const { tail, evicted } = takeRecentTail(messages, 8000, 2); + expect(tail.map((m) => m.content)).toEqual(["keep1", "keep2"]); + expect(evicted.length).toBe(1); + }); +}); + +describe("reduceMessagesToDigest", () => { + it("stubs tool bodies entirely and previews text", () => { + const digest = reduceMessagesToDigest([ + toolResultMessage("t0", "y".repeat(5000)), + userMessage("short request"), + ]); + expect(digest).toContain("- tool"); + expect(digest).not.toContain("yyyy"); + expect(digest).toContain("short request"); + }); + + it("keeps user previews well above the shorter assistant cap", () => { + const long = "w".repeat(2000); + // User budget is 1500 chars — larger than the 400-char assistant cap, so + // a long user request survives far more of its body than a long assistant + // reply would (requests matter most for summary accuracy). + const digestUser = reduceMessagesToDigest([userMessage(long)]); + const digestAssistant = reduceMessagesToDigest([assistantMessage(long)]); + expect(digestUser.length).toBeGreaterThan(800); + expect(digestAssistant.length).toBeLessThan(500); + }); + + it("respects the hard input cap", () => { + const many = []; + for (let i = 0; i < 200; i++) many.push(userMessage("r".repeat(300))); + const digest = reduceMessagesToDigest(many); + expect(digest.length).toBeLessThanOrEqual(20_000 + 40); + }); +}); + +describe("compactMessagesWithAi", () => { + it("keeps the recent tail and prepends an AI summary", async () => { + // Long enough to bypass the min-size guard, with an oversized OLD tool blob + // (should be evicted + stubbed, not kept verbatim). + const messages: ChatMessage[] = [ + userMessage("old request one"), + assistantMessage("old response"), + toolResultMessage("t0", "big".repeat(4000)), + ]; + for (let i = 0; i < 7; i++) messages.push(userMessage(`mid ${i}`)); + messages.push(toolResultMessage("t1", "tail-result")); + messages.push(userMessage("newest request")); + messages.push(assistantMessage("newest response")); + + const provider: ProviderService = { + async chat() { + return { message: { role: "assistant", content: "## Requests\n- old request one" }, usage: null }; + }, + async chatStream() { + return { message: { role: "assistant", content: null }, usage: null }; + }, + }; + await compactMessagesWithAi(messages, provider); + // Summary first, recent tail preserved verbatim, tool blobs gone. + expect(messages[0]?.content).toContain("[AI Summary of Previous Conversation]"); + expect(messages.some((m) => m.content === "newest request")).toBe(true); + expect(messages.some((m) => m.content === "newest response")).toBe(true); + expect(messages.some((m) => (m.content ?? "").includes("big".repeat(10)))).toBe(false); + }); + + it("degrades gracefully on summarizer failure, preserving the tail", async () => { + const messages = [ + userMessage("old"), + toolResultMessage("t0", "x".repeat(100)), + userMessage("recent request"), + ]; + const failing: ProviderService = { + async chat() { + throw new Error("network down"); + }, + async chatStream() { + return { message: { role: "assistant", content: null }, usage: null }; + }, + }; + await compactMessagesWithAi(messages, failing); + // The recent message survives even when the LLM call fails. + expect(messages.some((m) => m.content === "recent request")).toBe(true); + }); +}); diff --git a/src/features/agent/application/turn_service.ts b/src/features/agent/application/turn_service.ts index 822e5c7..7721a8b 100644 --- a/src/features/agent/application/turn_service.ts +++ b/src/features/agent/application/turn_service.ts @@ -8,6 +8,7 @@ import { type ToolCall, type ToolDef, type JsonValue, + Roles, systemMessage, userMessage, assistantMessage, @@ -40,6 +41,15 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000; const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"]; const COMPACT_KEEP_TAIL = 6; +/** Recent-tail budget (chars): the most recent history kept verbatim on compact. */ +const COMPACT_TAIL_CHARS = 8_000; +/** Caps what the summarizer actually receives, keeping input small and focused. */ +const COMPACT_MAX_INPUT_CHARS = 20_000; +/** Preview budget for a user request inside the digest (requests matter most). */ +const COMPACT_USER_PREVIEW_CHARS = 1_500; +/** Preview budget for assistant text inside the digest. */ +const COMPACT_TEXT_PREVIEW_CHARS = 400; + /** Tools that mutate the filesystem — after these, a verify run is expected. */ const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]); @@ -250,8 +260,75 @@ async function executeToolCallsInParallel( /* -------------------------------------------------------------------------- */ /** - * Compact oversized conversation history using AI summarisation. At most once - * per turn. Keeps the last COMPACT_KEEP_TAIL messages. + * Split history into a recent tail (kept verbatim, bound by CHAR budget so + * huge tool outputs don't monopolise it) and the evicted prefix to summarize. + */ +export function takeRecentTail( + messages: ChatMessage[], + tailChars = COMPACT_TAIL_CHARS, + minTail = COMPACT_KEEP_TAIL, +): { tail: ChatMessage[]; evicted: ChatMessage[] } { + if (messages.length <= minTail) return { tail: [...messages], evicted: [] }; + let used = 0; + let keep = 0; + // Walk from the newest message backward. Always keep at least minTail; then + // stop once the aggregated char budget is exceeded. + for (let i = messages.length - 1; i >= 0; i--) { + const len = messages[i]!.content?.length ?? 0; + if (keep >= minTail && used + len > tailChars) break; + keep += 1; + used += len; + } + const tail = messages.slice(messages.length - keep); + const evicted = messages.slice(0, messages.length - keep); + return { tail, evicted }; +} + +/** + * Reduce older history to a compact digest for the summarizer. Tool-result + * bodies are dropped entirely (they are noise for a summary); user requests + * get a generous preview; assistant text gets a shorter one. Caps the result + * at COMPACT_MAX_INPUT_CHARS so the summarizer sees a small, focused input. + */ +export function reduceMessagesToDigest(messages: ChatMessage[]): string { + let out = ""; + let budget = COMPACT_MAX_INPUT_CHARS; + for (const m of messages) { + if (budget <= 0) break; + if (m.role === Roles.Tool) { + const name = m.name ?? "tool"; + const line = `- tool ${name} executed\n`; + out += line; + budget -= line.length; + continue; + } + const text = m.content?.trim() ?? ""; + if (text === "") continue; // assistant messages that only carried tool calls + const cap = m.role === Roles.User ? COMPACT_USER_PREVIEW_CHARS : COMPACT_TEXT_PREVIEW_CHARS; + let preview = text.replace(/\s*\n+\s*/g, " "); + if (preview.length > cap) { + preview = `${preview.slice(0, cap)}…[+${text.length - cap} ch]`; + } + const line = `- ${m.role}: ${preview}\n`; + out += line; + budget -= line.length; + } + // Hard guarantee: never exceed the summarizer input cap. + if (out.length > COMPACT_MAX_INPUT_CHARS) return out.slice(0, COMPACT_MAX_INPUT_CHARS); + return out.trim(); +} + +/** + * Compact oversized conversation history efficiently yet accurately. + * + * - Only the OLDER history is summarized; the recent tail is kept verbatim + * (bounded by char budget, not message count, so tool bloat can't win the + * budget). + * - The evicted history is REDUCED to a small digest before the summarizer + * sees it, so the LLM works on a focused input (cheaper + more accurate) + * instead of a ~60k-char dump. + * - On failure it degrades gracefully: keeps the recent tail and a marker, + * rather than wiping context to a useless sentinel. */ export async function compactMessagesWithAi( messages: ChatMessage[], @@ -259,17 +336,35 @@ export async function compactMessagesWithAi( ): Promise { if (messages.length <= COMPACT_KEEP_TAIL + 2) return; - const splitIdx = messages.length - COMPACT_KEEP_TAIL; - const evicted = messages.splice(0, splitIdx); + const { tail, evicted } = takeRecentTail(messages); + const digest = reduceMessagesToDigest(evicted); + if (digest.trim() === "") { + // Nothing worth summarizing (e.g. only stubbed tool bodies) — keep it all. + return; + } - const summaryPrompt: ChatMessage[] = [systemMessage(compactionPrompt()), ...evicted, userMessage("Please summarise our previous conversation above for context continuity.")]; + const summaryPrompt: ChatMessage[] = [ + systemMessage(compactionPrompt()), + userMessage( + "The conversation below is the OLDER part of a session. Recent messages are kept separately, so do not preserve them. Compress the older part into the requested summary format.\n\n---\n" + digest, + ), + ]; try { const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3); - const summaryText = message.content ?? "Previous context summarised."; - messages.unshift(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText.trim()}`)); + const summaryText = message.content?.trim() ?? ""; + const rebuilt: ChatMessage[] = []; + if (summaryText !== "") { + rebuilt.push(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText}`)); + } else { + rebuilt.push(systemMessage("[Earlier conversation summarized]")); + } + rebuilt.push(...tail); + messages.splice(0, messages.length, ...rebuilt); } catch { - messages.unshift(systemMessage("[Earlier conversation messages compacted to save context window]")); + // Graceful degradation: never wipe recent state on a summarizer failure. + const rebuilt: ChatMessage[] = [systemMessage("[Earlier context compressed; recent messages follow.]"), ...tail]; + messages.splice(0, messages.length, ...rebuilt); } }