Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7e2eda3bb4 | ||
|
|
e6718fa118 | ||
|
|
f98030f28a | ||
|
|
93e5464370 | ||
|
|
f5b04214cf | ||
|
|
008f737e1f | ||
|
|
2100e7ab54 | ||
|
|
c53fdfa7c9 | ||
|
|
5912a35bbb | ||
|
|
d86d6aeef2 | ||
|
|
30cc579773 | ||
|
|
a54f1644f5 |
@@ -1,3 +1,30 @@
|
||||
# [1.24.0](https://github.com/asepharyana/zesdex/compare/v1.23.2...v1.24.0) (2026-09-03)
|
||||
|
||||
|
||||
### Features
|
||||
|
||||
* **tui:** add new flow helpers for bash exit code handling and verification prompts ([f98030f](https://github.com/asepharyana/zesdex/commit/f98030f28a20e3d9ba807bb98ddbf142511a3964))
|
||||
* **tui:** enhance autocomplete functionality and UI interactions ([008f737](https://github.com/asepharyana/zesdex/commit/008f737e1f5d0eef6a07ebe0fff2c9730bd44eb4))
|
||||
* **tui:** enhance command execution on autocomplete and improve UI responsiveness ([f5b0421](https://github.com/asepharyana/zesdex/commit/f5b04214cf385ba2e4a02f6ce202436ffab23c35))
|
||||
* **tui:** enhance model selection and improve UI components ([c53fdfa](https://github.com/asepharyana/zesdex/commit/c53fdfa7c99887e8f3dfe68e7f1b2e766de039b1))
|
||||
* **tui:** implement interactive menu system for model and config selection ([2100e7a](https://github.com/asepharyana/zesdex/commit/2100e7ab54460f28b67c3bb07855eef8f04eb9ec))
|
||||
* **tui:** implement live model fetching and enhance model selection menus ([93e5464](https://github.com/asepharyana/zesdex/commit/93e54643703ce7f18286bd6a988e0fab18cc9e66))
|
||||
* **tui:** implement message compaction and summarization logic for efficient conversation handling ([e6718fa](https://github.com/asepharyana/zesdex/commit/e6718fa11825cccc74a9d100755226b7dfefc106))
|
||||
|
||||
## [1.23.2](https://github.com/asepharyana/zesdex/compare/v1.23.1...v1.23.2) (2026-09-03)
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **tui:** sliding-window stream reconciler + UI polish ([d86d6ae](https://github.com/asepharyana/zesdex/commit/d86d6aeef2dcd4924ced14f21502a84f7e916a26))
|
||||
|
||||
## [1.23.1](https://github.com/asepharyana/zesdex/compare/v1.23.0...v1.23.1) (2026-09-03)
|
||||
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
* **tui:** repair layout overflow + streaming dup; polish UI/state ([a54f164](https://github.com/asepharyana/zesdex/commit/a54f1644f5943399568194b4f129185a9d841f1d))
|
||||
|
||||
# [1.23.0](https://github.com/asepharyana/zesdex/compare/v1.22.0...v1.23.0) (2026-09-03)
|
||||
|
||||
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "zesdex",
|
||||
"version": "1.23.0",
|
||||
"version": "1.24.0",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"packageManager": "bun@1.3.14",
|
||||
|
||||
@@ -4,8 +4,25 @@ import {
|
||||
conversationChars,
|
||||
ErrorTracker,
|
||||
truncateToolOutput,
|
||||
bashExitCode,
|
||||
isToolFailure,
|
||||
deriveVerifyCommand,
|
||||
AgentTurnServiceImpl,
|
||||
takeRecentTail,
|
||||
reduceMessagesToDigest,
|
||||
compactMessagesWithAi,
|
||||
} from "./turn_service.ts";
|
||||
import { type ChatMessage, newConversation, systemMessage, userMessage, toolResultMessage } from "@zesdex/domain";
|
||||
import {
|
||||
type ChatMessage,
|
||||
newConversation,
|
||||
systemMessage,
|
||||
userMessage,
|
||||
assistantMessage,
|
||||
toolResultMessage,
|
||||
} from "@zesdex/domain";
|
||||
import type { AgentTurnParams } from "@zesdex/agent";
|
||||
import type { ToolExecutor } from "./index.ts";
|
||||
import type { ProviderService } from "./ports.ts";
|
||||
|
||||
describe("truncateToolOutput", () => {
|
||||
it("short output is unchanged", () => {
|
||||
@@ -56,4 +73,312 @@ describe("conversationChars", () => {
|
||||
conv.messages.push(systemMessage("sys"), userMessage("hello world"), toolResultMessage("id", "output"));
|
||||
expect(conversationChars(conv.messages)).toBe(3 + 11 + 6);
|
||||
});
|
||||
});
|
||||
});
|
||||
/* ── New flow helpers ─────────────────────────────────────────────── */
|
||||
|
||||
describe("bashExitCode / isToolFailure", () => {
|
||||
it("extracts a non-zero exit code", () => {
|
||||
expect(bashExitCode("boom\n\nExit code: 1 (1s)")).toBe(1);
|
||||
expect(bashExitCode("done\n\nExit code: 0 (0.5s)")).toBe(0);
|
||||
});
|
||||
it("returns null when no exit-code line exists", () => {
|
||||
expect(bashExitCode("plain output")).toBeNull();
|
||||
});
|
||||
it("isToolFailure flags bash non-zero exits as failures", () => {
|
||||
expect(isToolFailure("bash", "nope\n\nExit code: 2 (1s)")).toBe(true);
|
||||
expect(isToolFailure("bash", "ok\n\nExit code: 0 (1s)")).toBe(false);
|
||||
});
|
||||
it("isToolFailure still flags Error: prefixes for other tools", () => {
|
||||
expect(isToolFailure("read", "Error: no such file")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("deriveVerifyCommand", () => {
|
||||
it("defaults to bun check+test for a Bun repo", () => {
|
||||
// No real manifests in a control dir we guarantee to not exist.
|
||||
expect(deriveVerifyCommand("/nonexistent-zesdex-dir")).toContain("bun");
|
||||
});
|
||||
});
|
||||
|
||||
/* ── runTurn flow tests (fakes, no network) ───────────────────────── */
|
||||
|
||||
interface ScriptedStep {
|
||||
content?: string | null;
|
||||
tools?: Array<{ name: string; args?: string; id?: string }>;
|
||||
}
|
||||
|
||||
/** A ProviderService that replays scripted chatStream responses. */
|
||||
function makeScriptedProvider(script: ScriptedStep[]): {
|
||||
provider: ProviderService;
|
||||
seen: Array<ChatMessage[]>;
|
||||
} {
|
||||
const seen: Array<ChatMessage[]> = [];
|
||||
let step = 0;
|
||||
const chatResponses: string[] = [];
|
||||
const provider: ProviderService = {
|
||||
async chat(_messages, _tools, _maxTokens, _temperature) {
|
||||
// Used by compaction and review. For review tests we script below.
|
||||
const text = chatResponses.shift() ?? "";
|
||||
return { message: { role: "assistant", content: text || null }, usage: [10, 5] };
|
||||
},
|
||||
async chatStream(messages, _tools, _max, _temp, onEvent) {
|
||||
// Snapshot a copy: runTurn freely mutates the live array (splice/compact).
|
||||
seen.push([...messages]);
|
||||
const s = script[Math.min(step, script.length - 1)] ?? { content: "done" };
|
||||
step += 1;
|
||||
if (s.content !== undefined && s.content !== null) onEvent({ kind: "token", content: s.content });
|
||||
if (s.tools && s.tools.length > 0) {
|
||||
const msg = assistantMessage(null) as ChatMessage;
|
||||
msg.tool_calls = s.tools.map((t, i) => ({
|
||||
id: t.id ?? `call-${i}`,
|
||||
type: "function",
|
||||
function: { name: t.name, arguments: t.args ?? "{}" },
|
||||
}));
|
||||
return { message: msg, usage: [10, 2] };
|
||||
}
|
||||
return {
|
||||
message: { role: "assistant", content: s.content ?? null },
|
||||
usage: [10, 2],
|
||||
};
|
||||
},
|
||||
// expose a way for tests to script chat replies
|
||||
} as ProviderService;
|
||||
(provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => {
|
||||
chatResponses.push(text);
|
||||
};
|
||||
return { provider, seen };
|
||||
}
|
||||
|
||||
/** A ToolExecutor that echoes fixed outputs per tool. */
|
||||
function fixedExecutor(outs: Record<string, string>): ToolExecutor {
|
||||
return {
|
||||
async execute(name) {
|
||||
return outs[name] ?? "ok";
|
||||
},
|
||||
isParallelSafe() {
|
||||
return false;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function buildParams(userText: string): {
|
||||
params: AgentTurnParams;
|
||||
events: Array<any>;
|
||||
} {
|
||||
const events: Array<any> = [];
|
||||
const params: AgentTurnParams = {
|
||||
messages: [userMessage(userText)],
|
||||
session_dir: "/tmp/zesdex-flow-test",
|
||||
workspace_roots: ["/nonexistent-zesdex-dir"], // no AGENTS.md/package.json side effects
|
||||
turn_events: { push: (e) => events.push(e), drain: () => [] },
|
||||
in_flight: { value: false },
|
||||
abort: new AbortController(),
|
||||
api_key: "k",
|
||||
model: "m",
|
||||
api_base: "https://x",
|
||||
};
|
||||
return { params, events };
|
||||
}
|
||||
|
||||
/** Extract the `[Flow]` steering system message from a captured batch. */
|
||||
function flowDirective(batch: ChatMessage[]): string {
|
||||
const m = batch.find((x) => (x.content ?? "").includes("[Flow]"));
|
||||
return m?.content ?? "";
|
||||
}
|
||||
|
||||
describe("runTurn complexity steering", () => {
|
||||
it("injects a complex-request directive for a complex prompt", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ content: "final answer" },
|
||||
]);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||
const { params } = buildParams("Please refactor the architecture across multiple files");
|
||||
await svc.runTurn(params);
|
||||
// The complexity directive system message must be present before the LLM.
|
||||
const flow = flowDirective(seen[0] ?? []);
|
||||
expect(flow).toContain("flagged as complex");
|
||||
});
|
||||
|
||||
it("injects a keep-it-simple directive instead", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([{ content: "hi" }]);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||
const { params } = buildParams("what is 2+2");
|
||||
await svc.runTurn(params);
|
||||
const flow = flowDirective(seen[0] ?? []);
|
||||
expect(flow).toContain("looks simple");
|
||||
expect(flow).not.toContain("complex");
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn verify-after-edit", () => {
|
||||
it("injects a verify nudge after a write, cleared after a successful bash", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ tools: [{ name: "write" }] },
|
||||
{ tools: [{ name: "bash" }] },
|
||||
{ content: "done" },
|
||||
]);
|
||||
const executor = fixedExecutor({ write: "wrote it", bash: "ok\n\nExit code: 0 (1s)" });
|
||||
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||
const { params } = buildParams("add a comment");
|
||||
await svc.runTurn(params);
|
||||
// The LLM call after the write should carry the verify nudge.
|
||||
const second = seen[1] ?? [];
|
||||
expect(second.some((m) => (m.content ?? "").includes("Run the verify command"))).toBe(true);
|
||||
// The successful bash exits 0, which clears pendingVerify — so the nudge is
|
||||
// injected exactly once, not repeated on every later call.
|
||||
const finalBatch = seen[seen.length - 1] ?? [];
|
||||
const nudgeCount = finalBatch.filter((m) => (m.content ?? "").includes("Run the verify command")).length;
|
||||
expect(nudgeCount).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn convergence guard", () => {
|
||||
it("stops after repeated identical read calls", async () => {
|
||||
const script: ScriptedStep[] = [];
|
||||
for (let i = 0; i < 6; i++) script.push({ tools: [{ name: "read" }] });
|
||||
const { provider, seen } = makeScriptedProvider(script);
|
||||
const executor = fixedExecutor({ read: "same content" });
|
||||
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||
const { params, events } = buildParams("read something");
|
||||
await svc.runTurn(params);
|
||||
const warned = events.some((e) => e.kind === "system_note" && /without any progress|repeated/.test(e.message ?? ""));
|
||||
expect(warned).toBe(true);
|
||||
// Not every call got issued — the guard broke early.
|
||||
expect(seen.length).toBeLessThan(script.length);
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn self-review", () => {
|
||||
it("feeds a reviewer critique back as a system message after a mutator", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ tools: [{ name: "edit" }] },
|
||||
{ content: "fixed" },
|
||||
]);
|
||||
const withChat = provider as unknown as { setChatReply(s: string): void };
|
||||
withChat.setChatReply("- [PRIORITY: high] handle empty input in parse()");
|
||||
const executor = fixedExecutor({ edit: "edited" });
|
||||
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||
const { params, events } = buildParams("fix parse()");
|
||||
await svc.runTurn(params);
|
||||
// The reviewer critique must have reached the next LLM call's history.
|
||||
const second = seen[1] ?? [];
|
||||
expect(second.some((m) => (m.content ?? "").includes("[Reviewer]"))).toBe(true);
|
||||
// review_usage emitted.
|
||||
expect(events.some((e) => e.kind === "review_usage")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
/* ── Compaction: efficient + accurate ─────────────────────────────── */
|
||||
|
||||
describe("takeRecentTail", () => {
|
||||
it("keeps the most recent messages and evicts an oversized older blob", () => {
|
||||
const messages = [
|
||||
toolResultMessage("t0", "y".repeat(5000)), // huge old tool output
|
||||
userMessage("old request"),
|
||||
assistantMessage("recent reply A"),
|
||||
userMessage("recent reply B"),
|
||||
];
|
||||
const { tail, evicted } = takeRecentTail(messages, 25, 2);
|
||||
expect(evicted.length).toBe(2); // the blob + the old request
|
||||
expect(tail.map((m) => m.content)).toEqual(["recent reply A", "recent reply B"]);
|
||||
});
|
||||
|
||||
it("gives the whole history as tail when it is small enough", () => {
|
||||
const messages = [userMessage("a"), assistantMessage("b")];
|
||||
const { tail, evicted } = takeRecentTail(messages, 1000, 6);
|
||||
expect(evicted).toEqual([]);
|
||||
expect(tail.length).toBe(2);
|
||||
});
|
||||
|
||||
it("does not let a huge tool output blow the tail past the min count", () => {
|
||||
// minTail 2: newest two kept regardless; the huge tool body stays evicted.
|
||||
const messages = [
|
||||
toolResultMessage("t0", "x".repeat(50000)),
|
||||
userMessage("keep1"),
|
||||
assistantMessage("keep2"),
|
||||
];
|
||||
const { tail, evicted } = takeRecentTail(messages, 8000, 2);
|
||||
expect(tail.map((m) => m.content)).toEqual(["keep1", "keep2"]);
|
||||
expect(evicted.length).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("reduceMessagesToDigest", () => {
|
||||
it("stubs tool bodies entirely and previews text", () => {
|
||||
const digest = reduceMessagesToDigest([
|
||||
toolResultMessage("t0", "y".repeat(5000)),
|
||||
userMessage("short request"),
|
||||
]);
|
||||
expect(digest).toContain("- tool");
|
||||
expect(digest).not.toContain("yyyy");
|
||||
expect(digest).toContain("short request");
|
||||
});
|
||||
|
||||
it("keeps user previews well above the shorter assistant cap", () => {
|
||||
const long = "w".repeat(2000);
|
||||
// User budget is 1500 chars — larger than the 400-char assistant cap, so
|
||||
// a long user request survives far more of its body than a long assistant
|
||||
// reply would (requests matter most for summary accuracy).
|
||||
const digestUser = reduceMessagesToDigest([userMessage(long)]);
|
||||
const digestAssistant = reduceMessagesToDigest([assistantMessage(long)]);
|
||||
expect(digestUser.length).toBeGreaterThan(800);
|
||||
expect(digestAssistant.length).toBeLessThan(500);
|
||||
});
|
||||
|
||||
it("respects the hard input cap", () => {
|
||||
const many = [];
|
||||
for (let i = 0; i < 200; i++) many.push(userMessage("r".repeat(300)));
|
||||
const digest = reduceMessagesToDigest(many);
|
||||
expect(digest.length).toBeLessThanOrEqual(20_000 + 40);
|
||||
});
|
||||
});
|
||||
|
||||
describe("compactMessagesWithAi", () => {
|
||||
it("keeps the recent tail and prepends an AI summary", async () => {
|
||||
// Long enough to bypass the min-size guard, with an oversized OLD tool blob
|
||||
// (should be evicted + stubbed, not kept verbatim).
|
||||
const messages: ChatMessage[] = [
|
||||
userMessage("old request one"),
|
||||
assistantMessage("old response"),
|
||||
toolResultMessage("t0", "big".repeat(4000)),
|
||||
];
|
||||
for (let i = 0; i < 7; i++) messages.push(userMessage(`mid ${i}`));
|
||||
messages.push(toolResultMessage("t1", "tail-result"));
|
||||
messages.push(userMessage("newest request"));
|
||||
messages.push(assistantMessage("newest response"));
|
||||
|
||||
const provider: ProviderService = {
|
||||
async chat() {
|
||||
return { message: { role: "assistant", content: "## Requests\n- old request one" }, usage: null };
|
||||
},
|
||||
async chatStream() {
|
||||
return { message: { role: "assistant", content: null }, usage: null };
|
||||
},
|
||||
};
|
||||
await compactMessagesWithAi(messages, provider);
|
||||
// Summary first, recent tail preserved verbatim, tool blobs gone.
|
||||
expect(messages[0]?.content).toContain("[AI Summary of Previous Conversation]");
|
||||
expect(messages.some((m) => m.content === "newest request")).toBe(true);
|
||||
expect(messages.some((m) => m.content === "newest response")).toBe(true);
|
||||
expect(messages.some((m) => (m.content ?? "").includes("big".repeat(10)))).toBe(false);
|
||||
});
|
||||
|
||||
it("degrades gracefully on summarizer failure, preserving the tail", async () => {
|
||||
const messages = [
|
||||
userMessage("old"),
|
||||
toolResultMessage("t0", "x".repeat(100)),
|
||||
userMessage("recent request"),
|
||||
];
|
||||
const failing: ProviderService = {
|
||||
async chat() {
|
||||
throw new Error("network down");
|
||||
},
|
||||
async chatStream() {
|
||||
return { message: { role: "assistant", content: null }, usage: null };
|
||||
},
|
||||
};
|
||||
await compactMessagesWithAi(messages, failing);
|
||||
// The recent message survives even when the LLM call fails.
|
||||
expect(messages.some((m) => m.content === "recent request")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
type ToolCall,
|
||||
type ToolDef,
|
||||
type JsonValue,
|
||||
Roles,
|
||||
systemMessage,
|
||||
userMessage,
|
||||
assistantMessage,
|
||||
@@ -20,7 +21,9 @@ import {
|
||||
mainAgentPromptWithProjectContext,
|
||||
compactionPrompt,
|
||||
errorRecoveryNote,
|
||||
reviewerPrompt,
|
||||
} from "@zesdex/agent";
|
||||
import { isComplexRequest } from "@zesdex/workflow";
|
||||
import type { ProviderService } from "./ports.ts";
|
||||
import type { ToolExecutor } from "./index.ts";
|
||||
|
||||
@@ -38,11 +41,89 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000;
|
||||
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
|
||||
const COMPACT_KEEP_TAIL = 6;
|
||||
|
||||
/** Recent-tail budget (chars): the most recent history kept verbatim on compact. */
|
||||
const COMPACT_TAIL_CHARS = 8_000;
|
||||
/** Caps what the summarizer actually receives, keeping input small and focused. */
|
||||
const COMPACT_MAX_INPUT_CHARS = 20_000;
|
||||
/** Preview budget for a user request inside the digest (requests matter most). */
|
||||
const COMPACT_USER_PREVIEW_CHARS = 1_500;
|
||||
/** Preview budget for assistant text inside the digest. */
|
||||
const COMPACT_TEXT_PREVIEW_CHARS = 400;
|
||||
|
||||
/** Tools that mutate the filesystem — after these, a verify run is expected. */
|
||||
const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]);
|
||||
|
||||
/** Consecutive identical, non-progressing tool iterations before the loop stops. */
|
||||
const MAX_NO_PROGRESS_STREAK = 4;
|
||||
|
||||
/** Self-review is bounded to this many passes per turn. */
|
||||
const MAX_REVIEW_PASSES = 1;
|
||||
|
||||
/** Whether the output string denotes a tool error. */
|
||||
function isErrorOutput(output: string): boolean {
|
||||
return output.startsWith("Error:");
|
||||
}
|
||||
|
||||
/** Extract the numeric exit code from a `bash` tool result, or null if not a failure (0). */
|
||||
export function bashExitCode(output: string): number | null {
|
||||
const match = output.match(/Exit code:\s*(\d+)/);
|
||||
if (!match) return null;
|
||||
const code = Number(match[1]);
|
||||
return Number.isInteger(code) && code >= 0 ? code : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a tool result is a real failure. `Error:` prefixes cover most tools;
|
||||
* a `bash` command that exits non-zero returns an `Exit code: N` line instead.
|
||||
*/
|
||||
export function isToolFailure(toolName: string, output: string): boolean {
|
||||
if (isErrorOutput(output)) return true;
|
||||
if (toolName === "bash" || toolName === "bash_output") {
|
||||
const code = bashExitCode(output);
|
||||
if (code === null) return false; // no exit-code line → no signal
|
||||
return code !== 0;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort derivation of the repository's verify command from convention
|
||||
* manifests. Defaults to a safe lint+test invocation for Bun.
|
||||
*/
|
||||
export function deriveVerifyCommand(root: string): string {
|
||||
const fs = require("node:fs");
|
||||
// Bun project: prefer an explicit `check` (typecheck) script then test.
|
||||
try {
|
||||
const pkg = JSON.parse(fs.readFileSync(`${root}/package.json`, "utf8")) as {
|
||||
scripts?: Record<string, string>;
|
||||
};
|
||||
const s = pkg?.scripts ?? {};
|
||||
const parts: string[] = [];
|
||||
if (s["check"] && typeof s["check"] === "string") parts.push(`bun run check`);
|
||||
else if (s["typecheck"] && typeof s["typecheck"] === "string") parts.push(`bun run typecheck`);
|
||||
if (s["lint"] && typeof s["lint"] === "string") parts.push(`bun run lint`);
|
||||
if (s["test"] && typeof s["test"] === "string") parts.push(`bun run test`);
|
||||
if (parts.length > 0) return parts.join(" && ");
|
||||
} catch {
|
||||
/* no package.json — fall through */
|
||||
}
|
||||
// Rust project.
|
||||
try {
|
||||
if (fs.existsSync(`${root}/Cargo.toml`)) return "cargo check && cargo test";
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
// Java/Maven.
|
||||
try {
|
||||
if (fs.existsSync(`${root}/pom.xml`) || fs.existsSync(`${root}/build.gradle`)) {
|
||||
return "mvn test";
|
||||
}
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
return "bun run check && bun test";
|
||||
}
|
||||
|
||||
/** Truncate a long tool output, preserving the head + truncation marker. */
|
||||
export function truncateToolOutput(output: string): string {
|
||||
if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output;
|
||||
@@ -136,7 +217,7 @@ async function executeToolCall(
|
||||
output = `Error: ${(e as Error).message}`;
|
||||
}
|
||||
|
||||
const isError = isErrorOutput(output);
|
||||
const isError = isToolFailure(name, output);
|
||||
const truncated = truncateToolOutput(output);
|
||||
|
||||
sink.push({
|
||||
@@ -179,8 +260,75 @@ async function executeToolCallsInParallel(
|
||||
/* -------------------------------------------------------------------------- */
|
||||
|
||||
/**
|
||||
* Compact oversized conversation history using AI summarisation. At most once
|
||||
* per turn. Keeps the last COMPACT_KEEP_TAIL messages.
|
||||
* Split history into a recent tail (kept verbatim, bound by CHAR budget so
|
||||
* huge tool outputs don't monopolise it) and the evicted prefix to summarize.
|
||||
*/
|
||||
export function takeRecentTail(
|
||||
messages: ChatMessage[],
|
||||
tailChars = COMPACT_TAIL_CHARS,
|
||||
minTail = COMPACT_KEEP_TAIL,
|
||||
): { tail: ChatMessage[]; evicted: ChatMessage[] } {
|
||||
if (messages.length <= minTail) return { tail: [...messages], evicted: [] };
|
||||
let used = 0;
|
||||
let keep = 0;
|
||||
// Walk from the newest message backward. Always keep at least minTail; then
|
||||
// stop once the aggregated char budget is exceeded.
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const len = messages[i]!.content?.length ?? 0;
|
||||
if (keep >= minTail && used + len > tailChars) break;
|
||||
keep += 1;
|
||||
used += len;
|
||||
}
|
||||
const tail = messages.slice(messages.length - keep);
|
||||
const evicted = messages.slice(0, messages.length - keep);
|
||||
return { tail, evicted };
|
||||
}
|
||||
|
||||
/**
|
||||
* Reduce older history to a compact digest for the summarizer. Tool-result
|
||||
* bodies are dropped entirely (they are noise for a summary); user requests
|
||||
* get a generous preview; assistant text gets a shorter one. Caps the result
|
||||
* at COMPACT_MAX_INPUT_CHARS so the summarizer sees a small, focused input.
|
||||
*/
|
||||
export function reduceMessagesToDigest(messages: ChatMessage[]): string {
|
||||
let out = "";
|
||||
let budget = COMPACT_MAX_INPUT_CHARS;
|
||||
for (const m of messages) {
|
||||
if (budget <= 0) break;
|
||||
if (m.role === Roles.Tool) {
|
||||
const name = m.name ?? "tool";
|
||||
const line = `- tool ${name} executed\n`;
|
||||
out += line;
|
||||
budget -= line.length;
|
||||
continue;
|
||||
}
|
||||
const text = m.content?.trim() ?? "";
|
||||
if (text === "") continue; // assistant messages that only carried tool calls
|
||||
const cap = m.role === Roles.User ? COMPACT_USER_PREVIEW_CHARS : COMPACT_TEXT_PREVIEW_CHARS;
|
||||
let preview = text.replace(/\s*\n+\s*/g, " ");
|
||||
if (preview.length > cap) {
|
||||
preview = `${preview.slice(0, cap)}…[+${text.length - cap} ch]`;
|
||||
}
|
||||
const line = `- ${m.role}: ${preview}\n`;
|
||||
out += line;
|
||||
budget -= line.length;
|
||||
}
|
||||
// Hard guarantee: never exceed the summarizer input cap.
|
||||
if (out.length > COMPACT_MAX_INPUT_CHARS) return out.slice(0, COMPACT_MAX_INPUT_CHARS);
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Compact oversized conversation history efficiently yet accurately.
|
||||
*
|
||||
* - Only the OLDER history is summarized; the recent tail is kept verbatim
|
||||
* (bounded by char budget, not message count, so tool bloat can't win the
|
||||
* budget).
|
||||
* - The evicted history is REDUCED to a small digest before the summarizer
|
||||
* sees it, so the LLM works on a focused input (cheaper + more accurate)
|
||||
* instead of a ~60k-char dump.
|
||||
* - On failure it degrades gracefully: keeps the recent tail and a marker,
|
||||
* rather than wiping context to a useless sentinel.
|
||||
*/
|
||||
export async function compactMessagesWithAi(
|
||||
messages: ChatMessage[],
|
||||
@@ -188,17 +336,35 @@ export async function compactMessagesWithAi(
|
||||
): Promise<void> {
|
||||
if (messages.length <= COMPACT_KEEP_TAIL + 2) return;
|
||||
|
||||
const splitIdx = messages.length - COMPACT_KEEP_TAIL;
|
||||
const evicted = messages.splice(0, splitIdx);
|
||||
const { tail, evicted } = takeRecentTail(messages);
|
||||
const digest = reduceMessagesToDigest(evicted);
|
||||
if (digest.trim() === "") {
|
||||
// Nothing worth summarizing (e.g. only stubbed tool bodies) — keep it all.
|
||||
return;
|
||||
}
|
||||
|
||||
const summaryPrompt: ChatMessage[] = [systemMessage(compactionPrompt()), ...evicted, userMessage("Please summarise our previous conversation above for context continuity.")];
|
||||
const summaryPrompt: ChatMessage[] = [
|
||||
systemMessage(compactionPrompt()),
|
||||
userMessage(
|
||||
"The conversation below is the OLDER part of a session. Recent messages are kept separately, so do not preserve them. Compress the older part into the requested summary format.\n\n---\n" + digest,
|
||||
),
|
||||
];
|
||||
|
||||
try {
|
||||
const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3);
|
||||
const summaryText = message.content ?? "Previous context summarised.";
|
||||
messages.unshift(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText.trim()}`));
|
||||
const summaryText = message.content?.trim() ?? "";
|
||||
const rebuilt: ChatMessage[] = [];
|
||||
if (summaryText !== "") {
|
||||
rebuilt.push(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText}`));
|
||||
} else {
|
||||
rebuilt.push(systemMessage("[Earlier conversation summarized]"));
|
||||
}
|
||||
rebuilt.push(...tail);
|
||||
messages.splice(0, messages.length, ...rebuilt);
|
||||
} catch {
|
||||
messages.unshift(systemMessage("[Earlier conversation messages compacted to save context window]"));
|
||||
// Graceful degradation: never wipe recent state on a summarizer failure.
|
||||
const rebuilt: ChatMessage[] = [systemMessage("[Earlier context compressed; recent messages follow.]"), ...tail];
|
||||
messages.splice(0, messages.length, ...rebuilt);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -268,18 +434,43 @@ export class AgentTurnServiceImpl {
|
||||
const abort = params.abort;
|
||||
const { in_flight } = params;
|
||||
|
||||
// Insert system prompt at index 0 with repo conventions loaded.
|
||||
const projectContext = buildProjectContext(params.workspace_roots[0] ?? ".");
|
||||
const systemPrompt = mainAgentPromptWithProjectContext(projectContext);
|
||||
// Insert system prompt at index 0 with repo conventions + verify command.
|
||||
const root = params.workspace_roots[0] ?? ".";
|
||||
const projectContext = buildProjectContext(root);
|
||||
const verifyCommand = deriveVerifyCommand(root);
|
||||
const systemPrompt = mainAgentPromptWithProjectContext(projectContext, verifyCommand);
|
||||
params.messages.unshift(systemMessage(systemPrompt));
|
||||
const originalCount = params.messages.length;
|
||||
|
||||
// Estimate request complexity from the last user message.
|
||||
const last = params.messages[params.messages.length - 1];
|
||||
const requestLen = last?.content?.length ?? 0;
|
||||
const userText = last?.content ?? "";
|
||||
|
||||
// Complexity steering: a one-shot directive so the model picks the right
|
||||
// depth instead of relying on prose memory. Pure and cheap.
|
||||
const complex = isComplexRequest(userText);
|
||||
this.push(sink, {
|
||||
kind: "system_note",
|
||||
systemKind: "info",
|
||||
message: complex ? "Complex request detected — plan before executing." : "Simple request — keep tool use minimal.",
|
||||
});
|
||||
params.messages.push(
|
||||
systemMessage(
|
||||
complex
|
||||
? "[Flow] This request is flagged as complex. Enter a short plan with `plan_enter` and track steps with `todowrite` before starting edits."
|
||||
: "[Flow] This request looks simple. If you can answer directly without tools, do so — do not spawn agents or workflows for it.",
|
||||
),
|
||||
);
|
||||
|
||||
const errors = new ErrorTracker();
|
||||
let sawToolCalls = false;
|
||||
let sawMutator = false;
|
||||
let pendingVerify = false;
|
||||
let verifyPrompted = false;
|
||||
let noProgressStreak = 0;
|
||||
let lastSignature: string | null = null;
|
||||
let reviewPasses = 0;
|
||||
|
||||
for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) {
|
||||
// Check abort flag.
|
||||
@@ -293,6 +484,26 @@ export class AgentTurnServiceImpl {
|
||||
break;
|
||||
}
|
||||
|
||||
// Convergence guard: bail out of a loop stuck re-issuing the same call.
|
||||
if (noProgressStreak >= MAX_NO_PROGRESS_STREAK) {
|
||||
this.push(sink, {
|
||||
kind: "system_note",
|
||||
systemKind: "warn",
|
||||
message: "Stopping: the same tool call is being repeated without any progress.",
|
||||
});
|
||||
break;
|
||||
}
|
||||
|
||||
// Verify-after-edit: nudge the model to run the check before concluding.
|
||||
if (pendingVerify && !verifyPrompted) {
|
||||
params.messages.push(
|
||||
systemMessage(
|
||||
`[Flow] You just modified files. Run the verify command via \`bash\` now (${verifyCommand}) and resolve any failures before concluding your turn.`,
|
||||
),
|
||||
);
|
||||
verifyPrompted = true;
|
||||
}
|
||||
|
||||
// Auto-compact oversized history before the LLM call.
|
||||
await this.autoCompactIfNeeded(params.messages);
|
||||
|
||||
@@ -343,12 +554,44 @@ export class AgentTurnServiceImpl {
|
||||
return seq;
|
||||
})();
|
||||
|
||||
let mutated = false;
|
||||
let verified = false;
|
||||
let batchSignature = "";
|
||||
for (let i = 0; i < toolCalls.length; i++) {
|
||||
const tc = toolCalls[i]!;
|
||||
const output = outputs[i]!;
|
||||
if (isErrorOutput(output)) errors.record(tc.function.name, output, params.messages);
|
||||
if (isToolFailure(tc.function.name, output)) errors.record(tc.function.name, output, params.messages);
|
||||
if (MUTATOR_TOOLS.has(tc.function.name)) {
|
||||
mutated = true;
|
||||
sawMutator = true;
|
||||
}
|
||||
if ((tc.function.name === "bash" || tc.function.name === "bash_output") && bashExitCode(output) === 0) {
|
||||
verified = true;
|
||||
}
|
||||
// Batch signature: concat of tool+output for no-progress detection.
|
||||
batchSignature += `${tc.function.name} | ||||