diff --git a/src/features/agent/application/turn_service.test.ts b/src/features/agent/application/turn_service.test.ts index 2346d80..af12c8a 100644 --- a/src/features/agent/application/turn_service.test.ts +++ b/src/features/agent/application/turn_service.test.ts @@ -111,13 +111,16 @@ interface ScriptedStep { function makeScriptedProvider(script: ScriptedStep[]): { provider: ProviderService; seen: Array; + chatCalls: Array; } { const seen: Array = []; + const chatCalls: Array = []; let step = 0; const chatResponses: string[] = []; const provider: ProviderService = { - async chat(_messages, _tools, _maxTokens, _temperature) { + async chat(messages, _tools, _maxTokens, _temperature) { // Used by compaction and review. For review tests we script below. + chatCalls.push([...messages]); const text = chatResponses.shift() ?? ""; return { message: { role: "assistant", content: text || null }, usage: [10, 5] }; }, @@ -146,7 +149,7 @@ function makeScriptedProvider(script: ScriptedStep[]): { (provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => { chatResponses.push(text); }; - return { provider, seen }; + return { provider, seen, chatCalls }; } /** A ToolExecutor that echoes fixed outputs per tool. */ @@ -180,33 +183,13 @@ function buildParams(userText: string): { return { params, events }; } -/** Extract the `[Flow]` steering system message from a captured batch. */ -function flowDirective(batch: ChatMessage[]): string { - const m = batch.find((x) => (x.content ?? "").includes("[Flow]")); - return m?.content ?? ""; -} - -describe("runTurn complexity steering", () => { - it("injects a complex-request directive for a complex prompt", async () => { - const { provider, seen } = makeScriptedProvider([ - { content: "final answer" }, - ]); +describe("runTurn", () => { + it("runs a simple one-shot prompt to completion", async () => { + const { provider } = makeScriptedProvider([{ content: "here is the answer" }]); const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never); - const { params } = buildParams("Please refactor the architecture across multiple files"); + const { params } = buildParams("hello"); await svc.runTurn(params); - // The complexity directive system message must be present before the LLM. - const flow = flowDirective(seen[0] ?? []); - expect(flow).toContain("flagged as complex"); - }); - - it("injects a keep-it-simple directive instead", async () => { - const { provider, seen } = makeScriptedProvider([{ content: "hi" }]); - const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never); - const { params } = buildParams("what is 2+2"); - await svc.runTurn(params); - const flow = flowDirective(seen[0] ?? []); - expect(flow).toContain("looks simple"); - expect(flow).not.toContain("complex"); + expect(params.in_flight.value).toBe(false); }); }); @@ -266,6 +249,25 @@ describe("runTurn self-review", () => { // review_usage emitted. expect(events.some((e) => e.kind === "review_usage")).toBe(true); }); + + it("passes the work-in-progress to the reviewer (not an empty slate)", async () => { + const { provider, chatCalls } = makeScriptedProvider([ + { tools: [{ name: "edit" }] }, + { content: "fixed" }, + ]); + (provider as unknown as { setChatReply(s: string): void }).setChatReply( + "- [PRIORITY: medium] add a null check", + ); + const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({ edit: "changed" }), [] as never); + const { params } = buildParams("harden the parser"); + await svc.runTurn(params); + // The REVIEWER chat call carries the digest in its user message. + const reviewerCall = chatCalls.find((c) => + c.some((m) => (m.content ?? "").includes("Review the work-in-progress below")), + ); + expect(reviewerCall).toBeTruthy(); + expect(reviewerCall!.some((m) => (m.content ?? "").includes("harden the parser"))).toBe(true); + }); }); /* ── Compaction: efficient + accurate ─────────────────────────────── */ diff --git a/src/features/agent/application/turn_service.ts b/src/features/agent/application/turn_service.ts index 7721a8b..3cfdf4b 100644 --- a/src/features/agent/application/turn_service.ts +++ b/src/features/agent/application/turn_service.ts @@ -23,7 +23,6 @@ import { errorRecoveryNote, reviewerPrompt, } from "@zesdex/agent"; -import { isComplexRequest } from "@zesdex/workflow"; import type { ProviderService } from "./ports.ts"; import type { ToolExecutor } from "./index.ts"; @@ -59,6 +58,9 @@ const MAX_NO_PROGRESS_STREAK = 4; /** Self-review is bounded to this many passes per turn. */ const MAX_REVIEW_PASSES = 1; +/** Per-tool-call wall-clock budget before the tool is considered hung. */ +const TOOL_TIMEOUT_MS = 90_000; + /** Whether the output string denotes a tool error. */ function isErrorOutput(output: string): boolean { return output.startsWith("Error:"); @@ -206,16 +208,30 @@ async function executeToolCall( executor: ToolExecutor, sink: TurnEventSink, tc: ToolCall, + signal?: AbortSignal, ): Promise { const name = tc.function.name; const args = sanitizeToolArguments(tc.function.arguments); - let output: string; - try { - output = await executor.execute(name, args as JsonValue); - } catch (e) { - output = `Error: ${(e as Error).message}`; - } + const run = executor.execute(name, args as JsonValue).catch((e) => `Error: ${(e as Error).message}`); + + // Correct async cancellation: enforce a per-tool time budget and stop + // immediately if the turn is aborted while the tool is still running. + let timer: ReturnType | undefined; + const output = await Promise.race([ + run, + new Promise((resolve) => { + const onAbort = () => resolve("Error: Turn aborted by user"); + timer = setTimeout(() => resolve(`Error: Tool timed out after ${TOOL_TIMEOUT_MS / 1000}s`), TOOL_TIMEOUT_MS); + signal?.addEventListener("abort", onAbort, { once: true }); + // Release the abort listener + timer once the tool settles either way. + void run.finally(() => { + clearTimeout(timer); + signal?.removeEventListener("abort", onAbort); + }); + }), + ]); + if (timer) clearTimeout(timer); const isError = isToolFailure(name, output); const truncated = truncateToolOutput(output); @@ -237,6 +253,7 @@ async function executeToolCallsInParallel( executor: ToolExecutor, sink: TurnEventSink, toolCalls: ToolCall[], + signal?: AbortSignal, ): Promise { // Simple bounded concurrency preserving input order. const results: string[] = new Array(toolCalls.length); @@ -246,7 +263,7 @@ async function executeToolCallsInParallel( while (true) { const idx = next++; if (idx >= toolCalls.length) return; - results[idx] = await executeToolCall(executor, sink, toolCalls[idx]!); + results[idx] = await executeToolCall(executor, sink, toolCalls[idx]!, signal); } } @@ -445,23 +462,6 @@ export class AgentTurnServiceImpl { // Estimate request complexity from the last user message. const last = params.messages[params.messages.length - 1]; const requestLen = last?.content?.length ?? 0; - const userText = last?.content ?? ""; - - // Complexity steering: a one-shot directive so the model picks the right - // depth instead of relying on prose memory. Pure and cheap. - const complex = isComplexRequest(userText); - this.push(sink, { - kind: "system_note", - systemKind: "info", - message: complex ? "Complex request detected — plan before executing." : "Simple request — keep tool use minimal.", - }); - params.messages.push( - systemMessage( - complex - ? "[Flow] This request is flagged as complex. Enter a short plan with `plan_enter` and track steps with `todowrite` before starting edits." - : "[Flow] This request looks simple. If you can answer directly without tools, do so — do not spawn agents or workflows for it.", - ), - ); const errors = new ErrorTracker(); let sawToolCalls = false; @@ -507,9 +507,10 @@ export class AgentTurnServiceImpl { // Auto-compact oversized history before the LLM call. await this.autoCompactIfNeeded(params.messages); - // Adaptive generation parameters. - const maxTokens = adaptiveMaxTokens(requestLen); - const temperature = sawToolCalls ? 0.2 : 0.7; + // Generation parameters: a config override wins, else fall back to the +// adaptive defaults (lower temperature once the model is doing tool work). + const maxTokens = params.max_tokens ?? adaptiveMaxTokens(requestLen); + const temperature = params.temperature ?? (sawToolCalls ? 0.2 : 0.7); this.push(sink, { kind: "stream_start" }); @@ -547,10 +548,10 @@ export class AgentTurnServiceImpl { const parallel = toolCalls.length > 1 && toolCalls.every(isParallelSafe); const outputs = parallel - ? await executeToolCallsInParallel(this.toolExecutor, sink, toolCalls) + ? await executeToolCallsInParallel(this.toolExecutor, sink, toolCalls, abort.signal) : await (async () => { const seq: string[] = []; - for (const tc of toolCalls) seq.push(await executeToolCall(this.toolExecutor, sink, tc)); + for (const tc of toolCalls) seq.push(await executeToolCall(this.toolExecutor, sink, tc, abort.signal)); return seq; })(); @@ -580,11 +581,12 @@ export class AgentTurnServiceImpl { if (mutated) { pendingVerify = true; - verifyPrompted = false; // allow a fresh nudge after the next edit batch + // Do NOT re-arm verifyPrompted: the verify instruction is already in + // context, so re-injecting it on every edit batch only bloats tokens. } if (verified) { pendingVerify = false; // a successful bash run satisfies the verify nudge - verifyPrompted = false; + verifyPrompted = false; // a later edit batch may nudge again } // Bounded self-review pass after file mutations (at most one per turn). @@ -608,9 +610,16 @@ export class AgentTurnServiceImpl { sink: TurnEventSink, ): Promise { if (abort.signal.aborted) return; + // Reduce the work-in-progress to a focused digest so the reviewer has real + // context to critique (not an empty slate that invites hallucinated issues). + const digest = reduceMessagesToDigest(messages.slice(-80)); + if (digest.trim() === "") return; try { const result = await this.provider.chat( - [systemMessage(reviewerPrompt())], + [ + systemMessage(reviewerPrompt()), + userMessage(`Review the work-in-progress below and list actionable issues:\n\n---\n${digest}`), + ], undefined, 1024, 0.3, diff --git a/src/features/agent/domain/mod.ts b/src/features/agent/domain/mod.ts index 7398ab3..fc112a1 100644 --- a/src/features/agent/domain/mod.ts +++ b/src/features/agent/domain/mod.ts @@ -62,7 +62,6 @@ export function agentStatusDisplay(s: AgentStatus, error?: string): string { /** Events emitted onto the turn-event queue while an agent turn runs. */ export type TurnEvent = - | { kind: "assistant_message"; message: ChatMessage } | { kind: "tool_result"; tool_call_id: string; tool_name: string; output: string; is_error: boolean; path: string | null } | { kind: "system_note"; systemKind: string; message: string } | { kind: "stream_start" } @@ -183,6 +182,10 @@ export interface AgentTurnParams { api_key: string; model: string; api_base?: string; + /** Optional override for the LLM max_tokens ceiling (defaults to adaptive). */ + max_tokens?: number; + /** Optional override for the LLM temperature (defaults to 0.7 / 0.2 tooling). */ + temperature?: number; } /** Minimal event-sink abstraction (backs the Rust `Arc>`). */ diff --git a/src/features/agent/infrastructure/tools/parallel_delegate.test.ts b/src/features/agent/infrastructure/tools/parallel_delegate.test.ts new file mode 100644 index 0000000..bae5de8 --- /dev/null +++ b/src/features/agent/infrastructure/tools/parallel_delegate.test.ts @@ -0,0 +1,35 @@ +/** + * Tests for the LLM-driven parallel-delegation splitter. The decomposition is + * decided by the AI (no hardcoded keyword guessing); these tests lock in the + * tolerant JSON parsing and the access-tier normalization. + */ +import { describe, expect, test } from "bun:test"; +import { parseDirectivesJson } from "./parallel_delegate.ts"; + +describe("parseDirectivesJson", () => { + test("parses a clean JSON array", () => { + const out = parseDirectivesJson( + `[{"directive":"Scan the config code","access":"read"},{"directive":"Implement the change","access":"full"}]`, + ); + expect(out.length).toBe(2); + expect(out[0]).toEqual({ directive: "Scan the config code", access: "read" }); + }); + + test("extracts the array from markdown fences with a preamble", () => { + const out = parseDirectivesJson( + 'Here are the subtasks:\n```json\n[{"directive":"A","access":"read"},{"directive":"B","access":"full"}]\n```\nDone.', + ); + expect(out.map((d) => d.directive)).toEqual(["A", "B"]); + }); + + test("normalises unknown access tiers to write", () => { + const out = parseDirectivesJson(`[{"directive":"do it","access":"admin"}]`); + expect(out[0]!.access).toBe("write"); + }); + + test("drops empty directives and malformed input", () => { + expect(parseDirectivesJson(`[{"directive":"","access":"write"}]`)).toEqual([]); + expect(parseDirectivesJson("not json at all")).toEqual([]); + expect(parseDirectivesJson("")).toEqual([]); + }); +}); \ No newline at end of file diff --git a/src/features/agent/infrastructure/tools/parallel_delegate.ts b/src/features/agent/infrastructure/tools/parallel_delegate.ts index f09a2c8..358ebe4 100644 --- a/src/features/agent/infrastructure/tools/parallel_delegate.ts +++ b/src/features/agent/infrastructure/tools/parallel_delegate.ts @@ -50,7 +50,8 @@ export class ParallelDelegate implements Tool { })) .filter((d) => d.directive !== ""); } else { - directives = fallbackSplit(task, maxParallel); + // Let the LLM decide the decomposition instead of guessing keywords. + directives = await aiDecomposeTask(task, maxParallel); } if (directives.length === 0) { @@ -73,22 +74,67 @@ export class ParallelDelegate implements Tool { } } -/** Fallback splitting when LLM is unavailable. */ -function fallbackSplit(task: string, maxParallel: number): Array<{ directive: string; access: string }> { - const directives: Array<{ directive: string; access: string }> = []; - if (task.includes("backend") || task.includes("api") || task.includes("server")) { - directives.push({ directive: `Implement the backend/API components for: ${task}`, access: "write" }); +/** + * Have the LLM decompose a task into focused, non-overlapping parallel + * sub-directives. This is the AI deciding how to split the work — no keyword + * guessing. On LLM failure (no provider / error), degrades to a single honest + * full-task directive rather than fabricated "part N" stubs. + */ +async function aiDecomposeTask( + task: string, + maxParallel: number, +): Promise> { + try { + const { resolveConfig } = await import("../../../subagent/infrastructure/config_resolver.ts"); + const { buildProviderService } = await import("../../../subagent/infrastructure/http_provider.ts"); + const { baseUrl, apiKey, model } = await resolveConfig(); + const svc = buildProviderService(baseUrl, apiKey, model); + + const systemPrompt = + "You decompose a large task into independent, non-overlapping subtasks that can be worked on " + + "in parallel. Return a JSON array of up to the requested count of objects, each with a " + + "'directive' (a concrete, self-contained action) and an 'access' tier of exactly one of " + + "read | write | full. read = inspect/search only; write = edit files; full = edit + run shell. " + + "Prefer fewer, genuinely parallel directives over many that touch the same files. " + + "Return ONLY the JSON array, no prose."; + + const { message } = await svc.chat( + [ + { role: "system", content: systemPrompt }, + { role: "user", content: `Task: ${task}\nMax subtasks: ${maxParallel}\n` }, + ], + undefined, + 1024, + 0.3, + ); + + const parsed = parseDirectivesJson(message.content ?? ""); + if (parsed.length > 0) return parsed.slice(0, maxParallel); + } catch { + /* fall through to the single-directive fallback */ + } + // Honest fallback: run the whole task as one agent rather than guessing a split. + return [{ directive: task, access: "full" }]; +} + +/** Parse an LLM JSON array of directives, tolerant of stray prose/markdown fences. */ +export function parseDirectivesJson(raw: string): Array<{ directive: string; access: string }> { + const fenced = raw.match(/```(?:json)?\s*([\s\S]*?)```/i); + const body = fenced ? fenced[1]! : raw; + const start = body.indexOf("["); + const end = body.lastIndexOf("]"); + if (start < 0 || end <= start) return []; + try { + const arr = JSON.parse(body.slice(start, end + 1)) as unknown; + if (!Array.isArray(arr)) return []; + return arr + .filter((d): d is Record => typeof d === "object" && d !== null) + .map((d) => ({ + directive: typeof d.directive === "string" ? d.directive.trim() : "", + access: ["read", "write", "full"].includes(String(d.access)) ? String(d.access) : "write", + })) + .filter((d) => d.directive !== ""); + } catch { + return []; } - if (task.includes("frontend") || task.includes("ui") || task.includes("client")) { - directives.push({ directive: `Implement the frontend/UI components for: ${task}`, access: "write" }); - } - if (task.includes("test") || task.includes("unit")) { - directives.push({ directive: `Write unit tests for: ${task}`, access: "read" }); - } - if (directives.length === 0) { - for (let i = 0; i < maxParallel; i++) { - directives.push({ directive: `Part ${i + 1} of parallel task: ${task}`, access: "write" }); - } - } - return directives.slice(0, maxParallel); } diff --git a/src/features/agent/infrastructure/tools/registry.test.ts b/src/features/agent/infrastructure/tools/registry.test.ts index ea4336b..8b44526 100644 --- a/src/features/agent/infrastructure/tools/registry.test.ts +++ b/src/features/agent/infrastructure/tools/registry.test.ts @@ -37,12 +37,12 @@ describe("allTools registry", () => { describe("read-only tools are parallel-safe", () => { test("parallel-safe list", () => { - for (const name of ["read", "grep", "glob", "semantic_search", "list_symbols", "web_search", "recall", "dir_list", "pong", "seq_think", "dir_cache_update"]) { + for (const name of ["read", "grep", "glob", "semantic_search", "list_symbols", "web_search", "recall", "dir_list", "pong", "seq_think"]) { expect(toolIsParallelSafe(name)).toBe(true); } }); - test("mutating tools are not parallel-safe", () => { - for (const name of ["edit", "write", "delete", "bash", "git_operator", "git_worktree", "remember", "forget", "todowrite", "plan_enter", "spawn_agents"]) { + test("mutating tools (incl. dir_cache_update, which writes the cache) are not parallel-safe", () => { + for (const name of ["edit", "write", "delete", "bash", "git_operator", "git_worktree", "remember", "forget", "todowrite", "plan_enter", "spawn_agents", "dir_cache_update"]) { expect(toolIsParallelSafe(name)).toBe(false); } }); diff --git a/src/features/agent/infrastructure/tools/registry.ts b/src/features/agent/infrastructure/tools/registry.ts index f6f84f7..711f38d 100644 --- a/src/features/agent/infrastructure/tools/registry.ts +++ b/src/features/agent/infrastructure/tools/registry.ts @@ -88,7 +88,6 @@ export function toolIsParallelSafe(name: string): boolean { "dir_list", "pong", "seq_think", - "dir_cache_update", ].includes(name); } diff --git a/src/features/workflow/infrastructure/complexity.ts b/src/features/workflow/infrastructure/complexity.ts deleted file mode 100644 index cb8de3f..0000000 --- a/src/features/workflow/infrastructure/complexity.ts +++ /dev/null @@ -1,22 +0,0 @@ -/** - * Complexity heuristics — determine whether a request is complex enough to - * warrant hive-mind orchestration. - * Mirrors `apps/infrastructure/src/workflow/hive_mind/complexity.rs`. - */ - -/** Heuristics to determine if a request is complex enough for hive-mind. */ -export function isComplexRequest(task: string): boolean { - const complexityIndicators = [ - "refactor", - "redesign", - "multiple files", - "architecture", - "migration", - "comprehensive", - "end-to-end", - "full-stack", - ]; - - const taskLower = task.toLowerCase(); - return complexityIndicators.some((indicator) => taskLower.includes(indicator)); -} diff --git a/src/features/workflow/infrastructure/index.ts b/src/features/workflow/infrastructure/index.ts index b20f04e..089e3ef 100644 --- a/src/features/workflow/infrastructure/index.ts +++ b/src/features/workflow/infrastructure/index.ts @@ -7,5 +7,4 @@ export { executeWorkflowScript, executePrimitive, formatWorkflowResult, MAX_NODE export { executeWorkflow } from "./orchestrate.ts"; export { executeCycle } from "./cycle.ts"; export { synthesizeConsensus } from "./synthesis.ts"; -export { isComplexRequest } from "./complexity.ts"; export { writeHiveMindConvergence } from "./docs.ts"; diff --git a/src/features/workflow/infrastructure/script.test.ts b/src/features/workflow/infrastructure/script.test.ts new file mode 100644 index 0000000..71b85f8 --- /dev/null +++ b/src/features/workflow/infrastructure/script.test.ts @@ -0,0 +1,61 @@ +/** + * Tests for the workflow YAML-subset parser, including the folded multi-line + * directive support. + */ +import { describe, expect, test } from "bun:test"; +import { parseWorkflowScript } from "./script.ts"; + +describe("parseWorkflowScript", () => { + test("parses a flat single-line workflow", () => { + const script = parseWorkflowScript(` +name: my-workflow +phases: + - name: research + directive: "Explore the codebase for the auth module." + - name: implement + directive: "Implement the change." +`); + expect(script.name).toBe("my-workflow"); + expect(script.phases.length).toBe(2); + expect(script.phases[0]!.name).toBe("research"); + expect(script.phases[0]!.directive).toBe("Explore the codebase for the auth module."); + }); + + test("folds an indented multi-line directive into one string", () => { + const script = parseWorkflowScript(` +name: audit +phases: + - name: analyze + directive: | + Look at two things: + first, how auth is wired; + second, where secrets are stored. +`); + expect(script.phases.length).toBe(1); + const d = script.phases[0]!.directive; + expect(d).toContain("Look at two things:"); + expect(d).toContain("second, where secrets are stored."); + // Line breaks are preserved as folded content, not flattened. + expect(d).toContain("\n"); + }); + + test("an indented continuation does not swallow the next phase", () => { + const script = parseWorkflowScript(` +phases: + - name: a + directive: first + and more + - name: b + directive: second +`); + expect(script.phases.length).toBe(2); + expect(script.phases[0]!.directive).toContain("and more"); + expect(script.phases[1]!.name).toBe("b"); + expect(script.phases[1]!.directive).toBe("second"); + }); + + test("empty or unparseable yaml yields no phases", () => { + expect(parseWorkflowScript("").phases.length).toBe(0); + expect(parseWorkflowScript("name: x\n").phases.length).toBe(0); + }); +}); \ No newline at end of file diff --git a/src/features/workflow/infrastructure/script.ts b/src/features/workflow/infrastructure/script.ts index 69e0cdc..dde58f5 100644 --- a/src/features/workflow/infrastructure/script.ts +++ b/src/features/workflow/infrastructure/script.ts @@ -31,21 +31,62 @@ export function parseWorkflowScript(yaml: string): WorkflowScript { let inPhases = false; let currentPhase: Partial | null = null; + /** Set when `directive:` was opened, to fold more-indented lines into it. */ + let directiveIndent: number | null = null; + + const flushPhase = (): void => { + if (currentPhase && (currentPhase.name || currentPhase.directive)) { + phases.push({ + name: currentPhase.name ?? "phase", + directive: currentPhase.directive ?? "", + }); + } + currentPhase = null; + directiveIndent = null; + }; + + const indentOf = (s: string): number => { + const m = s.match(/^(\s*)/); + return m ? m[1]!.length : 0; + }; + + /** Strip one surrounding pair of single/double quotes, YAML-scalar style. */ + const unquote = (v: string): string => { + const t = v.trim(); + if (t.length >= 2 && ((t.startsWith('"') && t.endsWith('"')) || (t.startsWith("'") && t.endsWith("'")))) { + return t.slice(1, -1); + } + return t; + }; for (const raw of lines) { const trimmed = raw.trim(); if (trimmed === "" || trimmed.startsWith("#")) continue; + // Fold continuation lines into the open directive (indented block text). + // A continuation is a non-blank line indented deeper than the directive + // key, that is not a new list item and not a `key:` assignment. + if (currentPhase && directiveIndent !== null) { + const indent = indentOf(raw); + const isListItem = trimmed.startsWith("- "); + const isKv = /^[\w-]+:\s/.test(trimmed); + if (indent > directiveIndent && !isListItem && !isKv) { + currentPhase.directive = `${currentPhase.directive ?? ""}\n${trimmed}`; + continue; + } + } + // Top-level key: value const topMatch = trimmed.match(/^(\w+):\s*(.*)$/); if (topMatch && !raw.startsWith(" ") && !raw.startsWith("-")) { const key = topMatch[1]!; - const val = topMatch[2]!.trim(); + const val = unquote(topMatch[2]!); if (key === "name" && val !== "") { name = val; } else if (key === "phases") { inPhases = true; } + directiveIndent = null; continue; } @@ -54,18 +95,17 @@ export function parseWorkflowScript(yaml: string): WorkflowScript { // List item: - name: foo or - directive: bar const listMatch = trimmed.match(/^-\s+(\w+):\s*(.*)$/); if (listMatch) { - // If there was a previous phase, push it - if (currentPhase && (currentPhase.name || currentPhase.directive)) { - phases.push({ - name: currentPhase.name ?? "phase", - directive: currentPhase.directive ?? "", - }); - } + flushPhase(); const key = listMatch[1]!; - const val = listMatch[2]!.trim(); + const val = unquote(listMatch[2]!); currentPhase = { name: undefined, directive: undefined }; if (key === "name") currentPhase.name = val; - if (key === "directive") currentPhase.directive = val; + if (key === "directive") { + currentPhase.directive = val; + directiveIndent = indentOf(raw); + } else { + directiveIndent = null; + } continue; } @@ -74,20 +114,19 @@ export function parseWorkflowScript(yaml: string): WorkflowScript { const kvMatch = trimmed.match(/^(\w+):\s*(.*)$/); if (kvMatch) { const key = kvMatch[1]!; - const val = kvMatch[2]!.trim(); + const val = unquote(kvMatch[2]!); if (key === "name") currentPhase.name = val; - if (key === "directive") currentPhase.directive = val; + if (key === "directive") { + currentPhase.directive = val; + directiveIndent = indentOf(raw); + } else { + directiveIndent = null; + } } } } - // Push the last phase - if (currentPhase && (currentPhase.name || currentPhase.directive)) { - phases.push({ - name: currentPhase.name ?? "phase", - directive: currentPhase.directive ?? "", - }); - } + flushPhase(); return { name, phases }; } diff --git a/src/interfaces/cli/compose.ts b/src/interfaces/cli/compose.ts index 4ef1a69..dcceb7c 100644 --- a/src/interfaces/cli/compose.ts +++ b/src/interfaces/cli/compose.ts @@ -42,6 +42,10 @@ export interface WiredRuntime { apiBase?: string; /** Fetch the provider's available model ids (via GET /models). */ listModels?: () => Promise; + /** Optional LLM max_tokens override from settings. */ + maxTokens?: number; + /** Optional LLM temperature override from settings. */ + temperature?: number; } /** @@ -86,6 +90,8 @@ export async function runSingleProcess(): Promise { model, apiBase: baseUrl, listModels: () => llmClient.listModels(), + maxTokens: settings.max_tokens, + temperature: settings.temperature, }; } @@ -121,6 +127,8 @@ export function buildTurnParams( api_key: runtime.apiKey, model: runtime.model, api_base: runtime.apiBase, + max_tokens: runtime.maxTokens, + temperature: runtime.temperature, }; } diff --git a/src/interfaces/tui/run.ts b/src/interfaces/tui/run.ts index a1d59ce..0273dd4 100644 --- a/src/interfaces/tui/run.ts +++ b/src/interfaces/tui/run.ts @@ -54,6 +54,8 @@ function makeTurnRunner(runtime: WiredRuntime, state: AppStateRest): TurnRunner api_key: runtime.apiKey, model: runtime.model, api_base: runtime.apiBase, + max_tokens: runtime.maxTokens, + temperature: runtime.temperature, }); }; }