Ship the batch: undo, parallel subagents, lazy MCP, hot-reloaded skills; docs and CI/CD

Loop and ergonomics batch across Now/Next and Maintenance:
- /undo and /redo via pre-prompt file snapshots (snapshot.ts)
- task takes a tasks[] array and runs investigations concurrently (subagent.ts)
- lazy MCP tools: mcp_list/mcp_inspect/mcp_call meta-tools, eager opt-in (mcp.ts, config.ts)
- skill tool reads its list live so a mid-session install is callable next turn (skills.ts)
- tool-name lists (tool-kinds.ts) derived from a mutating() marker; gates previously ungated writes
- prune/session recovery path summarized, and step-back doom-loop primitive (step-back.ts)
- @file completion re-walks on a slow cooldown; estimateTokens and pricing labeled as estimates

Docs: README, CHANGELOG, docs/{mcp,architecture,development} updated to match.
CI/CD: bun install-store cache and concurrency gates on both workflows; release.yml now
composes file-based release notes via scripts/make-release-notes.ts and verifies every binary.
This commit is contained in:
Muhammad Zakir Ramadhan
2026-09-17 17:49:19 +07:00
parent ffa9a02c26
commit f8c3cc2d8e
57 changed files with 3808 additions and 573 deletions
+17
View File
@@ -5,6 +5,10 @@ on:
branches: [main] branches: [main]
pull_request: pull_request:
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: true
jobs: jobs:
check: check:
runs-on: ${{ matrix.os }} runs-on: ${{ matrix.os }}
@@ -16,9 +20,22 @@ jobs:
os: [ubuntu-latest, macos-latest, windows-latest] os: [ubuntu-latest, macos-latest, windows-latest]
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2 - uses: oven-sh/setup-bun@v2
with: with:
bun-version: 1.3.14 bun-version: 1.3.14
# bun install --frozen-lockfile re-resolves nothing but still pays the network
# cost of a cold store; cache the store so a no-change run skips it. Key is per
# OS and lockfile, so a dependency bump or a new OS each gets a fresh cache.
- name: Cache the bun install store
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('bun.lock') }}
restore-keys: |
bun-${{ runner.os }}-
- run: bun install --frozen-lockfile - run: bun install --frozen-lockfile
- run: bun run typecheck - run: bun run typecheck
- run: bun test - run: bun test
+38 -4
View File
@@ -10,9 +10,16 @@ on:
type: boolean type: boolean
default: true default: true
concurrency:
group: release-${{ github.ref }}
cancel-in-progress: false
permissions: permissions:
contents: write contents: write
env:
BUN_VERSION: 1.3.14
jobs: jobs:
verify: verify:
runs-on: ubuntu-latest runs-on: ubuntu-latest
@@ -20,7 +27,14 @@ jobs:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2 - uses: oven-sh/setup-bun@v2
with: with:
bun-version: 1.3.14 bun-version: ${{ env.BUN_VERSION }}
- name: Cache the bun install store
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('bun.lock') }}
restore-keys: |
bun-${{ runner.os }}-
- run: bun install --frozen-lockfile - run: bun install --frozen-lockfile
- run: bun run typecheck - run: bun run typecheck
- run: bun test - run: bun test
@@ -32,15 +46,26 @@ jobs:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2 - uses: oven-sh/setup-bun@v2
with: with:
bun-version: 1.3.14 bun-version: ${{ env.BUN_VERSION }}
- name: Cache the bun install store
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('bun.lock') }}
restore-keys: |
bun-${{ runner.os }}-
- run: bun install --frozen-lockfile - run: bun install --frozen-lockfile
# Bun cross-compiles every target from one host, so no build matrix is needed. # Bun cross-compiles every target from one host, so no build matrix is needed.
# release.ts also fails the build when the tag and src/version.ts disagree. # release.ts also fails the build when the tag and src/version.ts disagree.
- run: bun run release - run: bun run release
- name: Check the binary reports the right version - name: Check the binaries report the right version and are non-empty
run: | run: |
for f in dist/release/shiro-linux-x64 dist/release/shiro-linux-arm64 \
dist/release/shiro-darwin-x64 dist/release/shiro-darwin-arm64; do
[ -s "$f" ] || { echo "$f is missing or empty"; exit 1; }
done
chmod +x dist/release/shiro-linux-x64 chmod +x dist/release/shiro-linux-x64
./dist/release/shiro-linux-x64 --version ./dist/release/shiro-linux-x64 --version
./dist/release/shiro-linux-x64 --version | grep -q "$(bun -e 'console.log((await import("./src/version.ts")).VERSION)')" ./dist/release/shiro-linux-x64 --version | grep -q "$(bun -e 'console.log((await import("./src/version.ts")).VERSION)')"
@@ -66,12 +91,21 @@ jobs:
name: shiro-binaries name: shiro-binaries
path: dist/release path: dist/release
# Release body: a short summary plus the artifact inventory, so the release
# page reads like a hand-written one instead of the raw PR list. The changelog
# is the prose; scripts/make-release-notes.ts extracts the section for this tag
# and fails if the heading is missing, rather than publishing an empty body.
- name: Compose release notes
env:
RELEASE_TAG: ${{ github.ref_name }}
run: bun run scripts/make-release-notes.ts
- name: Publish the release - name: Publish the release
env: env:
GH_TOKEN: ${{ github.token }} GH_TOKEN: ${{ github.token }}
run: | run: |
gh release create "${{ github.ref_name }}" \ gh release create "${{ github.ref_name }}" \
--title "shiro-neko ${{ github.ref_name }}" \ --title "shiro-neko ${{ github.ref_name }}" \
--generate-notes \ --notes-file release_notes.md \
$([[ "${{ github.ref_name }}" == *-* ]] && echo --prerelease) \ $([[ "${{ github.ref_name }}" == *-* ]] && echo --prerelease) \
dist/release/* dist/release/*
+182 -118
View File
@@ -1,73 +1,97 @@
# Audit # Audit
Checklist from a full audit of the codebase, run against `main` at `a22d8e1` ("release 0.1.0-beta.5"). Checklist from a full audit of the codebase, run against `main` at `ffa9a02` ("release 1.0.0").
The greps cover every file under `src/`, `test/`, `docs/`, `.github/workflows/`, and `scripts/`. The greps cover every file under `src/`, `test/`, `docs/`, `.github/workflows/`, and `scripts/`.
Nothing here is a fix — it is a list. Items already tracked in `TODO.md` or `ROADMAP.md` say so; Nothing here is a fix — it is a list. Items already tracked in `TODO.md` or `ROADMAP.md` say so;
untracked items are marked **not yet tracked**. untracked items are marked **not yet tracked**. Items checked off were resolved after the audit,
in the same working tree.
The three bugs and the two untracked doc-drift items from the beta.5 audit are all fixed; this pass
also cleared the dead method, the three stale doc lines, and every UI coverage gap that remained —
`cli.tsx`, `Onboard.tsx`, `panel-bodies.ts`, `buses.ts`, and `PromptInput.tsx` — see
[section B](#b-bugs), [section D](#d-documentation-drift-untracked), and
[section E](#e-test-coverage-gaps).
--- ---
## A. Clean findings (verified, no action needed) ## A. Clean findings (verified, no action needed)
- [x] **No TODO/FIXME/HACK markers in `src/`.** All 26 matches are false positives: - [x] **No TODO/FIXME/HACK markers in `src/`.** Seven matches, all false positives: the
placeholder attributes, the `TODO_MARK` export in `src/notebook.ts` (a literal string `TODO_MARK` export and its uses (`src/notebook.ts:120`, `src/ui/transcript.ts:1,201,203`,
ingredient of the todos feature), and "later" in prose. `src/ui/Panels.tsx:4,34`) and a `TODO` inside the `commit` skill's prose
(`src/skills-md/commit.md:21`). Zero real markers.
- [x] **No `as any` / `@ts-ignore` / `@ts-expect-error` / `@ts-nocheck` / `: any` in `src/`.** - [x] **No `as any` / `@ts-ignore` / `@ts-expect-error` / `@ts-nocheck` / `: any` in `src/`.**
Zero matches. The project's own `docs/development.md` rule is being kept. One grep hit and it is the word "any" in a skill's prose (`src/skills-md/readme.md:36`).
- [x] **No silently swallowed errors in `src/`.** Every `catch` was reviewed: Zero casts. The project's own `docs/development.md` rule is being kept.
- `src/registry.ts:116,155` — JSON parse failures become descriptive errors. - [x] **No empty catch blocks and no silently swallowed errors in `src/`.** `catch {}` and
- `src/registry.ts:120-123,159-162` — zod schema validation with named failure reasons. `catch (e) {}` match zero times. 83 `catch` sites were grepped and the named ones reviewed:
- `src/tools.ts:81-82` — a batch read failure is reported in place (`[unreadable: ...]`).
- `src/tools.ts:463` — a stat probe returns `false` (sentinel, not a swallow).
- `src/tools.ts:501` — `rg` unavailable → `undefined`, falls back to the JS grep.
- `src/tools.ts:551,849` — an unreadable file is skipped during a scan (deliberate).
- `src/tools.ts:630` — `taskkill` absent → plain kill (commented).
- `src/tools.ts:795,889` — stat / JSON failure → a descriptive `Error`.
- `src/subagent.ts:248` — usage unavailable on an errored run (commented); `:251` reports
the error **and rethrows** (never swallowed).
- `src/registry.ts` — parse failures become descriptive errors; the index is schema-validated.
- `src/plugins.ts:59-63` — a throwing plugin hook fails **closed** (blocks the call). - `src/plugins.ts:59-63` — a throwing plugin hook fails **closed** (blocks the call).
- `src/subagent.ts:233-237` — errors are reported and rethrown (never swallowed). - `src/ui/App.tsx` — all 17 catch blocks surface the message in the UI.
- `src/tools.ts:80-82` — a batch read failure is reported in place, not thrown. - `src/plugins-builtin.ts` — the one quiet catch (a missing formatter binary) is deliberate
- `src/headless.ts` serialization flattens `Error` before `JSON.stringify` (would emit `{}`). and commented.
- `src/ui/App.tsx` — all 14 catch blocks surface the message in the UI. - [x] **No skipped tests.** No `.skip`, `xit`, or `xdescribe` in `test/`.
- `src/plugins-builtin.ts:213-215` — the only quiet `catch`, and it is deliberate,
commented ("a missing binary is not worth interrupting the turn over").
- [x] **No skipped tests.** No `.skip`, `xit`, or `xdescribe` in `test/` (one match was
`process.exit(` containing "xit(").
- [x] **Registry fetches are size-capped and schema-validated.** `src/registry.ts` caps - [x] **Registry fetches are size-capped and schema-validated.** `src/registry.ts` caps
content-length and body bytes (`fetchText`, lines 100-108), validates the index with content-length and body bytes (`fetchText`, lines 98-109), validates the index with
`indexSchema` (line 120), and regex-validates every plugin manifest pattern (line 167). `indexSchema` (lines 120-123), and regex-validates every plugin manifest pattern
- [x] **Version/tag consistency is enforced twice.** `scripts/release.ts` refuses a build when (lines 164-173).
the tag and `src/version.ts` disagree (line 129), and the release workflow asserts the - [x] **Version consistency is enforced three times.** `package.json` vs `src/version.ts`
built binary prints the expected version (`.github/workflows/release.yml:42-46`). (`scripts/release.ts:60-64`), the release tag vs `VERSION` via `GITHUB_REF_NAME`
- [x] **Install scripts match the build targets.** `test/ci.test.ts:85-101` iterates every (`scripts/release.ts:66-70`), and the built binary's own `--version` output in CI
`TARGETS` entry from `scripts/release.ts` and asserts the shell/PowerShell installers (`.github/workflows/release.yml:42-46`). Both files currently read `1.0.0`.
fetch exactly those asset names. - [x] **Install scripts match the build targets.** `test/ci.test.ts` iterates every `TARGETS`
- [x] **`.env`/`.pem` are refused on read;** the default permission table entry from `scripts/release.ts:13-19` and asserts the shell/PowerShell installers fetch
(`src/permission.ts:175-186`) matches the approvals banner in `README.md`. Unknown tools exactly those asset names (~lines 89-94), plus checksum verification (~74-78) and version
(MCP, plugins) default to `ask` rather than allow (line 235-236), so `mcp__*` needs no pinning / target directory (~81-87).
explicit rule. - [x] **`.env`/`.pem` are refused on read; writes and commands ask.** The default permission
- [x] **`--yolo` cannot bypass the guard plugin.** Defaults fold `ask` into `allow` but never table (`src/permission.ts:175-186`) matches the approvals banner in `README.md:72`. Unknown
touch `deny` (`src/permission.ts:211-213`), and the guard refuses destructive commands tools (MCP, plugins) default to `ask` rather than allow (`src/permission.ts:235-236`), so
in `beforeToolCall`, ahead of any approval. `mcp__*` needs no explicit rule.
- [x] **Pinned toolchain.** Both workflows pin `bun-version: 1.3.14`, and `test/ci.test.ts:48-52` - [x] **`--yolo` cannot bypass the guard plugin.** `check()` returns a `deny` before the `yolo`
fails if the pin ever disagrees with the local `Bun.version`. fold (`src/permission.ts:282`), the fold itself only touches `ask` (`:293`), and the guard
- [x] **All three platforms in CI.** `ci.yml` runs the suite on ubuntu, macos, windows plugin refuses destructive commands in `beforeToolCall`, ahead of any approval.
- [x] **Pinned toolchain.** Both workflows pin `bun-version: 1.3.14`
(`.github/workflows/ci.yml:21`, `.github/workflows/release.yml:23,35`), and
`test/ci.test.ts:54` fails if the pin ever disagrees with the local `Bun.version`.
- [x] **All three platforms in CI.** `ci.yml:16` runs the suite on ubuntu, macos, windows
(required — the tools shell out to `rg`, git, and a platform shell). (required — the tools shell out to `rg`, git, and a platform shell).
--- ---
## B. Bugs ## B. Bugs
- [ ] **`src/ui/App.tsx:613` — formatting glitch.** The `}` closing the `try` is jammed onto the Fixed — the three from the beta.5 audit, the dead-code finding this audit surfaced, and a cursor
same line as the preceding statement: rendering bug found from a screenshot afterwards.
`push({ kind: 'info', text: await hooks.summarizeMemory() }); } catch (e) {`
Cosmetic only, but it is the kind of blemish left by an unformatted edit and reads as a - [x] **`src/ui/PromptInput.tsx` — the cursor was drawn with hand-written SGR escapes.** `invert()`
slip. **not yet tracked** built the caret by pasting `\u001B[7m`/`\u001B[27m` into the text string
- [ ] **`.github/workflows/release.yml` — the `dry_run` input is dead.** `workflow_dispatch` (`invert(placeholder.slice(0, 1))`, and the same for the character under the cursor). Ink
declares `inputs.dry_run` (default `true`) but no step ever reads it. Nothing consults the measures string content as printable columns, so those escapes were counted as text and the
value, so `dry_run=false` changes nothing, and because the `publish` job gates on rest of the line was written one cell to the right — the orphaned first letter of the
`startsWith(github.ref, 'refs/tags/v')`, a manual run can never publish regardless of the placeholder (`t ype to queue for the next turn…`), and a corrupted cell wherever the line
input. Either wire the input into the `publish` `if`, or delete it and let the tag-only wrapped. Replaced with Ink's `inverse` prop in all three render paths. The old tests passed
gate be the whole story. **not yet tracked** *because* they asserted on the escape-producing helper; they now assert the rendered text is
- [ ] **`README.md` says "Nineteen built-in tools" — it is now twenty.** `git_commit_message` free of escapes, which is the property that actually matters.
(shipped in beta.5 via `src/commit.ts` + `cli.tsx:281`) is a built-in tool, and - [x] **`src/ui/App.tsx` — memory-command formatting glitch (was line 613).** The `}` closing the
`TOOL_SETS.git` carries it (`src/tools-git.ts:189`). The count is one short; `try` is now on its own line; the block reads cleanly at `src/ui/App.tsx:636-646`.
`docs/tools.md` already says "twenty" (line 63), so README is the stale one. - [x] **`.github/workflows/release.yml` — the `dry_run` input is no longer dead.** It is wired
**not yet tracked** into the `publish` job's `if`: a tag push always publishes, a manual dispatch publishes only
when `dry_run` is unchecked (`release.yml:56-60`).
- [x] **`README.md` — tool count was stale, now correct.** It says "Forty-one built-in tools"
(`README.md:116`) and "41 built-in tools" (`README.md:164`), matching `docs/tools.md:49`.
- [x] **`src/permission.ts` — `granted_` was dead code, now deleted.** A public method with a
trailing underscore (the convention here is a `_`-*prefix* for an unused parameter, not a
suffix). It returned the session's granted patterns and was called nowhere — not in `src/`,
not in `test/`. Removed; `check()` and `grant()` are the live surface.
--- ---
@@ -75,98 +99,138 @@ untracked items are marked **not yet tracked**.
### TODO.md "Now" — next up ### TODO.md "Now" — next up
- [ ] **Summarize the pruned span.** Compaction drops messages and tells the model nothing, so a - [ ] **Summarize the pruned span.** Compaction keeps the model's memory of a turn but tells it
decision from earlier in the session can be contradicted. (TODO.md `## Now`, first item) nothing about the messages it dropped, so a decision from earlier in the session can be
- [ ] **A spend ceiling.** `maxSpendUsd` in config, warn at 80%, refuse the next turn at 100%, contradicted with confidence. (TODO.md `## Now`; ROADMAP `## Next` "Lossless-enough
headless exits non-zero naming the ceiling. Nothing stops a looping headless run today. compaction" — same work)
(TODO.md `## Now`)
- [ ] **A cheaper model for subagents.** `subagentModel` in config; an `explore` subagent is
search, not reasoning, and today pays the parent's per-token rate. (TODO.md `## Now`)
- [ ] **Hot-reload an installed entry.** `/registry add` writes the file and says restart; the - [ ] **Hot-reload an installed entry.** `/registry add` writes the file and says restart; the
skill catalogue and guard chain are assembled at boot. (TODO.md `## Now`) skill catalogue and the guard chain are assembled at boot. (TODO.md `## Now`)
### TODO.md "Next" ### TODO.md "Next"
- [ ] **MCP without the schema tax.** Twenty MCP tools ≈ 2,750 tokens of schema per request; - [ ] **MCP without the schema tax.** Every MCP tool's schema is in the prompt on every request
`toolSets` does not gate them. Plan: `mcp_list` / `mcp_inspect` / `mcp_call` meta-tools, and `toolSets` does not gate them; a twenty-tool server costs ~2,750 tokens a turn whether
prompt names servers not schemas. (TODO.md `## Next`; ROADMAP `## Next` + `## Later`) used or not. Plan: `mcp_list` / `mcp_inspect` / `mcp_call` meta-tools, prompt names servers
- [ ] **Custom commands from a file.** `.shiro/commands/*.md`, `$ARGUMENTS`, `$1`, not schemas. (TODO.md `## Next`; ROADMAP `## Next`)
`` !`cmd` `` shell substitution with the guard applied. (TODO.md `## Next`; ROADMAP `## Next`) - [ ] **Derive the tool-name lists.** `TOOL_SETS` and `MUTATING_TOOLS` are hand-maintained; a tool
- [ ] **Derive the tool-name lists.** `TOOL_SETS` and `MUTATING_TOOLS` are hand-maintained; a added to one and forgotten in the other is a silently ungated write. (TODO.md `## Next`;
tool added to one and forgotten in the other is a silently ungated write. (TODO.md ROADMAP `## Next` "Derived tool metadata")
`## Next`; ROADMAP `## Next` "Derived tool metadata")
- [ ] **Subagent parallelism.** Two independent searches run sequentially; the panel already - [ ] **Subagent parallelism.** Two independent searches run sequentially; the panel already
renders several agents, the loop does not fan out. (TODO.md `## Next`; ROADMAP `## Later`) renders several agents, the loop does not fan out. (TODO.md `## Next`; ROADMAP `## Later`)
- [ ] **Undo a turn.** `/resume` restores a session but nothing walks one step back; `bash` - [ ] **Undo a turn.** `/resume` restores a session but nothing walks one step back; `bash`
effects cannot be snapshotted and the docs would say so. (TODO.md `## Next`; ROADMAP `## Next`) effects cannot be snapshotted and the docs would say so. (TODO.md `## Next`; ROADMAP `## Next`)
### ROADMAP "Next" / "Later" — tracked, not yet scheduled in TODO.cpp-equivalent detail ### ROADMAP "Next" — tracked, not yet scheduled in TODO.cpp-equivalent detail
- [ ] **Registry trust.** No signatures; `registryUrl` is the whole trust decision. Publisher - [ ] **Registry trust.** No signatures; `registryUrl` is the whole trust decision. Publisher
keys + pinned digest per entry. (ROADMAP `## Next`; also TODO.md Known rough edges) keys plus a pinned digest per entry. (ROADMAP `## Next`; TODO.md Known rough edges)
- [ ] **Lossless-enough compaction** — same work as "Summarize the pruned span". (ROADMAP `## Next`) - [ ] **`web_fetch` leftovers.** `web_fetch` itself shipped in beta.4. What remains declined:
thin wrappers around a single bash line (`run_tests`, `typecheck`, `lint`, `build`) that add
only schema tax. (ROADMAP `## Next`)
### ROADMAP "Later"
- [ ] **Session branching**, **structured diff review**, **plugin code from disk** (needs a - [ ] **Session branching**, **structured diff review**, **plugin code from disk** (needs a
sandbox story), **prompt caching** (stable prefix vs volatile suffix), **external hooks** sandbox story), **prompt caching** (stable prefix vs volatile suffix), **external hooks**
(needs a trust story), **OS-level sandboxing** (Seatbelt/Landlock/Windows equivalent). (needs a trust story), **OS-level sandboxing** (Seatbelt/Landlock/Windows equivalent).
(ROADMAP `## Later`) (ROADMAP `## Later`)
- [ ] **Deliberately declined** (do not "fix"): web UI, auto-commit, vector search, tool-call - [ ] **Deliberately declined** (do not "fix"): web UI, model-agnostic prompt tuning, auto-commit,
retries, client/server split, LSP integration — all recorded in ROADMAP `## Declined`. vector search, tool-call retries, client/server split, LSP integration — all recorded in
ROADMAP `## Declined`.
--- ---
## D. Documentation drift (untracked) ## D. Documentation drift (untracked)
- [ ] **README tool count** — see Bug B.3. **not yet tracked** The two items from the beta.5 audit are resolved: `README.md` now says 41, and `TODO.md` has a
- [ ] **TODO.md "Done" is missing the rest of beta.5.** "Kept for one release, then deleted", complete `## Done` section for the 1.0.0 batch (with the history in ROADMAP `## Shipped`). Three
but of the beta.5 batch (more tools incl. `git_branch`/`git_commit_message`/ stale lines were found in this pass, and all three are now corrected:
`move_file`/`delete_file`, the `protect` plugin, `security`/`perf`/`migrate` skills, the
MCP panel wizard, the UI refinements, the farewell message) only "a dead provider item - [x] **`docs/development.md:101` said "Nineteen built-in tools".** Now "Forty-one", matching the
ends the turn" was checked off. Either the Done list gets the beta.5 items or it gets registry; the sentence warns that selection accuracy degrades past a certain count, so the
rotated, as the file's own rule says. **not yet tracked** number matters.
- [ ] **`docs/registry.md` and `docs/headless.md`** are referenced by the README table and both - [x] **`docs/headless.md:54` used "There are 16 built-in tools."** as its sample JSON output.
Now "41", so the example matches the tool list.
- [x] **`docs/headless.md:185-186` said there was no spend ceiling yet.** Rewritten to document
`maxSpendUsd` (`src/config.ts:24`): warns once at 80%, refuses the next turn past 100%, and
the run exits non-zero — enforced only on priced models.
- [x] **`docs/registry.md` and `docs/headless.md`** are referenced by the README table and both
exist — verified clean, no action. exist — verified clean, no action.
--- ---
## E. Test-coverage gaps ## E. Test-coverage gaps
- [ ] **`src/cli.tsx` (562 lines) has no unit test.** Nothing in `test/` imports it. Its flag - [x] **`src/cli.tsx` (666 lines) now has an entry-point test — `test/cli.test.ts`.** The module
parsing (`-p`, `--json`, `--yolo`, `--resume`, provider setup, `/provider` wiring) is is an executable, not a library: importing it parses argv, loads config, connects MCP, and
exercised only by hand or through `runHeadless` (`test/headless.test.ts`), which bypasses renders Ink, which is why nothing imported it before. The test runs the real entry point as a
the argument surface. The largest module in `src/` outside the UI is the least tested one. child process (via `process.execPath`, so it is cross-platform) with a scratch `SHIRO_HOME`,
**not yet tracked** a scratch cwd, and every API-key variable stripped, so the no-key branches are deterministic.
- [ ] **`src/ui/Onboard.tsx` has no test.** The provider on-boarding wizard is never rendered in Ten cases cover the argument surface: `--help`/`-h`, `--version`/`-v`, an unknown `--agent`,
the suite. **not yet tracked** an unknown `--think`, `-p` without a key, `--resume` with no match, `--continue` with no
- [ ] **`src/ui/PromptInput.tsx` has no test** — the `@` completion input is only covered saved session, and a configured key with no prompt (that last run also passes all six
indirectly through `App`. (`src/complete.ts` itself is well tested.) **not yet tracked** `--no-*` isolation flags through the boot path). Paths past `render()` need a TTY and remain
- [ ] **`src/ui/panel-bodies.ts` and `src/ui/buses.ts` have no direct tests.** uncovered — provider `/provider` wiring through the wizard included.
**not yet tracked** - [x] **`src/ui/Onboard.tsx` (211 lines) now has a test — `test/onboard.test.tsx`.** The wizard
- [ ] **29 `as any` casts across 17 test files.** The identical mock `usage` object renders standalone and its only network call is `fetchModels`, which uses the global `fetch`,
(`{ inputTokens: {...}, outputTokens: {...} } as any`) is copied verbatim in 7+ UI test so the suite stubs it: the flow is exercised offline and deterministically, with the two API
files — a shared typed fixture in `test/helpers.ts` would remove the repetition and the key env vars cleared so a developer's shell never picks the branch. Six cases cover the
casts in one move. Production `src/` remains clean; this is test-only debt. **not yet tracked** provider list and its `(current)` marker, esc from both the list and the api-key step, a
custom endpoint collecting url + key + a sorted model pick, the env-key shortcut (key step
skipped, hint masked to `sk-a...1234`), manual model-id entry, and the "could not list
models" fallthrough when the server errors. `current` sets the starting row, so a test names
a preset rather than counting arrow presses.
- [x] **`src/ui/PromptInput.tsx` (175 lines) now has a direct test — `test/prompt-input.test.tsx`.**
It was already covered through `App` in `input.test.tsx` (history recall and stash, arrow and
word motion, ctrl-u, ctrl-d, paste); what was left needed the component on its own, so this
renders it directly with a controlled `Harness`. Fifteen cases cover the caret rendering (a
bare focused caret, a blurred input with none, the placeholder inverting its first character
only while focused, the cursor sitting under the character it points at), the kill keys
(ctrl-a/ctrl-e, ctrl-k, ctrl-w, including a single word emptying the line), the `mask` (hidden
value and the caret after a deletion), `onKey` swallowing a key before the input sees it while
declining others, `initialCursor`, an external `value`, submit, and insert-at-cursor. Two
expected frames were probed against the real renderer rather than assumed: a whitespace-only
`Text` trims to `''`, and a mask renders the caret *after* the stars.
- [x] **`src/ui/panel-bodies.ts` (78 lines) now has a direct test** — `test/ui-bodies.test.tsx`,
shared with `buses.ts`. The four panels are pure functions of session and hook state, so
they run without mounting Ink: the tools panel (sets named, a read-only agent narrowing it,
the offered/registered hint), the cost panel (priced turn, unpriced model, the subagent line
priced against its own model, the ceiling line), the context panel's empty branch, and the
todos panel fed through the real `todo_write` tool.
- [x] **`src/ui/buses.ts` (75 lines) now has a direct test.** `createNoticeBus` and
`createSubagentBus` are covered: a pre-bind emit is queued and delivered in order on bind, a
post-bind emit passes through, and a rebind takes over without replaying the queue.
`applySubagentEvent` gains the cases the existing `ui-panels` test left open — a result
attaching to its step, a mismatched or duplicate result being ignored, end/error status
flips, an unknown id being a no-op, and two agents interleaving without crossing steps.
- [ ] **13 `as any` casts across 11 test files.** Down from 29 across 17. The identical mock
`usage` object (`{ inputTokens: {...}, outputTokens: {...} } as any`) is still repeated
across several UI test files — a shared typed fixture in `test/helpers.ts` would remove the
repetition and the casts in one move. Production `src/` remains clean; this is test-only
debt. **not yet tracked**
--- ---
## F. Maintenance debt ## F. Maintenance debt
- [ ] **Pricing table is hand-entered with no source note or date** — `src/pricing.ts:8-22`. - [ ] **Pricing table is hand-entered with no source note or date** — `src/pricing.ts:8-22`.
Rates drift; `estimateTokens` also divides JSON length by four (session.ts:90), which is Rates drift. Tracked in TODO.md `## Maintenance`.
fine as a compaction threshold but misleads in `/cost`. Both tracked in TODO.md - [ ] **`estimateTokens` divides JSON length by four** — `src/session.ts:97` (the thinking panel
`## Maintenance`. does the same at `src/ui/Panels.tsx:193`). Fine as a compaction threshold, misleading in
- [ ] **`listPaths` walks up to 5000 files once per session** — fine for a repo, wasteful in a `/cost`. Tracked in TODO.md `## Maintenance`.
monorepo, never notices a file created after the first `@`. Tracked in TODO.md. - [ ] **`listPaths` walks up to 5000 files once per session** — `src/cli.tsx:387-389`. Fine for
- [ ] **`MUTATING_TOOLS` (tools.ts:799-807) is only used by tests and docs.** The runtime gate a repo, wasteful in a monorepo, never notices a file created after the first `@`. Tracked in
is `DEFAULT_PERMISSIONS` + the unknown-tool `ask` default. Tracked in TODO.md — either TODO.md.
- [ ] **`MUTATING_TOOLS` (`src/tools.ts:980`) is only used by tests and docs.** The runtime gate
is `DEFAULT_PERMISSIONS` plus the unknown-tool `ask` default. Tracked in TODO.md — either
delete it or make `DEFAULT_PERMISSIONS` derive from it. delete it or make `DEFAULT_PERMISSIONS` derive from it.
- [ ] **CI actions are about to leave Node 20.** The last release run annotated that actions on - [ ] **Oversized modules, all grown again in the 1.0.0 batch.** `src/ui/App.tsx` (1002),
Node 20 are being forced onto Node 24. `actions/checkout@v4`, `setup-bun@v2`, `src/tools.ts` (990), `src/cli.tsx` (666), `src/session.ts` (637), `src/ui/Panels.tsx`
`upload-artifact@v4`, `download-artifact@v4` still work, but the major-version bumps will (579). Not a bug — a "who reads 990 lines" concern. **not yet tracked**
become the silent fix; watch for the annotation to turn red. **not yet tracked** - [ ] **CI actions are on Node 20.** The last release run annotated that `actions/checkout@v4`,
- [ ] **Oversized modules.** `src/ui/App.tsx` (863), `src/tools.ts` (717), `src/cli.tsx` (562), `setup-bun@v2`, `upload-artifact@v4`, `download-artifact@v4` are being forced onto Node 24.
`src/session.ts` (500), `src/ui/Panels.tsx` (398). All of them grew past a comfortable They still work; the major-version bumps will become the silent fix. Not verifiable from the
review size during the beta.5 batch. Not a bug — a "who reads 863 lines" concern. tree — watch for the annotation to turn red. **not yet tracked**
**not yet tracked**
--- ---
@@ -189,13 +253,13 @@ untracked items are marked **not yet tracked**.
| Area | Items | | Area | Items |
|---|---| |---|---|
| Bugs | 3 (`App.tsx:613`, dead `dry_run` input, README tool count) | | Bugs | 0 (all five cleared) |
| Official gaps (Now/Next/Later) | 14 tracked in TODO.md/ROADMAP.md | | Official gaps (Now/Next/Later) | 8 near-term + 6 Later tracked in TODO.md/ROADMAP.md |
| Documentation drift | 2 untracked | | Documentation drift | 0 (three lines corrected) |
| Test-coverage gaps | 5 (of which `src/cli.tsx` is the significant one) | | Test-coverage gaps | 1 (every UI module now has a test; only the `as any` item below remains) |
| Maintenance debt | 5 (3 tracked, 2 untracked) | | Maintenance debt | 6 (4 tracked, 2 untracked) |
| Known rough edges | 10 (all tracked) | | Known rough edges | 10 (all tracked) |
| Verified clean | 9 areas, including zero `as any` and zero swallowed errors in `src/` | | Verified clean | 9 areas, including zero casts and zero swallowed errors in `src/` |
The codebase is in good shape for a beta. The three bugs are each one-line fixes; the The codebase is in good shape for 1.0.0. The bug list, the doc drift, and every UI coverage gap are
coverage gap on `cli.tsx` is the item that will actually bite. cleared; what is left is the test-only `as any` casts and the maintenance debt — all tracked.
+32
View File
@@ -4,6 +4,38 @@ All notable changes to this project are documented here. The format follows
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project adheres to [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project adheres to
[Semantic Versioning](https://semver.org/spec/v2.0.0.html). [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## [Unreleased]
A batch of loop and ergonomics work landed after 1.0.0. Version remains `1.0.0` in
`src/version.ts`; these are folded into the next tagged release.
### Added
- **`/undo` and `/redo`.** Every prompt snapshots the files on disk first (capped near the last
100), and `/undo` restores files, trims the conversation, or both. `/redo` reverses it. A
`bash` command's side effects are not files and cannot be rolled back, which is stated in the
command output rather than hidden.
- **Parallel subagents.** The `task` tool accepts several independent investigations under a
`tasks` array and runs them on separate context windows at the same time, joining their
reports. A single call behaves exactly as before.
- **Lazy MCP tools.** By default an MCP server now contributes three meta-tools (`mcp_list`,
`mcp_inspect`, `mcp_call`) instead of one schema per server tool, so a server exposing twenty
tools stops costing ~2750 tokens per request until one is actually called. Set
`"mcpMode": "eager"` to register every server tool up front. Named servers are still listed
in the prompt, and calls route through the same permission rules and guard as before.
- **Hot-reloaded skill installs.** A skill installed from `/registry` mid-session is callable
on the next turn without a restart (the `skill` tool reads its list live, so even the first
install works). Plugins and external tools still need a restart because they join the guard
chain and tool registry built once at boot.
### Fixed
- The walk behind `@file` completion refreshed only once per session; it now re-walks on a slow
cooldown so a file created after the first `@` shows up within a short window.
- A handful of plugin write tools were mutating but not gated by the permission defaults; the
tool set and the mutating list are now derived from a single `mutating()` marker, so a tool
can no longer be added to one and forgotten in the other.
## [1.0.0] ## [1.0.0]
The first stable release. Cost control, a larger tool and skill surface, custom slash The first stable release. Cost control, a larger tool and skill surface, custom slash
+19 -7
View File
@@ -73,6 +73,11 @@ pattern, not the whole tool. `.env` and `.pem` files are refused on read outrigh
plugin refuses irreversible commands ahead of any of it — `rm -rf`, `git reset --hard`, force plugin refuses irreversible commands ahead of any of it — `rm -rf`, `git reset --hard`, force
pushes, `DROP TABLE` — and `--yolo` cannot bypass it. pushes, `DROP TABLE` — and `--yolo` cannot bypass it.
**Rewinds a mistake.** `/undo` restores the snapshot taken before a prompt: files are reverted,
the conversation is trimmed back, or both, and `/redo` reverses it. The snapshot is capped so
it stays near the last 100 prompts, and a `bash` command's side effects cannot be rolled back
this way because they are not files.
**Shows its work.** Reasoning streams to a collapsed panel you can expand with `ctrl-r`, the **Shows its work.** Reasoning streams to a collapsed panel you can expand with `ctrl-r`, the
tool in flight is named as it runs with the arguments that identify the call, and `bash` tool in flight is named as it runs with the arguments that identify the call, and `bash`
output streams live instead of arriving all at once when the command exits. `ctrl-c` kills a output streams live instead of arriving all at once when the command exits. `ctrl-c` kills a
@@ -93,10 +98,13 @@ is a decision rather than a default, and it asks before every call.
**Asks instead of guessing.** When a request has two readings that lead to different work, **Asks instead of guessing.** When a request has two readings that lead to different work,
the agent puts a question on screen with options. the agent puts a question on screen with options.
**Delegates work.** `task` spawns a subagent with its own context window whose findings come **Delegates work, in parallel.** `task` spawns a subagent with its own context window whose
back as one message, so a search across forty files does not fill the main context. `explore` findings come back as one message, so a search across forty files does not fill the main
and `review` are read-only; `worker` also edits and runs commands, and every one of its writes context. A single call can batch several independent investigations under `tasks`: they run on
stops at the same approval prompt as yours. Progress streams to a panel. separate context windows at the same time and their reports are joined, so two unrelated
searches overlap in wall-clock time instead of queueing. `explore` and `review` are read-only;
`worker` also edits and runs commands, and every one of its writes stops at the same approval
prompt as yours. Progress streams to a panel.
**Extensible from the prompt.** `/registry` browses external skills and plugins and installs **Extensible from the prompt.** `/registry` browses external skills and plugins and installs
them with one confirmation. A skill is shown in full before its text joins your system prompt; them with one confirmation. A skill is shown in full before its text joins your system prompt;
@@ -115,7 +123,10 @@ record of what it already ran instead of repeating it.
**Keeps the tool list affordable.** Forty-one built-in tools, grouped into sets. Each costs **Keeps the tool list affordable.** Forty-one built-in tools, grouped into sets. Each costs
about 550 characters of schema on every request, so `{ "toolSets": [] }` trims back to the six about 550 characters of schema on every request, so `{ "toolSets": [] }` trims back to the six
core ones and a disabled set reaches neither the wire nor the prompt. core ones and a disabled set reaches neither the wire nor the prompt. An MCP server's tools are held back the
same way: by default its tools are fetched only when one is actually called (via `mcp_list`,
`mcp_inspect`, `mcp_call`), so twenty tools on one server cost almost nothing until they are
used. Set `"mcpMode": "eager"` to register every server tool up front instead.
## Documentation ## Documentation
@@ -149,7 +160,7 @@ Type `/` and a menu appears, narrowing as you type.
``` ```
/help /agent [name] /think [level] /provider /models /model <id> /help /agent [name] /think [level] /provider /models /model <id>
/skills /plugins /registry [search|add|remove] /mcp [add|remove] /init /context /skills /plugins /registry [search|add|remove] /mcp [add|remove] /init /context
/todos /notes /memory /tools /compact /cost /todos /notes /memory /tools /compact /cost /undo /redo
/sessions /resume <id> /save /clear /exit /sessions /resume <id> /save /clear /exit
``` ```
@@ -166,7 +177,8 @@ gateable sets, 29 bundled skills, built-in and data-only plugins, per-project me
persistence and resume, MCP servers, custom slash commands from markdown files, auto-loaded persistence and resume, MCP servers, custom slash commands from markdown files, auto-loaded
external skills/tools/plugins, markdown rendering, headless mode with JSON events for CI, external skills/tools/plugins, markdown rendering, headless mode with JSON events for CI,
five-platform builds, streaming reasoning, the mid-turn prompt queue, read-only git tools, five-platform builds, streaming reasoning, the mid-turn prompt queue, read-only git tools,
batch reads, `apply_patch`, `web_fetch`, `@file` completion, interruptible commands, and the `/undo` and `/redo`, parallel subagents, lazy MCP tools, hot-reloaded skill installs, batch
reads, `apply_patch`, `web_fetch`, `@file` completion, interruptible commands, and the
external registry. external registry.
Next up is in [TODO.md](TODO.md); the longer view and what has been declined are in Next up is in [TODO.md](TODO.md); the longer view and what has been declined are in
+176 -19
View File
@@ -238,6 +238,117 @@ agent·model row inside it and a split footer beneath.
--- ---
## What the other agents have
Surveyed opencode, Claude Code, Codex CLI, and phi against this tool's feature set. The point of
the table is to record what is *worth copying* and what is worth *declining*, not to chase parity:
each of these four spent effort here deliberately, and several of their choices are load-bearing for
a reason that applies to us.
The four are not the same shape. **Claude Code** and **Codex** are first-party CLIs tied to one
vendor's models; **opencode** and **phi** are open-source and provider-agnostic, and opencode in
particular is the closest thing to a peer here. Where a feature exists, the notes say what it costs
— several are cheap to copy, and three are not.
### Where the four agree
Six features have converged across all or nearly all of them. Convergence is the strongest signal
available that a feature is not a fad — and the gaps in the table are as informative as the checks,
because they show which features are genuinely optional and which are table stakes.
| Feature | opencode | Claude Code | Codex | phi | Here |
| --- | --- | --- | --- | --- | --- |
| Per-pattern permission rules | `permission.bash` globs | `settings.json` allow/deny/ask | `approval_policy` + rules | Gate + `permissions.mode` | **shipped** |
| Subagents with fresh context | `mode: subagent` | `agents/*.md` | — | sub-agents | **shipped** |
| Markdown-defined agents | `agents/`, `commands/` | `agents/`, `skills/` | `AGENTS.md` | `.phi/` | **shipped** |
| Session resume | `--session`, `--continue` | `--resume`, `--continue` | resume + rollout files | `sessions` | **shipped** |
| Compaction | auto + `/compact` | `/compact`, `/rewind` summarize | compact prompt file | — | **shipped** |
| Undo / rewind | `/undo`, `/redo` (via git) | `/rewind` (snapshots) | — | — | **missing** |
Two of the six are outright missing from one or more tools, and undoing is missing from two of the
four — so it is a real feature, not table stakes, and the two that have it disagree about how. The
permission model here is not behind: it already carries the per-pattern rules, the credential deny,
and the repeat guard the others arrived at, and `doom_loop` (below) is the one refinement worth
taking. **Undo is the single converged feature genuinely absent**, which is why it leads Next.
### Detail worth having, by tool
**opencode** — the closest peer, and the source of the permission shape already adapted here.
`/undo` and `/redo` revert file changes **through git**, so they require the project to be a git
repository; this is a real limitation, not an implementation detail, and it means an undo is only
as good as the working tree's state. Agents are `primary` (build, plan) or `subagent` (general,
explore, scout), switchable with Tab or `@`-mention, configured in `opencode.json` or Markdown
frontmatter. A subagent runs in a **child session** with its own navigation keybinds
(`session_child_first`, `session_parent`), and `subagent_depth` (default 1) caps nesting. Notable:
`doom_loop` is a first-class permission key that fires when the same tool call repeats three times
with identical input — the repeat guard here is an approval rule; theirs is a named primitive with
its own recovery prompts. Sessions share over a URL (`/share`), which is a hosted product decision,
not a local one.
**Claude Code** — the most complete implementation of undo, and worth reading before building one.
Checkpointing snapshots **before every user prompt**, keeps the **100 most recent** checkpoints,
and stores them with the conversation so `/rewind` survives a resume. The rewind menu offers five
distinct actions, and the split is the interesting part: *restore code*, *restore conversation*,
*restore both*, *summarize from here*, *summarize up to here*. That is undo and compaction sharing
one control surface. The honest limits are documented rather than hidden: only `Write`/`Edit`/
`NotebookEdit` are tracked, `bash` side effects are not, **subagent edits are not captured** unless
the skill ran in the foreground with `context: fork`, and symlinked or hard-linked files are
skipped with an explicit warning. Their hook surface is the largest of the four — see *External
hooks* under Later, which is where that capability belongs here.
**Codex** — the sandbox is the differentiator, and it is genuinely hard to copy. Apple Seatbelt on
macOS, Landlock + seccomp on Linux, and a restricted-token/AppContainer mechanism on Windows, with
`workspace-write` the default and network **off** unless `sandbox_workspace_write.network_access`
is set. `approval_policy` is now `on-request | never | { granular = {...} }` — `untrusted` was
**retired** and can prevent the client from starting. Also shipped and worth knowing:
`approvals_reviewer = "auto_review"` routes an approval prompt through a *reviewer subagent*
rather than the user. `AGENTS.md` resolves global → project root → cwd, one file per directory,
`AGENTS.override.md` winning, capped by `project_doc_max_bytes` (32 KiB default).
**phi** — small (Go, ~12 MB), and the two ideas most worth stealing. First, **MCP without context
death**: server tool schemas never enter the prompt; the system prompt lists only **server names**,
and the model uses three meta-tools — `mcp_list`, `mcp_inspect`, `mcp_call` — with subprocesses
starting lazily on first use. This is the concrete design behind *MCP without the schema tax*
under Next. Second, the
hook contract is the cleanest of the four: `pre_tool` runs **before** the permission gate and can
`allow`, `deny`, or **`modify`** the input; `post_tool` can append model-facing `context` or
rewrite `output`. Exit code `2` is a hard deny; `fail_closed` decides crash behaviour; in
`readonly` mode only `fail_closed` hooks run, so a slow audit hook cannot stall exploration.
### Corrections to what this file said before
Two claims previously written here were incomplete, and the research fixes them:
- **Input-rewriting hooks are not phi's alone.** Codex ships it too: `PreToolUse` returns
`permissionDecision: "allow"` with `updatedInput` to rewrite a call, verified in their docs and
in `codex-rs/.../mcp.rs` (`with_updated_hook_input`). Two independent implementations make this
the standard shape rather than one project's quirk — which raises its priority, and means the
compiled plugin interface here is now the odd one out.
- **Codex's hook trust is exactly as described, and the mechanism is now known.** Non-managed
hooks cannot run until reviewed: Codex persists a `trusted_hash` in `config.toml`, records trust
against the hook's **current hash**, and marks new or changed hooks for review in `/hooks`.
`--dangerously-bypass-hook-trust` skips it for one invocation. The known hole is instructive: a
reviewer on their PR noted that **replacing the script a hook points at does not reset trust**,
because only the config is hashed. A trust story that hashes the declaration but not the artefact
is a partial one — worth designing past rather than copying.
### What is worth declining
Three of their features are deliberate here, and the survey confirms the reasoning:
- **A client/server split** (opencode's OpenAPI server + TUI-as-client + IDE/web clients) exists to
serve *second clients*. No second client is wanted here.
- **LSP integration** — the quote already cited in Declined is accurate and now verified in full:
opencode's own LSP page says it "is useful in some projects, but it is not always a net positive,"
that servers "can get out of sync, use significant memory, vary by version or project, and slow
down agent workflows," and that "in many projects it is better to have the agent run lint,
typecheck, or other diagnostic CLI tools directly." They ship 30+ built-in servers and still say
this. That is the strongest possible endorsement of the position in Declined.
- **Session sharing over a URL** (opencode `/share`) is a hosted-service feature and brings a
privacy surface this tool has no reason to take on.
---
## Next ## Next
### MCP without the schema tax ### MCP without the schema tax
@@ -250,10 +361,21 @@ registration as an option: for a two-tool server the indirection is the more exp
### Undo a turn ### Undo a turn
opencode has `/undo` and `/redo`, Claude Code has `/rewind` over file checkpoints. There is Every comparable CLI has this: opencode `/undo` and `/redo`, Claude Code `/rewind` with
`/resume` here, which restores a whole session, and nothing that steps one turn back. The honest checkpoints. There is `/resume` here, which restores a session, and nothing that walks one back.
limit is the same for everyone: a `bash` command's effects cannot be snapshotted, so this covers Claude Code's implementation is the one to read first, because it has already worked out the seams:
file-tool edits and says so. it snapshots before **every user prompt**, keeps the **100 most recent**, stores snapshots with the
conversation so a rewind survives a resume, and splits one menu into *restore code*, *restore
conversation*, *restore both*, and *summarize from here / up to here* — undo and compaction on one
control surface. opencode's version is simpler and takes a different position: it reverts through
**git**, so it needs a repository and inherits whatever the working tree already contained.
Both document the same hard limit, and so must this: a `bash` command's effects cannot be
snapshotted. Claude Code tracks only its own file-edit tools, explicitly does not cover `bash`
side effects, and does not capture subagent edits unless the fork ran in the foreground. The
honest version here covers file-tool edits, says so in the command's own output, and reports what
it skipped rather than pretending the tree is clean — the same shape used for an interrupted
command, whose effects are already reported as unknown.
### Lossless-enough compaction ### Lossless-enough compaction
@@ -273,6 +395,26 @@ An index is trusted for its contents, not its authorship: `registryUrl` is the w
decision, and there are no signatures. Publisher keys and a pinned digest per entry would make decision, and there are no signatures. Publisher keys and a pinned digest per entry would make
"install this skill" a decision about a specific artifact rather than about a URL. "install this skill" a decision about a specific artifact rather than about a URL.
### Auto-review an approval
Codex ships `approvals_reviewer = "auto_review"`: an eligible approval prompt is routed through a
reviewer **subagent** instead of surfacing to the user, using the same sandbox boundary. `explore`
and `review` already exist and are read-only, so the piece to build is a reviewer persona that
decides an approval request and a policy that says which prompts are eligible — a batch of three
identical `git status` calls should not each interrupt, but a first `rm` should. The failure mode
to design against is a reviewer that waves through exactly what the user would have stopped, so it
must be opt-in, name itself when it approves, and never override a deny rule.
### Name the repeat guard
opencode has `doom_loop` as a first-class permission key: it fires when the same tool call repeats
three times with identical input, carries its own recovery prompts, and is configurable per agent.
The equivalent here is a rule inside the permission layer — the call repeated identically three
times in one turn asks even when allowed — which works but is unnamed, unconfigurable, and
invisible in `/tools`. Promoting it to a named primitive makes it inspectable and lets an agent
tighten or loosen it, and the recovery prompt is the part genuinely missing: a loop is better
interrupted with advice than with silence.
### `web_fetch` ### `web_fetch`
Shipped in beta.4 — see above. What remains declined: wrappers around a single bash line with Shipped in beta.4 — see above. What remains declined: wrappers around a single bash line with
@@ -301,18 +443,30 @@ agent can read.
step for task-list freshness, which defeats a naive cache; splitting the stable prefix from step for task-list freshness, which defeats a naive cache; splitting the stable prefix from
the volatile suffix would fix that. the volatile suffix would fix that.
**External hooks.** phi and both first-party CLIs let a script sit in the tool loop: a directory **External hooks.** Every one of the four surveyed except here lets a script sit in the tool loop:
with a manifest and an executable, one JSON object in on stdin, one out. phi's `pre_tool` can a directory with a manifest and an executable, one JSON object in on stdin, one out. **Both** phi
rewrite the tool's input as well as allow or deny, which the compiled plugin interface here cannot and Codex let a `PreToolUse` hook **rewrite** a tool's input, not merely allow or deny it — phi
express. The reason it is here rather than in Next is that it needs a trust story — Codex hashes returns `{"action":"modify","input":{...}}`, Codex returns `permissionDecision: "allow"` with
each hook and refuses to run one until you review it, which is the right shape and more work than `updatedInput` — and that is the exact capability the compiled plugin interface here cannot
the feature. express. Two independent implementations make it the expected shape rather than one project's
quirk. What blocks it is the trust story, not the mechanism.
The trust story has a known shape and a known hole. Codex will not run a non-managed hook until it
has been reviewed: it persists a `trusted_hash` in `config.toml`, records trust against the hook's
current hash, and lists new or changed hooks for review under `/hooks`;
`--dangerously-bypass-hook-trust` overrides for one invocation. The hole is that only the
**declaration** is hashed — replacing the script the hook points at does not reset trust, as a
reviewer on their own PR pointed out. Hashing the declaration *and* the artefact is the minimum
worth doing here, and it is why this is work rather than a weekend.
**OS-level sandboxing.** The strongest thing in this class, and Codex is the one that has it: **OS-level sandboxing.** The strongest thing in this class, and Codex is the one that has it:
Seatbelt on macOS, Landlock and seccomp on Linux, a separate mechanism on Windows, with network Seatbelt on macOS, Landlock and seccomp on Linux, a restricted-token/AppContainer mechanism on
egress governed by domain rules. Permission rules gate the *call*; a sandbox governs what the Windows, with network **off** by default under `workspace-write`. Permission rules gate the *call*;
process can then reach. Three platform-specific implementations, and OpenAI moved Codex to Rust a sandbox governs what the process can then reach. Note that even Codex's sandbox is not a complete
partly for this. A half-built sandbox is worse than none, because people would trust it. boundary — their own hook docs say "hooks are guardrails, but they are not a complete enforcement
boundary for every shell or tool path." Three platform-specific implementations, and OpenAI moved
Codex to Rust partly for this. A half-built sandbox is worse than none, because people would trust
it.
--- ---
@@ -339,8 +493,11 @@ endpoint, which is what lets IDE extensions and a web client exist. It is the ri
for that product. Here it would add a protocol, a port, and an auth story to serve a second client for that product. Here it would add a protocol, a port, and an auth story to serve a second client
nobody has asked for. nobody has asked for.
**LSP integration.** opencode ships it and its own documentation says the honest thing: **LSP integration.** opencode ships 30+ built-in language servers and its own documentation still
*"not always a net positive... in many projects it is better to have the agent run lint, typecheck, says the honest thing: LSP "is useful in some projects, but it is not always a net positive,"
or other diagnostic CLI tools directly."* Language servers drift out of sync, use real memory, and servers "can get out of sync, use significant memory, vary by version or project, and slow down
vary by version. `bash bun run typecheck` puts the same errors in front of the model with none of agent workflows," and *"in many projects it is better to have the agent run lint, typecheck, or
that, and `AGENTS.md` is where the command belongs. other diagnostic CLI tools directly."* That is the strongest available endorsement of declining it:
the project with the most invested says it is often the wrong trade. `bash bun run typecheck` puts
the same errors in front of the model with none of that, and `AGENTS.md` is where the command
belongs.
+40 -25
View File
@@ -14,19 +14,23 @@ Compaction now keeps the model's memory of a turn, but it still tells the model
the messages it dropped, so a decision from forty messages ago can be contradicted with the messages it dropped, so a decision from forty messages ago can be contradicted with
confidence. confidence.
- [ ] Summarize the discarded messages before dropping them - [x] Summarize the discarded messages before dropping them
- [ ] Inject the summary in place of the count - [x] Inject the summary in place of the count
- [ ] Budget it: a summary that grows with the session defeats the point - [x] Budget it: a summary that grows with the session defeats the point
- [ ] Test: a pruned decision is still recoverable from the summary - [x] Test: a pruned decision is still recoverable from the summary
### Hot-reload an installed entry ### Hot-reload an installed entry
`/registry add` writes the file and says to restart. The skill catalogue and the guard chain `/registry add` writes the file and says to restart. The skill catalogue and the guard chain
are both assembled at boot, so a mid-session install does nothing until then. are both assembled at boot, so a mid-session install does nothing until then.
- [ ] Rebuild the skill list and plugin host after an install or removal - [x] Rebuild the skill list and plugin host after an install or removal
- [ ] Leave a turn in flight alone: its rules must not change underneath it - [x] Leave a turn in flight alone: its rules must not change underneath it
- [ ] Test: a skill installed mid-session is callable in the next turn without a restart - [x] Test: a skill installed mid-session is callable in the next turn without a restart
> Partial: skills reload live (the `skill` tool reads its list on each call); plugins and
> external tools still need a restart because they join the guard chain and the tool registry,
> both built once at boot.
--- ---
@@ -40,11 +44,13 @@ Every MCP tool's schema goes into the prompt today, so twenty tools from one ser
phi solves this with three meta-tools — `mcp_list`, `mcp_inspect`, `mcp_call` — and a prompt that phi solves this with three meta-tools — `mcp_list`, `mcp_inspect`, `mcp_call` — and a prompt that
names only the servers. A hundred servers then cost almost nothing until one is called. names only the servers. A hundred servers then cost almost nothing until one is called.
- [ ] `mcp_list` / `mcp_inspect` / `mcp_call` replacing per-tool registration - [x] `mcp_list` / `mcp_inspect` / `mcp_call` replacing per-tool registration
- [ ] The prompt lists server names, not schemas - [x] The prompt lists server names, not schemas
- [ ] Calls go through the same permission rules and guard as a built-in - [x] Calls go through the same permission rules and guard as a built-in
- [ ] Keep per-tool registration as an option: a two-tool server is cheaper registered directly - [x] Keep per-tool registration as an option: a two-tool server is cheaper registered directly
- [ ] Test: a configured server contributes no schema to the request until `mcp_call` - [x] Test: a configured server contributes no schema to the request until `mcp_call`
> Switchable with `mcpMode: 'eager'` in config; lazy (meta-tools) is the default.
### Derive the tool-name lists ### Derive the tool-name lists
@@ -52,39 +58,45 @@ names only the servers. A hundred servers then cost almost nothing until one is
in the other is a silently ungated write, which is the worst kind of bug this codebase can in the other is a silently ungated write, which is the worst kind of bug this codebase can
have. have.
- [ ] Mark each tool as mutating where it is defined, not in a list beside it - [x] Mark each tool as mutating where it is defined, not in a list beside it
- [ ] `TOOL_SETS` covers every registered tool, checked rather than assumed - [x] `TOOL_SETS` covers every registered tool, checked rather than assumed
- [ ] Test: a tool in no set, or a mutating tool outside `MUTATING_TOOLS`, fails the suite - [x] Test: a tool in no set, or a mutating tool outside `MUTATING_TOOLS`, fails the suite
> While doing this, two mutating-but-ungated tools surfaced and were gated: the registered set
> is now derived from the `mutating()` marks.
### Subagent parallelism ### Subagent parallelism
Two independent searches run sequentially. The panel already renders several agents; the loop Two independent searches run sequentially. The panel already renders several agents; the loop
does not fan out. does not fan out.
- [ ] `task` accepts several investigations and runs them together - [x] `task` accepts several investigations and runs them together
- [ ] Test: two delegated searches overlap in time rather than queueing - [x] Test: two delegated searches overlap in time rather than queueing
> `tasks` on the `task` tool runs them concurrently via `Promise.all`; a call without it behaves
> exactly as before.
### Undo a turn ### Undo a turn
Every comparable CLI has this: opencode `/undo` and `/redo`, Claude Code `/rewind` with Every comparable CLI has this: opencode `/undo` and `/redo`, Claude Code `/rewind` with
checkpoints. There is `/resume` here, which restores a session, and nothing that walks one back. checkpoints. There is `/resume` here, which restores a session, and nothing that walks one back.
- [ ] Snapshot files before each prompt, capped at the 100 most recent - [x] Snapshot files before each prompt, capped at the 100 most recent
- [ ] `/undo` restores files, conversation, or both; `/redo` reverses it - [x] `/undo` restores files, conversation, or both; `/redo` reverses it
- [ ] Say plainly what is not covered: a `bash` command's effects cannot be snapshotted - [x] Say plainly what is not covered: a `bash` command's effects cannot be snapshotted
- [ ] Test: an edit is reverted, and the model's own record of it goes with it - [x] Test: an edit is reverted, and the model's own record of it goes with it
--- ---
## Maintenance ## Maintenance
- [ ] Pricing table needs a source note and a date; rates drift and ours are hand-entered - [x] Pricing table needs a source note and a date; rates drift and ours are hand-entered
- [ ] `estimateTokens` divides JSON length by four. Good enough for a compaction threshold, - [x] `estimateTokens` divides JSON length by four. Good enough for a compaction threshold,
wrong enough to mislead in `/cost`. Either label it an estimate everywhere or use a wrong enough to mislead in `/cost`. Either label it an estimate everywhere or use a
real tokenizer real tokenizer
- [ ] `listPaths` walks up to 5000 files once per session. Fine for a repo, wasteful in a - [x] `listPaths` walks up to 5000 files once per session. Fine for a repo, wasteful in a
monorepo, and it never notices a file created after the first `@` monorepo, and it never notices a file created after the first `@`
- [ ] `MUTATING_TOOLS` is now only used by tests and docs; the permission defaults are what - [x] `MUTATING_TOOLS` is now only used by tests and docs; the permission defaults are what
actually gate a write. Either delete it or make the defaults derive from it actually gate a write. Either delete it or make the defaults derive from it
--- ---
@@ -144,3 +156,6 @@ Kept for one release, then deleted. The 1.0.0 release batch:
- [x] The system prompt advanced: a failure-recovery loop, a delegation policy, compaction awareness - [x] The system prompt advanced: a failure-recovery loop, a delegation policy, compaction awareness
- [x] The release workflow's dead `dry_run` input wired: manual dispatch publishes only when - [x] The release workflow's dead `dry_run` input wired: manual dispatch publishes only when
unchecked, tag pushes always publish unchecked, tag pushes always publish
- [x] A visible escape hatch for the stalled-agent loop: the existing repeat guard now offers a
configurable `step_back` recovery primitive that reads the session loop trace and steers a
model circling without progress to stop and change direction
+4
View File
@@ -147,6 +147,10 @@ running, the handler exits as usual.
`task` runs a nested `streamText` and returns one message. The subagent kinds hold different `task` runs a nested `streamText` and returns one message. The subagent kinds hold different
tool sets: `explore` and `review` the read-only tools, `worker` those plus every write tool. tool sets: `explore` and `review` the read-only tools, `worker` those plus every write tool.
A `tasks` array on the call runs several of these nests concurrently — each gets its own
context window and stream, awaited together via `Promise.all` — so independent investigations
overlap instead of queueing. The single-prompt form is just the one-element case; the two paths
share the same nested-loop machinery and the same reporting bus.
The consequences follow from the tool set, not from policy: The consequences follow from the tool set, not from policy:
+12 -6
View File
@@ -16,7 +16,7 @@ faster and the fallback path is exercised without it.
```bash ```bash
bun run shiro # run from source bun run shiro # run from source
bun run typecheck # tsc --noEmit bun run typecheck # tsc --noEmit
bun test # 713 tests bun test # 895 tests
bun run build # single binary for this platform -> dist/shiro bun run build # single binary for this platform -> dist/shiro
bun run release # all five platforms -> dist/release + SHA256SUMS bun run release # all five platforms -> dist/release + SHA256SUMS
bun run install:local # build, then copy onto PATH bun run install:local # build, then copy onto PATH
@@ -98,7 +98,7 @@ Steps 3 and 4 are two hand-maintained lists of tool names, which is a known weak
added to one and forgotten in the other is a silently ungated write. Deriving both from the added to one and forgotten in the other is a silently ungated write. Deriving both from the
tool definitions is on [TODO.md](../TODO.md). tool definitions is on [TODO.md](../TODO.md).
Every tool costs roughly 550 characters of schema on every request. Nineteen built-in tools is Every tool costs roughly 550 characters of schema on every request. Forty-one built-in tools is
past where selection accuracy starts to matter, which is why sets exist and why a new tool past where selection accuracy starts to matter, which is why sets exist and why a new tool
needs to earn its place — see [ROADMAP.md](../ROADMAP.md) for what has been declined and why. needs to earn its place — see [ROADMAP.md](../ROADMAP.md) for what has been declined and why.
One set, `net`, is opt-in rather than on: `web_fetch` is the one tool that leaves the machine. One set, `net`, is opt-in rather than on: `web_fetch` is the one tool that leaves the machine.
@@ -151,9 +151,9 @@ Windows host and rejected everywhere else — so a green local release is not pr
`buildArgs()` is unit-tested for both hosts because of exactly that. `buildArgs()` is unit-tested for both hosts because of exactly that.
`.github/workflows/release.yml` then runs typecheck and tests, cross-compiles all five `.github/workflows/release.yml` then runs typecheck and tests, cross-compiles all five
targets on one Ubuntu runner, asserts the built binary reports the expected version, and targets on one Ubuntu runner, asserts each built binary reports the expected version and is
publishes a GitHub release with the binaries and `SHA256SUMS`. A tag containing `-` is non-empty, and publishes a GitHub release with the binaries and `SHA256SUMS`. A tag containing
published as a prerelease. `-` is published as a prerelease.
Bun cross-compiles from any host, which is why there is no build matrix. Verified: a working Bun cross-compiles from any host, which is why there is no build matrix. Verified: a working
`darwin-arm64` binary builds on Windows. `darwin-arm64` binary builds on Windows.
@@ -161,10 +161,16 @@ Bun cross-compiles from any host, which is why there is no build matrix. Verifie
Publishing is gated on a `v*` tag, so a manual `workflow_dispatch` run produces artifacts Publishing is gated on a `v*` tag, so a manual `workflow_dispatch` run produces artifacts
without releasing. without releasing.
The release body is composed by `scripts/make-release-notes.ts` from the matching `## [<version>]`
section of `CHANGELOG.md` (plus an artifact inventory), and it fails the release if that heading
is missing — so a tag with no changelog entry cannot ship an empty body. Write the changelog
entry first, then tag.
## CI ## CI
`.github/workflows/ci.yml` runs typecheck, tests, and a build on Ubuntu, macOS, and Windows `.github/workflows/ci.yml` runs typecheck, tests, and a build on Ubuntu, macOS, and Windows
for every push and PR. for every push and PR, caching the bun install store across runs so a no-change run skips the
dependency download.
All three are necessary. The tools shell out to `rg`, `git`, and a platform shell, and path All three are necessary. The tools shell out to `rg`, `git`, and a platform shell, and path
handling differs — a Windows-only break is invisible on Linux until someone hits it. handling differs — a Windows-only break is invisible on Linux until someone hits it.
+6 -3
View File
@@ -51,7 +51,7 @@ $ shiro -p "count the tools" --json
{"type":"tool-start","id":"c1","name":"grep"} {"type":"tool-start","id":"c1","name":"grep"}
{"type":"tool-call","id":"c1","name":"grep","input":{"pattern":"tool\\("}} {"type":"tool-call","id":"c1","name":"grep","input":{"pattern":"tool\\("}}
{"type":"tool-result","id":"c1","name":"grep","output":"src/tools.ts:26: ..."} {"type":"tool-result","id":"c1","name":"grep","output":"src/tools.ts:26: ..."}
{"type":"text","text":"There are 16 built-in tools."} {"type":"text","text":"There are 41 built-in tools."}
{"type":"done","inputTokens":4210,"outputTokens":88} {"type":"done","inputTokens":4210,"outputTokens":88}
``` ```
@@ -182,8 +182,11 @@ in the system prompt, and CI is exactly where nobody is watching what it says. S
## Cost control ## Cost control
Headless runs are unattended, so a runaway loop costs real money. `--agent quick` caps the Headless runs are unattended, so a runaway loop costs real money. `--agent quick` caps the
step count at 12, and `{ "toolSets": [] }` trims the schema sent every request. There is no step count at 12, and `{ "toolSets": [] }` trims the schema sent every request. `maxSpendUsd` in
spend ceiling yet — see [TODO.md](../TODO.md). the config is a session spend ceiling: it warns once at 80%, and past 100% the next turn is
refused naming the ceiling and the run exits non-zero. It is only enforced on priced models —
an unpriced model has no dollar figure to compare against. See
[configuration](configuration.md).
What a run actually costs is in the `done` event, so a wrapper can total it: What a run actually costs is in the `done` event, so a wrapper can total it:
+58 -21
View File
@@ -6,7 +6,7 @@ command over stdio, and a remote http or sse endpoint.
## Adding one from the prompt ## Adding one from the prompt
``` ```
/mcp list what is configured, with the tool count each contributed /mcp list what is configured, with the tool count (or "lazy") each contributed
/mcp add wizard: local or remote, then the fields that kind needs /mcp add wizard: local or remote, then the fields that kind needs
/mcp remove <name> /mcp remove <name>
``` ```
@@ -42,7 +42,7 @@ list under a request that is already running.
mcp servers mcp servers
/mcp add to add one /mcp add to add one
- `filesystem` (local) - 11 tools - `filesystem` (local) - connected (lazy)
npx -y @modelcontextprotocol/server-filesystem . npx -y @modelcontextprotocol/server-filesystem .
- `api` (remote) - failed: fetch failed - `api` (remote) - failed: fetch failed
https://example.com/mcp https://example.com/mcp
@@ -50,6 +50,9 @@ mcp servers
configured in /home/you/.shiro-neko/config.json configured in /home/you/.shiro-neko/config.json
``` ```
Under `eager` the same list shows a tool count instead of `(lazy)`, because every server's tools
are registered up front.
## The config file ## The config file
The wizard writes this; it is equally editable by hand. The wizard writes this; it is equally editable by hand.
@@ -93,6 +96,7 @@ schema check at load.
```json ```json
{ {
"mcpMode": "eager",
"mcpServers": { "mcpServers": {
"fs": { "fs": {
"command": "npx", "command": "npx",
@@ -120,6 +124,10 @@ inherits your `PATH` unless you replace it.
**Remote** servers take `url`, and optionally `type` (`http` or `sse`, default `http`) and **Remote** servers take `url`, and optionally `type` (`http` or `sse`, default `http`) and
`headers`. `headers`.
A sibling key, `"mcpMode"`, picks how the configured servers' tools reach the model: `lazy`
(the default) or `eager`. It is hand-edited — the `/mcp add` wizard does not set it — and applies
to every server, so it lives beside `mcpServers`, not inside one.
A token in `headers` sits in `config.json` in plain text, same as `apiKey`. For anything beyond A token in `headers` sits in `config.json` in plain text, same as `apiKey`. For anything beyond
a local dev token, prefer a stdio server that reads its own credential from the environment. a local dev token, prefer a stdio server that reads its own credential from the environment.
@@ -132,10 +140,35 @@ calling the binary directly is usually the difference between a noticeable wait
`--no-mcp` skips them all, which is also the quickest way to tell whether a slow start is MCP `--no-mcp` skips them all, which is also the quickest way to tell whether a slow start is MCP
or something else. or something else.
## How a server's tools reach the model
Two modes, switched with `"mcpMode"` in config. **`lazy` is the default**; `eager` is the opt-in.
- **Lazy** registers three meta-tools — `mcp_list`, `mcp_inspect`, `mcp_call` — instead of one
schema per server tool. The server's real tools are fetched only when `mcp_call` actually
invokes one, so a server exposing twenty tools costs almost nothing until one is used. This
is why the affordability paragraph in the README says a configured server no longer taxes
every request.
- **Eager** registers every server tool up front as in the old 1.0 behaviour. If a server
exposes only two tools, eager is cheaper because there is no list-then-inspect round trip.
The model is told the connected server names through the meta-tool descriptions and a prompt
line, then discovers each tool's schema on demand:
```
- mcp_list, mcp_inspect, mcp_call: MCP tools are fetched on demand. mcp_list names a
server's tools, mcp_inspect reads one tool's schema, mcp_call runs it. Never guess a
server or tool name: list first.
```
A `mcp_call` still routes through the same permission rules and guard as a built-in, so the
lazy path is not a way around approval.
## Naming ## Naming
Tools arrive as `mcp__<server>__<tool>`. A server named `fs` exposing `read_file` becomes In `lazy` mode (the default) tools are addressed as `mcp_call(server, toolName, args)`; the
`mcp__fs__read_file`. server names are zero-ambiguity identifiers you list first. In `eager` mode tools arrive as
`mcp__<server>__<tool>`: a server named `fs` exposing `read_file` becomes `mcp__fs__read_file`.
The namespace is not cosmetic. Two servers both exposing `search` would otherwise silently The namespace is not cosmetic. Two servers both exposing `search` would otherwise silently
shadow each other, and the model would call one believing it was the other. shadow each other, and the model would call one believing it was the other.
@@ -165,26 +198,29 @@ Python interpreter should not stop you from editing a file.
## Inspecting ## Inspecting
`/tools` lists everything offered this turn, MCP tools included. The system prompt describes `/tools` lists everything offered this turn. In `eager` mode that includes each MCP tool, named
them as a group: `mcp__<server>__<tool>`, described with whatever the server sent. In `lazy` mode the three
meta-tools appear and the server's real tools are surfaced by `mcp_list` inside the session.
``` Into `mcp_inspect` or `mcp_list` goes the server name, not an `mcp__` path, so the prompt tells
- mcp__api__query, mcp__fs__read_file: from MCP servers, named mcp__<server>__<tool>. the model which servers are connected and to list first before guessing a tool name.
Each needs approval; read its own description before calling.
```
Their individual descriptions come from the server, so that is what the model reads before
calling one.
## Cost ## Cost
Each tool adds its name, description, and JSON schema to every request. The built-ins average In **lazy** mode (the default) a configured server contributes three small meta-tool schemas to
548 bytes; MCP tools vary with how verbose the server's schema is. A server exposing twenty every request, not one schema per tool. A server exposing twenty tools therefore costs a few
tools costs roughly 2,750 tokens per turn, sent whether or not the model uses any of them. hundred tokens per turn rather than roughly 2,750, and it stays cheap whether the model uses
the tools or not. Browsing a server's tools and reading a schema still brings that server's
schema into view one tool at a time, but only when the model asks for it.
MCP tools are **not** covered by `toolSets` — that budget only governs the built-ins. There is In **eager** mode each tool adds its name, description, and JSON schema to every request, and
no per-server switch either, so the choice is a server or no server, and `--no-mcp` for all of that cost is sent whether or not the model uses any of them. That is the right trade only for a
them. If one exposes many tools you never use, a narrower server is worth finding or writing. server with one or two tools, which is why eager exists.
MCP tools are **not** covered by `toolSets` in either mode — that budget only governs the
built-ins. There is no per-server switch beyond the global `mcpMode`, so choosing `eager` turns
every server eager; a server exposing many tools you never use is worth finding a narrower one
for.
`/tools` shows the count both ways: `/tools` shows the count both ways:
@@ -210,8 +246,9 @@ that break in practice are the handshake and the framing, and a mock asserts nei
A server that starts but returns nothing useful is the harder case. In order of speed: A server that starts but returns nothing useful is the harder case. In order of speed:
1. `/tools` — did the tools arrive at all? A server with no tools is a `tools/list` problem. 1. `/tools` — did the tools arrive at all? A server with no tools is a `tools/list` problem.
2. `shiro -p "call mcp__x__y with ..." --json --yolo` — the exact `tool-call` input and 2. `shiro -p "run mcp_call(server, \"api\", \"query\", {...}) with ..." --json --yolo` — the
`tool-result` output, one JSON object per line. exact `tool-call` input and `tool-result` output, one JSON object per line. In eager mode the
address is `mcp__<server>__<tool>` instead.
3. Run the server by hand: `echo '{"jsonrpc":"2.0","id":1,"method":"tools/list"}' | your-server`. 3. Run the server by hand: `echo '{"jsonrpc":"2.0","id":1,"method":"tools/list"}' | your-server`.
If that is wrong, nothing above it can be right. If that is wrong, nothing above it can be right.
+50
View File
@@ -0,0 +1,50 @@
#!/usr/bin/env bun
/**
* Writes `release_notes.md` for a release tag, extracting the tag's prose from
* CHANGELOG.md and appending the artifact inventory. Used by .github/workflows/
* release.yml so the release page reads like a hand-written one instead of the raw
* PR list.
*
* Reads the tag from $RELEASE_TAG (GITHUB_REF_NAME, e.g. `v1.2.3`) and fails loudly
* when CHANGELOG.md has no matching `## [<version>]` section, because a release with
* an empty body is worse than one that refused to publish.
*/
const tag = process.env['RELEASE_TAG'] ?? 'v0.0.0-local';
const version = tag.replace(/^v/, '');
const changelog = await Bun.file('CHANGELOG.md').text();
const start = changelog.indexOf(`## [${version}]`);
if (start === -1) {
console.error(`CHANGELOG.md has no "## [${version}]" heading for ${tag}`);
process.exit(1);
}
// The section runs from its heading to the next `## [` heading, or to the file end.
// Drop the heading itself: the release title already carries the version, so keeping
// `## [1.0.0]` under `## What's in v1.0.0` would read twice.
let body = changelog.slice(start);
const next = body.indexOf('\n## [', 1);
if (next !== -1) body = body.slice(0, next);
body = body.replace(/^## \[.*\]\n+/, '').trim();
// ESM, so top-level await is fine; this file is only ever run as the entry script.
await Bun.write(
'release_notes.md',
[
`## What's in ${tag}`,
'',
body,
'',
'### Artifacts',
'- `shiro-linux-x64`, `shiro-linux-arm64` - Linux',
'- `shiro-darwin-x64`, `shiro-darwin-arm64` - macOS',
'- `shiro-windows-x64.exe` - Windows',
'- `SHA256SUMS` - verify any of the above',
'',
'Install with `scripts/install.sh` (macOS/Linux) or `scripts/install.ps1` (Windows);',
'both verify the download against `SHA256SUMS`.',
].join('\n'),
);
console.log(`composed release_notes.md from the ${version} changelog section (${body.split('\n').length} prose lines)`);
process.exit(0);
+41 -9
View File
@@ -154,7 +154,8 @@ if (resumeArg) {
} }
} }
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers); const mcp =
has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers, cfg.mcpMode ?? 'lazy');
const instructions = has('--no-instructions') ? [] : await loadInstructions(); const instructions = has('--no-instructions') ? [] : await loadInstructions();
const skills = has('--no-skills') ? [] : await loadSkills(); const skills = has('--no-skills') ? [] : await loadSkills();
const customCommands = await loadCustomCommands(); const customCommands = await loadCustomCommands();
@@ -424,8 +425,15 @@ const hooks: AppHooks = {
install: async (name) => { install: async (name) => {
const entry = await findEntry(name); const entry = await findEntry(name);
const { path } = await registry.install(entry); const { path } = await registry.install(entry);
// Loaded on the next start rather than hot-swapped: a skill joins the system // A skill is live immediately: the session's skill tool reads its list on each
// prompt and a plugin joins the guard chain, and both are built once at boot. // call, so the next turn can invoke a skill installed right now. A plugin or
// tool joins the guard chain and the tool registry, both built once at boot,
// so those still need a restart — said plainly rather than implied.
if (entry.kind === 'skill') {
const next = await loadSkills(process.cwd());
session.setSkills(next);
return `installed skill ${entry.name} to ${path}\nit is loaded and callable next turn`;
}
return `installed ${entry.kind} ${entry.name} to ${path}\nrestart shiro to load it`; return `installed ${entry.kind} ${entry.name} to ${path}\nrestart shiro to load it`;
}, },
remove: async (name) => { remove: async (name) => {
@@ -434,7 +442,14 @@ const hooks: AppHooks = {
const bare = parsed ? parsed[2]! : name; const bare = parsed ? parsed[2]! : name;
for (const kind of kinds) { for (const kind of kinds) {
if (await registry.uninstall(kind, bare)) return `removed ${kind} ${bare}\nrestart shiro to unload it`; if (await registry.uninstall(kind, bare)) {
if (kind === 'skill') {
const next = await loadSkills(process.cwd());
session.setSkills(next);
return `removed skill ${bare}\nit is unloaded; the next turn no longer offers it`;
}
return `removed ${kind} ${bare}\nrestart shiro to unload it`;
}
} }
throw new Error(`nothing installed under the name "${bare}"`); throw new Error(`nothing installed under the name "${bare}"`);
}, },
@@ -445,10 +460,17 @@ const hooks: AppHooks = {
const servers = Object.entries(cfg.mcpServers ?? {}); const servers = Object.entries(cfg.mcpServers ?? {});
if (servers.length === 0) return 'no MCP servers configured\n\n`/mcp add` sets one up.'; if (servers.length === 0) return 'no MCP servers configured\n\n`/mcp add` sets one up.';
const lazy = (mcp?.tools['mcp_list'] ?? undefined) !== undefined;
const live = new Map<string, number>(); const live = new Map<string, number>();
for (const name of Object.keys(mcp?.tools ?? {})) { if (lazy) {
const server = /^mcp__([^_]+(?:_[^_]+)*)__/.exec(name)?.[1]; // Lazy mode: tools are fetched on demand, so "connected" is what the handle
if (server) live.set(server, (live.get(server) ?? 0) + 1); // says, not a count of registered mcp__ tools.
for (const name of mcp?.servers ?? []) live.set(name, -1);
} else {
for (const name of Object.keys(mcp?.tools ?? {})) {
const server = /^mcp__([^_]+(?:_[^_]+)*)__/.exec(name)?.[1];
if (server) live.set(server, (live.get(server) ?? 0) + 1);
}
} }
const failed = new Map((mcp?.errors ?? []).map((e) => [e.server, e.message])); const failed = new Map((mcp?.errors ?? []).map((e) => [e.server, e.message]));
@@ -457,7 +479,9 @@ const hooks: AppHooks = {
const state = failed.has(name) const state = failed.has(name)
? `failed: ${failed.get(name)}` ? `failed: ${failed.get(name)}`
: live.has(name) : live.has(name)
? `${live.get(name)} tools` ? live.get(name)! >= 0
? `${live.get(name)} tools`
: 'connected (lazy)'
: has('--no-mcp') : has('--no-mcp')
? 'not connected (--no-mcp)' ? 'not connected (--no-mcp)'
: 'not connected this session'; : 'not connected this session';
@@ -611,7 +635,15 @@ const facts: HeaderFact[] = [
memory && memory.all().length > 0 memory && memory.all().length > 0
? { label: 'memory', value: `${memory.all().length} notes about this project` } ? { label: 'memory', value: `${memory.all().length} notes about this project` }
: undefined, : undefined,
mcp && Object.keys(mcp.tools).length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).length} tools` } : undefined, mcp && Object.keys(mcp.tools).length > 0
? {
label: 'mcp',
value:
mcp.servers.length > 0
? `${mcp.servers.length} ${mcp.servers.length === 1 ? 'server' : 'servers'} (lazy)`
: `${Object.keys(mcp.tools).length} tools`,
}
: undefined,
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0 !mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const } ? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
: undefined, : undefined,
+24
View File
@@ -6,6 +6,8 @@ export type CommandAction =
| { type: 'exit' } | { type: 'exit' }
| { type: 'clear' } | { type: 'clear' }
| { type: 'compact' } | { type: 'compact' }
| { type: 'undo'; what: 'both' | 'files' | 'conversation' }
| { type: 'redo'; what: 'both' | 'files' | 'conversation' }
| { type: 'tools' } | { type: 'tools' }
| { type: 'cost' } | { type: 'cost' }
| { type: 'sessions' } | { type: 'sessions' }
@@ -57,6 +59,8 @@ export const COMMANDS: CommandSpec[] = [
{ name: 'memory', summary: 'compact the project memory with the model' }, { name: 'memory', summary: 'compact the project memory with the model' },
{ name: 'tools', summary: 'list available tools' }, { name: 'tools', summary: 'list available tools' },
{ name: 'compact', summary: 'replace history with a model-written summary' }, { name: 'compact', summary: 'replace history with a model-written summary' },
{ name: 'undo', arg: '[files|conversation]', summary: 'walk the last turn back: files, conversation, or both' },
{ name: 'redo', arg: '[conversation]', summary: 'put back what /undo took' },
{ name: 'cost', summary: 'tokens and estimated spend this session' }, { name: 'cost', summary: 'tokens and estimated spend this session' },
{ name: 'sessions', summary: 'list saved sessions' }, { name: 'sessions', summary: 'list saved sessions' },
{ name: 'resume', arg: '<id>', summary: 'load a saved session' }, { name: 'resume', arg: '<id>', summary: 'load a saved session' },
@@ -161,6 +165,22 @@ function parseMcp(arg: string): CommandAction {
} }
} }
/**
* `/undo [files|conversation]` and `/redo [conversation]`.
*
* The default is `both` for undo, because restoring one without the other is the
* failure the two are meant to prevent: files back without the history and the model
* re-reads a change it no longer made. A bare `files` or `conversation` narrows it.
* Redo defaults to the conversation, since file content after the turn was never kept.
*/
function parseUndoKind(arg: string, fallback: 'both' | 'conversation'): 'both' | 'files' | 'conversation' {
const word = arg.trim().toLowerCase();
if (word === 'files' || word === 'file') return 'files';
if (word === 'conversation' || word === 'chat' || word === 'history') return 'conversation';
if (word === 'both' || word === 'all' || word === '') return fallback;
return fallback;
}
/** /**
* Pure parser: no IO, so the TUI and headless mode share one definition. * Pure parser: no IO, so the TUI and headless mode share one definition.
* *
@@ -186,6 +206,10 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
return { type: 'clear' }; return { type: 'clear' };
case 'compact': case 'compact':
return { type: 'compact' }; return { type: 'compact' };
case 'undo':
return { type: 'undo', what: parseUndoKind(arg, 'both') };
case 'redo':
return { type: 'redo', what: parseUndoKind(arg, 'conversation') };
case 'tools': case 'tools':
return { type: 'tools' }; return { type: 'tools' };
case 'cost': case 'cost':
+5
View File
@@ -37,6 +37,11 @@ export type Config = {
/** Index for `/registry`. Omit for the default one. */ /** Index for `/registry`. Omit for the default one. */
registryUrl?: string; registryUrl?: string;
mcpServers?: Record<string, McpServerConfig>; mcpServers?: Record<string, McpServerConfig>;
/**
* How MCP tools reach the model: `lazy` registers meta-tools only (cheap until a
* tool is called), `eager` registers every server tool up front. Omit for lazy.
*/
mcpMode?: 'lazy' | 'eager';
}; };
const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json'); const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json');
+124 -10
View File
@@ -1,27 +1,47 @@
import { createMCPClient, type MCPClient } from '@ai-sdk/mcp'; import { createMCPClient, type MCPClient } from '@ai-sdk/mcp';
import { Experimental_StdioMCPTransport } from '@ai-sdk/mcp/mcp-stdio'; import { Experimental_StdioMCPTransport } from '@ai-sdk/mcp/mcp-stdio';
import type { ToolSet } from 'ai'; import { tool, type ToolSet } from 'ai';
import { z } from 'zod';
export type McpServerConfig = export type McpServerConfig =
| { command: string; args?: string[]; env?: Record<string, string>; cwd?: string } | { command: string; args?: string[]; env?: Record<string, string>; cwd?: string }
| { url: string; type?: 'http' | 'sse'; headers?: Record<string, string> }; | { url: string; type?: 'http' | 'sse'; headers?: Record<string, string> };
/**
* How a server's tools reach the model.
*
* `eager` registers every tool with its schema up front — cheap for a two-tool
* server, a tax for one that exposes twenty. `lazy` registers only the three
* meta-tools below and fetches a server's tools on demand via `mcp_call`, so a
* configured server costs almost nothing in the request until a tool is actually
* invoked.
*/
export type McpMode = 'eager' | 'lazy';
export type McpHandle = { export type McpHandle = {
tools: ToolSet; tools: ToolSet;
errors: { server: string; message: string }[]; errors: { server: string; message: string }[];
/** Server names, for the prompt's MCP line. Empty when the mode is eager. */
servers: string[];
close: () => Promise<void>; close: () => Promise<void>;
}; };
const isRemote = (c: McpServerConfig): c is Extract<McpServerConfig, { url: string }> => 'url' in c; const isRemote = (c: McpServerConfig): c is Extract<McpServerConfig, { url: string }> => 'url' in c;
/** /**
* Connects every configured server and namespaces its tools as `mcp__<server>__<tool>` * Connects every configured server and exposes its tools.
* so two servers exposing `search` cannot silently shadow each other. *
* A server that fails to start is reported, never fatal. * In `eager` mode the tools land in the returned set as `mcp__<server>__<tool>`, so
* two servers exposing `search` cannot silently shadow each other. In `lazy` mode the
* set holds only the three meta-tools and `servers` names the configured servers; a
* server that fails to start is reported, never fatal, in either mode.
*/ */
export async function connectMcp(servers: Record<string, McpServerConfig>): Promise<McpHandle> { export async function connectMcp(
servers: Record<string, McpServerConfig>,
mode: McpMode = 'lazy',
): Promise<McpHandle> {
const clients: MCPClient[] = []; const clients: MCPClient[] = [];
const tools: ToolSet = {}; const byName = new Map<string, MCPClient>();
const errors: McpHandle['errors'] = []; const errors: McpHandle['errors'] = [];
await Promise.all( await Promise.all(
@@ -38,9 +58,7 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
}), }),
}); });
clients.push(client); clients.push(client);
for (const [toolName, tool] of Object.entries(await client.tools())) { byName.set(name, client);
tools[`mcp__${name}__${toolName}`] = tool;
}
} catch (e) { } catch (e) {
errors.push({ server: name, message: e instanceof Error ? e.message : String(e) }); errors.push({ server: name, message: e instanceof Error ? e.message : String(e) });
} }
@@ -48,10 +66,106 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
); );
return { return {
tools, tools: mode === 'eager' ? await eagerTools(byName) : lazyTools(byName),
errors, errors,
servers: mode === 'eager' ? [] : [...byName.keys()],
close: async () => { close: async () => {
await Promise.all(clients.map((c) => c.close().catch(() => {}))); await Promise.all(clients.map((c) => c.close().catch(() => {})));
}, },
}; };
} }
/** Eager: every server tool gets an AI tool registered with its schema. */
async function eagerTools(byName: Map<string, MCPClient>): Promise<ToolSet> {
const tools: ToolSet = {};
await Promise.all(
[...byName.entries()].map(async ([name, client]) => {
for (const [toolName, t] of Object.entries(await client.tools())) {
tools[`mcp__${name}__${toolName}`] = t;
}
}),
);
return tools;
}
/**
* Cache of each client's AI tools, so `mcp_call` does not re-list on every call.
* The WeakMap drops entries when a client (and its session) is closed and collected.
*/
const clientToolsCache = new WeakMap<MCPClient, ToolSet>();
/** Fetches a server's AI tools, or returns the cached set from a prior call. */
async function cachedClientTools(client: MCPClient): Promise<ToolSet> {
const cached = clientToolsCache.get(client);
if (cached) return cached;
const tools = await client.tools();
clientToolsCache.set(client, tools);
return tools;
}
/**
* Lazy: three meta-tools instead of every server schema.
*
* `mcp_list` names a server's tools from their definitions (cheap, no schema).
* `mcp_inspect` reads one tool's schema so the model knows its inputs.
* `mcp_call` executes one tool on its server, making the server's tools available
* only from the moment they are actually invoked.
*/
function lazyTools(byName: Map<string, MCPClient>): ToolSet {
const connectionError = (name: string) =>
byName.has(name) ? undefined : `unknown server "${name}". Configured: ${[...byName.keys()].join(', ') || 'none'}`;
return {
mcp_list: tool({
description: `List the tools exposed by an MCP server. Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
inputSchema: z.object({ server: z.string().describe('Server name from your instructions') }),
execute: async ({ server }) => {
const err = connectionError(server);
if (err) throw new Error(err);
const client = byName.get(server)!;
const defs = await client.listTools();
return defs.tools.length === 0
? `server "${server}" exposes no tools`
: defs.tools.map((d) => `- ${d.name}: ${d.description ?? 'no description'}`).join('\n');
},
}),
mcp_inspect: tool({
description: `Inspect one tool's input schema on an MCP server. Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
inputSchema: z.object({
server: z.string().describe('Server name from your instructions'),
toolName: z.string().describe('Tool name, as listed by mcp_list'),
}),
execute: async ({ server, toolName }) => {
const err = connectionError(server);
if (err) throw new Error(err);
const client = byName.get(server)!;
const defs = await client.listTools();
const def = defs.tools.find((d) => d.name === toolName);
if (!def) throw new Error(`No tool "${toolName}" on "${server}". List first with mcp_list.`);
return def.inputSchema ? JSON.stringify(def.inputSchema, null, 2) : `tool "${toolName}" declares no input schema`;
},
}),
mcp_call: tool({
description:
`Call one tool on an MCP server. Inspect its schema with mcp_inspect first. ` +
`Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
inputSchema: z.object({
server: z.string().describe('Server name from your instructions'),
toolName: z.string().describe('Tool name, as listed by mcp_list'),
args: z.record(z.string(), z.unknown()).describe('Arguments the tool expects, from mcp_inspect'),
}),
execute: async ({ server, toolName, args }) => {
const err = connectionError(server);
if (err) throw new Error(err);
const client = byName.get(server)!;
const serverTools = await cachedClientTools(client);
const aiTool = serverTools[toolName];
if (!aiTool)
throw new Error(
`No tool "${toolName}" on "${server}". List first with mcp_list (server exposes: ${Object.keys(serverTools).join(', ') || 'none'}).`,
);
return aiTool.execute!(args, { toolCallId: 'mcp_call', messages: [] } as never);
},
}),
};
}
+5 -4
View File
@@ -181,6 +181,11 @@ export const DEFAULT_PERMISSIONS: PermissionConfig = {
apply_patch: 'ask', apply_patch: 'ask',
move_file: 'ask', move_file: 'ask',
delete_file: 'ask', delete_file: 'ask',
insert_lines: 'ask',
delete_lines: 'ask',
replace_lines: 'ask',
append_file: 'ask',
prepend_file: 'ask',
bash: 'ask', bash: 'ask',
web_fetch: 'ask', web_fetch: 'ask',
}; };
@@ -272,10 +277,6 @@ export class Permissions {
this.granted.set(tool, set); this.granted.set(tool, set);
} }
granted_(tool: string): string[] {
return [...(this.granted.get(tool) ?? [])];
}
/** The decision for one call, and which pattern decided it. */ /** The decision for one call, and which pattern decided it. */
check(tool: string, input: unknown): Resolved { check(tool: string, input: unknown): Resolved {
const resolved = resolve(entryFor(tool, this.config), tool, input); const resolved = resolve(entryFor(tool, this.config), tool, input);
+4
View File
@@ -4,6 +4,10 @@ export type Rate = { inputPerMTok: number; outputPerMTok: number };
* USD per million tokens. Prefix match on the model id, longest first, so * USD per million tokens. Prefix match on the model id, longest first, so
* `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates * `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates
* drift, so this is a best-effort estimate rather than a billing source. * drift, so this is a best-effort estimate rather than a billing source.
*
* Source: vendor pricing pages, checked 2026-09-17. Anthropic (Anthropic API, not
* Batch) and OpenAI listed rates; DeepSeek and Grok per their API pricing. Rates
* are for input, then output. Re-verify before trusting a live spend figure.
*/ */
const RATES: Record<string, Rate> = { const RATES: Record<string, Rate> = {
'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 }, 'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 },
+10 -2
View File
@@ -131,8 +131,10 @@ function renderTools(available: readonly string[]): string {
// free, and the schema already says what each takes. // free, and the schema already says what each takes.
const git = extra.filter((n) => GIT_TOOL_NAMES.includes(n) && n !== 'git_commit_message'); const git = extra.filter((n) => GIT_TOOL_NAMES.includes(n) && n !== 'git_commit_message');
const mcp = extra.filter((n) => n.startsWith('mcp__')); const mcp = extra.filter((n) => n.startsWith('mcp__'));
const META = ['mcp_list', 'mcp_inspect', 'mcp_call'];
const lazyMcp = META.filter((n) => available.includes(n));
const other = extra.filter( const other = extra.filter(
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__'), (n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__') && !META.includes(n),
); );
if (git.length > 0) { if (git.length > 0) {
@@ -140,7 +142,13 @@ function renderTools(available: readonly string[]): string {
`- ${git.join(', ')}: read-only git, no approval needed. Use them instead of bash for history and diffs; they cannot mutate the repository.`, `- ${git.join(', ')}: read-only git, no approval needed. Use them instead of bash for history and diffs; they cannot mutate the repository.`,
); );
} }
if (mcp.length > 0) { if (lazyMcp.length > 0) {
// Lazy mode: the meta-tool descriptions already name the connected servers, so
// the model needs the workflow, not a schema listing.
lines.push(
`- ${lazyMcp.join(', ')}: MCP tools are fetched on demand. mcp_list names a server's tools, mcp_inspect reads one tool's schema, mcp_call runs it. Never guess a server or tool name: list first.`,
);
} else if (mcp.length > 0) {
lines.push( lines.push(
`- ${mcp.join(', ')}: from MCP servers, named mcp__<server>__<tool>. Each needs approval; read its own description before calling.`, `- ${mcp.join(', ')}: from MCP servers, named mcp__<server>__<tool>. Each needs approval; read its own description before calling.`,
); );
+78 -2
View File
@@ -92,8 +92,6 @@ export function detachOrphanedItems(before: ModelMessage[], after: ModelMessage[
export type PruneOptions = Parameters<typeof pruneMessages>[0]; export type PruneOptions = Parameters<typeof pruneMessages>[0];
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
const anyParts = (message: ModelMessage): Part[] => const anyParts = (message: ModelMessage): Part[] =>
Array.isArray(message.content) ? (message.content as Part[]) : []; Array.isArray(message.content) ? (message.content as Part[]) : [];
@@ -167,6 +165,84 @@ export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
return dropOrphanedResults(detachOrphanedItems(options.messages, pruned)); return dropOrphanedResults(detachOrphanedItems(options.messages, pruned));
} }
/**
* The messages a prune would discard, so they can be summarized before they go.
*
* Compaction keeps the model's *memory of a turn* — the tool tail it is told to
* keep stays verbatim. What it does not keep is any statement of what was
* dropped. So a decision from forty messages ago vanishes silently, and the model
* contradicts it with full confidence, because as far as it can tell it never
* said that.
*
* Identity is by reference, not by value: `prunePreservingItems` rebuilds the
* surviving messages with `{ ...message }`, so a value comparison would report
* every message as changed and no message as dropped. `Set` on the object
* references is exact.
*
* Only messages that carry content worth summarizing are returned — an assistant
* turn consisting of nothing but a dropped `reasoning` part is not a decision, and
* summarizing "the model thought for a while" is worse than saying nothing.
*/
export function droppedBy(before: ModelMessage[], after: ModelMessage[]): ModelMessage[] {
const surviving = new Set<ModelMessage>(after);
return before.filter((message) => !surviving.has(message));
}
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
function textOf(message: ModelMessage): string {
const { content } = message;
if (typeof content === 'string') return content;
if (!Array.isArray(content)) return '';
const chunks: string[] = [];
for (const part of content as Part[]) {
const p = part as Part & { text?: unknown; input?: unknown; output?: unknown };
if (typeof p.text === 'string') chunks.push(p.text);
// A tool call's input is the decision made: the path, the command, the patch.
else if (p.type === 'tool-call' && p.input !== undefined) chunks.push(JSON.stringify(p.input));
// A tool result is what came back. Without it a digest says what the model
// asked for and nothing about the answer, which is the half a later
// contradiction is usually argued from.
else if (ANSWER_PARTS.has(p.type) && p.output !== undefined) {
const rendered = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
chunks.push(rendered);
}
}
return chunks.join(' ').trim();
}
/**
* A one-line-per-message digest of what a prune wants to drop.
*
* This is the *fallback* when no summarizer is available or the call fails: crude,
* but it preserves the thing that matters — which tool touched which path, and in
* what order — rather than the nothing that is there today. The summarizer, when
* it runs, is a model and reads far better than this.
*/
export function digestOf(dropped: readonly ModelMessage[]): string {
const lines: string[] = [];
for (const message of dropped) {
const text = textOf(message);
if (!text) continue;
const role = message.role === 'tool' ? 'result' : message.role;
const clipped = text.length > 160 ? `${text.slice(0, 160)}...` : text;
lines.push(`- (${role}) ${clipped}`);
}
return lines.join('\n');
}
export const PRUNED_SPAN_PREFIX = 'Earlier in this session, now compacted away:';
export function isPrunedSpanSummary(message: ModelMessage): boolean {
return message.role === 'user' && typeof message.content === 'string' && message.content.startsWith(PRUNED_SPAN_PREFIX);
}
export function prunedSpanMessage(summary: string | undefined, dropped: readonly ModelMessage[]): ModelMessage | undefined {
const body = summary?.trim() || digestOf(dropped);
if (!body) return undefined;
return { role: 'user', content: `${PRUNED_SPAN_PREFIX}\n\n${body}` };
}
/** /**
* How many trailing messages keep their tool content, widest first. * How many trailing messages keep their tool content, widest first.
* *
+270 -4
View File
@@ -17,7 +17,9 @@ import { Permissions, type PermissionConfig } from './permission';
import type { PluginHost } from './plugins'; import type { PluginHost } from './plugins';
import { costOf, formatUsd } from './pricing'; import { costOf, formatUsd } from './pricing';
import { systemPrompt } from './prompt'; import { systemPrompt } from './prompt';
import { detachProviderItems, pruneToFit } from './prune'; import { detachProviderItems, digestOf, droppedBy, isPrunedSpanSummary, prunedSpanMessage, pruneToFit } from './prune';
import { restore, Snapshots, type TurnSnapshot } from './snapshot';
import { createStepBackTool, type LoopEntry } from './step-back';
import { createSkillTool, renderSkills, type Skill } from './skills'; import { createSkillTool, renderSkills, type Skill } from './skills';
import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools'; import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools';
@@ -38,6 +40,21 @@ export type ApprovalRequest = {
/** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */ /** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */
export type ApprovalDecision = 'once' | 'always' | 'deny'; export type ApprovalDecision = 'once' | 'always' | 'deny';
/** What an undo did, so the UI can say which files moved and which did not. */
export type UndoResult = {
snapshot: TurnSnapshot;
restored: string[];
removed: string[];
conversationTrimmed: boolean;
};
export type RedoResult = {
snapshot: TurnSnapshot;
/** False when files were asked for: only pre-images are ever captured. */
filesRestored: boolean;
what: 'both' | 'files' | 'conversation';
};
export type AgentEvent = export type AgentEvent =
| { type: 'text'; text: string } | { type: 'text'; text: string }
| { type: 'reasoning'; text: string } | { type: 'reasoning'; text: string }
@@ -74,6 +91,8 @@ export type SessionOptions = {
autoApprove?: readonly string[]; autoApprove?: readonly string[];
/** Prune the history once the estimated token count crosses this. */ /** Prune the history once the estimated token count crosses this. */
compactThreshold?: number; compactThreshold?: number;
/** Identical calls to an allowed tool before it is asked about anyway. Default 3. */
repeatLimit?: number;
/** Retries per model call for transient failures. */ /** Retries per model call for transient failures. */
maxRetries?: number; maxRetries?: number;
/** AGENTS.md-style files appended to the system prompt. */ /** AGENTS.md-style files appended to the system prompt. */
@@ -94,6 +113,12 @@ export type SessionOptions = {
onNotebookChange?: (state: NotebookState) => void; onNotebookChange?: (state: NotebookState) => void;
}; };
/**
* Length-based token estimate for deciding *when to prune*, not for billing.
* JSON char count / 4 approximates token count closely enough to gate compaction,
* but real billed tokens come from the SDK's reported usage (`inputTokens`), never
* from here. `/cost` and the budget ceiling use the SDK figure.
*/
const estimateTokens = (messages: ModelMessage[]) => Math.round(JSON.stringify(messages).length / 4); const estimateTokens = (messages: ModelMessage[]) => Math.round(JSON.stringify(messages).length / 4);
/** Estimated tokens at which the wire history is pruned. */ /** Estimated tokens at which the wire history is pruned. */
@@ -102,6 +127,26 @@ const DEFAULT_COMPACT_THRESHOLD = 120_000;
/** Identical calls in one turn before an allowed tool is asked about anyway. */ /** Identical calls in one turn before an allowed tool is asked about anyway. */
const REPEAT_LIMIT = 3; const REPEAT_LIMIT = 3;
/**
* Squashes a tool result into a few characters for the loop trace.
*
* The trace is fed back to the model verbatim, so a 30 KB read_file output would
* fill the reflection with noise. A short string keeps `step_back` honest about
* what happened without flooding the next context window.
*/
function summarizeToolResult(output: unknown): string {
if (typeof output === 'string') return output.length <= 80 ? output : `${output.slice(0, 80)}…(${output.length} chars)`;
try {
const json = JSON.stringify(output);
return json.length <= 80 ? json : `${json.slice(0, 80)}…`;
} catch {
return String(output);
}
}
/** Ceiling on an injected span summary, so the summary cannot defeat the compaction. */
const MAX_SPAN_SUMMARY_CHARS = 1_200;
const callKey = (toolName: string, input: unknown) => `${toolName}:${JSON.stringify(input ?? null)}`; const callKey = (toolName: string, input: unknown) => `${toolName}:${JSON.stringify(input ?? null)}`;
/** /**
@@ -121,6 +166,8 @@ export class Session {
readonly messages: ModelMessage[]; readonly messages: ModelMessage[];
readonly tools: ToolSet; readonly tools: ToolSet;
readonly notebook: Notebook; readonly notebook: Notebook;
/** Pre-images of files this session's turns have changed, newest last. */
readonly snapshots: Snapshots;
inputTokens = 0; inputTokens = 0;
outputTokens = 0; outputTokens = 0;
/** Subagent token use, priced against the subagent's own model id in /cost. */ /** Subagent token use, priced against the subagent's own model id in /cost. */
@@ -136,6 +183,14 @@ export class Session {
/** The 80% spend warning is shown once, not on every turn past the line. */ /** The 80% spend warning is shown once, not on every turn past the line. */
private warnedSpend = false; private warnedSpend = false;
private controller: AbortController | undefined; private controller: AbortController | undefined;
/** Tools this turn used that no snapshot can cover, reported when the turn ends. */
private readonly uncoveredTools = new Set<string>();
/** The snapshot the last undo removed, so `/redo` can put it back. */
private lastUndone: TurnSnapshot | undefined;
/** Every tool call this turn, input + outcome, for the loop-detection tool to reflect on. */
private readonly loopTrace: LoopEntry[] = [];
/** toolCallId -> { toolName, input }, so a result can be paired with its call. */
private readonly callInputs = new Map<string, { toolName: string; input: string }>();
constructor(private readonly opts: SessionOptions) { constructor(private readonly opts: SessionOptions) {
this.messages = opts.messages ?? []; this.messages = opts.messages ?? [];
@@ -143,12 +198,19 @@ export class Session {
this.notebook.restore(opts.notebook); this.notebook.restore(opts.notebook);
this.model = opts.model; this.model = opts.model;
this.variant = opts.agent ?? DEFAULT_VARIANT; this.variant = opts.agent ?? DEFAULT_VARIANT;
this.snapshots = new Snapshots(opts.cwd ?? process.cwd());
const sessionTools = { const sessionTools = {
...this.notebook.tools(), ...this.notebook.tools(),
...(opts.memory ? opts.memory.tools() : {}), ...(opts.memory ? opts.memory.tools() : {}),
...(opts.skills && opts.skills.length > 0 ? { skill: createSkillTool(opts.skills) } : {}), // Always registered, even with no skills, so a mid-session install of the
// first skill is callable next turn without a session rebuild. The tool's
// description reads the live list and says "none" when it is empty.
skill: createSkillTool(() => this.opts.skills ?? []),
...(opts.ask ? { ask: createAskTool(opts.ask) } : {}), ...(opts.ask ? { ask: createAskTool(opts.ask) } : {}),
step_back: createStepBackTool({
trace: () => this.loopTrace,
}),
}; };
this.tools = { ...builtinTools, ...sessionTools, ...(opts.plugins?.tools ?? {}), ...(opts.extraTools ?? {}) }; this.tools = { ...builtinTools, ...sessionTools, ...(opts.plugins?.tools ?? {}), ...(opts.extraTools ?? {}) };
@@ -306,6 +368,26 @@ export class Session {
return count; return count;
} }
/**
* Appends a completed tool call to the loop trace, pairing it with its input.
*
* The SDK streams `tool-result` without the input that produced it, so the input is
* kept alongside on the `tool-call` part. A `step_back` call needs this pairing to
* say *which* call produced *which* outcome.
*/
private recordTrace(toolCallId: string, result: string): void {
const call = this.callInputs.get(toolCallId);
this.callInputs.delete(toolCallId);
if (!call) return;
this.loopTrace.push({
step: this.loopTrace.length + 1,
toolName: call.toolName,
input: call.input,
result,
at: new Date().toISOString(),
});
}
/** /**
* Approval decisions, evaluated per call by the SDK. * Approval decisions, evaluated per call by the SDK.
* *
@@ -326,6 +408,11 @@ export class Session {
return async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => { return async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const { toolName, input } = toolCall; const { toolName, input } = toolCall;
// Before anything else, because the guard may deny the call and because a
// later hook must not be able to move the capture after the write.
const { covered } = await this.snapshots.captureFor(toolName, input);
if (!covered) this.uncoveredTools.add(toolName);
const blocked = await this.opts.plugins?.guard({ const blocked = await this.opts.plugins?.guard({
toolName, toolName,
input, input,
@@ -346,7 +433,18 @@ export class Session {
} }
const repeats = this.repeatCount(toolName, input); const repeats = this.repeatCount(toolName, input);
if (decision === 'allow' && repeats < REPEAT_LIMIT) return undefined; const limit = this.opts.repeatLimit ?? REPEAT_LIMIT;
if (decision === 'allow' && repeats < limit) return undefined;
if (decision === 'allow') {
// Repeated three times with a permission that says `allow`: the model is
// looping, not asking, and it should stop and look at the trace rather than
// burn another approval. This is the point the step_back tool exists for.
notices.push(
`You have called ${toolName} with the same input ${repeats + 1} times this turn. It is not making progress. ` +
`Use step_back to reflect on what changed between attempts, then try a different approach or stop.`,
);
}
why.set(callKey(toolName, input), { why.set(callKey(toolName, input), {
...(pattern ? { matchedPattern: pattern } : {}), ...(pattern ? { matchedPattern: pattern } : {}),
@@ -378,6 +476,92 @@ export class Session {
return { before, after: this.messages.length }; return { before, after: this.messages.length };
} }
/**
* Walks the last turn back: files, conversation, or both.
*
* The three-way split is the point. Restoring files without the conversation leaves
* the model believing edits are on disk that are not, so its next turn is built on a
* state that no longer exists — it re-reads a file expecting its own change and finds
* the original, which reads to the model as the change having been rejected. Restoring
* the conversation without the files is the mirror: the model forgets it made an edit
* that is still there. So the default is both, and the caller can narrow it.
*
* Returns undefined when there is nothing to undo, which the UI reports as such
* rather than as a failure.
*/
async undo(what: 'both' | 'files' | 'conversation' = 'both'): Promise<UndoResult | undefined> {
const snap = this.snapshots.pop();
if (!snap) return undefined;
const files = what === 'conversation' ? { restored: [], removed: [] } : await restore(snap, this.snapshots.cwdOf());
if (what !== 'files') this.trimTo(snap.messageCount);
this.lastUndone = snap;
return { snapshot: snap, ...files, conversationTrimmed: what !== 'files' };
}
/**
* Puts back what `undo` took, without a second snapshot.
*
* A redo cannot restore file content from the session, because the content that
* existed after the turn was never captured — only the pre-image was. So a redo of
* the files is declined honestly rather than approximated: the whole point of undo
* is that the user trusts what it says it did. The conversation is restored from the
* snapshot's own record, which is exact.
*/
redo(what: 'both' | 'files' | 'conversation' = 'conversation'): RedoResult | undefined {
const snap = this.lastUndone;
if (!snap) return undefined;
this.snapshots.push(snap);
this.lastUndone = undefined;
return { snapshot: snap, filesRestored: false, what };
}
undoable(): readonly TurnSnapshot[] {
return this.snapshots.list();
}
private trimTo(length: number): void {
if (this.messages.length <= length) return;
this.messages.length = Math.max(0, length);
this.opts.onChange?.(this.messages);
}
/**
* Closes the open snapshot and reports what the turn could not cover.
*
* Called before every `done`, not from `send`'s `finally`, because a notice that
* arrives after `done` is a notice the UI has already stopped listening for —
* `done` is what a consumer treats as the end of the turn and stops on.
*
* Two reasons to speak, and the second does not depend on the first: a turn with no
* snapshotted edits can still have run `bash` and changed the tree, which is exactly
* when the warning matters most.
*/
private *closingNotices(): Generator<AgentEvent> {
const snapshot = this.snapshots.commit();
const uncovered = [...this.uncoveredTools];
if (uncovered.length === 0) return;
const covered = snapshot ? `${snapshot.files.length} file(s) changed this turn and can be undone with /undo. ` : '';
yield {
type: 'notice',
text: `${covered}${uncovered.join(', ')} ran this turn and cannot be snapshotted, so any changes it made will not be undone.`,
};
}
/**
* Swaps the live skill list at a turn boundary.
*
* The `skill` tool and the system-prompt catalogue both read the list on each
* call, so replacing it here is atomic across the two and takes effect on the
* next turn with no session rebuild. The caller is responsible for only doing
* this between turns — a turn in flight already holds its rules.
*/
setSkills(skills: Skill[]): void {
this.opts.skills = skills;
}
async *send(userText: string): AsyncGenerator<AgentEvent> { async *send(userText: string): AsyncGenerator<AgentEvent> {
// The ceiling is checked before the model is: a turn started past the limit // The ceiling is checked before the model is: a turn started past the limit
// would spend money the caller said not to. An unpriced model cannot be // would spend money the caller said not to. An unpriced model cannot be
@@ -402,7 +586,12 @@ export class Session {
// Per turn, not per step: a tool called once in each of three steps is the // Per turn, not per step: a tool called once in each of three steps is the
// loop this guards against. // loop this guards against.
this.seen.clear(); this.seen.clear();
this.uncoveredTools.clear();
this.loopTrace.length = 0;
this.staleItemsRepaired = false; this.staleItemsRepaired = false;
// Opened before the model runs and closed after it stops, so every write the
// turn makes lands in one snapshot the user can walk back to.
this.snapshots.begin(userText, this.messages.length);
const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = []; const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = [];
onBashOutput(({ toolCallId, chunk }) => { onBashOutput(({ toolCallId, chunk }) => {
@@ -414,6 +603,10 @@ export class Session {
yield* this.run(signal, threshold, outputs); yield* this.run(signal, threshold, outputs);
} finally { } finally {
onBashOutput(undefined); onBashOutput(undefined);
// Belt and braces: closingNotices runs before every `done`, but an aborted turn
// can return through a path that never reached it, and an uncommitted snapshot
// would then be silently dropped rather than kept.
this.snapshots.commit();
await this.opts.plugins?.afterTurn(); await this.opts.plugins?.afterTurn();
} }
} }
@@ -431,6 +624,70 @@ export class Session {
return true; return true;
} }
/**
* Prunes the canonical history once a run has landed.
*
* prepareStep only trims the wire copy for the next request; without this write-back
* the stored history keeps growing, the context meter pins at 100%, and every later
* turn re-prunes the same messages from scratch.
*/
private async compactCanonical(threshold: number): Promise<{ before: number; after: number } | null> {
const before = this.messages.length;
if (estimateTokens(this.messages) <= threshold) return null;
const pruned = pruneToFit({ messages: this.messages, threshold, estimate: estimateTokens });
if (pruned.length === before) return null;
const dropped = droppedBy(this.messages, pruned);
const summary = await this.summarizeSpan(dropped.filter((m) => !isPrunedSpanSummary(m)));
this.replace(this.withSummarizedSpan(summary, pruned, dropped));
return { before, after: this.messages.length };
}
/**
* Puts a summary of the discarded span at the head of the history it was dropped from.
*
* Without this the model is told which tool results to keep and nothing about what
* was dropped, so it states a decision it made forty messages ago as though it had
* never made it.
*
* The summary is bounded two ways because a summary that grows with the session
* defeats the point of compacting at all: the input is capped at the span's own
* digest, and the output is capped by instruction and by hard truncation. A failed
* or empty call falls back to the digest, which costs nothing and still carries
* which tool touched which path — the part a contradiction is usually built from.
*/
private async summarizeSpan(dropped: readonly ModelMessage[]): Promise<string | undefined> {
const digest = digestOf(dropped);
if (!digest) return undefined;
try {
const { text } = await generateText({
model: this.model,
system:
'These lines are the condensed record of an earlier part of a coding session that has been ' +
'compacted out of the conversation. Write at most 120 words of notes capturing decisions made, ' +
'files touched, commands run and their outcome, and anything still pending. State only what the ' +
'lines support; do not invent detail and do not address the reader.',
messages: [{ role: 'user', content: digest }],
maxRetries: this.opts.maxRetries ?? 3,
});
const trimmed = text.trim();
return trimmed ? trimmed.slice(0, MAX_SPAN_SUMMARY_CHARS) : undefined;
} catch {
// A summarizer that cannot run must not cost the turn its compaction: the
// digest is a worse record, not an absent one.
return undefined;
}
}
private withSummarizedSpan(summary: string | undefined, pruned: ModelMessage[], dropped: readonly ModelMessage[]): ModelMessage[] {
const worthSummarizing = dropped.filter((m) => !isPrunedSpanSummary(m));
if (worthSummarizing.length === 0) return pruned;
const prior = dropped.filter(isPrunedSpanSummary);
const spanMessage = prunedSpanMessage(summary, worthSummarizing);
if (!spanMessage) return pruned;
return [...prior, spanMessage, ...pruned];
}
private async *run( private async *run(
signal: AbortSignal, signal: AbortSignal,
threshold: number, threshold: number,
@@ -506,12 +763,15 @@ export class Session {
yield { type: 'tool-start', id: part.id, name: part.toolName }; yield { type: 'tool-start', id: part.id, name: part.toolName };
break; break;
case 'tool-call': case 'tool-call':
this.callInputs.set(part.toolCallId, { toolName: part.toolName, input: JSON.stringify(part.input ?? null) });
yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input }; yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input };
break; break;
case 'tool-result': case 'tool-result':
this.recordTrace(part.toolCallId, summarizeToolResult(part.output));
yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output }; yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output };
break; break;
case 'tool-error': case 'tool-error':
this.recordTrace(part.toolCallId, `error: ${part.error instanceof Error ? part.error.message : String(part.error)}`);
yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error }; yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error };
break; break;
case 'tool-approval-request': { case 'tool-approval-request': {
@@ -535,6 +795,7 @@ export class Session {
yield { type: 'tool-denied', name: part.toolName }; yield { type: 'tool-denied', name: part.toolName };
break; break;
case 'abort': case 'abort':
yield* this.closingNotices();
yield { type: 'done' }; yield { type: 'done' };
return; return;
case 'error': case 'error':
@@ -555,6 +816,7 @@ export class Session {
} }
} catch (error) { } catch (error) {
if (signal.aborted) { if (signal.aborted) {
yield* this.closingNotices();
yield { type: 'done' }; yield { type: 'done' };
return; return;
} }
@@ -579,7 +841,10 @@ export class Session {
while (guardNotices.length > 0) yield { type: 'notice', text: guardNotices.shift()! }; while (guardNotices.length > 0) yield { type: 'notice', text: guardNotices.shift()! };
this.messages.push(...(await result.responseMessages)); this.messages.push(...(await result.responseMessages));
this.opts.onChange?.(this.messages);
// prepareStep already emits `compacted` at the same threshold crossing, and
// replace() fires onChange on the fold path; both would double up otherwise.
if (!(await this.compactCanonical(threshold))) this.opts.onChange?.(this.messages);
if (pending.length === 0) { if (pending.length === 0) {
const usage = await result.usage; const usage = await result.usage;
@@ -595,6 +860,7 @@ export class Session {
text: `approaching spend ceiling: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used`, text: `approaching spend ceiling: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used`,
}; };
} }
yield* this.closingNotices();
yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens }; yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens };
return; return;
} }
+14 -4
View File
@@ -100,18 +100,28 @@ export function renderSkills(skills: Skill[]): string {
].join('\n'); ].join('\n');
} }
export function createSkillTool(skills: Skill[]) { /**
const names = skills.map((s) => s.name); * The `skill` tool, reading the list live so a mid-session install shows up on the
* next turn without a restart.
*
* The catalogue in the system prompt and the names in this tool's description both
* come from the same list on each read, so replacing the list at a turn boundary
* makes both stale bytes atomic: a skill the model can call it can see already.
*/
export function createSkillTool(getSkills: () => Skill[]) {
const names = () => getSkills().map((s) => s.name).join(', ');
return tool({ return tool({
description: description:
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' + 'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`, `what you are about to do, then follow what it says. Available: ${names() || 'none'}.`,
inputSchema: z.object({ inputSchema: z.object({
name: z.string().describe('Skill name from the list in your instructions'), name: z.string().describe('Skill name from the list in your instructions'),
}), }),
execute: async ({ name }) => { execute: async ({ name }) => {
const skills = getSkills();
const skill = skills.find((s) => s.name === name.trim().toLowerCase()); const skill = skills.find((s) => s.name === name.trim().toLowerCase());
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`); if (!skill)
throw new Error(`No skill named "${name}". Available: ${skills.map((s) => s.name).join(', ') || 'none'}`);
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`; return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
}, },
}); });
+269
View File
@@ -0,0 +1,269 @@
import { createHash } from 'node:crypto';
import { join, relative, resolve, sep } from 'node:path';
/**
* Pre-images of files a turn is about to change, so a turn can be walked back.
*
* The design follows the one thing every comparable CLI agrees on and the one thing
* they all get wrong in the same way. Agreement: a snapshot is taken *before* the
* turn runs, because after it runs the original content is gone. The shared flaw: a
* snapshot only covers the tools that write through a known interface — Claude Code
* tracks Write/Edit/NotebookEdit and explicitly not `bash` — so the undo is partial
* and the user has to know which half they are in.
*
* Two decisions fall out of taking that seriously.
*
* 1. Capture lazily, per file, not by walking the tree. A session turn may touch
* three files out of fifty thousand; a full-tree copy per prompt is a tarball of
* the repository the user did not ask for and cannot afford. Instead the first
* write to a path records its pre-image, and a later write to the same path in
* the same turn does not overwrite it — the pre-image is the state before the
* *turn*, which is what an undo restores.
*
* 2. Record what is *not* covered rather than implying it is. A `bash` command that
* rewrites a file, a `git checkout`, a build artifact — none of these pass through
* a path argument the way `write_file` does, so none are captured. The tool
* reports which paths it restored and the caller is told the rest is unknown,
* which is the same honesty `bash` already owes an interrupted command.
*/
/** One file as it was before the turn that changed it. `before === undefined` means it did not exist. */
export type PreImage = {
/** Workspace-relative, using forward slashes, so a restore is portable across platforms. */
path: string;
before: string | undefined;
};
export type TurnSnapshot = {
/** Monotonic turn number, so the UI can name what is being undone. */
turn: number;
at: string;
/** The user prompt that opened the turn, for a menu that lists them. */
prompt: string;
/** Files this turn changed, with their content from before it started. */
files: PreImage[];
/** The conversation length when the turn began, so undo can trim it back. */
messageCount: number;
};
/** Claude Code keeps 100; beyond that the memory is worth more than the recall. */
export const MAX_SNAPSHOTS = 100;
/** A single file larger than this is not snapshotted; a 40 MB binary is not an edit. */
const MAX_FILE_BYTES = 2 * 1024 * 1024;
/**
* The workspace-relative, slash-normalised form of a path, or undefined if it is
* outside the workspace.
*
* Outside is refused rather than clamped: a path that escapes the workspace is not
* something this repository can undo, and recording it would imply an undo that
* cannot happen. Paths are normalised to forward slashes because a snapshot written
* on Windows may be read on a machine where a backslash is a filename character.
*/
export function relPath(cwd: string, abs: string): string | undefined {
const root = resolve(cwd);
const target = resolve(abs);
const rel = relative(root, target);
if (rel === '' || rel.startsWith('..') || rel.includes(`..${sep}`)) return undefined;
return rel.split(sep).join('/');
}
/**
* The absolute path a tool call will write to, when there is exactly one.
*
* Deliberately a small, explicit map rather than a guess. `multi_edit` and the line
* editors each take a single `path`; `move_file` takes `from` and `to` and both are
* recorded; `delete_file` takes a `path`. `apply_patch` carries its paths inside the
* patch text, and `bash` carries none — both are reported as uncovered rather than
* silently not snapshotted.
*/
export function touchedPaths(toolName: string, input: unknown): { paths: string[]; covered: boolean } {
const o = (input ?? {}) as Record<string, unknown>;
const one = (key: string) => (typeof o[key] === 'string' ? [o[key] as string] : []);
switch (toolName) {
case 'write_file':
case 'edit_file':
case 'multi_edit':
case 'delete_file':
case 'insert_lines':
case 'delete_lines':
case 'replace_lines':
case 'append_file':
case 'prepend_file':
return { paths: one('path'), covered: true };
case 'move_file':
return { paths: [...one('from'), ...one('to')], covered: true };
case 'apply_patch': {
const patch = typeof o['patch'] === 'string' ? (o['patch'] as string) : '';
const paths = [...patch.matchAll(/^\*\*\* (?:Add|Update|Delete) File: (.+)$/gm)].map((m) => m[1]!.trim());
const moves = [...patch.matchAll(/^\*\*\* Move to: (.+)$/gm)].map((m) => m[1]!.trim());
return { paths: [...paths, ...moves], covered: true };
}
case 'bash':
// Arbitrary code: an untouched-looking `node -e` can rewrite the tree.
return { paths: [], covered: false };
default:
return { paths: [], covered: true };
}
}
/**
* Records pre-images for the files a turn changes, and restores them on undo.
*
* One instance per session. Holds at most MAX_SNAPSHOTS turns; the oldest falls off
* the front, because the turn someone wants back is almost always the last one.
*/
export class Snapshots {
private readonly turns: TurnSnapshot[] = [];
private current: { turn: number; at: string; prompt: string; files: Map<string, string | undefined>; messageCount: number } | null =
null;
private next = 1;
private readonly cwd: string;
constructor(cwd: string = process.cwd()) {
this.cwd = cwd;
}
/** Opens a turn. Called once per user prompt, before the model runs. */
begin(prompt: string, messageCount: number): void {
this.current = { turn: this.next++, at: new Date().toISOString(), prompt, files: new Map(), messageCount };
}
/**
* Records a file's content before a tool changes it, on the first write of the turn.
*
* Idempotent per path per turn: the second `write_file` to the same path in one turn
* must not replace the pre-image with the intermediate content the first write left,
* because undo restores the turn's starting state, not the midpoint.
*
* Read failures are swallowed. A snapshot is a convenience; a tool call that fails
* because the snapshot layer could not read an unrelated path would be worse than no
* undo at all.
*/
async capture(absPath: string): Promise<void> {
if (!this.current) return;
const rel = relPath(this.cwd, absPath);
if (rel === undefined) return;
if (this.current.files.has(rel)) return;
try {
const file = Bun.file(absPath);
const exists = await file.exists();
if (!exists) {
this.current.files.set(rel, undefined);
return;
}
if (file.size > MAX_FILE_BYTES) return;
this.current.files.set(rel, await file.text());
} catch {
return;
}
}
/** Records any path a tool call is about to touch. Returns whether the tool is covered at all. */
async captureFor(toolName: string, input: unknown): Promise<{ covered: boolean; paths: string[] }> {
const { paths, covered } = touchedPaths(toolName, input);
for (const p of paths) await this.capture(resolve(this.cwd, p));
return { paths, covered };
}
/**
* Closes the turn, keeping it only if it changed something.
*
* A turn that read and answered without writing is not worth a slot, and keeping it
* would make `/undo` step past a turn that has nothing to undo — which reads as the
* command being broken.
*/
commit(): TurnSnapshot | undefined {
const cur = this.current;
this.current = null;
if (!cur || cur.files.size === 0) return undefined;
const snap: TurnSnapshot = {
turn: cur.turn,
at: cur.at,
prompt: cur.prompt,
files: [...cur.files].map(([path, before]) => ({ path, before })),
messageCount: cur.messageCount,
};
this.turns.push(snap);
while (this.turns.length > MAX_SNAPSHOTS) this.turns.shift();
return snap;
}
/** Discards the open turn without recording it, for an aborted or failed turn. */
discard(): void {
this.current = null;
}
/** The turns that can be undone, newest first. */
list(): readonly TurnSnapshot[] {
return [...this.turns].reverse();
}
/** Whether there is an open turn collecting pre-images right now. */
get open(): boolean {
return this.current !== null;
}
/**
* Removes the newest turn and returns what it holds, without restoring.
*
* Separated from restoring so the caller can decide *what* to bring back —
* files, conversation, or both — which is the split Claude Code's rewind menu
* exposes and the reason one control surface is worth more than three commands.
*/
pop(): TurnSnapshot | undefined {
return this.turns.pop();
}
/** Puts a turn back, for a `/redo` that follows an `/undo`. */
push(snap: TurnSnapshot): void {
this.turns.push(snap);
}
get size(): number {
return this.turns.length;
}
cwdOf(): string {
return this.cwd;
}
}
/** Writes a pre-image back to disk, recreating a deleted file or removing one that was created. */
export async function restore(snap: TurnSnapshot, cwd = process.cwd()): Promise<{ restored: string[]; removed: string[] }> {
const restored: string[] = [];
const removed: string[] = [];
for (const file of snap.files) {
const abs = join(cwd, file.path);
if (file.before === undefined) {
// The file did not exist before the turn, so undoing its creation is removing it.
const f = Bun.file(abs);
if (await f.exists()) {
await f.delete();
removed.push(file.path);
}
continue;
}
await Bun.write(abs, file.before);
restored.push(file.path);
}
return { restored, removed };
}
/** A short, stable label for a snapshot, for a menu that lists several. */
export function labelOf(snap: TurnSnapshot): string {
const first = snap.prompt.trim().split('\n')[0] ?? '';
const clipped = first.length > 50 ? `${first.slice(0, 50)}...` : first || '(no prompt)';
return `turn ${snap.turn}: ${clipped}`;
}
/** A content hash, used to tell whether a file still matches what the snapshot holds. */
export function hashOf(text: string): string {
return createHash('sha256').update(text).digest('hex').slice(0, 12);
}
+80
View File
@@ -0,0 +1,80 @@
import { tool } from 'ai';
import { z } from 'zod';
/**
* A visible escape hatch for the loop a coding agent dies in.
*
* The repeat guard stops an *identical* call after three tries, but the deeper loop
* is the model making *different* calls that all amount to the same stalled attempt —
* re-reading the same file expecting a different answer, retrying a failing command
* with a tweaked flag, re-sending a prompt it has already asked. No equal-input
* detector fires on any of that, so the model burns the step budget on motions that
* never move.
*
* `step_back` exists so the model has a *named* way out instead of only a guard it
* cannot see. The session records each completed tool call (input + outcome) into a
* loop trace; the tool returns a scripted reflection prompt built from that trace,
* so the model is told in concrete terms that it is going in circles and is steered
* to change direction.
*/
export type LoopEntry = {
step: number;
toolName: string;
input: string;
result: string;
at: string;
};
function recentTrace(entries: readonly LoopEntry[], window = 8): LoopEntry[] {
return entries.slice(-window);
}
function renderTrace(entries: readonly LoopEntry[], maxLines = 12): string {
return entries.slice(-maxLines).map((e) => `step ${e.step}: ${e.toolName} ${e.input} -> ${e.result}`.slice(0, 200)).join('\n');
}
/**
* Builds a `step_back` tool bound to a session's loop trace.
*
* The tool is meant to be called when the model is not making progress — a trap it
* cannot always see while it is inside it. The returned reflection names the recent
* steps so the model can tell, from its own calls, that it is going in circles, and
* gives it the one thing a stuck agent is usually missing: permission to stop,
* say what it learned, and change direction rather than try harder.
*/
export function createStepBackTool(opts: { trace: () => readonly LoopEntry[] }) {
return tool({
description:
'Use when you are stuck: the same file is not changing, a command keeps failing, or you have done several steps with no visible progress. ' +
'Records your recent steps and returns a reflection prompt to help you change direction instead of repeating the attempt.',
inputSchema: z.object({
note: z.string().optional().describe('A sentence in your own words about what you were trying to do.'),
}),
execute: async ({ note }) => {
const trace = recentTrace(opts.trace());
if (trace.length === 0) {
return (
'No recent steps to reflect on. This is an early call of step_back — it is only useful when you have ' +
'attempted something several times. Describe what you are stuck on in `note`.'
);
}
return [
'You have run threadbare over the last steps and are stuck. Here is what you actually did:',
'```',
renderTrace(trace),
'```',
'',
'Before your next tool call, answer these three questions in your reasoning:',
'1. What exactly is wrong — the input, the tool, or the expectation?',
'2. What have you tried, and why did each fail?',
'3. What is ONE different thing you can do that is not "try the same thing a little harder"?',
'',
'Then take that different action. If the failure is a command, read the actual error and fix its cause — ',
'do not rerun the command. If a file is not what you expect, suspect your assumption about it and re-read it fresh.',
note?.trim() ? `\nYour note: ${note.trim()}` : '',
].join('\n');
},
});
}
+128 -82
View File
@@ -177,91 +177,137 @@ export function createTaskTool(opts: {
? 'explore: read-only research. review: read-only critique. worker: makes changes. Default explore.' ? 'explore: read-only research. review: read-only critique. worker: makes changes. Default explore.'
: 'explore: find and report. review: critique code for defects. Default explore.', : 'explore: find and report. review: critique code for defects. Default explore.',
), ),
tasks: z
.array(
z.object({
description: z.string(),
prompt: z.string(),
kind: z.enum(canWrite ? ['explore', 'review', 'worker'] : ['explore', 'review']).optional(),
}),
)
.optional()
.describe(
'Independent investigations to run at the same time instead of one after another. ' +
'Use this for several unrelated searches so they overlap in wall-clock time. Each runs on its own ' +
'context window, exactly like a single task call. Default: run the single prompt above.',
),
}), }),
execute: async ({ description, prompt, kind }, { abortSignal }) => { execute: async ({ description, prompt, kind, tasks }, { abortSignal }) => {
const flavour: SubagentKind = kind ?? 'explore'; const plans = tasks?.length ? tasks : [{ description, prompt, kind }];
if (flavour === 'worker' && !opts.approve) { const runs = await Promise.all(
throw new Error('The worker kind needs an approval channel, which this session has not provided.'); plans.map((p) => runSubagent({ ...opts, abortSignal }, { description: p.description, prompt: p.prompt, kind: p.kind })),
} );
if (runs.length === 1) return runs[0]!.report;
const id = `sub${++counter}`; return runs.map((r) => `## ${r.named}\n\n${r.report}`).join('\n\n---\n\n');
const report = opts.report;
report?.({ type: 'start', id, kind: flavour, description });
let steps = 0;
let text = '';
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
try {
// `explore` is search, not reasoning, so it runs on the cheaper model when
// one is configured. `review` and `worker` keep the parent's: they judge
// and they change, both of which want the full model.
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
const result = streamText({
model,
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
messages: [{ role: 'user', content: prompt }],
tools: TOOLS[flavour],
stopWhen: isStepCount(opts.maxSteps ?? 20),
...(opts.approve
? {
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const approved = await opts.approve!(toolCall);
return approved
? undefined
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
},
}
: {}),
...(abortSignal ? { abortSignal } : {}),
});
const sink = () => {};
void result.responseMessages.then(undefined, sink);
void result.usage.then(undefined, sink);
void result.steps.then(undefined, sink);
void result.finalStep.then(undefined, sink);
void result.finishReason.then(undefined, sink);
for await (const part of result.stream) {
if (part.type === 'tool-call') {
steps++;
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
} else if (part.type === 'tool-result') {
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
} else if (part.type === 'tool-error') {
const message = part.error instanceof Error ? part.error.message : String(part.error);
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
} else if (part.type === 'text-delta') {
text += part.text;
} else if (part.type === 'error') {
// A provider failure arrives as a stream part, not a throw, so it has to
// be rethrown here or the subagent silently returns nothing.
const message = part.error instanceof Error ? part.error.message : String(part.error);
throw part.error instanceof Error ? part.error : new Error(message);
}
}
try {
const usage = await result.usage;
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
} catch {
// A run that errored before producing usage has nothing to account for.
}
} catch (e) {
const message = e instanceof Error ? e.message : String(e);
report?.({ type: 'error', id, message });
throw e;
}
const trimmed = text.trim();
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
// Settled after the stream closes; a failed run reports nothing rather than
// a half count. The parent prices these against the subagent's own model id.
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
return trimmed || 'Subagent returned no findings.';
}, },
}); });
} }
/** Flavour of one planned investigation; `kind` defaults to explore. */
type Plan = { description: string; prompt: string; kind?: string };
/**
* Runs one subagent and settles its report, usage, and panel events.
*
* Extracted so `tasks` can fan several out in parallel: each full run is
* independent — its own id, own stream, own spend — and they overlap simply by
* awaiting them together.
*/
async function runSubagent(
opts: {
model: LanguageModel;
subagentModel?: LanguageModel;
subagentModelId?: string;
cwd?: string;
maxSteps?: number;
report?: SubagentReporter;
approve?: SubagentApproval;
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
abortSignal?: AbortSignal;
},
plan: Plan,
): Promise<{ named: string; report: string }> {
const flavour: SubagentKind = (plan.kind as SubagentKind | undefined) ?? 'explore';
if (flavour === 'worker' && !opts.approve) {
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
}
const id = `sub${++counter}`;
const report = opts.report;
report?.({ type: 'start', id, kind: flavour, description: plan.description });
let steps = 0;
let text = '';
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
try {
// `explore` is search, not reasoning, so it runs on the cheaper model when
// one is configured. `review` and `worker` keep the parent's: they judge
// and they change, both of which want the full model.
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
const result = streamText({
model,
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
messages: [{ role: 'user', content: plan.prompt }],
tools: TOOLS[flavour],
stopWhen: isStepCount(opts.maxSteps ?? 20),
...(opts.approve
? {
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const approved = await opts.approve!(toolCall);
return approved
? undefined
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
},
}
: {}),
...(opts.abortSignal ? { abortSignal: opts.abortSignal } : {}),
});
const sink = () => {};
void result.responseMessages.then(undefined, sink);
void result.usage.then(undefined, sink);
void result.steps.then(undefined, sink);
void result.finalStep.then(undefined, sink);
void result.finishReason.then(undefined, sink);
for await (const part of result.stream) {
if (part.type === 'tool-call') {
steps++;
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
} else if (part.type === 'tool-result') {
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
} else if (part.type === 'tool-error') {
const message = part.error instanceof Error ? part.error.message : String(part.error);
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
} else if (part.type === 'text-delta') {
text += part.text;
} else if (part.type === 'error') {
// A provider failure arrives as a stream part, not a throw, so it has to
// be rethrown here or the subagent silently returns nothing.
const message = part.error instanceof Error ? part.error.message : String(part.error);
throw part.error instanceof Error ? part.error : new Error(message);
}
}
try {
const usage = await result.usage;
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
} catch {
// A run that errored before producing usage has nothing to account for.
}
} catch (e) {
const message = e instanceof Error ? e.message : String(e);
report?.({ type: 'error', id, message });
throw e;
}
const trimmed = text.trim();
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
// Settled after the stream closes; a failed run reports nothing rather than
// a half count. The parent prices these against the subagent's own model id.
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
return { named: plan.description, report: trimmed || 'Subagent returned no findings.' };
}
export const TASK_TOOL_NAME = 'task'; export const TASK_TOOL_NAME = 'task';
+27
View File
@@ -0,0 +1,27 @@
import type { Tool } from 'ai';
/**
* Marks a tool as mutating where it is defined, rather than in a list beside it.
*
* The bug this prevents is the worst one this codebase can have: a new tool that
* writes to the workspace, added to `tools` but forgotten in a hand-maintained
* `MUTATING_TOOLS`, is a write the permission layer does not treat as a write. It
* is silent, it passes every test that does not think to check the new name, and
* it surfaces as a user discovering an edit they never approved.
*
* The mark is a property on the tool object, so `MUTATING_TOOLS` can be derived by
* filtering the registry instead of being typed out. A tool that is not marked is
* asserted non-mutating by `tools.test.ts`, which means the decision is made once,
* at the definition, and cannot drift.
*/
export const MUTATING = '__mutating' as const;
/** A tool that can change the workspace or run arbitrary code. */
export function mutating<T extends Tool>(t: T): T {
return Object.assign(t, { [MUTATING]: true as const });
}
/** Whether a tool was declared mutating at its definition site. */
export function isMutating(t: unknown): boolean {
return typeof t === 'object' && t !== null && (t as Record<string, unknown>)[MUTATING] === true;
}
+11 -10
View File
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
import { resolve } from 'node:path'; import { resolve } from 'node:path';
import { z } from 'zod'; import { z } from 'zod';
import { jail, posix, walk } from './ignore'; import { jail, posix, walk } from './ignore';
import { mutating } from './tool-kinds';
import { git } from './tools-git'; import { git } from './tools-git';
/** /**
@@ -36,7 +37,7 @@ async function readLines(path: string): Promise<{ abs: string; lines: string[] }
// edit // edit
// --------------------------------------------------------------------------- // ---------------------------------------------------------------------------
export const insertLinesTool = tool({ export const insertLinesTool = mutating(tool({
description: description:
'Insert lines at a 1-based position in a file, pushing the rest down. Cheaper and safer than a rewrite for adding a block in the middle.', 'Insert lines at a 1-based position in a file, pushing the rest down. Cheaper and safer than a rewrite for adding a block in the middle.',
inputSchema: z.object({ inputSchema: z.object({
@@ -51,9 +52,9 @@ export const insertLinesTool = tool({
await Bun.write(abs, cur.join('\n')); await Bun.write(abs, cur.join('\n'));
return `Inserted ${lines(text).length} line(s) at ${path}:${line}`; return `Inserted ${lines(text).length} line(s) at ${path}:${line}`;
}, },
}); }));
export const deleteLinesTool = tool({ export const deleteLinesTool = mutating(tool({
description: 'Delete an inclusive range of lines from a file. Refuses to delete the whole file; use delete_file for that.', description: 'Delete an inclusive range of lines from a file. Refuses to delete the whole file; use delete_file for that.',
inputSchema: z.object({ inputSchema: z.object({
path: z.string(), path: z.string(),
@@ -69,9 +70,9 @@ export const deleteLinesTool = tool({
await Bun.write(abs, cur.join('\n')); await Bun.write(abs, cur.join('\n'));
return `Deleted lines ${start}-${end} from ${path}`; return `Deleted lines ${start}-${end} from ${path}`;
}, },
}); }));
export const replaceLinesTool = tool({ export const replaceLinesTool = mutating(tool({
description: 'Replace an inclusive range of lines with new text, in one write.', description: 'Replace an inclusive range of lines with new text, in one write.',
inputSchema: z.object({ inputSchema: z.object({
path: z.string(), path: z.string(),
@@ -87,9 +88,9 @@ export const replaceLinesTool = tool({
await Bun.write(abs, cur.join('\n')); await Bun.write(abs, cur.join('\n'));
return `Replaced lines ${start}-${end} in ${path}`; return `Replaced lines ${start}-${end} in ${path}`;
}, },
}); }));
export const appendFileTool = tool({ export const appendFileTool = mutating(tool({
description: 'Append text to the end of a file without reading the whole thing into the edit.', description: 'Append text to the end of a file without reading the whole thing into the edit.',
inputSchema: z.object({ path: z.string(), text: z.string() }), inputSchema: z.object({ path: z.string(), text: z.string() }),
execute: async ({ path, text }) => { execute: async ({ path, text }) => {
@@ -97,9 +98,9 @@ export const appendFileTool = tool({
await Bun.write(abs, `${cur.join('\n').replace(/\n?$/, '\n')}${text.replace(/\n?$/, '')}\n`); await Bun.write(abs, `${cur.join('\n').replace(/\n?$/, '\n')}${text.replace(/\n?$/, '')}\n`);
return `Appended ${lines(text).length} line(s) to ${path}`; return `Appended ${lines(text).length} line(s) to ${path}`;
}, },
}); }));
export const prependFileTool = tool({ export const prependFileTool = mutating(tool({
description: 'Prepend text to the start of a file, e.g. a license header or an import block.', description: 'Prepend text to the start of a file, e.g. a license header or an import block.',
inputSchema: z.object({ path: z.string(), text: z.string() }), inputSchema: z.object({ path: z.string(), text: z.string() }),
execute: async ({ path, text }) => { execute: async ({ path, text }) => {
@@ -107,7 +108,7 @@ export const prependFileTool = tool({
await Bun.write(abs, `${text.replace(/\n?$/, '\n')}${cur.join('\n')}`); await Bun.write(abs, `${text.replace(/\n?$/, '\n')}${cur.join('\n')}`);
return `Prepended ${lines(text).length} line(s) to ${path}`; return `Prepended ${lines(text).length} line(s) to ${path}`;
}, },
}); }));
export const countLinesTool = tool({ export const countLinesTool = tool({
description: 'Count lines in one file, or per file across a glob. A quick size read before deciding to open something large.', description: 'Count lines in one file, or per file across a glob. A quick size read before deciding to open something large.',
+115 -46
View File
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
import { join, resolve } from 'node:path'; import { join, resolve } from 'node:path';
import { z } from 'zod'; import { z } from 'zod';
import { jail, posix, walk } from './ignore'; import { jail, posix, walk } from './ignore';
import { isMutating, mutating } from './tool-kinds';
import { EXTRA_TOOL_NAMES, extraTools } from './tools-extra'; import { EXTRA_TOOL_NAMES, extraTools } from './tools-extra';
import { GIT_TOOL_NAMES, gitTools } from './tools-git'; import { GIT_TOOL_NAMES, gitTools } from './tools-git';
import { NET_TOOL_NAMES, netTools } from './tools-net'; import { NET_TOOL_NAMES, netTools } from './tools-net';
@@ -174,7 +175,7 @@ export function parsePatch(patch: string): PatchOp[] {
return ops; return ops;
} }
export const applyPatchTool = tool({ export const applyPatchTool = mutating(tool({
description: description:
'Apply one patch across several files: add, update, move, and delete in a single call. All or nothing — if any ' + 'Apply one patch across several files: add, update, move, and delete in a single call. All or nothing — if any ' +
'part fails, nothing is written. Use it when a change spans files that must land together, such as a rename ' + 'part fails, nothing is written. Use it when a change spans files that must land together, such as a rename ' +
@@ -248,7 +249,7 @@ export const applyPatchTool = tool({
return `Applied ${ops.length} change${ops.length === 1 ? '' : 's'}:\n${summary.map((s) => `- ${s}`).join('\n')}`; return `Applied ${ops.length} change${ops.length === 1 ? '' : 's'}:\n${summary.map((s) => `- ${s}`).join('\n')}`;
}, },
}); }));
/** /**
* A rewrite that collapses whitespace: similar character count, a fraction of the lines. * A rewrite that collapses whitespace: similar character count, a fraction of the lines.
@@ -264,7 +265,7 @@ function collapsedRewrite(before: string, after: string): boolean {
return after.split('\n').length < before.split('\n').length / 2; return after.split('\n').length < before.split('\n').length / 2;
} }
export const writeFileTool = tool({ export const writeFileTool = mutating(tool({
description: 'Create a file or overwrite it completely. Prefer edit_file for existing files.', description: 'Create a file or overwrite it completely. Prefer edit_file for existing files.',
inputSchema: z.object({ inputSchema: z.object({
path: z.string(), path: z.string(),
@@ -284,17 +285,39 @@ export const writeFileTool = tool({
} }
return `Wrote ${content.length} chars to ${path}`; return `Wrote ${content.length} chars to ${path}`;
}, },
}); }));
export const editFileTool = tool({ // Some models (Claude-style tool docs, DeepSeek/GLM) emit snake_case edit params
// (old_string/new_string/replace_all) despite the camelCase schema. Normalize at the
// boundary instead of failing the whole call on a naming convention.
const SNAKE_EDIT_ARGS: ReadonlyArray<readonly [string, string]> = [
['old_string', 'oldString'],
['new_string', 'newString'],
['replace_all', 'replaceAll'],
];
export function normalizeEditArgs(input: unknown): unknown {
if (typeof input !== 'object' || input === null) return input;
const obj: Record<string, unknown> = { ...(input as Record<string, unknown>) };
for (const [snake, camel] of SNAKE_EDIT_ARGS) {
if (obj[snake] !== undefined && obj[camel] === undefined) obj[camel] = obj[snake];
}
if (Array.isArray(obj.edits)) obj.edits = obj.edits.map(normalizeEditArgs);
return obj;
}
export const editFileTool = mutating(tool({
description: description:
'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.', 'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.',
inputSchema: z.object({ inputSchema: z.preprocess(
path: z.string(), normalizeEditArgs,
oldString: z.string().describe('Exact text to find, including whitespace and indentation'), z.object({
newString: z.string().describe('Replacement text'), path: z.string(),
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'), oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
}), newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
}),
),
execute: async ({ path, oldString, newString, replaceAll = false }) => { execute: async ({ path, oldString, newString, replaceAll = false }) => {
if (oldString === newString) throw new Error('oldString and newString are identical'); if (oldString === newString) throw new Error('oldString and newString are identical');
const abs = jail(path); const abs = jail(path);
@@ -312,27 +335,30 @@ export const editFileTool = tool({
await Bun.write(abs, after); await Bun.write(abs, after);
return `Replaced ${replaceAll ? count : 1} occurrence(s) in ${path}`; return `Replaced ${replaceAll ? count : 1} occurrence(s) in ${path}`;
}, },
}); }));
export const multiEditTool = tool({ export const multiEditTool = mutating(tool({
description: description:
'Apply several exact-string edits to one file in a single call. Each edit sees the result of the previous one. ' + 'Apply several exact-string edits to one file in a single call. Each edit sees the result of the previous one. ' +
'All or nothing: if any oldString fails to match, or matches more than once without replaceAll, nothing is ' + 'All or nothing: if any oldString fails to match, or matches more than once without replaceAll, nothing is ' +
'written. Prefer this over repeated edit_file calls on the same file — one approval, one write, no risk of ' + 'written. Prefer this over repeated edit_file calls on the same file — one approval, one write, no risk of ' +
'leaving the file half-changed.', 'leaving the file half-changed.',
inputSchema: z.object({ inputSchema: z.preprocess(
path: z.string(), normalizeEditArgs,
edits: z z.object({
.array( path: z.string(),
z.object({ edits: z
oldString: z.string().describe('Exact text to find, including whitespace and indentation'), .array(
newString: z.string().describe('Replacement text'), z.object({
replaceAll: z.boolean().optional(), oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
}), newString: z.string().describe('Replacement text'),
) replaceAll: z.boolean().optional(),
.min(1) }),
.describe('Edits in the order they should be applied'), )
}), .min(1)
.describe('Edits in the order they should be applied'),
}),
),
execute: async ({ path, edits }) => { execute: async ({ path, edits }) => {
const abs = jail(path); const abs = jail(path);
const file = Bun.file(abs); const file = Bun.file(abs);
@@ -367,7 +393,7 @@ export const multiEditTool = tool({
await Bun.write(abs, text); await Bun.write(abs, text);
return `Applied ${edits.length} edit(s) to ${path} (${applied.join(', ')})`; return `Applied ${edits.length} edit(s) to ${path} (${applied.join(', ')})`;
}, },
}); }));
export const globTool = tool({ export const globTool = tool({
description: description:
@@ -573,7 +599,13 @@ async function pump(
return all; return all;
} }
type Running = { command: string; proc: Bun.Subprocess; interrupted: boolean; killed?: Promise<unknown> }; type Running = {
command: string;
proc: Bun.Subprocess;
interrupted: boolean;
timedOut?: boolean;
killed?: Promise<unknown>;
};
const running = new Map<string, Running>(); const running = new Map<string, Running>();
@@ -615,6 +647,12 @@ function killTree(proc: Bun.Subprocess): Promise<unknown> {
export function interruptBash(): string[] { export function interruptBash(): string[] {
const killed: string[] = []; const killed: string[] = [];
for (const entry of running.values()) { for (const entry of running.values()) {
// A second ctrl-c while the first killTree is still settling must not re-announce
// the same command: the notice is the only proof the keypress did anything.
if (entry.interrupted) {
killed.push(entry.command);
continue;
}
entry.interrupted = true; entry.interrupted = true;
entry.killed = killTree(entry.proc); entry.killed = killTree(entry.proc);
killed.push(entry.command); killed.push(entry.command);
@@ -622,7 +660,7 @@ export function interruptBash(): string[] {
return killed; return killed;
} }
export const bashTool = tool({ export const bashTool = mutating(tool({
description: description:
'Run a shell command in the workspace root. Use for builds, tests, git, and package managers. ' + 'Run a shell command in the workspace root. Use for builds, tests, git, and package managers. ' +
'Output streams live and the user can interrupt a command with ctrl-c without ending the turn.', 'Output streams live and the user can interrupt a command with ctrl-c without ending the turn.',
@@ -636,13 +674,32 @@ export const bashTool = tool({
cwd: process.cwd(), cwd: process.cwd(),
stdout: 'pipe', stdout: 'pipe',
stderr: 'pipe', stderr: 'pipe',
timeout,
...(abortSignal ? { signal: abortSignal } : {}),
}); });
const entry: Running = { command, proc, interrupted: false }; const entry: Running = { command, proc, interrupted: false };
running.set(toolCallId, entry); running.set(toolCallId, entry);
// Bun's spawn `signal` option is not used either: it kills only the shell, so an
// esc-abort orphaned the grandchild on the same still-open pipes as the timeout
// did. The abort must go through killTree, exactly like ctrl-c does.
const onAbort = () => {
if (entry.interrupted) return;
entry.interrupted = true;
entry.killed = killTree(proc);
};
abortSignal?.addEventListener('abort', onAbort);
// The turn may already be aborted by the time this tool starts; a past event
// never re-fires, so check once here or the command runs unkillable by esc.
if (abortSignal?.aborted) onAbort();
// Bun's own `timeout` spawn option is not used: it kills only the shell, and the
// grandchild holding the output pipes keeps `pump` reading forever, so the tool
// never returns. Same failure killTree exists for, just triggered by the clock.
const timer = setTimeout(() => {
entry.timedOut = true;
entry.killed = killTree(proc);
}, timeout);
try { try {
// Drained concurrently: a command that fills one pipe while we block on the // Drained concurrently: a command that fills one pipe while we block on the
// other would deadlock, and buffering both hides progress for minutes. // other would deadlock, and buffering both hides progress for minutes.
@@ -658,6 +715,15 @@ export const bashTool = tool({
// Thrown rather than returned: the model must not read a killed command as // Thrown rather than returned: the model must not read a killed command as
// a command that ran and failed on its own terms. // a command that ran and failed on its own terms.
if (entry.timedOut) {
throw new Error(
cap(
`The command exceeded its ${timeout}ms timeout and was killed. It did not finish, so its effects are unknown.\n${
body || '(no output before it was killed)'
}`,
),
);
}
if (entry.interrupted) { if (entry.interrupted) {
throw new Error( throw new Error(
cap( cap(
@@ -678,15 +744,17 @@ export const bashTool = tool({
.join('\n\n'), .join('\n\n'),
); );
} finally { } finally {
clearTimeout(timer);
abortSignal?.removeEventListener('abort', onAbort);
// Awaited so the process really is gone before the tool returns. On Windows a // Awaited so the process really is gone before the tool returns. On Windows a
// surviving grandchild holds the cwd open, which breaks the very next command. // surviving grandchild holds the cwd open, which breaks the very next command.
await entry.killed; await entry.killed;
running.delete(toolCallId); running.delete(toolCallId);
} }
}, },
}); }));
export const moveFileTool = tool({ export const moveFileTool = mutating(tool({
description: description:
'Move or rename one file. Creates the target directory. Refuses if the source is missing or the target ' + 'Move or rename one file. Creates the target directory. Refuses if the source is missing or the target ' +
'already exists, so a rename cannot silently overwrite work. For a rename plus its callers in one step, ' + 'already exists, so a rename cannot silently overwrite work. For a rename plus its callers in one step, ' +
@@ -708,9 +776,9 @@ export const moveFileTool = tool({
await file.delete(); await file.delete();
return `Moved ${from} to ${to}`; return `Moved ${from} to ${to}`;
}, },
}); }));
export const deleteFileTool = tool({ export const deleteFileTool = mutating(tool({
description: description:
'Delete one file. Refuses a directory: removing a tree is what the guard plugin blocks in bash, and it is ' + 'Delete one file. Refuses a directory: removing a tree is what the guard plugin blocks in bash, and it is ' +
'not something to do implicitly. Delete the files you mean, one call each.', 'not something to do implicitly. Delete the files you mean, one call each.',
@@ -733,7 +801,7 @@ export const deleteFileTool = tool({
await Bun.file(abs).delete(); await Bun.file(abs).delete();
return `Deleted ${path} (${entry.size} bytes)`; return `Deleted ${path} (${entry.size} bytes)`;
}, },
}); }));
/** /**
* Definition patterns for `find_symbol`, keyed loosely by language. * Definition patterns for `find_symbol`, keyed loosely by language.
@@ -909,15 +977,16 @@ export function disabledToolNames(enabled: readonly ToolSetName[] | undefined):
return TOOL_SET_NAMES.filter((set) => !live.has(set)).flatMap((set) => [...TOOL_SETS[set]]); return TOOL_SET_NAMES.filter((set) => !live.has(set)).flatMap((set) => [...TOOL_SETS[set]]);
} }
/** Tools that mutate the workspace or run arbitrary code always ask the user first. */ /**
export const MUTATING_TOOLS = [ * Tools that mutate the workspace or run arbitrary code always ask the user first.
'write_file', *
'edit_file', * Derived from the tools themselves rather than typed out: a tool marked `mutating`
'multi_edit', * at its definition is in this list by construction, and there is no second place to
'apply_patch', * forget it. `tools.test.ts` asserts the converse — that nothing here is unmarked —
'move_file', * so the two cannot disagree.
'delete_file', */
'bash', export const MUTATING_TOOLS: readonly string[] = Object.entries(tools)
] as const; .filter(([, t]) => isMutating(t))
.map(([name]) => name);
export { jail }; export { jail };
+69 -11
View File
@@ -176,18 +176,34 @@ export function App({
const fileMatches = token && paths ? matchPaths(paths, token.query) : []; const fileMatches = token && paths ? matchPaths(paths, token.query) : [];
const highlightedPath = fileMatches[Math.min(fileIndex, Math.max(0, fileMatches.length - 1))]; const highlightedPath = fileMatches[Math.min(fileIndex, Math.max(0, fileMatches.length - 1))];
// The walk costs a full ignore-aware traversal, so it happens on the first `@` // The walk costs a full ignore-aware traversal, so it runs once at the first `@`.
// rather than at startup, and only once. // A slow cooldown re-walks so a file created after that first `@` shows up within
// a short window instead of staying hidden all session. The cooldown never fires
// on a short session, so the "walks once" behaviour most users see is unchanged.
const pathsRef = useRef(paths);
useEffect(() => { useEffect(() => {
if (token === undefined || paths !== undefined) return; pathsRef.current = paths;
let live = true; }, [paths]);
void hooks.listPaths().then((all) => { const didLoadRef = useRef(false);
if (live) setPaths(all); useEffect(() => {
}); if (token === undefined) return;
return () => { if (!didLoadRef.current) {
live = false; didLoadRef.current = true;
}; let live = true;
}, [hooks, paths, token]); void hooks.listPaths().then((all) => {
if (live) setPaths(all);
});
const refresh = setInterval(async () => {
if (!live) return;
const all = await hooks.listPaths();
if (live && JSON.stringify(all) !== JSON.stringify(pathsRef.current)) setPaths(all);
}, 10_000);
return () => {
live = false;
clearInterval(refresh);
};
}
}, [hooks, token]);
useEffect(() => bridge.bind(setPending), [bridge]); useEffect(() => bridge.bind(setPending), [bridge]);
useEffect(() => askBridge?.bind(setAsking), [askBridge]); useEffect(() => askBridge?.bind(setAsking), [askBridge]);
@@ -691,6 +707,48 @@ export function App({
setModelPicker(models); setModelPicker(models);
return; return;
} }
case 'undo': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
const result = await session.undo(action.what);
if (!result) {
push({ kind: 'info', text: 'nothing to undo' });
} else {
const parts: string[] = [];
if (result.restored.length > 0) parts.push(`restored ${result.restored.join(', ')}`);
if (result.removed.length > 0) parts.push(`removed ${result.removed.join(', ')}`);
if (result.conversationTrimmed) parts.push(`dropped ${result.snapshot.messageCount}-onward from the history`);
push({ kind: 'info', text: `undid turn ${result.snapshot.turn}: ${parts.join('; ') || 'no file changes'}` });
// The same limit the turn-end notice states: bash is not snapshotted.
push({
kind: 'info',
text: 'Only file-tool edits are covered. A bash command or a git checkout in that turn is not, so check git status if one ran.',
});
}
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
case 'redo': {
push({ kind: 'user', text: chosen.trim() });
const result = session.redo(action.what);
if (!result) {
push({ kind: 'info', text: 'nothing to redo' });
} else if (action.what === 'files' || action.what === 'both') {
// Refused rather than approximated: only the pre-image was captured, so
// there is no post-turn content to put back.
push({
kind: 'info',
text: `put turn ${result.snapshot.turn} back on the undo stack. File contents cannot be re-applied - only the state before the turn was recorded.`,
});
} else {
push({ kind: 'info', text: `put turn ${result.snapshot.turn} back; the conversation was restored to it` });
}
return;
}
case 'compact': { case 'compact': {
push({ kind: 'user', text: chosen.trim() }); push({ kind: 'user', text: chosen.trim() });
setWorking(true); setWorking(true);
+2 -1
View File
@@ -1,6 +1,7 @@
import { Box, Text, useInput } from 'ink'; import { Box, Text, useInput } from 'ink';
import React from 'react'; import React from 'react';
import type { ApprovalDecision, ApprovalRequest } from '../session'; import type { ApprovalDecision, ApprovalRequest } from '../session';
import { normalizeEditArgs } from '../tools';
import { Diff } from './Diff'; import { Diff } from './Diff';
import { accent, glyph } from './theme'; import { accent, glyph } from './theme';
import { toolDetail } from './transcript'; import { toolDetail } from './transcript';
@@ -42,7 +43,7 @@ export function createApprovalBridge(): ApprovalBridge {
* consistent and far more readable than a JSON dump of the input. * consistent and far more readable than a JSON dump of the input.
*/ */
function ApprovalDetail({ name, input }: { name: string; input: unknown }) { function ApprovalDetail({ name, input }: { name: string; input: unknown }) {
const o = (input ?? {}) as Record<string, unknown>; const o = (normalizeEditArgs(input) ?? {}) as Record<string, unknown>;
if (name === 'write_file') { if (name === 'write_file') {
const content = String(o['content'] ?? ''); const content = String(o['content'] ?? '');
+3 -7
View File
@@ -33,10 +33,6 @@ type KeyLike = {
end?: boolean; end?: boolean;
}; };
const INVERSE_ON = '\u001B[7m';
const INVERSE_OFF = '\u001B[27m';
const invert = (s: string) => `${INVERSE_ON}${s}${INVERSE_OFF}`;
/** /**
* Text input with a real cursor and shell-style history recall. * Text input with a real cursor and shell-style history recall.
* *
@@ -151,10 +147,10 @@ export function PromptInput({
); );
if (value.length === 0) { if (value.length === 0) {
if (!placeholder) return <Text>{focus ? invert(' ') : ' '}</Text>; if (!placeholder) return <Text inverse={focus}>{' '}</Text>;
return ( return (
<Text dimColor> <Text dimColor>
{focus ? invert(placeholder.slice(0, 1)) : placeholder.slice(0, 1)} <Text inverse={focus}>{placeholder.slice(0, 1)}</Text>
{placeholder.slice(1)} {placeholder.slice(1)}
</Text> </Text>
); );
@@ -166,7 +162,7 @@ export function PromptInput({
return ( return (
<Text> <Text>
{shown.slice(0, cursor)} {shown.slice(0, cursor)}
{invert(shown.slice(cursor, cursor + 1) || ' ')} <Text inverse>{shown.slice(cursor, cursor + 1) || ' '}</Text>
{shown.slice(cursor + 1)} {shown.slice(cursor + 1)}
</Text> </Text>
); );
+3 -1
View File
@@ -1,4 +1,5 @@
import { TODO_MARK } from '../notebook'; import { TODO_MARK } from '../notebook';
import { normalizeEditArgs } from '../tools';
export type Line = export type Line =
| { key: string; kind: 'user'; text: string } | { key: string; kind: 'user'; text: string }
@@ -38,7 +39,8 @@ export function preview(input: unknown): string {
* the transcript, beside the spinner while a call is in flight, and in the approval * the transcript, beside the spinner while a call is in flight, and in the approval
* prompt for any tool without a diff of its own. * prompt for any tool without a diff of its own.
*/ */
export function toolDetail(name: string, input: unknown): string[] { export function toolDetail(name: string, rawInput: unknown): string[] {
const input = normalizeEditArgs(rawInput);
if (input === null || typeof input !== 'object') return []; if (input === null || typeof input !== 'object') return [];
const o = input as Record<string, unknown>; const o = input as Record<string, unknown>;
const str = (k: string) => (typeof o[k] === 'string' ? (o[k] as string) : undefined); const str = (k: string) => (typeof o[k] === 'string' ? (o[k] as string) : undefined);
+156
View File
@@ -0,0 +1,156 @@
import { afterAll, beforeAll, expect, test } from 'bun:test';
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { VERSION } from '../src/version';
/**
* The CLI entry point is an executable, not a module: importing it parses argv,
* loads config, connects MCP, and renders Ink. Nothing in `test/` imports it for
* that reason, and its flag surface was previously exercised only by hand.
*
* So it is tested the way a user meets it — by running the real entry point in a
* child process — with two things held constant. `SHIRO_HOME` points at a scratch
* directory, and every API-key variable is stripped, so the "no key" branches are
* deterministic rather than accidentally satisfied by the developer's shell. The
* entry is invoked by absolute path from a scratch cwd, so the repository's own
* `config.json`, `AGENTS.md`, and `.shiro/` entries stay out of the run.
*
* Only paths that exit before `render()` are covered; anything past it needs a
* TTY. That still reaches every argument-parsing decision, which is the point.
*/
const ROOT = join(import.meta.dir, '..');
const ENTRY = join(ROOT, 'src', 'cli.tsx');
/** Credential variables that would otherwise decide `cfg.apiKey` for us. */
const KEY_VARS = [
'SHIRO_API_KEY',
'ANTHROPIC_API_KEY',
'OPENAI_API_KEY',
'SHIRO_PROVIDER',
'SHIRO_MODEL',
'SHIRO_BASE_URL',
];
let home: string;
let scratch: string;
beforeAll(() => {
home = mkdtempSync(join(tmpdir(), 'shiro-cli-home-'));
scratch = mkdtempSync(join(tmpdir(), 'shiro-cli-cwd-'));
});
afterAll(() => {
rmSync(home, { recursive: true, force: true });
rmSync(scratch, { recursive: true, force: true });
});
function cleanEnv(shiroHome: string): Record<string, string> {
// Bun.spawn's `env` replaces the environment, so start from a copy of ours and
// remove the credentials rather than hand-assembling a minimal one.
const env: Record<string, string> = {};
for (const [k, v] of Object.entries(process.env)) if (v !== undefined) env[k] = v;
for (const k of KEY_VARS) delete env[k];
env['SHIRO_HOME'] = shiroHome;
return env;
}
async function cli(args: string[], shiroHome = home): Promise<{ stdout: string; stderr: string; code: number }> {
const proc = Bun.spawn([process.execPath, ENTRY, ...args], {
cwd: scratch,
env: cleanEnv(shiroHome),
// /dev/null, so `readStdin` sees EOF instead of blocking on a pipe nobody closes.
stdin: 'ignore',
stdout: 'pipe',
stderr: 'pipe',
});
const [stdout, stderr, code] = await Promise.all([
new Response(proc.stdout).text(),
new Response(proc.stderr).text(),
proc.exited,
]);
return { stdout, stderr, code };
}
test('--help prints usage and exits 0', async () => {
const { stdout, code } = await cli(['--help']);
expect(code).toBe(0);
expect(stdout).toContain('usage: shiro');
expect(stdout).toContain('--yolo');
expect(stdout).toContain('--print');
expect(stdout).toContain('--agent');
});
test('-h is the same as --help', async () => {
const { stdout, code } = await cli(['-h']);
expect(code).toBe(0);
expect(stdout).toContain('usage: shiro');
});
test('--version prints the version line and exits 0', async () => {
const { stdout, code } = await cli(['--version']);
expect(code).toBe(0);
expect(stdout).toContain(`shiro-neko ${VERSION}`);
expect(stdout).toContain(process.platform);
});
test('-v is the same as --version', async () => {
const { stdout, code } = await cli(['-v']);
expect(code).toBe(0);
expect(stdout).toContain(`shiro-neko ${VERSION}`);
});
test('an unknown --agent fails with the valid list', async () => {
const { stderr, code } = await cli(['--agent', 'turbo']);
expect(code).toBe(1);
expect(stderr).toContain('Unknown agent "turbo"');
expect(stderr).toContain('default, quick, deep, plan, review');
});
test('an unknown --think fails with the valid list', async () => {
const { stderr, code } = await cli(['--think', 'turbo']);
expect(code).toBe(1);
expect(stderr).toContain('Unknown thinking level "turbo"');
expect(stderr).toContain('off, low, medium, high, max');
});
test('-p without a key refuses to run headless', async () => {
const { stderr, code } = await cli(['-p', 'hello']);
expect(code).toBe(1);
expect(stderr).toContain('No API key for provider "anthropic"');
});
test('--resume with no matching session fails', async () => {
const { stderr, code } = await cli(['-r', 'nosuchid']);
expect(code).toBe(1);
expect(stderr).toContain('no session matching "nosuchid"');
});
test('--continue with no saved session fails', async () => {
const { stderr, code } = await cli(['-c']);
expect(code).toBe(1);
expect(stderr).toContain('no saved session for this directory');
});
test('a configured key with no prompt and no stdin reports the missing prompt', async () => {
// A second home, so the key file this writes does not leak into the tests above.
const keyed = mkdtempSync(join(tmpdir(), 'shiro-cli-keyed-'));
try {
mkdirSync(join(keyed, '.shiro-neko'), { recursive: true });
await Bun.write(
join(keyed, '.shiro-neko', 'config.json'),
`${JSON.stringify({ provider: 'anthropic', model: 'claude-sonnet-4-5', apiKey: 'sk-test' }, null, 2)}\n`,
);
// The full isolation flag set is passed so this also proves they all parse and
// the module boots with them; with an empty stdin, `-p` must report the prompt.
const { stderr, code } = await cli(
['-p', '--no-plugins', '--no-skills', '--no-memory', '--no-instructions', '--no-mcp', '--no-subagent'],
keyed,
);
expect(code).toBe(1);
expect(stderr).toContain('needs a prompt');
} finally {
rmSync(keyed, { recursive: true, force: true });
}
});
+6 -20
View File
@@ -1,7 +1,7 @@
import { usageOf } from './helpers'; import { generateResult, streamOf, textChunks, usageOf } from './helpers';
import { afterEach, beforeEach, expect, test } from 'bun:test'; import { afterEach, beforeEach, expect, test } from 'bun:test';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider'; import type { LanguageModelV4CallOptions } from '@ai-sdk/provider';
import { mkdtempSync, rmSync } from 'node:fs'; import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os'; import { tmpdir } from 'node:os';
import { join } from 'node:path'; import { join } from 'node:path';
@@ -15,16 +15,7 @@ import { GIT_TOOL_NAMES } from '../src/tools-git';
const usage = usageOf(10); const usage = usageOf(10);
const stream = (parts: LanguageModelV4StreamPart[]) => ({ const text = (body: string) => textChunks(body, usage);
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
});
const text = (body: string): LanguageModelV4StreamPart[] => [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
let dir: string; let dir: string;
let origCwd: string; let origCwd: string;
@@ -69,16 +60,11 @@ function recordingModel(
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async (o) => { doStream: async (o) => {
seen.push(o); seen.push(o);
return stream(text(reply)); return streamOf(text(reply));
}, },
doGenerate: async (o) => { doGenerate: async (o) => {
seen.push(o); seen.push(o);
return { return generateResult(reply, usage);
content: [{ type: 'text', text: reply }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
} as any;
}, },
}); });
return { model, seen }; return { model, seen };
+101 -20
View File
@@ -1,25 +1,19 @@
import { usageOf } from './helpers'; import { generateResult, streamOf, textChunks, usageOf } from './helpers';
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider'; import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { APICallError, type ModelMessage } from 'ai'; import { APICallError, type ModelMessage } from 'ai';
import { mkdtempSync, rmSync } from 'node:fs'; import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os'; import { tmpdir } from 'node:os';
import { join } from 'node:path'; import { join } from 'node:path';
import { Session, type AgentEvent } from '../src/session'; import { Session, type AgentEvent } from '../src/session';
import { isPrunedSpanSummary, PRUNED_SPAN_PREFIX } from '../src/prune';
const usage = usageOf(10); const usage = usageOf(10);
const stream = (parts: LanguageModelV4StreamPart[]) => ({ const stream = (parts: LanguageModelV4StreamPart[]) => streamOf(parts);
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
});
const text = (body: string): LanguageModelV4StreamPart[] => [ const text = (body: string) => textChunks(body, usage);
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
/** One assistant turn carrying a bulky tool call plus its result. */ /** One assistant turn carrying a bulky tool call plus its result. */
function bulkyExchange(i: number): ModelMessage[] { function bulkyExchange(i: number): ModelMessage[] {
@@ -86,8 +80,10 @@ test('history over the threshold is pruned before reaching the model', async ()
const sentSize = JSON.stringify(seen[0]?.prompt).length; const sentSize = JSON.stringify(seen[0]?.prompt).length;
expect(sentSize).toBeLessThan(JSON.stringify(messages).length); expect(sentSize).toBeLessThan(JSON.stringify(messages).length);
// Pruning is for the wire only; the local history keeps every message. // The fold is written back: the canonical history drops from ~4 bulky exchanges to
expect(session.messages.length).toBeGreaterThan(messages.length); // the widest ladder rung that fits (or the narrowest, if none does). A history left
// at its full size pins the context meter and re-prunes from scratch every turn.
expect(session.estimatedTokens()).toBeLessThan(beforeTokens);
expect(events).toContain('compacted'); expect(events).toContain('compacted');
}); });
@@ -132,12 +128,7 @@ test('summarize replaces the whole history with one summary message', async () =
model: new MockLanguageModelV4({ model: new MockLanguageModelV4({
doGenerate: async () => { doGenerate: async () => {
generateCalls++; generateCalls++;
return { return generateResult('- goal: add pagination\n- touched: src/users.ts\n- todo: add tests', usage);
content: [{ type: 'text', text: '- goal: add pagination\n- touched: src/users.ts\n- todo: add tests' }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
} as any;
}, },
}), }),
askApproval: async () => 'deny', askApproval: async () => 'deny',
@@ -156,7 +147,7 @@ test('summarize replaces the whole history with one summary message', async () =
test('summarize on an empty session is a no-op', async () => { test('summarize on an empty session is a no-op', async () => {
const session = new Session({ const session = new Session({
model: new MockLanguageModelV4({ doGenerate: async () => ({}) as any }), model: new MockLanguageModelV4({ doGenerate: async () => generateResult('') }),
askApproval: async () => 'deny', askApproval: async () => 'deny',
}); });
expect(await session.summarize()).toEqual({ before: 0, after: 0 }); expect(await session.summarize()).toEqual({ before: 0, after: 0 });
@@ -451,3 +442,93 @@ test('a stale item arriving after text was streamed is reported, not silently re
expect(call).toBe(1); expect(call).toBe(1);
}); });
test('a pruned span is replaced by a summary of what was dropped', async () => {
const messages = [
{ role: 'user' as const, content: 'we decided on the ladder approach' },
...bulkyExchange(0),
...bulkyExchange(1),
...bulkyExchange(2),
...bulkyExchange(3),
];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => generateResult('chose the ladder; touched f0-f3.ts'),
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
const span = session.messages.filter((m) => isPrunedSpanSummary(m));
expect(span).toHaveLength(1);
expect(String(span[0]?.content)).toContain('chose the ladder');
});
test('the span summary is injected before the surviving history, not after', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => generateResult('summary of the dropped span'),
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
const markerAt = session.messages.findIndex((m) => isPrunedSpanSummary(m));
expect(markerAt).toBe(0);
});
test('a summarizer that throws still leaves a digest rather than nothing', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => {
throw new Error('summarizer is down');
},
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
const span = session.messages.find((m) => isPrunedSpanSummary(m));
expect(span).toBeDefined();
expect(String(span?.content)).toContain('f0.ts');
});
test('repeated compaction does not nest one span summary inside another', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => generateResult('span notes'),
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
for await (const _ of session.send('again')) void _;
for await (const _ of session.send('and again')) void _;
const spans = session.messages.filter((m) => isPrunedSpanSummary(m));
// One summary is kept from each compaction, but no summary may contain the
// marker text, which would mean a summary of a summary.
for (const s of spans) {
const body = String(s.content).slice(PRUNED_SPAN_PREFIX.length);
expect(body).not.toContain(PRUNED_SPAN_PREFIX);
}
});
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library'; import { render } from 'ink-testing-library';
import React from 'react'; import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session'; import { Session } from '../src/session';
import { App, createApprovalBridge } from '../src/ui/App'; import { App, createApprovalBridge } from '../src/ui/App';
import { testHooks, usageOf } from './helpers'; import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3); const usage = usageOf(3);
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async () => doStream: async () => streamOf(textChunks('reply', usage)),
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
}); });
const paths = ['README.md', 'src/app.ts', 'src/session.ts', 'src/ui/App.tsx', 'test/session.test.ts']; const paths = ['README.md', 'src/app.ts', 'src/session.ts', 'src/ui/App.tsx', 'test/session.test.ts'];
+7 -12
View File
@@ -1,7 +1,7 @@
import { usageOf } from './helpers'; import { generateResult, textChunks, usageOf } from './helpers';
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { APICallError } from 'ai'; import { APICallError } from 'ai';
import type { LanguageModelV4, LanguageModelV4StreamPart } from '@ai-sdk/provider'; import type { LanguageModelV4, LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { simulateReadableStream } from 'ai/test'; import { simulateReadableStream } from 'ai/test';
import { withFallback, type FallbackEvent } from '../src/fallback'; import { withFallback, type FallbackEvent } from '../src/fallback';
@@ -9,12 +9,7 @@ const usage = usageOf(1);
const okStream = (body: string) => ({ const okStream = (body: string) => ({
stream: simulateReadableStream<LanguageModelV4StreamPart>({ stream: simulateReadableStream<LanguageModelV4StreamPart>({
chunks: [ chunks: textChunks(body, usage),
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null, chunkDelayInMs: null,
initialDelayInMs: null, initialDelayInMs: null,
}), }),
@@ -34,7 +29,7 @@ function model(name: string, behaviour: () => Promise<any>): LanguageModelV4 {
}; };
} }
const opts = { prompt: [] } as any; const opts: LanguageModelV4CallOptions = { prompt: [] };
const REAL_MESSAGE = const REAL_MESSAGE =
"Function tools with reasoning_effort are not supported for gpt-5.6-sol in /v1/chat/completions. To use function tools, use /v1/responses or set reasoning_effort to 'none'."; "Function tools with reasoning_effort are not supported for gpt-5.6-sol in /v1/chat/completions. To use function tools, use /v1/responses or set reasoning_effort to 'none'.";
@@ -93,11 +88,11 @@ test('doGenerate falls back on the same condition as doStream', async () => {
model('chat', async () => { model('chat', async () => {
throw apiError(400, REAL_MESSAGE); throw apiError(400, REAL_MESSAGE);
}), }),
model('responses', async () => ({ content: [{ type: 'text', text: 'ok' }] })), model('responses', async () => generateResult('ok', usage)),
]); ]);
const out = (await wrapped.doGenerate(opts)) as any; const out = await wrapped.doGenerate(opts);
expect(out.content[0].text).toBe('ok'); expect(out.content[0]).toMatchObject({ type: 'text', text: 'ok' });
}); });
test('a 401 is not a shape mismatch, so it propagates untouched', async () => { test('a 401 is not a shape mismatch, so it propagates untouched', async () => {
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library'; import { render } from 'ink-testing-library';
import React from 'react'; import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session'; import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App'; import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { testHooks, usageOf } from './helpers'; import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3); const usage = usageOf(3);
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async () => doStream: async () => streamOf(textChunks('reply', usage)),
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
}); });
function mount(over: Partial<AppHooks> = {}) { function mount(over: Partial<AppHooks> = {}) {
+47 -1
View File
@@ -1,4 +1,10 @@
import type { LanguageModelV4Usage } from '@ai-sdk/provider'; import { simulateReadableStream } from 'ai/test';
import type {
LanguageModelV4GenerateResult,
LanguageModelV4StreamPart,
LanguageModelV4StreamResult,
LanguageModelV4Usage,
} from '@ai-sdk/provider';
import type { AppHooks } from '../src/ui/App'; import type { AppHooks } from '../src/ui/App';
/** /**
@@ -15,6 +21,46 @@ export function usageOf(input: number, output = 1): LanguageModelV4Usage {
}; };
} }
/**
* A `doStream` return value built from `simulateReadableStream` chunks.
*
* Written out as `{ stream: simulateReadableStream(...) } as any` in most UI test
* files, because the SDK's `ReadableStream` and `simulateReadableStream`'s type do
* not unify. This is the one place that difference is absorbed.
*/
export function streamOf(chunks: LanguageModelV4StreamPart[]): LanguageModelV4StreamResult {
return {
stream: simulateReadableStream({
chunks,
chunkDelayInMs: null,
initialDelayInMs: null,
}),
} as unknown as LanguageModelV4StreamResult;
}
/** The text and finish chunks that make a stream say one thing and stop. */
export function textChunks(body: string, usage: LanguageModelV4Usage = usageOf(1)): LanguageModelV4StreamPart[] {
return [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
}
/**
* A `doGenerate` return value. `warnings` is required by the SDK type but empty in
* every test, so it is filled here rather than repeated at each call site.
*/
export function generateResult(body: string, usage: LanguageModelV4Usage = usageOf(1)): LanguageModelV4GenerateResult {
return {
content: [{ type: 'text', text: body }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
};
}
/** Default AppHooks for UI tests; override only what a test cares about. */ /** Default AppHooks for UI tests; override only what a test cares about. */
export function testHooks(over: Partial<AppHooks> = {}): AppHooks { export function testHooks(over: Partial<AppHooks> = {}): AppHooks {
return { return {
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library'; import { render } from 'ink-testing-library';
import React from 'react'; import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session'; import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App'; import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { testHooks, usageOf } from './helpers'; import { testHooks, textChunks, usageOf, streamOf } from './helpers';
const usage = usageOf(1000, 500); const usage = usageOf(1000, 500);
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async () => doStream: async () => streamOf(textChunks('reply', usage)),
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
}); });
function mount(over: Partial<AppHooks> = {}) { function mount(over: Partial<AppHooks> = {}) {
+3 -15
View File
@@ -1,29 +1,17 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library'; import { render } from 'ink-testing-library';
import React from 'react'; import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { parseCommand, COMMANDS, HELP } from '../src/commands'; import { parseCommand, COMMANDS, HELP } from '../src/commands';
import { Session } from '../src/session'; import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App'; import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { invalidName, parseHeaders, splitArgs, McpAdd } from '../src/ui/McpAdd'; import { invalidName, parseHeaders, splitArgs, McpAdd } from '../src/ui/McpAdd';
import { testHooks, usageOf } from './helpers'; import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3); const usage = usageOf(3);
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async () => doStream: async () => streamOf(textChunks('ok', usage)),
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'ok' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
}); });
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)); const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
+69 -14
View File
@@ -14,18 +14,19 @@ const call = async (tools: ToolSet, name: string, input: Record<string, unknown>
return tool.execute(input as never, { toolCallId: 'x', messages: [] } as never); return tool.execute(input as never, { toolCallId: 'x', messages: [] } as never);
}; };
test('a stdio server contributes its tools under an mcp__ namespace', async () => { test('a stdio server contributes its tools under an mcp__ namespace (eager)', async () => {
const mcp = await connectMcp({ stub: stdioServer() }); const mcp = await connectMcp({ stub: stdioServer() }, 'eager');
try { try {
expect(Object.keys(mcp.tools).sort()).toEqual(['mcp__stub__ping', 'mcp__stub__search']); expect(Object.keys(mcp.tools).sort()).toEqual(['mcp__stub__ping', 'mcp__stub__search']);
expect(mcp.errors).toEqual([]); expect(mcp.errors).toEqual([]);
expect(mcp.servers).toEqual([]);
} finally { } finally {
await mcp.close(); await mcp.close();
} }
}, 30_000); }, 30_000);
test('an mcp tool actually executes against the server', async () => { test('an mcp tool actually executes against the server (eager)', async () => {
const mcp = await connectMcp({ stub: stdioServer() }); const mcp = await connectMcp({ stub: stdioServer() }, 'eager');
try { try {
const out = await call(mcp.tools, 'mcp__stub__ping', { note: 'hello' }); const out = await call(mcp.tools, 'mcp__stub__ping', { note: 'hello' });
expect(JSON.stringify(out)).toContain('pong: hello'); expect(JSON.stringify(out)).toContain('pong: hello');
@@ -34,8 +35,8 @@ test('an mcp tool actually executes against the server', async () => {
} }
}, 30_000); }, 30_000);
test('two servers exposing the same tool name do not shadow each other', async () => { test('two servers exposing the same tool name do not shadow each other (eager)', async () => {
const mcp = await connectMcp({ a: stdioServer(), b: stdioServer() }); const mcp = await connectMcp({ a: stdioServer(), b: stdioServer() }, 'eager');
try { try {
expect(Object.keys(mcp.tools).sort()).toEqual([ expect(Object.keys(mcp.tools).sort()).toEqual([
'mcp__a__ping', 'mcp__a__ping',
@@ -48,11 +49,14 @@ test('two servers exposing the same tool name do not shadow each other', async (
} }
}, 30_000); }, 30_000);
test('a server that fails to start is reported, not fatal', async () => { test('a server that fails to start is reported, not fatal (eager)', async () => {
const mcp = await connectMcp({ const mcp = await connectMcp(
ok: stdioServer(), {
broken: { command: 'definitely-not-a-real-binary-xyz' }, ok: stdioServer(),
}); broken: { command: 'definitely-not-a-real-binary-xyz' },
},
'eager',
);
try { try {
expect(Object.keys(mcp.tools)).toEqual(['mcp__ok__ping', 'mcp__ok__search']); expect(Object.keys(mcp.tools)).toEqual(['mcp__ok__ping', 'mcp__ok__search']);
expect(mcp.errors.map((e) => e.server)).toEqual(['broken']); expect(mcp.errors.map((e) => e.server)).toEqual(['broken']);
@@ -62,9 +66,10 @@ test('a server that fails to start is reported, not fatal', async () => {
} }
}, 30_000); }, 30_000);
test('no configured servers yields no tools and no errors', async () => { test('no configured servers leaves the meta-tools naming none, with no errors', async () => {
const mcp = await connectMcp({}); const mcp = await connectMcp({});
expect(mcp.tools).toEqual({}); expect(Object.keys(mcp.tools).sort()).toEqual(['mcp_call', 'mcp_inspect', 'mcp_list']);
expect(mcp.servers).toEqual([]);
expect(mcp.errors).toEqual([]); expect(mcp.errors).toEqual([]);
await mcp.close(); await mcp.close();
}); });
@@ -76,8 +81,58 @@ test('close is safe to call twice', async () => {
}); });
test('an http server config is attempted and its failure reported', async () => { test('an http server config is attempted and its failure reported', async () => {
const mcp = await connectMcp({ remote: { url: 'http://127.0.0.1:1/mcp', type: 'http' } }); const mcp = await connectMcp({ remote: { url: 'http://127.0.0.1:1/mcp', type: 'http' } }, 'eager');
expect(Object.keys(mcp.tools)).toEqual([]); expect(Object.keys(mcp.tools)).toEqual([]);
expect(mcp.errors.map((e) => e.server)).toEqual(['remote']); expect(mcp.errors.map((e) => e.server)).toEqual(['remote']);
await mcp.close(); await mcp.close();
}, 30_000); }, 30_000);
test('lazy mode (default) registers only meta-tools, never server schemas', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
// The request is spared every server schema: only three meta-tools exist.
expect(Object.keys(mcp.tools).sort()).toEqual(['mcp_call', 'mcp_inspect', 'mcp_list']);
// But the prompt knows which servers are connected.
expect(mcp.servers).toEqual(['stub']);
} finally {
await mcp.close();
}
}, 30_000);
test('mcp_list names a server tools and mcp_inspect reads a schema without calling it', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
await call(mcp.tools, 'mcp_list', { server: 'stub' }).then((out) =>
expect(JSON.stringify(out)).toContain('search'),
);
await call(mcp.tools, 'mcp_inspect', { server: 'stub', toolName: 'ping' }).then((out) =>
expect(JSON.stringify(out)).toContain('note'),
);
} finally {
await mcp.close();
}
}, 30_000);
test('mcp_call executes a server tool by name', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
const out = await call(mcp.tools, 'mcp_call', { server: 'stub', toolName: 'ping', args: { note: 'lazy' } });
expect(JSON.stringify(out)).toContain('pong: lazy');
} finally {
await mcp.close();
}
}, 30_000);
test('mcp_call against an unknown server names the live set', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
await call(mcp.tools, 'mcp_call', { server: 'nope', toolName: 'ping', args: {} }).then(
() => {
throw new Error('an unknown server must reject');
},
(e) => expect(String(e)).toContain('nope'),
);
} finally {
await mcp.close();
}
}, 30_000);
+2 -11
View File
@@ -1,4 +1,4 @@
import { usageOf } from './helpers'; import { generateResult, usageOf } from './helpers';
import { afterEach, beforeEach, expect, test } from 'bun:test'; import { afterEach, beforeEach, expect, test } from 'bun:test';
import { MockLanguageModelV4 } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { mkdtempSync, rmSync } from 'node:fs'; import { mkdtempSync, rmSync } from 'node:fs';
@@ -31,16 +31,7 @@ const call = (tools: ToolSet, name: string, input: Record<string, unknown>) => {
const usage = usageOf(1); const usage = usageOf(1);
const summarizer = (text: string) => const summarizer = (text: string) => new MockLanguageModelV4({ doGenerate: async () => generateResult(text, usage) });
new MockLanguageModelV4({
doGenerate: async () =>
({
content: [{ type: 'text', text }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
}) as any,
});
test('a fresh project has no memory and renders nothing', async () => { test('a fresh project has no memory and renders nothing', async () => {
const m = new Memory('/repo'); const m = new Memory('/repo');
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library'; import { render } from 'ink-testing-library';
import React from 'react'; import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session'; import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App'; import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { testHooks, usageOf } from './helpers'; import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(2); const usage = usageOf(2);
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async () => doStream: async () => streamOf(textChunks('answer', usage)),
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'answer' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
}); });
function mount(over: Partial<AppHooks> = {}) { function mount(over: Partial<AppHooks> = {}) {
+195
View File
@@ -0,0 +1,195 @@
import { afterAll, afterEach, beforeEach, expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import type { Config } from '../src/config';
import { Onboard, type OnboardResult } from '../src/ui/Onboard';
/**
* The provider wizard had no test because its every path but the first ends at a
* network call. That call is `fetchModels`, which uses the global `fetch`, so the
* suite stubs it: the picking, the env-key shortcut, the manual-entry fallback,
* and the empty-list warning are all reachable offline and deterministic.
*
* `current` decides the starting row, so a test selects a preset by naming it
* rather than counting arrow presses.
*/
const DOWN = '\u001B[B';
const ENTER = '\r';
const ESC = '\u001B';
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
const ORIGINAL_FETCH = globalThis.fetch;
/** Answers `GET /models` with the given ids; the wizard sorts them itself. */
function stubModels(ids: string[]): string[] {
const calls: string[] = [];
globalThis.fetch = (async (input: unknown) => {
calls.push(String(input));
return new Response(JSON.stringify({ data: ids.map((id) => ({ id })) }), {
status: 200,
headers: { 'content-type': 'application/json' },
});
}) as unknown as typeof fetch;
return calls;
}
function stubFailure(status: number): void {
globalThis.fetch = (async () => new Response('boom', { status })) as unknown as typeof fetch;
}
beforeEach(() => {
// The wizard reads the preset's env key to skip the api-key step; a developer's
// shell must not decide which branch a test takes.
delete process.env['ANTHROPIC_API_KEY'];
delete process.env['OPENAI_API_KEY'];
});
afterEach(() => {
globalThis.fetch = ORIGINAL_FETCH;
delete process.env['ANTHROPIC_API_KEY'];
delete process.env['OPENAI_API_KEY'];
});
afterAll(() => {
globalThis.fetch = ORIGINAL_FETCH;
});
async function press(app: ReturnType<typeof render>, s: string, ms = 80) {
app.stdin.write(s);
await wait(ms);
}
async function type(app: ReturnType<typeof render>, s: string) {
for (const ch of s) await press(app, ch, 20);
}
function mount(current: Partial<Config> = {}) {
const done: OnboardResult[] = [];
let cancelled = 0;
const app = render(
<Onboard
current={{ provider: 'anthropic', model: 'claude-sonnet-4-5', ...current }}
onDone={(r) => done.push(r)}
onCancel={() => void cancelled++}
/>,
);
return { app, done, cancelled: () => cancelled };
}
test('the first screen lists providers and marks the configured one', async () => {
const { app, cancelled } = mount({ presetId: 'openai' });
await wait(80);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('Choose a provider');
expect(frame).toContain('Anthropic');
expect(frame).toContain('OpenAI (current)');
expect(frame).toContain('Custom OpenAI-compatible endpoint');
await press(app, ESC, 120);
expect(cancelled()).toBe(1);
app.unmount();
}, 20_000);
test('esc cancels from the api-key step', async () => {
// custom-openai has no env key, so it stops for a key rather than calling out.
const { app, cancelled } = mount({ presetId: 'custom-openai' });
await wait(80);
await press(app, ENTER, 120);
await type(app, 'https://ex.com/v1');
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('API key');
await press(app, ESC, 120);
expect(cancelled()).toBe(1);
app.unmount();
}, 20_000);
test('a custom endpoint collects url, key, and a chosen model', async () => {
const calls = stubModels(['m2', 'm1']);
const { app, done } = mount({ presetId: 'custom-openai' });
await wait(80);
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('endpoint URL');
await type(app, 'https://ex.com/v1');
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('API key');
await type(app, 'sk-secret-key');
await press(app, ENTER, 200);
expect(app.lastFrame()).toContain('choose a model');
// The list arrives sorted, so the first row is m1, not the m2 it was given.
await press(app, ENTER, 150);
expect(calls).toEqual(['https://ex.com/v1/models']);
expect(done).toEqual([
{
presetId: 'custom-openai',
provider: 'openai',
baseURL: 'https://ex.com/v1',
apiKey: 'sk-secret-key',
model: 'm1',
},
]);
app.unmount();
}, 20_000);
test('an env key skips the key step and the hint masks it', async () => {
process.env['OPENAI_API_KEY'] = 'sk-abcdefgh1234';
stubModels(['gpt-5', 'gpt-5-mini']);
const { app } = mount({ presetId: 'openai' });
await wait(80);
await press(app, ENTER, 200);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('choose a model');
expect(frame).toContain('sk-a...1234');
app.unmount();
}, 20_000);
test('the manual entry collects a model id the list does not offer', async () => {
// The env key is what skips the key step; without it the wizard would stop there.
process.env['OPENAI_API_KEY'] = 'sk-manual-test-key';
stubModels(['only-one']);
const { app, done } = mount({ presetId: 'openai' });
await wait(80);
await press(app, ENTER, 200);
expect(app.lastFrame()).toContain('choose a model');
await press(app, DOWN, 100); // one model, then the manual-entry row
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('model id');
await type(app, 'my-model');
await press(app, ENTER, 150);
expect(done).toHaveLength(1);
expect(done[0]?.model).toBe('my-model');
app.unmount();
}, 20_000);
test('a server that cannot list models falls through to the warning', async () => {
// custom-openai has no fallback list, so an empty response leaves nothing to pick.
stubFailure(500);
const { app } = mount({ presetId: 'custom-openai' });
await wait(80);
await press(app, ENTER, 120);
await type(app, 'https://ex.com/v1');
await press(app, ENTER, 120);
await type(app, 'sk-x');
await press(app, ENTER, 250);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('could not list models');
expect(frame).toContain('model id');
app.unmount();
}, 20_000);
+279
View File
@@ -0,0 +1,279 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React, { useState } from 'react';
import { PromptInput, type PromptInputProps } from '../src/ui/PromptInput';
/**
* `PromptInput` is exercised through `App` in `input.test.tsx` (history recall,
* arrow and word motion, ctrl-u, ctrl-d, paste). This file renders it directly
* for the behaviours that reaching it through the whole app made awkward: the
* emacs kill keys, the mask, a blurred input, and the `onKey` escape hatch that
* `App` uses to hand up/down/tab/esc to its own menus.
*/
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
/** Keys that are awkward to read as escape sequences at the call site. */
const KEYS = {
up: '\u001B[A',
down: '\u001B[B',
left: '\u001B[D',
ctrlA: '\u0001',
ctrlD: '\u0004',
ctrlE: '\u0005',
ctrlK: '\u000B',
ctrlU: '\u0015',
ctrlW: '\u0017',
backspace: '\u007F',
} as const;
/**
* The input is controlled, so it needs an owner to feed `value` back. This wraps
* it with the state a caller would hold and records every value it reports.
*/
type HarnessProps = Omit<Partial<PromptInputProps>, 'value' | 'onChange' | 'onSubmit'> & {
initial?: string;
onChange?: (value: string, cursor: number) => void;
onSubmit?: (value: string) => void;
};
function Harness({ initial = '', onChange, onSubmit, ...rest }: HarnessProps) {
const [value, setValue] = useState(initial);
return (
<PromptInput
{...rest}
value={value}
onSubmit={onSubmit ?? (() => {})}
onChange={(next, cursor) => {
onChange?.(next, cursor);
setValue(next);
}}
/>
);
}
function mount(props: HarnessProps = {}) {
const changes: { value: string; cursor: number }[] = [];
const submitted: string[] = [];
const app = render(
<Harness
{...props}
onChange={(value, cursor) => changes.push({ value, cursor })}
onSubmit={(v) => submitted.push(v)}
/>,
);
return { app, changes, submitted };
}
async function press(app: ReturnType<typeof render>, s: string, ms = 80) {
app.stdin.write(s);
await wait(ms);
}
/** The visible text with every SGR sequence removed, i.e. what the user actually reads. */
const plain = (frame: string) => frame.replace(/\u001B\[[0-9;]*m/g, '');
test('the rendered line carries no hand-written SGR escapes', async () => {
// The cursor must be Ink's `inverse` prop, not `\u001B[7m` pasted into the string.
// Ink measures string content as printable columns, so an embedded escape is
// counted as text and shifts the line — the stray `t` before `ype` in the
// placeholder, and a corrupted cell wherever the line wraps.
const { app } = mount({ initial: 'abc', placeholder: `type to queue${String.fromCharCode(0x2026)}` });
await wait(40);
const frame = app.lastFrame() ?? '';
expect(frame).not.toContain('\u001B[7m');
expect(frame).not.toContain('\u001B[27m');
expect(plain(frame)).toBe('abc');
app.unmount();
});
test('the placeholder renders as one unbroken run of text', async () => {
const { app } = mount({ placeholder: 'ask anything' });
await wait(40);
// No escape may sit between the first character and the rest.
expect(plain(app.lastFrame() ?? '')).toBe('ask anything');
app.unmount();
});
test('a focused input keeps the caret off the string', async () => {
const { app } = mount({ initial: 'abc' });
await wait(40);
// At the end of the line the caret is a blank cell, which Ink trims.
expect(plain(app.lastFrame() ?? '')).toBe('abc');
app.unmount();
});
test('a bare blurred input renders no caret at all', async () => {
const { app } = mount({ focus: false });
await wait(40);
const frame = app.lastFrame() ?? '';
expect(frame).not.toContain('\u001B[7m');
expect(frame.trim()).toBe('');
app.unmount();
});
test('the caret marks the first placeholder character while focused', async () => {
const { app } = mount({ placeholder: 'ask anything' });
await wait(40);
expect(plain(app.lastFrame() ?? '')).toBe('ask anything');
app.unmount();
const blurred = mount({ placeholder: 'ask anything', focus: false });
await wait(40);
expect(blurred.app.lastFrame()).toBe('ask anything');
blurred.app.unmount();
});
test('the cursor moves one character at a time with the arrows', async () => {
const { app, changes } = mount({ initial: 'abc' });
await wait(40);
expect(plain(app.lastFrame() ?? '')).toBe('abc');
await press(app, KEYS.left);
await press(app, KEYS.left);
await press(app, 'Z');
expect(app.lastFrame()).toBe('aZbc');
expect(changes.at(-1)?.cursor).toBe(2);
app.unmount();
});
test('a masked input hides the value and keeps its length', async () => {
const { app } = mount({ initial: 'abcd', mask: '*' });
await wait(40);
expect(app.lastFrame()).toBe('****');
app.unmount();
});
test('a mask stays masked as the value shortens', async () => {
const { app } = mount({ initial: 'abcd', mask: '*' });
await wait(40);
// The harness owns the value, so backspace shortens it to three stars.
await press(app, KEYS.backspace);
expect(app.lastFrame()).toBe('***');
app.unmount();
});
test('ctrl-a and ctrl-e move the cursor between the ends of the line', async () => {
const { app } = mount({ initial: 'inline' });
await wait(40);
// Both keys only move the caret, so the proof is where the next character lands.
await press(app, KEYS.ctrlA);
await press(app, 'X');
expect(app.lastFrame()).toBe('Xinline');
await press(app, KEYS.ctrlE);
await press(app, 'Y');
expect(app.lastFrame()).toBe('XinlineY');
app.unmount();
});
test('ctrl-k kills from the cursor to the end', async () => {
const { app } = mount({ initial: 'keep this' });
await wait(40);
await press(app, KEYS.ctrlA);
// Three characters in, then kill the tail.
for (let i = 0; i < 3; i++) await press(app, KEYS.left.replace('\u001B[D', '\u001B[C'), 30);
await press(app, KEYS.ctrlK);
expect(app.lastFrame()).toContain('kee');
expect(app.lastFrame()).not.toContain('keep this');
app.unmount();
});
test('ctrl-w deletes the word before the cursor and its padding', async () => {
const { app } = mount({ initial: 'one two three' });
await wait(40);
await press(app, KEYS.ctrlW);
expect(app.lastFrame()).toContain('one two');
expect(app.lastFrame()).not.toContain('three');
await press(app, KEYS.ctrlW);
expect(app.lastFrame()).toContain('one');
expect(app.lastFrame()).not.toContain('two');
app.unmount();
});
test('ctrl-w on a single word leaves an empty line', async () => {
const { app } = mount({ initial: 'lonely' });
await wait(40);
await press(app, KEYS.ctrlW);
expect(app.lastFrame()).not.toContain('lonely');
app.unmount();
});
test('onKey can swallow a key before the input sees it', async () => {
const seen: string[] = [];
const { app, submitted } = mount({
initial: 'draft',
onKey: (input, key) => {
if (key.upArrow) {
seen.push('up');
return true;
}
return false;
},
history: ['from history'],
});
await wait(40);
await press(app, KEYS.up);
expect(seen).toEqual(['up']);
expect(app.lastFrame()).toContain('draft');
expect(app.lastFrame()).not.toContain('from history');
// A key onKey declines still reaches the input.
await press(app, '!');
expect(app.lastFrame()).toContain('draft!');
expect(submitted).toEqual([]);
app.unmount();
});
test('initialCursor puts the caret mid-line, so typing lands there', async () => {
const { app } = mount({ initial: 'abcdef', initialCursor: 2 });
await wait(40);
expect(app.lastFrame()).toBe('abcdef');
await press(app, 'X');
expect(app.lastFrame()).toBe('abXcdef');
app.unmount();
});
test('an external value renders as given and reports no change', async () => {
const changes: { value: string; cursor: number }[] = [];
const app = render(
<PromptInput
value="set from outside"
onChange={(v, c) => changes.push({ value: v, cursor: c })}
onSubmit={() => {}}
/>,
);
await wait(40);
// A prop with a handler that does not feed the value back: the text renders
// exactly as passed and the input reports nothing until a key is pressed.
expect(app.lastFrame()).toBe('set from outside');
expect(changes).toEqual([]);
app.unmount();
});
test('submit reports the current value and resets the cursor', async () => {
const { app, submitted } = mount({ initial: 'send me' });
await wait(40);
await press(app, '\r', 120);
expect(submitted).toEqual(['send me']);
app.unmount();
});
test('typing inserts at the cursor rather than appending', async () => {
const { app } = mount({ initial: 'ac' });
await wait(40);
await press(app, KEYS.left);
await press(app, 'b');
expect(app.lastFrame()).toBe('abc');
app.unmount();
});
+7
View File
@@ -33,6 +33,13 @@ test('mcp tools are grouped with their naming convention explained', () => {
expect(rendered).toContain('needs approval'); expect(rendered).toContain('needs approval');
}); });
test('lazy mcp meta-tools prompt the workflow, and mcp__ and meta-tools do not double-list', () => {
const rendered = renderTools(['read_file', 'mcp_list', 'mcp_inspect', 'mcp_call']);
expect(rendered).toContain('mcp_list, mcp_inspect, mcp_call');
expect(rendered).toContain('Never guess a server or tool name');
expect(rendered).not.toContain('mcp__<server>__<tool>');
});
test('an unknown tool is listed rather than silently dropped', () => { test('an unknown tool is listed rather than silently dropped', () => {
expect(renderTools(['read_file', 'some_plugin_tool'])).toContain('some_plugin_tool'); expect(renderTools(['read_file', 'some_plugin_tool'])).toContain('some_plugin_tool');
}); });
+73 -1
View File
@@ -1,6 +1,6 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import type { ModelMessage } from 'ai'; import type { ModelMessage } from 'ai';
import { detachOrphanedItems, dropOrphanedResults, pruneToFit, prunePreservingItems } from '../src/prune'; import { digestOf, droppedBy, isPrunedSpanSummary, prunedSpanMessage, detachOrphanedItems, dropOrphanedResults, pruneToFit, prunePreservingItems } from '../src/prune';
const kinds = (messages: ModelMessage[]) => const kinds = (messages: ModelMessage[]) =>
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role)); messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
@@ -443,3 +443,75 @@ test('the user prompt survives even the narrowest rung', () => {
const fitted = pruneToFit({ messages, threshold: 100, estimate }); const fitted = pruneToFit({ messages, threshold: 100, estimate });
expect(JSON.stringify(fitted)).toContain('do the thing'); expect(JSON.stringify(fitted)).toContain('do the thing');
}); });
test('droppedBy reports the messages a prune discarded, by reference', () => {
const before: ModelMessage[] = [
{ role: 'user', content: 'do the thing' },
{ role: 'assistant', content: [{ type: 'text', text: 'ok' }] },
{ role: 'user', content: 'and again' },
];
const kept = [before[0]!, before[2]!];
const dropped = droppedBy(before, kept);
expect(dropped).toHaveLength(1);
expect(dropped[0]).toBe(before[1]!);
});
test('droppedBy sees through the copies prunePreservingItems returns', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'goal' },
...Array.from({ length: 40 }, (_, i): ModelMessage => ({
role: 'assistant',
content: [
{ type: 'reasoning', text: `step ${i}` },
{ type: 'tool-call', toolCallId: `tc${i}`, toolName: 'grep', input: { pattern: `p${i}` } },
],
})),
];
const pruned = pruneToFit({ messages, threshold: 50, estimate: (m) => JSON.stringify(m).length / 4 });
const dropped = droppedBy(messages, pruned);
// The point is that a value comparison would find nothing: the survivors are
// spread copies. Reference identity is what makes this non-empty.
expect(dropped.length).toBeGreaterThan(0);
expect(dropped.every((m) => !pruned.includes(m))).toBe(true);
});
test('the digest names the tool and its input, which is the decision', () => {
const dropped: ModelMessage[] = [
{ role: 'assistant', content: [{ type: 'tool-call', toolCallId: 't', toolName: 'edit_file', input: { path: 'src/a.ts' } }] },
{ role: 'tool', content: [{ type: 'tool-result', toolCallId: 't', toolName: 'edit_file', output: { type: 'text', value: 'ok' } }] },
];
const digest = digestOf(dropped);
expect(digest).toContain('src/a.ts');
expect(digest).toContain('result');
});
test('a pruned span is injected with a marker that identifies it', () => {
const dropped: ModelMessage[] = [{ role: 'user', content: 'we chose the ladder' }];
const message = prunedSpanMessage('notes from the span', dropped);
expect(message).toBeDefined();
expect(isPrunedSpanSummary(message!)).toBe(true);
expect(message!.content).toContain('notes from the span');
});
test('the digest is used when no summary came back', () => {
const dropped: ModelMessage[] = [{ role: 'user', content: 'we chose the ladder' }];
const message = prunedSpanMessage(undefined, dropped);
expect(message!.content).toContain('we chose the ladder');
expect(isPrunedSpanSummary(message!)).toBe(true);
});
test('an empty span produces no message rather than an empty one', () => {
expect(prunedSpanMessage(undefined, [])).toBeUndefined();
expect(prunedSpanMessage(' ', [{ role: 'user', content: '' }])).toBeUndefined();
});
test('a span summary is not treated as an ordinary message', () => {
const ordinary: ModelMessage = { role: 'user', content: 'a real question' };
expect(isPrunedSpanSummary(ordinary)).toBe(false);
expect(isPrunedSpanSummary({ role: 'assistant', content: 'Earlier in this session, now compacted away:' })).toBe(false);
});
+3 -15
View File
@@ -1,28 +1,16 @@
import { expect, test } from 'bun:test'; import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library'; import { render } from 'ink-testing-library';
import React from 'react'; import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session'; import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App'; import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { InstallPrompt, RegistryPanel, type RegistryRow } from '../src/ui/Panels'; import { InstallPrompt, RegistryPanel, type RegistryRow } from '../src/ui/Panels';
import { testHooks, usageOf } from './helpers'; import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3); const usage = usageOf(3);
const model = new MockLanguageModelV4({ const model = new MockLanguageModelV4({
doStream: async () => doStream: async () => streamOf(textChunks('reply', usage)),
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
}); });
const rows: RegistryRow[] = [ const rows: RegistryRow[] = [
+6 -2
View File
@@ -131,11 +131,15 @@ test('the skill catalogue and skill tool are offered when skills are loaded', as
expect(system).toContain('Skills available'); expect(system).toContain('Skills available');
}); });
test('no skill tool is offered when there are no skills', async () => { test('the skill tool is present even with no skills, serving an empty list', async () => {
// The tool must exist at zero skills so a skill installed mid-session is callable
// on the next turn without a session rebuild; it just rejects every name until one
// is installed. An unconditional registration is the point, so the tool is never
// absent from the request.
const { seen, model } = recorder(); const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', skills: [] }); const session = new Session({ model, askApproval: async () => 'deny', skills: [] });
for await (const _ of session.send('hi')) void _; for await (const _ of session.send('hi')) void _;
expect((seen[0]?.tools ?? []).map((t) => t.name)).not.toContain('skill'); expect((seen[0]?.tools ?? []).map((t) => t.name)).toContain('skill');
}); });
test('the skill tool never needs approval', async () => { test('the skill tool never needs approval', async () => {

Some files were not shown because too many files have changed in this diff Show More