Initial commit: shiro-neko 0.1.0-beta.1
Agentic coding CLI on Bun, Ink, and the AI SDK. Core: streamText loop with SDK-level tool approval so a denied call provably never executes; endpoint fallback for OpenAI reasoning models; retry with backoff. Tools: read/write/edit/glob/grep/bash, path-jailed, gitignore-aware, ripgrep with a JS fallback, binary rejection, live bash streaming. Agents: five variants crossing thinking level with tool restriction; plan and review withhold mutating tools from the model. Extensibility: frontmatter skills with on-demand bodies, plugin host with blocking hooks, MCP stdio and HTTP, read-only subagents. State: durable per-project memory, session task lists, session persistence, compaction that repairs provider-item dependencies. Distribution: five-platform cross-compiled binaries with checksums, install scripts, CI on three operating systems. 404 tests, typecheck clean.
This commit is contained in:
@@ -0,0 +1,25 @@
|
||||
name: ci
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
check:
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# The tools shell out to rg, git, and a platform shell, so the suite has to
|
||||
# run on all three; a Windows-only path break is otherwise invisible.
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: 1.3.14
|
||||
- run: bun install --frozen-lockfile
|
||||
- run: bun run typecheck
|
||||
- run: bun test
|
||||
- run: bun run build
|
||||
@@ -0,0 +1,73 @@
|
||||
name: release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ['v*']
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
dry_run:
|
||||
description: Build the artifacts without publishing a release
|
||||
type: boolean
|
||||
default: true
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
jobs:
|
||||
verify:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: 1.3.14
|
||||
- run: bun install --frozen-lockfile
|
||||
- run: bun run typecheck
|
||||
- run: bun test
|
||||
|
||||
build:
|
||||
needs: verify
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: 1.3.14
|
||||
- run: bun install --frozen-lockfile
|
||||
|
||||
# Bun cross-compiles every target from one host, so no build matrix is needed.
|
||||
# release.ts also fails the build when the tag and src/version.ts disagree.
|
||||
- run: bun run release
|
||||
|
||||
- name: Check the binary reports the right version
|
||||
run: |
|
||||
chmod +x dist/release/shiro-linux-x64
|
||||
./dist/release/shiro-linux-x64 --version
|
||||
./dist/release/shiro-linux-x64 --version | grep -q "$(bun -e 'console.log((await import("./src/version.ts")).VERSION)')"
|
||||
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: shiro-binaries
|
||||
path: dist/release/
|
||||
retention-days: 7
|
||||
|
||||
publish:
|
||||
needs: build
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: shiro-binaries
|
||||
path: dist/release
|
||||
|
||||
- name: Publish the release
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
gh release create "${{ github.ref_name }}" \
|
||||
--title "shiro-neko ${{ github.ref_name }}" \
|
||||
--generate-notes \
|
||||
$([[ "${{ github.ref_name }}" == *-* ]] && echo --prerelease) \
|
||||
dist/release/*
|
||||
+40
@@ -0,0 +1,40 @@
|
||||
# dependencies (bun install)
|
||||
node_modules
|
||||
|
||||
# build output
|
||||
out
|
||||
dist
|
||||
*.tgz
|
||||
|
||||
# code coverage
|
||||
coverage
|
||||
*.lcov
|
||||
|
||||
# logs
|
||||
logs
|
||||
*.log
|
||||
report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json
|
||||
|
||||
# environment
|
||||
.env
|
||||
.env.local
|
||||
.env.development.local
|
||||
.env.test.local
|
||||
.env.production.local
|
||||
|
||||
# caches
|
||||
.eslintcache
|
||||
.cache
|
||||
*.tsbuildinfo
|
||||
|
||||
# editors
|
||||
.idea
|
||||
.vscode
|
||||
.DS_Store
|
||||
|
||||
# agent tooling state, not part of the project
|
||||
.omo/
|
||||
.codegraph/
|
||||
|
||||
# scratch files from local debugging
|
||||
tmp-*
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Muhammad Zakir Ramadhan
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,125 @@
|
||||
# shiro-neko
|
||||
|
||||
An agentic coding CLI. It reads your code, edits it, runs your tests, and asks when the
|
||||
request is ambiguous — in a terminal UI, with every mutating action gated behind an
|
||||
approval prompt.
|
||||
|
||||
Built on [Bun](https://bun.com), [Ink](https://github.com/vadimdemedes/ink), and the
|
||||
[AI SDK](https://ai-sdk.dev). Works with Anthropic, OpenAI, and any OpenAI- or
|
||||
Anthropic-compatible endpoint: OpenRouter, Groq, DeepSeek, xAI, Ollama, LM Studio, vLLM.
|
||||
|
||||
`0.1.0-beta.1` — usable, but the interfaces may still move.
|
||||
|
||||
## Install
|
||||
|
||||
A single prebuilt binary. No runtime, no `node_modules`.
|
||||
|
||||
```bash
|
||||
# macOS, Linux
|
||||
curl -fsSL https://raw.githubusercontent.com/zakirkun/shiro-neko/main/scripts/install.sh | sh
|
||||
|
||||
# Windows
|
||||
irm https://raw.githubusercontent.com/zakirkun/shiro-neko/main/scripts/install.ps1 | iex
|
||||
```
|
||||
|
||||
Both verify the download against the release checksums before installing. Builds are
|
||||
published for `linux-x64`, `linux-arm64`, `darwin-x64`, `darwin-arm64`, and `windows-x64`.
|
||||
|
||||
Or from source:
|
||||
|
||||
```bash
|
||||
git clone https://github.com/zakirkun/shiro-neko
|
||||
cd shiro-neko
|
||||
bun install
|
||||
bun run install:local # builds and puts `shiro` on PATH
|
||||
```
|
||||
|
||||
## First run
|
||||
|
||||
```bash
|
||||
shiro
|
||||
```
|
||||
|
||||
With no API key configured it opens provider setup: pick an endpoint, paste a key, choose
|
||||
from the models that endpoint actually reports. Settings land in
|
||||
`~/.shiro-neko/config.json`. Run `/provider` any time to change them.
|
||||
|
||||
```
|
||||
shiro-neko 0.1.0-beta.1 openai/gpt-5 session 0193ab2c
|
||||
agent: default thinking: medium
|
||||
cwd: /home/you/project
|
||||
skills: debug, refactor, review, test
|
||||
plugins: guard, time
|
||||
approvals: on for write_file, edit_file, bash, mcp__*
|
||||
/help for commands
|
||||
|
||||
> why does the pagination test fail?
|
||||
```
|
||||
|
||||
## What it does
|
||||
|
||||
**Answers about your code, grounded in your code.** `grep` goes through ripgrep when it is
|
||||
installed and honours `.gitignore`. `read_file` refuses binaries rather than filling the
|
||||
context with mojibake.
|
||||
|
||||
**Edits with your approval.** Every `write_file`, `edit_file`, and `bash` call stops for a
|
||||
`y`/`a`/`n` decision, with a coloured diff for edits. The `guard` plugin refuses
|
||||
irreversible commands outright — `rm -rf`, `git reset --hard`, force pushes, `DROP TABLE` —
|
||||
and `--yolo` cannot bypass it.
|
||||
|
||||
**Asks instead of guessing.** When a request has two readings that lead to different work,
|
||||
the agent puts a question on screen with options.
|
||||
|
||||
**Delegates searches.** `task` spawns a read-only subagent whose findings come back as one
|
||||
message, so a search across forty files does not fill the main context. Its progress
|
||||
streams to a panel.
|
||||
|
||||
**Remembers between sessions.** Decisions, working commands, and traps go into per-project
|
||||
memory that is injected at the start of every future session.
|
||||
|
||||
**Survives long tasks.** The task list and project memory live outside the message array,
|
||||
so they survive both automatic pruning and `/compact`.
|
||||
|
||||
**Runs headless.** `shiro -p "review this diff" --json` for scripts and CI.
|
||||
|
||||
## Documentation
|
||||
|
||||
| Guide | Contents |
|
||||
|---|---|
|
||||
| [Configuration](docs/configuration.md) | config file, environment variables, every flag |
|
||||
| [Tools](docs/tools.md) | every tool, the approval model, the guard |
|
||||
| [Agents and thinking](docs/agents.md) | variants, thinking levels, read-only modes |
|
||||
| [Skills](docs/skills.md) | the bundled skills and writing your own |
|
||||
| [Plugins](docs/plugins.md) | the plugin interface and the builtins |
|
||||
| [Memory and state](docs/memory.md) | memory, task lists, sessions, compaction |
|
||||
| [MCP](docs/mcp.md) | connecting Model Context Protocol servers |
|
||||
| [Headless mode](docs/headless.md) | `-p`, JSON events, exit codes, CI recipes |
|
||||
| [Architecture](docs/architecture.md) | how the loop works and why it is built this way |
|
||||
| [Development](docs/development.md) | building, testing, releasing |
|
||||
| [Roadmap](ROADMAP.md) | what is next and what has been declined |
|
||||
| [TODO](TODO.md) | the current work list |
|
||||
|
||||
## Commands
|
||||
|
||||
Type `/` and a menu appears, narrowing as you type.
|
||||
|
||||
```
|
||||
/help /agent [name] /think [level] /provider /models /model <id>
|
||||
/skills /plugins /init /context /todos /notes /memory
|
||||
/tools /compact /cost /sessions /resume <id> /save /clear /exit
|
||||
```
|
||||
|
||||
`esc` dismisses a panel or interrupts a running turn. Up and down recall earlier prompts.
|
||||
|
||||
## Status
|
||||
|
||||
Working: the agent loop, tool approvals, subagents, skills, plugins, per-project memory,
|
||||
session persistence, MCP, markdown rendering, headless mode, five-platform builds.
|
||||
|
||||
Next up is in [TODO.md](TODO.md); the longer view and what has been declined are in
|
||||
[ROADMAP.md](ROADMAP.md). The short version of what is missing: streaming reasoning display,
|
||||
a message queue for prompts typed mid-turn, `@file` completion, and git-aware tools.
|
||||
|
||||
## License
|
||||
|
||||
MIT. See [LICENSE](LICENSE).
|
||||
+135
@@ -0,0 +1,135 @@
|
||||
# Roadmap
|
||||
|
||||
What is built, what is next, and what has been deliberately declined. Reordered when
|
||||
evidence says the order is wrong.
|
||||
|
||||
Nothing here is a date. Items move to [TODO.md](TODO.md) when they are next up.
|
||||
|
||||
---
|
||||
|
||||
## Shipped
|
||||
|
||||
### 0.1.0-beta.1
|
||||
|
||||
**Core loop** — `streamText` with tool approvals suspended and resumed through the SDK's
|
||||
`toolApproval`, so a denied tool provably never executes. Endpoint fallback for OpenAI
|
||||
reasoning models that reject function tools on `/v1/chat/completions`. Retry with backoff
|
||||
for transient failures.
|
||||
|
||||
**Tools** — `read_file` `write_file` `edit_file` `glob` `grep` `bash`, all path-jailed to
|
||||
the workspace. ripgrep bridge with a JavaScript fallback. `.gitignore` and `.shiroignore`
|
||||
aware walking. Binary rejection. Live-streaming `bash` output.
|
||||
|
||||
**Interface** — Ink TUI with markdown rendering, slash command menu, readline input with
|
||||
per-project prompt history, coloured diffs in approval prompts, and panels for tasks,
|
||||
subagents, command output, questions, and command results.
|
||||
|
||||
**Agents** — five variants crossing thinking level with tool restrictions. `plan` and
|
||||
`review` withhold mutating tools from the model rather than discouraging them.
|
||||
|
||||
**Skills** — frontmatter markdown, catalogue in the prompt and body on demand. Four bundled,
|
||||
overridable per user and per project.
|
||||
|
||||
**Plugins** — tool contribution, auto-approval, `beforeToolCall` blocking, `afterTurn`
|
||||
hooks, prompt appendices. `guard` refuses irreversible shell commands ahead of any approval,
|
||||
including under `--yolo`.
|
||||
|
||||
**Memory and state** — durable per-project memory with hit-counted recall and model-driven
|
||||
compaction. Session task lists with four states. Session persistence with resume. Context
|
||||
compaction that repairs the provider-item dependencies pruning breaks.
|
||||
|
||||
**Subagents** — read-only `task` with `explore` and `review` flavours, progress streamed to
|
||||
a panel.
|
||||
|
||||
**Asking** — the `ask` tool, withheld in headless runs rather than left to hang.
|
||||
|
||||
**MCP** — stdio and HTTP servers, tools namespaced `mcp__<server>__<tool>`, a failing server
|
||||
reported rather than fatal.
|
||||
|
||||
**Distribution** — five-platform cross-compiled binaries, checksums, install scripts, CI on
|
||||
three operating systems, tag-driven releases.
|
||||
|
||||
---
|
||||
|
||||
## Next
|
||||
|
||||
### Visible process
|
||||
|
||||
The agent's reasoning is discarded. `reasoning-delta` already arrives from the session; the
|
||||
transcript drops it. A collapsed panel showing what the model is thinking, expandable with a
|
||||
key, is the largest gap between this and a tool that feels responsive on a slow turn.
|
||||
|
||||
Also missing: which file is being read or written as it happens. `tool-call` events carry
|
||||
the path but the transcript only shows a one-line summary after the fact.
|
||||
|
||||
### Message queue
|
||||
|
||||
Typing during a turn does nothing. It should queue and run when the turn ends. Interrupting
|
||||
with `esc` then retyping loses the thought. Requires input to stay live while `busy`, which
|
||||
means the prompt and the spinner have to coexist rather than swap.
|
||||
|
||||
### More tools
|
||||
|
||||
Measured cost: 553 characters of schema per tool, sent every request. Thirteen live tools is
|
||||
already at the point where selection accuracy starts to matter, so the next additions need
|
||||
`activeTools` gating per set before the count grows.
|
||||
|
||||
Ordered by value per line of code:
|
||||
|
||||
- `multi_edit` — several edits to one file atomically, killing read-edit-read-edit churn
|
||||
- `list_dir` — a tree view, so the model stops globbing blindly to orient
|
||||
- `read_many_files` — batch reads in one round trip
|
||||
- git read-only set — `git_status`, `git_diff`, `git_log`, `git_show`, `git_blame`, all
|
||||
approval-free because they cannot mutate
|
||||
- `web_fetch` — URL to markdown
|
||||
|
||||
Declined: wrappers around a single bash line with no added guarantee. `run_tests`,
|
||||
`typecheck`, `lint`, `build` are five tools of pure schema tax when the real commands are
|
||||
already in `AGENTS.md`.
|
||||
|
||||
### `@file` completion
|
||||
|
||||
Typing `@src/` should complete paths. The last real ergonomic gap in the input.
|
||||
|
||||
---
|
||||
|
||||
## Later
|
||||
|
||||
**Subagent parallelism.** Two independent searches run sequentially today. The panel already
|
||||
handles multiple agents; the loop does not fan out.
|
||||
|
||||
**Session branching.** Fork a session at a message to try a different approach without
|
||||
losing the original.
|
||||
|
||||
**Cost budgets.** A per-session ceiling that warns, then stops. Pricing and accounting exist;
|
||||
the limit does not.
|
||||
|
||||
**Structured diff review.** Approve or reject individual hunks of an `edit_file` call rather
|
||||
than the whole thing.
|
||||
|
||||
**Plugin loading from disk.** Plugins are compiled in. Loading `.shiro/plugins/*.ts` needs a
|
||||
sandbox story first — a plugin that can block tool calls can also lie about blocking them.
|
||||
|
||||
**Prompt caching.** Anthropic and OpenAI both support it. The system prompt is rebuilt every
|
||||
step for task-list freshness, which defeats a naive cache; splitting the stable prefix from
|
||||
the volatile suffix would fix that.
|
||||
|
||||
---
|
||||
|
||||
## Declined
|
||||
|
||||
**A web UI.** This is a terminal tool. A browser front end doubles the surface area and
|
||||
serves a different product.
|
||||
|
||||
**Model-agnostic prompt tuning.** Per-model prompt variants are a maintenance treadmill for
|
||||
gains that evaporate on the next model release.
|
||||
|
||||
**Auto-commit.** The agent should never write git history without being asked. Commits are
|
||||
the user's record of their own work.
|
||||
|
||||
**Vector search over the codebase.** ripgrep answers a scoped question in 135 ms with no
|
||||
index to build, invalidate, or ship. An embedding store is a large amount of machinery for a
|
||||
worse answer on a codebase that fits in a grep.
|
||||
|
||||
**Tool call retries on model error.** A model that produced a malformed call will usually
|
||||
produce it again. Surfacing the error teaches it more than a silent retry.
|
||||
@@ -0,0 +1,109 @@
|
||||
# TODO
|
||||
|
||||
Next up. One item, one outcome, verifiable when done.
|
||||
|
||||
Longer-term direction lives in [ROADMAP.md](ROADMAP.md).
|
||||
|
||||
---
|
||||
|
||||
## Now
|
||||
|
||||
### Show reasoning in the transcript
|
||||
|
||||
`src/session.ts` already yields `{ type: 'reasoning' }`; `src/ui/App.tsx` ignores it.
|
||||
|
||||
- [ ] Accumulate reasoning deltas into their own buffer, separate from `text`
|
||||
- [ ] Render as a dim collapsed panel: `thinking... 412 tokens`, expandable with a key
|
||||
- [ ] Drop it from the transcript when the turn ends — reasoning is not part of the answer
|
||||
- [ ] Test: a model emitting `reasoning-delta` puts text on screen before any `text-delta`
|
||||
|
||||
### Show the file being touched
|
||||
|
||||
`tool-call` carries the path but the transcript only shows a summary after the call returns.
|
||||
|
||||
- [ ] Render an active-tool line while a call is in flight: `read src/session.ts`
|
||||
- [ ] Clear it on `tool-result` or `tool-error`
|
||||
- [ ] Test: a slow tool leaves its line on screen for the duration
|
||||
|
||||
### Queue prompts typed during a turn
|
||||
|
||||
- [ ] Keep `PromptInput` mounted while `busy`, alongside the spinner
|
||||
- [ ] Submitting while busy appends to a queue and shows `queued: 2`
|
||||
- [ ] Drain the queue in order when the turn ends
|
||||
- [ ] `esc` clears the queue as well as aborting
|
||||
- [ ] Test: two prompts typed during a turn run in order afterwards
|
||||
|
||||
---
|
||||
|
||||
## Next
|
||||
|
||||
### `activeTools` gating per tool set
|
||||
|
||||
Needed before the tool count grows. Measured at 553 chars of schema per tool per request.
|
||||
|
||||
- [ ] `toolSets` in config: which sets are live
|
||||
- [ ] Sets: `core`, `git`, `edit-plus`, `net`
|
||||
- [ ] `prepareStep` narrows `activeTools` to the enabled sets
|
||||
- [ ] `/tools` shows which set each tool came from
|
||||
- [ ] Test: a disabled set's tools reach neither the wire nor the prompt
|
||||
|
||||
### `multi_edit`
|
||||
|
||||
- [ ] Several `{ oldString, newString }` edits against one file
|
||||
- [ ] Atomic: any failing match aborts the whole call, file untouched
|
||||
- [ ] Each edit applied to the result of the previous one
|
||||
- [ ] Approval prompt shows one combined diff
|
||||
- [ ] Test: a failing second edit leaves the file exactly as it was
|
||||
|
||||
### `list_dir`
|
||||
|
||||
- [ ] Tree view honouring `.gitignore`, depth-limited, entry-capped
|
||||
- [ ] Marks directories and shows file sizes
|
||||
- [ ] Test: respects ignore rules, stops at the depth limit
|
||||
|
||||
### Git read-only tools
|
||||
|
||||
All approval-free, since none can mutate.
|
||||
|
||||
- [ ] `git_status`, `git_diff`, `git_log`, `git_show`, `git_blame`
|
||||
- [ ] Structured output, not raw porcelain
|
||||
- [ ] Fail clearly outside a repo instead of returning git's error text
|
||||
- [ ] Test: each returns something usable in a temp repo, and a clean error outside one
|
||||
|
||||
### `@file` completion
|
||||
|
||||
- [ ] `@` in the input opens a path picker fed by the ignore-aware walker
|
||||
- [ ] Tab completes, continued typing narrows
|
||||
- [ ] Completed path inserted as a plain relative path
|
||||
- [ ] Test: `@src/` narrows to files under `src/`
|
||||
|
||||
---
|
||||
|
||||
## Maintenance
|
||||
|
||||
- [ ] Pricing table needs a source note and a date; rates drift and ours are hand-entered
|
||||
- [ ] `estimateTokens` divides JSON length by four. Good enough for a compaction threshold,
|
||||
wrong enough to mislead in `/cost`. Either label it an estimate everywhere or use a
|
||||
real tokenizer
|
||||
- [ ] The subagent shares the parent's model. A cheaper model for search would cut cost
|
||||
substantially on `explore` runs
|
||||
- [ ] No spend ceiling. A headless run that loops costs real money with nothing to stop it
|
||||
|
||||
---
|
||||
|
||||
## Known rough edges
|
||||
|
||||
Not bugs exactly, but things that will bite someone.
|
||||
|
||||
- **`/clear` wipes the terminal scrollback.** `<Static>` output is already committed, so
|
||||
clearing React state alone leaves it on screen. The escape sequence works but takes the
|
||||
user's earlier terminal history with it.
|
||||
- **Compaction is lossy in a way the model cannot see.** It is told the history was pruned,
|
||||
but not what was in the pruned part. A summary of the discarded span would be better than
|
||||
a count.
|
||||
- **Memory has no conflict resolution.** Two contradictory notes both persist and both get
|
||||
injected. `/memory` may merge them, or may keep both.
|
||||
- **`bash` cannot be interrupted independently.** `esc` aborts the whole turn, killing the
|
||||
command. There is no way to stop a runaway command and keep the turn.
|
||||
- **Windows `cmd /c` differs from `bash -lc`.** A command the model writes for one shell may
|
||||
fail on the other. The prompt states the platform; it does not translate.
|
||||
@@ -0,0 +1,85 @@
|
||||
# Agents and thinking
|
||||
|
||||
An agent variant sets three things: how much the model deliberates, which tools it is
|
||||
offered, and a behaviour appendix in the system prompt.
|
||||
|
||||
```bash
|
||||
shiro --agent deep # at launch
|
||||
shiro --agent plan --think low # variant with an overridden level
|
||||
```
|
||||
|
||||
```
|
||||
/agent picker
|
||||
/agent review direct
|
||||
/think picker
|
||||
/think max direct
|
||||
```
|
||||
|
||||
## The variants
|
||||
|
||||
| Variant | Thinking | Tools | Steps | For |
|
||||
|---|---|---|---|---|
|
||||
| `default` | medium | all | 50 | ordinary work |
|
||||
| `quick` | off | all | 12 | small, well-scoped edits |
|
||||
| `deep` | max | all | 80 | hard problems, unclear causes |
|
||||
| `plan` | high | read-only | 50 | investigate and propose |
|
||||
| `review` | high | read-only | 50 | critique a change |
|
||||
|
||||
**`quick`** tells the model not to deliberate, not to write a task list, and not to explore
|
||||
beyond what the change needs. Good for a rename or a one-line fix where thinking budget is
|
||||
pure latency.
|
||||
|
||||
**`deep`** asks for more than one hypothesis before acting, more reading before concluding,
|
||||
and findings recorded with `remember` so they survive compaction.
|
||||
|
||||
**`plan`** and **`review`** are genuinely read-only. `write_file`, `edit_file`, and `bash`
|
||||
are withheld from the model, not merely discouraged in prose — a model that cannot see a
|
||||
tool cannot call it. Their prompts also forbid describing edits as if they had been made.
|
||||
|
||||
## Thinking levels
|
||||
|
||||
`off`, `low`, `medium`, `high`, `max`. They map to whatever the provider actually supports:
|
||||
|
||||
| Level | OpenAI `reasoning_effort` | Anthropic `thinking` |
|
||||
|---|---|---|
|
||||
| `off` | `none` | `{ type: "disabled" }` |
|
||||
| `low` | `low` | `budget_tokens: 6400` |
|
||||
| `medium` | `medium` | proportional budget |
|
||||
| `high` | `high` | `budget_tokens: 38400` |
|
||||
| `max` | `xhigh` | maximum budget |
|
||||
|
||||
Verified against both wire formats rather than assumed.
|
||||
|
||||
Higher costs more and takes longer. `off` on a hard problem produces confident wrong
|
||||
answers; `max` on a rename wastes a few cents and several seconds. The variants pick
|
||||
sensible defaults, so reach for `/think` only when a specific turn needs something else.
|
||||
|
||||
## Overriding
|
||||
|
||||
`--agent deep --think low` gives you `deep`'s tools, steps, and appendix with a low thinking
|
||||
budget. The override clones the preset rather than mutating it, so a later `/agent deep` in
|
||||
the same session still gets `max`.
|
||||
|
||||
## Defaults in config
|
||||
|
||||
```json
|
||||
{ "agent": "deep", "thinking": "high" }
|
||||
```
|
||||
|
||||
A flag beats the config file. An unknown name fails at startup with the valid list rather
|
||||
than silently falling back:
|
||||
|
||||
```
|
||||
$ shiro --agent turbo
|
||||
shiro: Unknown agent "turbo". Available: default, quick, deep, plan, review
|
||||
```
|
||||
|
||||
## What the variant changes in the prompt
|
||||
|
||||
The system prompt describes only the tools actually offered, and the workflow rules adapt.
|
||||
Under `plan` the model is told it has no tools that change anything and that it cannot run
|
||||
commands, so it should say what to run rather than claim it passed. Under `default` it is
|
||||
told which tools need approval and to verify with the project's tests.
|
||||
|
||||
A prompt that describes a withheld tool teaches the model to attempt calls that cannot
|
||||
succeed, which is why the description is generated from the live tool set.
|
||||
@@ -0,0 +1,170 @@
|
||||
# Architecture
|
||||
|
||||
## The loop
|
||||
|
||||
One turn is a `streamText` call whose stream is translated into UI events.
|
||||
|
||||
```
|
||||
user prompt
|
||||
→ messages.push({ role: 'user', ... })
|
||||
→ streamText({ model, system, messages, tools, activeTools, reasoning, toolApproval })
|
||||
→ for each stream part → yield an AgentEvent
|
||||
→ if any tool needs approval, the stream ends suspended
|
||||
→ collect decisions from the UI
|
||||
→ push a tool message with the approval responses
|
||||
→ loop
|
||||
→ otherwise done
|
||||
```
|
||||
|
||||
`src/session.ts` is an async generator. The UI consumes events; it never touches the SDK.
|
||||
That is what lets the same session drive the Ink app, the headless printer, and the tests.
|
||||
|
||||
## Why approval goes through the SDK
|
||||
|
||||
An obvious design is a promise inside each tool's `execute`, resolved when the user answers.
|
||||
That was rejected: it makes "denied" a convention the tool must remember to honour, and one
|
||||
tool forgetting it is a silent security hole.
|
||||
|
||||
Instead the SDK's `toolApproval` is used. A denied call **provably never executes** — the SDK
|
||||
never reaches `execute`. The tool cannot opt out because the tool is not consulted.
|
||||
|
||||
```ts
|
||||
toolApproval: async ({ toolCall }) => {
|
||||
const blocked = await plugins?.guard({ toolName: toolCall.toolName, input: toolCall.input, cwd });
|
||||
if (blocked) return { type: 'denied', reason: blocked }; // --yolo cannot reach this
|
||||
if (yolo) return undefined;
|
||||
if (!needsApproval(toolCall.toolName)) return undefined;
|
||||
return 'user-approval';
|
||||
}
|
||||
```
|
||||
|
||||
Guards are checked first, so `--yolo` skips prompts but not refusals.
|
||||
|
||||
One subtlety: when this function denies, the SDK emits `tool-approval-request` with
|
||||
`isAutomatic: true` and answers it itself. Queueing that would prompt the user for a call
|
||||
that is already settled, so automatic requests are skipped and denial is surfaced from
|
||||
`tool-approval-response` instead.
|
||||
|
||||
## Where state lives
|
||||
|
||||
The system prompt is rebuilt on **every step**, not once per turn:
|
||||
|
||||
```ts
|
||||
prepareStep: ({ messages }) => {
|
||||
const instructions = this.systemFor(); // task list, memory, skills, agent
|
||||
if (estimateTokens(messages) <= threshold) return { instructions };
|
||||
return { instructions, messages: prunePreservingItems({ messages, reasoning: 'all', ... }) };
|
||||
}
|
||||
```
|
||||
|
||||
That is not an optimisation. A `todo_write` on step one must be visible to step two, and
|
||||
`system:` on `streamText` is bound once for the whole run. Returning `instructions` from
|
||||
`prepareStep` is the only place per-step state can enter.
|
||||
|
||||
The prompt also describes only the tools actually offered this turn. A prompt that mentions a
|
||||
withheld tool teaches the model to attempt impossible calls.
|
||||
|
||||
## Rendering
|
||||
|
||||
Ink re-renders the whole tree on every `setState`. At 50 tokens a second that is 50 full
|
||||
renders and a visibly flickering terminal.
|
||||
|
||||
Two things fix it:
|
||||
|
||||
- Finished lines go into `<Static>`, rendered once and never redrawn.
|
||||
- Token deltas accumulate in a ref and flush on a 60 ms interval, not per token.
|
||||
|
||||
Markdown is parsed on every flush. An unclosed fence renders as a code block that grows,
|
||||
which is what a reader expects while text is still arriving.
|
||||
|
||||
## Input
|
||||
|
||||
`ink-text-input` was replaced. It discards up and down before its own handler, so history
|
||||
recall is impossible, and it only ever *shrinks* its internal cursor offset, so an externally
|
||||
set value leaves the cursor stranded mid-string.
|
||||
|
||||
`src/ui/PromptInput.tsx` owns the cursor. That also gives home, end, and ctrl-a/e/k/u/w for
|
||||
free. It hands up, down, tab, and escape to a parent callback first, so the command menu and
|
||||
open panels can claim them before the input treats them as editing keys.
|
||||
|
||||
## Subagents
|
||||
|
||||
`task` runs a nested `streamText` with only `read_file`, `glob`, and `grep`. It returns one
|
||||
message.
|
||||
|
||||
Two consequences follow from the tool set, not from policy:
|
||||
|
||||
- It can never need approval, because it has no gated tools.
|
||||
- The parent's context holds the findings, not the search transcript.
|
||||
|
||||
Progress is reported through a callback, wired to a bus the panel subscribes to. Without the
|
||||
bus the panel would need a reference to the tool, and the tool would need one to React.
|
||||
|
||||
## Provider differences
|
||||
|
||||
Two are handled explicitly.
|
||||
|
||||
**Thinking levels.** `off`/`low`/`medium`/`high`/`max` become `reasoning_effort` on OpenAI and
|
||||
a `thinking` token budget on Anthropic. The SDK does the mapping; `src/agents.ts` only picks
|
||||
the level.
|
||||
|
||||
**Endpoint fallback.** Newer OpenAI models reject function tools on `/v1/chat/completions`
|
||||
and require `/v1/responses`. `src/fallback.ts` presents both as one model and switches when
|
||||
the first rejects the request *shape* — 400, 404, 405, 415, 422, 501 with `isRetryable` false.
|
||||
Retryable failures are left to the SDK's backoff.
|
||||
|
||||
The switch is sticky. Once an endpoint rejects the shape it will reject every later step too,
|
||||
so re-probing it each turn would waste a round trip per step.
|
||||
|
||||
Only `api.openai.com` gets the chain. Third-party endpoints do not implement `/v1/responses`.
|
||||
|
||||
## Compaction and its repair
|
||||
|
||||
`pruneMessages({ reasoning: 'all' })` strips a reasoning item and keeps the message item from
|
||||
the same response. The responses API treats the message as that reasoning item's dependent
|
||||
and returns 400.
|
||||
|
||||
The two carry different ids, so they cannot be matched by id. What links them is the assistant
|
||||
message they arrived in: one message is one response, and its reasoning item covers every
|
||||
other item in it. `src/prune.ts` drops the dependent parts of any turn whose reasoning was
|
||||
removed — which costs nothing, since pruning was already discarding those turns.
|
||||
|
||||
## Module map
|
||||
|
||||
| Module | Responsibility |
|
||||
|---|---|
|
||||
| `session.ts` | the loop, approvals, compaction, event stream |
|
||||
| `tools.ts` | file and shell tools, ripgrep bridge, bash streaming |
|
||||
| `ignore.ts` | gitignore-aware walker, path jail |
|
||||
| `prompt.ts` | system prompt assembly from live state |
|
||||
| `agents.ts` | variants, thinking levels |
|
||||
| `skills.ts` | discovery, catalogue, `skill` tool |
|
||||
| `memory.ts` | durable notes, search, model compaction |
|
||||
| `notebook.ts` | session task list |
|
||||
| `plugins.ts` | host, hooks, guard chain |
|
||||
| `subagent.ts` | `task` tool and progress events |
|
||||
| `ask.ts` | the `ask` tool |
|
||||
| `mcp.ts` | MCP clients and namespacing |
|
||||
| `fallback.ts` | endpoint chain |
|
||||
| `prune.ts` | provider-item repair |
|
||||
| `markdown.ts` | parser, no dependency |
|
||||
| `store.ts` | sessions, prompt history |
|
||||
| `config.ts` | resolution, model construction |
|
||||
| `providers.ts` | presets, `/models` fetch |
|
||||
| `pricing.ts` | USD rates |
|
||||
| `commands.ts` | slash registry, parsing, menu matching |
|
||||
| `headless.ts` | `-p` mode |
|
||||
| `cli.tsx` | argv, wiring, lifecycle |
|
||||
| `ui/*` | Ink components |
|
||||
|
||||
Every module is pure of the UI except `ui/`, and `ui/` never touches the SDK. The seam is the
|
||||
`AgentEvent` stream.
|
||||
|
||||
## Testing
|
||||
|
||||
404 tests, no mocking framework. `MockLanguageModelV4` from `ai/test` drives the loop;
|
||||
`ink-testing-library` drives the UI with real keystrokes; MCP is tested against a real stdio
|
||||
server subprocess; provider wire formats are tested against a local HTTP server.
|
||||
|
||||
The pattern throughout is to assert on what actually crossed a boundary — what went on the
|
||||
wire, what is on screen, what is on disk — rather than on internal calls.
|
||||
@@ -0,0 +1,140 @@
|
||||
# Configuration
|
||||
|
||||
Settings come from three places. Later wins:
|
||||
|
||||
1. `~/.shiro-neko/config.json`
|
||||
2. environment variables
|
||||
3. command-line flags
|
||||
|
||||
## The config file
|
||||
|
||||
Written by `/provider`, editable by hand. Every field is optional.
|
||||
|
||||
```json
|
||||
{
|
||||
"provider": "openai",
|
||||
"model": "gpt-5",
|
||||
"baseURL": "https://api.openai.com/v1",
|
||||
"apiKey": "sk-...",
|
||||
"presetId": "openai",
|
||||
"agent": "default",
|
||||
"thinking": "medium",
|
||||
"maxRetries": 3,
|
||||
"plugins": ["guard", "time"],
|
||||
"mcpServers": {
|
||||
"fs": { "command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", "."] }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
| Field | Meaning |
|
||||
|---|---|
|
||||
| `provider` | wire protocol: `anthropic` or `openai`. Not the vendor — Groq, OpenRouter, and Ollama all speak `openai` |
|
||||
| `model` | model id as the endpoint names it |
|
||||
| `baseURL` | API root. Defaults to the official endpoint for the provider |
|
||||
| `apiKey` | sent as `Authorization: Bearer` for `openai`, `x-api-key` for `anthropic` |
|
||||
| `presetId` | which preset `/provider` chose, so it can show what is configured |
|
||||
| `agent` | default variant: `default`, `quick`, `deep`, `plan`, `review` |
|
||||
| `thinking` | default level: `off`, `low`, `medium`, `high`, `max` |
|
||||
| `maxRetries` | retries per model call for transient failures. Default 3 |
|
||||
| `plugins` | which plugins to enable. Omit for `["guard", "time"]` |
|
||||
| `mcpServers` | see [MCP](mcp.md) |
|
||||
|
||||
## Provider presets
|
||||
|
||||
`/provider` offers these. Each sets `baseURL` and the wire protocol for you.
|
||||
|
||||
| Preset | Protocol | Endpoint |
|
||||
|---|---|---|
|
||||
| Anthropic | `anthropic` | `api.anthropic.com/v1` |
|
||||
| OpenAI | `openai` | `api.openai.com/v1` |
|
||||
| OpenRouter | `openai` | `openrouter.ai/api/v1` |
|
||||
| Groq | `openai` | `api.groq.com/openai/v1` |
|
||||
| DeepSeek | `openai` | `api.deepseek.com/v1` |
|
||||
| xAI | `openai` | `api.x.ai/v1` |
|
||||
| Ollama | `openai` | `localhost:11434/v1` |
|
||||
| LM Studio | `openai` | `localhost:1234/v1` |
|
||||
| Custom OpenAI-compatible | `openai` | you supply it |
|
||||
| Custom Anthropic-compatible | `anthropic` | you supply it |
|
||||
|
||||
After the key is entered, `GET /v1/models` is called and the list becomes a picker. If the
|
||||
endpoint does not implement it, you type the model id instead — the setup still completes.
|
||||
|
||||
## Environment variables
|
||||
|
||||
| Variable | Effect |
|
||||
|---|---|
|
||||
| `SHIRO_PROVIDER` | overrides `provider` |
|
||||
| `SHIRO_MODEL` | overrides `model` |
|
||||
| `SHIRO_BASE_URL` | overrides `baseURL` |
|
||||
| `SHIRO_API_KEY` | overrides `apiKey` |
|
||||
| `ANTHROPIC_API_KEY` | used when `provider` is `anthropic` and no key is set |
|
||||
| `OPENAI_API_KEY` | used when `provider` is `openai` and no key is set |
|
||||
| `SHIRO_HOME` | relocates config, sessions, memory, history, and user skills |
|
||||
| `SHIRO_INSTALL_DIR` | where `install:local` and the installers put the binary |
|
||||
| `SHIRO_REPO` | which GitHub repo the installers download from |
|
||||
| `SHIRO_VERSION` | pins the version the installers fetch |
|
||||
|
||||
`SHIRO_HOME` is what the test suite uses to keep a run out of your real config.
|
||||
|
||||
## Flags
|
||||
|
||||
```
|
||||
shiro [options]
|
||||
shiro -p "prompt" headless, prints to stdout
|
||||
cat file | shiro -p prompt read from stdin
|
||||
```
|
||||
|
||||
| Flag | Effect |
|
||||
|---|---|
|
||||
| `-p`, `--print [prompt]` | headless mode. Needs `--yolo` for tool use |
|
||||
| `--json` | with `-p`, one JSON event per line |
|
||||
| `-c`, `--continue` | resume the newest session for this directory |
|
||||
| `-r`, `--resume <id>` | resume by session id or unique prefix |
|
||||
| `--agent <name>` | `default`, `quick`, `deep`, `plan`, `review` |
|
||||
| `--think <level>` | `off`, `low`, `medium`, `high`, `max` |
|
||||
| `--provider <name>` | `anthropic` or `openai` |
|
||||
| `--model <id>` | model id |
|
||||
| `--base-url <url>` | API root |
|
||||
| `--no-mcp` | skip MCP servers |
|
||||
| `--no-subagent` | omit the `task` tool |
|
||||
| `--no-instructions` | ignore `AGENTS.md` and friends |
|
||||
| `--no-skills` | ignore builtin and project skills |
|
||||
| `--no-plugins` | disable all plugins, including the guard |
|
||||
| `--no-memory` | do not load or write project memory |
|
||||
| `--yolo` | skip every approval prompt |
|
||||
| `-v`, `--version` | version, bun version, platform, source or compiled |
|
||||
| `-h`, `--help` | usage |
|
||||
|
||||
## Where things live
|
||||
|
||||
```
|
||||
~/.shiro-neko/
|
||||
config.json provider, model, key, defaults
|
||||
sessions/<uuid>.json transcripts, token counts, cost, task list
|
||||
memory/<hash>.json durable per-project notes
|
||||
history/<hash>.json prompt history for up-arrow recall
|
||||
skills/*.md your own skills
|
||||
```
|
||||
|
||||
Project files:
|
||||
|
||||
```
|
||||
<project>/
|
||||
AGENTS.md instructions injected into the system prompt
|
||||
.shiro/skills/*.md project skills, override user and builtin
|
||||
.shiroignore extra ignore rules on top of .gitignore
|
||||
```
|
||||
|
||||
Memory and history file names are SHA-256 prefixes of the absolute project path, because a
|
||||
path is not a safe filename.
|
||||
|
||||
## OpenAI reasoning models
|
||||
|
||||
Newer OpenAI models reject function tools on `/v1/chat/completions` and require
|
||||
`/v1/responses`. For `api.openai.com` both are chained: a 400, 404, 405, 415, 422, or 501
|
||||
on the first switches to the second, sticks for the rest of the session, and prints one
|
||||
notice. Retryable failures — 429 and 5xx — are left to the SDK's backoff instead.
|
||||
|
||||
Third-party endpoints get a plain chat-completions model with no fallback probe, since they
|
||||
do not implement `/v1/responses`.
|
||||
@@ -0,0 +1,158 @@
|
||||
# Development
|
||||
|
||||
## Setup
|
||||
|
||||
```bash
|
||||
git clone https://github.com/zakirkun/shiro-neko
|
||||
cd shiro-neko
|
||||
bun install
|
||||
```
|
||||
|
||||
Bun 1.3.14 or newer. Nothing else is required, though `rg` on PATH makes `grep` about 15x
|
||||
faster and the fallback path is exercised without it.
|
||||
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
bun run shiro # run from source
|
||||
bun run typecheck # tsc --noEmit
|
||||
bun test # 404 tests
|
||||
bun run build # single binary for this platform -> dist/shiro
|
||||
bun run release # all five platforms -> dist/release + SHA256SUMS
|
||||
bun run install:local # build, then copy onto PATH
|
||||
```
|
||||
|
||||
`bun run install:local` copies the compiled binary. Do not use `bun link`: it writes a shim
|
||||
that re-execs `bun`, which fails on any machine where bun was installed without `bun.exe` on
|
||||
PATH — an npm install of bun, for instance. The compiled binary embeds its own runtime.
|
||||
|
||||
`SHIRO_INSTALL_DIR` overrides the target directory.
|
||||
|
||||
## Testing
|
||||
|
||||
No mocking framework. Everything is driven through a real boundary.
|
||||
|
||||
```ts
|
||||
// The loop: a mock provider, asserting what crossed the wire.
|
||||
const seen: LanguageModelV4CallOptions[] = [];
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async (o) => { seen.push(o); return stream(text('ok')); },
|
||||
});
|
||||
const session = new Session({ model, askApproval: async () => 'deny', agent: variantByName('plan') });
|
||||
for await (const _ of session.send('investigate')) void _;
|
||||
|
||||
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
|
||||
expect(offered).not.toContain('write_file');
|
||||
```
|
||||
|
||||
```ts
|
||||
// The UI: real keystrokes, asserting what is on screen.
|
||||
const app = render(<App session={session} bridge={bridge} hooks={testHooks()} header="hdr" />);
|
||||
app.stdin.write('/');
|
||||
await wait(120);
|
||||
expect(app.lastFrame()).toContain('/compact');
|
||||
```
|
||||
|
||||
Provider wire formats are tested against a local `Bun.serve`. MCP is tested against a real
|
||||
stdio subprocess. Tools are tested in a temp directory with `process.chdir`.
|
||||
|
||||
`SHIRO_HOME` points config, sessions, memory, and history at a temp directory, so a test run
|
||||
never touches your real state.
|
||||
|
||||
### What to assert
|
||||
|
||||
Assert on what crossed a boundary: the request body, the rendered frame, the file on disk.
|
||||
Not on internal calls.
|
||||
|
||||
That is not style. Several real bugs were caught this way and would have passed a
|
||||
mock-verification test:
|
||||
|
||||
- `pruneMessages` leaving a message item without its reasoning item — visible only in the
|
||||
request body
|
||||
- `--json` serialising `Error` as `{}` — visible only in the printed output
|
||||
- Automatic approval requests prompting the user — visible only in the event sequence
|
||||
|
||||
## Adding a tool
|
||||
|
||||
1. Define it in `src/tools.ts` with a `zod` schema. Descriptions are read by the model, so
|
||||
write them as guidance, not as documentation.
|
||||
2. Add it to the `tools` object.
|
||||
3. If it mutates anything, add it to `MUTATING_TOOLS` so it requires approval.
|
||||
4. Add a line to `TOOL_DOCS` in `src/prompt.ts` saying *when* to reach for it.
|
||||
5. Test the behaviour in a temp directory, including the failure path.
|
||||
|
||||
Every tool costs roughly 550 characters of schema on every request. Thirteen live tools is
|
||||
already where selection accuracy starts to matter, so a new tool needs to earn its place —
|
||||
see [ROADMAP.md](../ROADMAP.md) for what has been declined and why.
|
||||
|
||||
## Adding a slash command
|
||||
|
||||
`src/commands.ts` is the single source of truth. Add a `CommandSpec` to `COMMANDS`, a case to
|
||||
`parseCommand`, and a case in `App.tsx`. The menu, `/help`, and the parser all read from that
|
||||
one array, and a test asserts every entry parses and appears in help — they cannot drift.
|
||||
|
||||
## Code conventions
|
||||
|
||||
Sample a neighbouring file before inventing a pattern. Broadly:
|
||||
|
||||
- No comment that restates the code. Comments explain *why*, and usually only where something
|
||||
non-obvious was forced by an external constraint.
|
||||
- No `as any`, no `@ts-ignore`. `tsconfig.json` runs strict with
|
||||
`noUncheckedIndexedAccess`.
|
||||
- Validate at trust boundaries — model output, file contents, network responses. Not between
|
||||
internal functions.
|
||||
- Duplication over premature abstraction. No interface with one implementation.
|
||||
- Errors carry what the reader needs to act. `oldString appears 3 times in src/x.ts` beats
|
||||
`edit failed`.
|
||||
|
||||
## Releasing
|
||||
|
||||
The version lives in `src/version.ts`, compiled into the binary. `package.json` carries it too
|
||||
for tooling, and `bun run release` refuses to build if the two disagree, or if a git tag
|
||||
disagrees with either:
|
||||
|
||||
```
|
||||
$ GITHUB_REF_NAME=v9.9.9 bun run release
|
||||
tag v9.9.9 does not match src/version.ts (0.1.0-beta.1). Bump the version or retag.
|
||||
```
|
||||
|
||||
A binary reporting the wrong version is worse than a failed release.
|
||||
|
||||
To cut one:
|
||||
|
||||
```bash
|
||||
# bump src/version.ts and package.json to the same value
|
||||
git commit -am "release 0.1.0-beta.2"
|
||||
git tag v0.1.0-beta.2
|
||||
git push --follow-tags
|
||||
```
|
||||
|
||||
`.github/workflows/release.yml` then runs typecheck and tests, cross-compiles all five
|
||||
targets on one Ubuntu runner, asserts the built binary reports the expected version, and
|
||||
publishes a GitHub release with the binaries and `SHA256SUMS`. A tag containing `-` is
|
||||
published as a prerelease.
|
||||
|
||||
Bun cross-compiles from any host, which is why there is no build matrix. Verified: a working
|
||||
`darwin-arm64` binary builds on Windows.
|
||||
|
||||
Publishing is gated on a `v*` tag, so a manual `workflow_dispatch` run produces artifacts
|
||||
without releasing.
|
||||
|
||||
## CI
|
||||
|
||||
`.github/workflows/ci.yml` runs typecheck, tests, and a build on Ubuntu, macOS, and Windows
|
||||
for every push and PR.
|
||||
|
||||
All three are necessary. The tools shell out to `rg`, `git`, and a platform shell, and path
|
||||
handling differs — a Windows-only break is invisible on Linux until someone hits it.
|
||||
|
||||
## Debugging the agent itself
|
||||
|
||||
`--no-plugins --no-skills --no-memory --no-instructions --no-subagent --no-mcp` strips it to
|
||||
the seven core tools, which isolates whether a problem is the loop or something layered on it.
|
||||
|
||||
`--json` in headless mode shows the exact event sequence.
|
||||
|
||||
For provider issues, a local `Bun.serve` that logs the request body and returns a canned SSE
|
||||
stream answers "what did we actually send" faster than any amount of reading. Several bugs in
|
||||
this codebase were found that way.
|
||||
@@ -0,0 +1,128 @@
|
||||
# Headless mode
|
||||
|
||||
`-p` runs one prompt without the TUI. For scripts, CI, and piping.
|
||||
|
||||
```bash
|
||||
shiro -p "list every route and its handler"
|
||||
git diff | shiro -p "review this diff" --yolo
|
||||
shiro -p "fix the failing test" --yolo --agent deep
|
||||
```
|
||||
|
||||
The prompt comes from the argument, or from stdin when the argument is omitted.
|
||||
|
||||
## Tool use needs `--yolo`
|
||||
|
||||
There is no terminal to approve on, so every gated tool is denied unless `--yolo` is passed:
|
||||
|
||||
```
|
||||
$ shiro -p "add a test for paginate()"
|
||||
shiro: headless denies write_file, edit_file, bash and mcp tools unless --yolo is passed
|
||||
[tool] write_file {"path":"test/paginate.test.ts",...}
|
||||
[denied] write_file (run with --yolo to allow tool use in headless mode)
|
||||
```
|
||||
|
||||
Read-only tools work either way, so `-p` without `--yolo` is a safe way to ask questions
|
||||
about a codebase from a script.
|
||||
|
||||
**`--yolo` does not disable plugin guards.** `rm -rf` is still refused.
|
||||
|
||||
## Output
|
||||
|
||||
### Text mode (default)
|
||||
|
||||
Assistant text to stdout, everything else to stderr. Pipe-friendly:
|
||||
|
||||
```bash
|
||||
shiro -p "one-line summary of src/session.ts" > summary.txt
|
||||
```
|
||||
|
||||
```
|
||||
$ shiro -p "what does prune.ts do?" 2>/dev/null
|
||||
src/prune.ts repairs provider-item dependencies after pruneMessages strips reasoning items.
|
||||
```
|
||||
|
||||
### JSON mode
|
||||
|
||||
`--json` emits one event per line:
|
||||
|
||||
```bash
|
||||
$ shiro -p "count the tools" --json
|
||||
{"type":"tool-call","id":"c1","name":"grep","input":{"pattern":"tool\\("}}
|
||||
{"type":"tool-result","id":"c1","name":"grep","output":"src/tools.ts:26: ..."}
|
||||
{"type":"text","text":"There are 6 built-in file and shell tools."}
|
||||
{"type":"done","inputTokens":4210,"outputTokens":88}
|
||||
```
|
||||
|
||||
Event types: `text`, `reasoning`, `tool-call`, `tool-output`, `tool-result`, `tool-error`,
|
||||
`tool-denied`, `compacted`, `notice`, `error`, `done`.
|
||||
|
||||
Errors are flattened to message strings, because `JSON.stringify` turns an `Error` into `{}`
|
||||
and a JSON stream that reports failures as empty objects is useless for the one case it
|
||||
matters.
|
||||
|
||||
## Exit codes
|
||||
|
||||
`0` on success, `1` on a model or stream error. A denied tool is not a failure — the model was
|
||||
told and can respond to it.
|
||||
|
||||
```bash
|
||||
if shiro -p "does this build?" --yolo; then echo ok; else echo failed; fi
|
||||
```
|
||||
|
||||
## Sessions
|
||||
|
||||
Headless runs save like interactive ones, so `-c` picks up where one left off:
|
||||
|
||||
```bash
|
||||
shiro -p "start the refactor" --yolo
|
||||
shiro -p "now update the tests" --yolo -c
|
||||
```
|
||||
|
||||
## What is withheld
|
||||
|
||||
The `ask` tool is not offered at all, rather than being offered and left to hang. The model
|
||||
is told to decide and state its assumption instead.
|
||||
|
||||
Subagent progress events are not emitted; the report still comes back.
|
||||
|
||||
## CI recipes
|
||||
|
||||
Review a pull request diff:
|
||||
|
||||
```yaml
|
||||
- run: |
|
||||
git diff origin/main...HEAD > /tmp/diff
|
||||
shiro -p "Review this diff. Report defects with file and line. Say so if it is clean." \
|
||||
--agent review < /tmp/diff
|
||||
```
|
||||
|
||||
`--agent review` is read-only, so no `--yolo` is needed and nothing can be modified.
|
||||
|
||||
Fail the build on a specific finding:
|
||||
|
||||
```yaml
|
||||
- run: |
|
||||
shiro -p "Does any handler skip input validation? Answer only YES or NO." --json \
|
||||
| jq -r 'select(.type=="text") | .text' | grep -qv YES
|
||||
```
|
||||
|
||||
Generate a changelog entry:
|
||||
|
||||
```yaml
|
||||
- run: |
|
||||
git log --oneline "$(git describe --tags --abbrev=0)"..HEAD \
|
||||
| shiro -p "Write a changelog entry from these commits. Group by user-facing change." \
|
||||
>> CHANGELOG.md
|
||||
```
|
||||
|
||||
Pass the key as a secret:
|
||||
|
||||
```yaml
|
||||
env:
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
```
|
||||
|
||||
## Cost control
|
||||
|
||||
Headless runs are unattended, so a runaway loop costs real money. `--agent quick` caps the
|
||||
step count at 12. There is no spend ceiling yet — see [ROADMAP.md](../ROADMAP.md).
|
||||
+92
@@ -0,0 +1,92 @@
|
||||
# MCP
|
||||
|
||||
[Model Context Protocol](https://modelcontextprotocol.io) servers contribute tools. Configure
|
||||
them in `~/.shiro-neko/config.json` and they appear alongside the builtins.
|
||||
|
||||
## Configuration
|
||||
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
"fs": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "@modelcontextprotocol/server-filesystem", "."]
|
||||
},
|
||||
"db": {
|
||||
"command": "python",
|
||||
"args": ["-m", "my_mcp_server"],
|
||||
"env": { "DATABASE_URL": "postgres://localhost/dev" },
|
||||
"cwd": "/home/you/tools"
|
||||
},
|
||||
"api": {
|
||||
"url": "http://localhost:3000/mcp",
|
||||
"type": "http",
|
||||
"headers": { "Authorization": "Bearer local-dev-token" }
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
**stdio** servers take `command`, and optionally `args`, `env`, `cwd`. The process is spawned
|
||||
at startup and closed on exit.
|
||||
|
||||
**Remote** servers take `url`, and optionally `type` (`http` or `sse`, default `http`) and
|
||||
`headers`.
|
||||
|
||||
## Naming
|
||||
|
||||
Tools arrive as `mcp__<server>__<tool>`. A server named `fs` exposing `read_file` becomes
|
||||
`mcp__fs__read_file`.
|
||||
|
||||
The namespace is not cosmetic. Two servers both exposing `search` would otherwise silently
|
||||
shadow each other, and the model would call one believing it was the other.
|
||||
|
||||
## Approval
|
||||
|
||||
**Every MCP tool requires approval on every call.** They are third-party code with unknown
|
||||
side effects, so they are treated like `bash` rather than like `read_file`. `a` whitelists
|
||||
one tool for the session.
|
||||
|
||||
`--yolo` skips these prompts, as it does for the builtins. Plugin guards still apply.
|
||||
|
||||
## Failure handling
|
||||
|
||||
A server that fails to start is reported and the session continues:
|
||||
|
||||
```
|
||||
shiro-neko 0.1.0-beta.1 openai/gpt-5 session 0193ab2c
|
||||
mcp: 4 tools
|
||||
mcp db failed: spawn python ENOENT
|
||||
```
|
||||
|
||||
Nothing else is lost — the other servers still load, the builtins still work. A missing
|
||||
Python interpreter should not stop you from editing a file.
|
||||
|
||||
`--no-mcp` skips them all.
|
||||
|
||||
## Inspecting
|
||||
|
||||
`/tools` lists everything offered this turn, MCP tools included. The system prompt describes
|
||||
them as a group:
|
||||
|
||||
```
|
||||
- mcp__api__query, mcp__fs__read_file: from MCP servers, named mcp__<server>__<tool>.
|
||||
Each needs approval; read its own description before calling.
|
||||
```
|
||||
|
||||
Their individual descriptions come from the server, so that is what the model reads before
|
||||
calling one.
|
||||
|
||||
## Cost
|
||||
|
||||
Each tool adds roughly 550 characters of schema to every request. A server exposing twenty
|
||||
tools costs about 2,750 tokens per turn, sent whether or not the model uses any of them.
|
||||
|
||||
Prefer servers with a focused tool set. If one exposes many tools you never use, it is worth
|
||||
finding a narrower server or writing one.
|
||||
|
||||
## Writing a server
|
||||
|
||||
Any MCP-compliant server works. A minimal stdio one needs three methods: `initialize`,
|
||||
`tools/list`, and `tools/call`. The test suite includes one at
|
||||
`test/fixtures/mcp-stub.ts` — about 50 lines, and useful as a starting point.
|
||||
+157
@@ -0,0 +1,157 @@
|
||||
# Memory and state
|
||||
|
||||
Four kinds of state, each with a different lifetime.
|
||||
|
||||
| State | Lives in | Survives |
|
||||
|---|---|---|
|
||||
| transcript | the message array | until compaction or `/clear` |
|
||||
| task list | the system prompt, rebuilt each step | pruning and `/compact` |
|
||||
| project memory | `~/.shiro-neko/memory/<hash>.json` | across sessions, forever |
|
||||
| session record | `~/.shiro-neko/sessions/<uuid>.json` | until you delete it |
|
||||
|
||||
The split exists because compaction is destructive. `pruneMessages` deletes tool results and
|
||||
`/compact` deletes the whole transcript, so anything recorded only in messages is lost
|
||||
exactly when a long task needs it most.
|
||||
|
||||
## Project memory
|
||||
|
||||
Durable notes about the codebase, injected at the start of every session.
|
||||
|
||||
### `remember`
|
||||
|
||||
```
|
||||
kind fact | decision | gotcha | command
|
||||
text one self-contained line
|
||||
```
|
||||
|
||||
- **fact** — how something is. "The API is versioned under `/v2`."
|
||||
- **decision** — what was chosen and why. "We use snake_case for DB columns; the ORM
|
||||
expects it."
|
||||
- **gotcha** — a trap. "The migration must run before the seed or the FK fails."
|
||||
- **command** — an invocation that works. "Tests run with `bun test`, not `npm test`."
|
||||
|
||||
Duplicates are refused. Text is capped at 400 characters, the store at 300 entries.
|
||||
|
||||
### `recall`
|
||||
|
||||
Every term must appear. A match increments that entry's hit count, which protects it from
|
||||
compaction later — an entry the agent actually uses is worth keeping verbatim.
|
||||
|
||||
### `forget`
|
||||
|
||||
Removes by substring, for a note that turned out wrong.
|
||||
|
||||
### What the agent sees
|
||||
|
||||
```
|
||||
What you learned about this project in earlier sessions. Trust it, but verify anything
|
||||
that contradicts what you can see in the code now:
|
||||
- (command) tests run with bun test, not npm test
|
||||
- (gotcha) the migration must run before the seed
|
||||
- (decision) snake_case for DB columns, the ORM expects it
|
||||
```
|
||||
|
||||
Top 20 by hit count, then recency. The "verify anything that contradicts" line matters:
|
||||
memory goes stale and a confidently wrong note is worse than none.
|
||||
|
||||
### Compacting
|
||||
|
||||
`/memory` has the model merge entries. Two rules make it safe:
|
||||
|
||||
- Entries with at least one recall are kept verbatim and never merged.
|
||||
- A model returning nothing parseable leaves the store untouched.
|
||||
|
||||
Without the second rule a bad response wipes everything the agent has learned.
|
||||
|
||||
`/notes` lists the store with hit counts. `--no-memory` disables loading and writing.
|
||||
|
||||
## Task list
|
||||
|
||||
`todo_write` replaces the whole list each call. Four states:
|
||||
|
||||
```
|
||||
tasks ##########.............. 1/4 1 blocked
|
||||
[x] read the pagination code
|
||||
[~] fix the boundary
|
||||
[ ] add a test
|
||||
[!] update the docs (no write access to the wiki)
|
||||
```
|
||||
|
||||
`blocked` requires a note saying what is blocking it. The tool warns when more than one task
|
||||
is `in_progress`, when nothing is `in_progress` while work remains, or when a `blocked` task
|
||||
has no note.
|
||||
|
||||
The list is re-rendered into the system prompt on **every step**, not once per turn — a
|
||||
`todo_write` on step one has to be visible to step two. It is saved with the session and
|
||||
restored by `-c` or `/resume`.
|
||||
|
||||
`/todos` shows it. The panel above the input shows it live.
|
||||
|
||||
## Sessions
|
||||
|
||||
Every turn autosaves, debounced 400 ms so a long tool loop does not hit the disk each step.
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "0193ab2c-…",
|
||||
"createdAt": "…", "updatedAt": "…",
|
||||
"cwd": "/home/you/project",
|
||||
"provider": "openai", "model": "gpt-5",
|
||||
"title": "why does the pagination test fail?",
|
||||
"inputTokens": 48210, "outputTokens": 3105,
|
||||
"costUsd": 0.0913,
|
||||
"notebook": { "todos": [ … ] },
|
||||
"messages": [ … ]
|
||||
}
|
||||
```
|
||||
|
||||
```bash
|
||||
shiro -c # newest session for this directory
|
||||
shiro -r 0193ab2c # by id or unique prefix
|
||||
```
|
||||
|
||||
```
|
||||
/sessions list the last 15
|
||||
/resume <id>
|
||||
/save write now instead of waiting for the debounce
|
||||
```
|
||||
|
||||
A corrupt session file is skipped rather than crashing the list.
|
||||
|
||||
## Compaction
|
||||
|
||||
Two mechanisms.
|
||||
|
||||
**Automatic**, at roughly 120k estimated tokens: `pruneMessages` strips reasoning and older
|
||||
tool calls from what goes on the wire. Local history is untouched, so the transcript on your
|
||||
screen stays complete. The turn reports it:
|
||||
|
||||
```
|
||||
context compacted: 192 messages pruned to 15 on the wire
|
||||
```
|
||||
|
||||
**Manual**, `/compact`: the model writes a summary — goal, files touched, decisions, commands
|
||||
and outcomes, what remains — and it replaces the transcript entirely.
|
||||
|
||||
### The pruning repair
|
||||
|
||||
`pruneMessages({ reasoning: 'all' })` strips a reasoning item and keeps the message item from
|
||||
the same response. The OpenAI responses API treats the message as a dependent of that
|
||||
reasoning item and rejects the request:
|
||||
|
||||
```
|
||||
400 Item 'msg_…' of type 'message' was provided without its required 'reasoning' item: 'rs_…'
|
||||
```
|
||||
|
||||
The two carry different ids, so they cannot be matched by id. What links them is the
|
||||
assistant message they arrived in — one message is one response. `src/prune.ts` drops the
|
||||
dependent parts of any turn whose reasoning was removed. That costs nothing, because pruning
|
||||
was already discarding those turns.
|
||||
|
||||
## Prompt history
|
||||
|
||||
Per-directory, capped at 200, deduplicated against the previous entry. Up and down in the
|
||||
input walk it; down past the newest restores what you were typing.
|
||||
|
||||
Stored at `~/.shiro-neko/history/<hash>.json`, where the hash is a SHA-256 prefix of the
|
||||
project path.
|
||||
+120
@@ -0,0 +1,120 @@
|
||||
# Plugins
|
||||
|
||||
A plugin extends the agent in four ways: it can add tools, mark tools auto-approved, block
|
||||
a tool call before it runs, and append to the system prompt. It can also run something after
|
||||
each turn.
|
||||
|
||||
Plugins are compiled into the binary. Loading them from disk is deliberately not supported
|
||||
yet — see [ROADMAP.md](../ROADMAP.md).
|
||||
|
||||
## Enabling
|
||||
|
||||
```json
|
||||
{ "plugins": ["guard", "time"] }
|
||||
```
|
||||
|
||||
That is also the default when the field is absent. `--no-plugins` disables all of them,
|
||||
including the guard. `/plugins` lists what is active and reports any name that did not
|
||||
resolve.
|
||||
|
||||
## The interface
|
||||
|
||||
```ts
|
||||
export type Plugin = {
|
||||
name: string;
|
||||
description: string;
|
||||
tools?: ToolSet;
|
||||
autoApprove?: readonly string[];
|
||||
beforeToolCall?: (ctx: { toolName: string; input: unknown; cwd: string }) => string | undefined | Promise<string | undefined>;
|
||||
afterTurn?: () => void | Promise<void>;
|
||||
appendix?: string;
|
||||
};
|
||||
```
|
||||
|
||||
`beforeToolCall` returning a string **blocks** the call, and the string is given to the model
|
||||
as the reason. Returning `undefined` allows it.
|
||||
|
||||
Two decisions worth knowing about:
|
||||
|
||||
**A throwing hook blocks.** A guard that crashes must fail closed. Treating an exception as
|
||||
"allow" would mean a bug in a security plugin silently disables it.
|
||||
|
||||
**Blocks are checked before approval.** `--yolo` skips prompts; it does not skip guards. A
|
||||
plugin block is a refusal, not a permission question.
|
||||
|
||||
## Builtins
|
||||
|
||||
### `guard` (default on)
|
||||
|
||||
Refuses irreversible shell commands outright. Approval alone is a weak defence here: a user
|
||||
holding `a` through a batch of edits will approve one of these without reading it.
|
||||
|
||||
| Pattern | Why |
|
||||
|---|---|
|
||||
| `rm -rf`, `rm -f` | recursive or forced delete |
|
||||
| `git reset --hard` | discards uncommitted work |
|
||||
| `git clean -f` | deletes untracked files |
|
||||
| `git push --force`, `-f` | rewrites remote history |
|
||||
| `git branch -D` | deletes a branch without a merge check |
|
||||
| `DROP TABLE`, `TRUNCATE` | destroys database data |
|
||||
| `mkfs`, `dd of=/dev/…` | writes to a raw device |
|
||||
| `chmod 777` | makes files world-writable |
|
||||
| `shutdown`, `reboot`, `halt` | affects the whole machine |
|
||||
| `:(){ :\|:& };:` | fork bomb |
|
||||
| `curl … \| sh`, `wget … \| sh` | pipes a download into a shell |
|
||||
|
||||
```
|
||||
Blocked by the guard plugin: refusing "rm -rf build" (recursive or forced delete).
|
||||
Ask the user to run it themselves if it is really needed.
|
||||
```
|
||||
|
||||
The model is told to relay the command rather than work around it. `rm build/one-file.js`,
|
||||
`git push origin feature`, and `git commit` all pass — the patterns target irreversibility,
|
||||
not the commands themselves.
|
||||
|
||||
### `time` (default on)
|
||||
|
||||
Adds `current_time`, returning ISO 8601 plus the local string. Auto-approved; it reads
|
||||
nothing. Useful because models are confidently wrong about the date.
|
||||
|
||||
### `bell` (opt in)
|
||||
|
||||
Writes `\u0007` to stderr when a turn ends. Off by default — a bell after every turn is
|
||||
intrusive, but it is genuinely useful when a turn takes minutes.
|
||||
|
||||
```json
|
||||
{ "plugins": ["guard", "time", "bell"] }
|
||||
```
|
||||
|
||||
## Writing one
|
||||
|
||||
Plugins live in `src/plugins-builtin.ts` and are registered in `BUILTIN_PLUGINS`.
|
||||
|
||||
```ts
|
||||
export const noSecretsPlugin: Plugin = {
|
||||
name: 'no-secrets',
|
||||
description: 'refuses to write files that look like credentials',
|
||||
appendix:
|
||||
'The no-secrets plugin refuses writes to .env and credential files. Ask the user to ' +
|
||||
'add secrets themselves rather than working around it.',
|
||||
beforeToolCall: ({ toolName, input }) => {
|
||||
if (toolName !== 'write_file' && toolName !== 'edit_file') return undefined;
|
||||
const path = String((input as { path?: unknown } | null)?.path ?? '');
|
||||
if (/(^|\/)\.env|credentials|\.pem$/.test(path)) {
|
||||
return `refusing to write ${path}; add secrets yourself`;
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
};
|
||||
```
|
||||
|
||||
Then add it to `BUILTIN_PLUGINS` and, if it should be on by default, `DEFAULT_ENABLED`.
|
||||
|
||||
Write the `appendix` whenever the plugin can block something. Without it the model hits a
|
||||
refusal it was never told about and tries to route around it.
|
||||
|
||||
## Ordering
|
||||
|
||||
Plugins run in the order they are enabled. The first `beforeToolCall` to block wins;
|
||||
later hooks are not consulted. `afterTurn` runs every hook, and one throwing does not stop
|
||||
the rest.
|
||||
+123
@@ -0,0 +1,123 @@
|
||||
# Skills
|
||||
|
||||
A skill is a markdown file with instructions for one kind of task. Only its name and
|
||||
description sit in the system prompt; the body is loaded on demand.
|
||||
|
||||
That split matters. Four bundled skills are 4,659 characters of body but 681 characters of
|
||||
catalogue. Putting every body in the prompt would cost that on every request, for
|
||||
instructions that are relevant to one turn in twenty.
|
||||
|
||||
## Format
|
||||
|
||||
```markdown
|
||||
---
|
||||
name: deploy
|
||||
description: Ship a release. Use when asked to deploy, cut a release, or publish a build.
|
||||
---
|
||||
|
||||
# Deploy
|
||||
|
||||
1. Confirm the tests pass. Do not deploy on a red suite.
|
||||
2. Tag with the version from `src/version.ts`, not by hand.
|
||||
3. Push the tag. CI builds and publishes.
|
||||
|
||||
Never deploy from a dirty working tree.
|
||||
```
|
||||
|
||||
`name` and `description` are both required; a file missing either is skipped. The
|
||||
description is what the model matches against, so write it as a trigger — "use when asked
|
||||
to X" — not as a summary.
|
||||
|
||||
## Where they load from
|
||||
|
||||
Three sources, later overriding earlier by name:
|
||||
|
||||
1. **builtin** — compiled into the binary
|
||||
2. **user** — `~/.shiro-neko/skills/*.md`
|
||||
3. **project** — `.shiro/skills/*.md`
|
||||
|
||||
A project skill named `debug` replaces the bundled one entirely. `/skills` shows what
|
||||
loaded and where each came from.
|
||||
|
||||
`--no-skills` skips all of them, builtin included.
|
||||
|
||||
## The bundled skills
|
||||
|
||||
**`debug`** — reproduce first, form three hypotheses, disprove them cheapest-first, fix the
|
||||
cause not the symptom, write a test that failed before. After two failed attempts: re-read
|
||||
the error literally and check whether the code you think is running is the code that is
|
||||
running.
|
||||
|
||||
**`review`** — severity order: incorrect behaviour, missing validation at trust boundaries,
|
||||
security, resource handling, then clarity. Say plainly when something is fine. Do not invent
|
||||
findings to look thorough.
|
||||
|
||||
**`refactor`** — establish a safety net first, move in small steps with tests green between
|
||||
each, do not fix bugs while refactoring, do not add abstraction for a single caller.
|
||||
|
||||
**`test`** — read two existing test files first and match them, assert on behaviour not
|
||||
implementation, never weaken an assertion to make a test pass, a flaky test is a shared-state
|
||||
problem and not something to retry around.
|
||||
|
||||
They are string constants in `src/skills-builtin.ts` rather than files, because
|
||||
`bun build --compile` only embeds modules reachable through imports. A directory of `.md`
|
||||
files would be missing from the shipped binary.
|
||||
|
||||
## How the agent uses one
|
||||
|
||||
The catalogue appears in the system prompt:
|
||||
|
||||
```
|
||||
Skills available through the skill tool. Load one when its description matches the task,
|
||||
before you start working, and follow it as if the user had written it:
|
||||
- debug: Track down a bug whose cause is not obvious. Use when a test fails for unclear...
|
||||
- refactor: Restructure code without changing behaviour. Use when asked to refactor...
|
||||
```
|
||||
|
||||
When the model calls `skill({ name: "debug" })` it gets the full body back and is told to
|
||||
follow it for this task. The call needs no approval — it reads nothing outside the binary.
|
||||
|
||||
## Writing a good one
|
||||
|
||||
Skills work when they encode what a newcomer to *your* project would get wrong. The bundled
|
||||
ones are generic on purpose; yours should not be.
|
||||
|
||||
Useful:
|
||||
|
||||
```markdown
|
||||
---
|
||||
name: migration
|
||||
description: Write or run a database migration. Use when the schema changes.
|
||||
---
|
||||
|
||||
Migrations live in `db/migrations/` and are timestamped, never renumbered.
|
||||
|
||||
Run `bun run db:migrate` locally first. Staging runs them automatically on deploy;
|
||||
production needs `bun run db:migrate --env=prod` by hand, after the deploy is green.
|
||||
|
||||
Never edit a migration that has run anywhere. Write a new one.
|
||||
```
|
||||
|
||||
Not useful:
|
||||
|
||||
```markdown
|
||||
---
|
||||
name: quality
|
||||
description: Write good code.
|
||||
---
|
||||
|
||||
Follow best practices. Write clean, maintainable code with good naming.
|
||||
```
|
||||
|
||||
The second costs tokens and changes nothing.
|
||||
|
||||
## Skill or AGENTS.md?
|
||||
|
||||
`AGENTS.md` is always in the prompt. A skill is loaded when its description matches.
|
||||
|
||||
Put standing facts in `AGENTS.md`: build commands, layout, conventions that apply to every
|
||||
change. Put task-specific procedure in a skill: how to deploy, how to add a migration, how
|
||||
this project debugs its worker queue.
|
||||
|
||||
If it applies to every turn, it belongs in `AGENTS.md`. If it applies to one kind of turn,
|
||||
make it a skill.
|
||||
+172
@@ -0,0 +1,172 @@
|
||||
# Tools
|
||||
|
||||
## The approval model
|
||||
|
||||
Three categories.
|
||||
|
||||
**Free.** Read-only, no prompt: `read_file`, `glob`, `grep`, `task`.
|
||||
|
||||
**Session tools.** Also free, because they touch the agent's own state rather than your
|
||||
files: `todo_write`, `remember`, `recall`, `forget`, `skill`, `ask`, and anything a plugin
|
||||
marks auto-approved.
|
||||
|
||||
**Gated.** Every call stops for a decision: `write_file`, `edit_file`, `bash`, and every
|
||||
`mcp__*` tool.
|
||||
|
||||
```
|
||||
edit_file wants to run
|
||||
src/users.ts +2 -1
|
||||
export function paginate(offset: number, total: number) {
|
||||
- if (offset < total) return next();
|
||||
+ if (offset <= total) return next();
|
||||
}
|
||||
y allow once | a always allow edit_file | n deny
|
||||
```
|
||||
|
||||
`a` whitelists that tool for the rest of the session. `n` tells the model it was denied and
|
||||
to ask what to do instead. `--yolo` skips all prompts.
|
||||
|
||||
**The guard runs before all of this.** It is not an approval — it is a refusal, and `--yolo`
|
||||
does not reach it. See [plugins](plugins.md).
|
||||
|
||||
## File tools
|
||||
|
||||
### `read_file`
|
||||
|
||||
```
|
||||
path file path relative to the workspace root
|
||||
offset first line, 1-based
|
||||
limit max lines, default 2000
|
||||
```
|
||||
|
||||
Returns contents with 1-based line numbers. Refuses binaries: a NUL byte in the first 8 KB
|
||||
means the file is not text, and a model that reads a 90 MB executable has burned its whole
|
||||
context on nothing.
|
||||
|
||||
### `write_file`
|
||||
|
||||
```
|
||||
path file path
|
||||
content full contents
|
||||
```
|
||||
|
||||
New files and full rewrites only. Creates parent directories.
|
||||
|
||||
### `edit_file`
|
||||
|
||||
```
|
||||
path file path
|
||||
oldString exact text to find, whitespace and indentation included
|
||||
newString replacement
|
||||
replaceAll replace every occurrence instead of requiring exactly one
|
||||
```
|
||||
|
||||
`oldString` must match byte-for-byte and appear exactly once unless `replaceAll` is set.
|
||||
An ambiguous match is an error naming the count, which pushes the model to add surrounding
|
||||
context rather than guessing which occurrence it meant.
|
||||
|
||||
### `glob`
|
||||
|
||||
```
|
||||
pattern e.g. "src/**/*.ts"
|
||||
limit max paths, default 200
|
||||
includeIgnored also return files git ignores
|
||||
```
|
||||
|
||||
Walks the tree honouring `.gitignore` and `.shiroignore`, skipping `.git` and
|
||||
`node_modules` unconditionally. Nested ignore files apply only within their own directory,
|
||||
as git does. Returns posix paths relative to the workspace root.
|
||||
|
||||
### `grep`
|
||||
|
||||
```
|
||||
pattern regex source
|
||||
include glob limiting the search, default "**/*"
|
||||
ignoreCase case-insensitive
|
||||
includeIgnored also search files git ignores
|
||||
```
|
||||
|
||||
Shells out to ripgrep when it is on PATH — roughly 15x faster on a real repo — and falls
|
||||
back to a JavaScript walker otherwise. Output is `path:line: text` either way, so the model
|
||||
sees one format regardless. Skips binaries. Caps at 200 hits.
|
||||
|
||||
### `bash`
|
||||
|
||||
```
|
||||
command shell command
|
||||
timeout ms, default 120000, max 600000
|
||||
```
|
||||
|
||||
Runs in the workspace root through `bash -lc` or `cmd /c`. Output streams live to the panel
|
||||
above the input rather than appearing all at once when the command exits — a two-minute test
|
||||
run is otherwise indistinguishable from a hang. Both pipes are drained concurrently, since a
|
||||
command that fills one while you block on the other deadlocks.
|
||||
|
||||
Returns exit code, stdout, stderr, and a note if a signal killed it.
|
||||
|
||||
## Agent tools
|
||||
|
||||
### `task`
|
||||
|
||||
```
|
||||
description short label shown to you
|
||||
prompt self-contained instructions
|
||||
kind "explore" (default) or "review"
|
||||
```
|
||||
|
||||
Spawns a read-only subagent with `read_file`, `glob`, and `grep` only. It returns one
|
||||
report, so the parent pays for findings rather than the whole search transcript. It sees
|
||||
none of the parent conversation, so its prompt has to stand alone.
|
||||
|
||||
`explore` finds and reports. `review` critiques code in severity order. Progress streams to
|
||||
the subagent panel.
|
||||
|
||||
### `ask`
|
||||
|
||||
```
|
||||
question one specific question
|
||||
options choices, recommendation first, each with an optional detail
|
||||
multiple allow more than one
|
||||
```
|
||||
|
||||
Stops the turn and puts the question on screen. With options it is a picker; without, free
|
||||
text. `esc` skips, which tells the model to decide and state its assumption.
|
||||
|
||||
Withheld entirely in headless mode — a question with no one to answer it would hang.
|
||||
|
||||
### `todo_write`
|
||||
|
||||
```
|
||||
todos the complete list: content, status, optional note
|
||||
```
|
||||
|
||||
Statuses: `pending`, `in_progress`, `done`, `blocked`. Send the whole list each time; it
|
||||
replaces the previous one. Warns when more than one task is `in_progress`, when nothing is
|
||||
`in_progress` while work remains, or when a `blocked` task has no note.
|
||||
|
||||
### `remember`, `recall`, `forget`
|
||||
|
||||
Durable per-project notes. See [memory](memory.md).
|
||||
|
||||
### `skill`
|
||||
|
||||
```
|
||||
name skill name from the catalogue
|
||||
```
|
||||
|
||||
Loads the body of a skill. See [skills](skills.md).
|
||||
|
||||
## Path safety
|
||||
|
||||
Every path a tool receives goes through a jail: resolved against the workspace root, then
|
||||
checked that it did not escape. `../../etc/passwd` and absolute paths outside the root are
|
||||
both refused before any filesystem call.
|
||||
|
||||
The model's output is a trust boundary. It can emit any string, so the check happens on
|
||||
every call rather than being assumed.
|
||||
|
||||
## Output caps
|
||||
|
||||
Any single tool result is truncated at 30,000 characters with a note saying how much was
|
||||
cut. `grep` stops at 200 hits, `glob` at 200 paths, `read_file` at 2000 lines by default.
|
||||
Without caps one `grep` for `function` can end a session.
|
||||
@@ -0,0 +1,39 @@
|
||||
{
|
||||
"name": "shiro-neko",
|
||||
"version": "0.1.0-beta.1",
|
||||
"type": "module",
|
||||
"private": true,
|
||||
"bin": {
|
||||
"shiro": "./src/cli.tsx"
|
||||
},
|
||||
"scripts": {
|
||||
"shiro": "bun run src/cli.tsx",
|
||||
"typecheck": "tsc --noEmit",
|
||||
"test": "bun test",
|
||||
"build": "bun build --compile --minify src/cli.tsx --outfile dist/shiro",
|
||||
"release": "bun run scripts/release.ts",
|
||||
"install:local": "bun run build && bun run scripts/install.ts"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/bun": "^1.4.0",
|
||||
"@types/react": "19.2.18",
|
||||
"ink-testing-library": "4.0.0",
|
||||
"react-devtools-core": "^7.0.1"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"typescript": "^5"
|
||||
},
|
||||
"dependencies": {
|
||||
"@ai-sdk/anthropic": "4.0.46",
|
||||
"@ai-sdk/mcp": "2.0.41",
|
||||
"@ai-sdk/openai": "4.0.53",
|
||||
"@ai-sdk/openai-compatible": "3.0.41",
|
||||
"ai": "7.0.87",
|
||||
"ink": "7.1.1",
|
||||
"ink-select-input": "6.2.0",
|
||||
"ink-spinner": "^5.0.0",
|
||||
"ink-text-input": "6.0.0",
|
||||
"react": "19.2.8",
|
||||
"zod": "4.5.4"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
# Installs shiro-neko from a GitHub release.
|
||||
#
|
||||
# irm https://raw.githubusercontent.com/zakirkun/shiro-neko/main/scripts/install.ps1 | iex
|
||||
#
|
||||
# Set $env:SHIRO_VERSION to pin a version, $env:SHIRO_INSTALL_DIR to change the target.
|
||||
$ErrorActionPreference = 'Stop'
|
||||
|
||||
$repo = if ($env:SHIRO_REPO) { $env:SHIRO_REPO } else { 'zakirkun/shiro-neko' }
|
||||
$installDir = if ($env:SHIRO_INSTALL_DIR) { $env:SHIRO_INSTALL_DIR } else { Join-Path $HOME '.bun\bin' }
|
||||
|
||||
if ([Environment]::Is64BitOperatingSystem -ne $true) {
|
||||
Write-Error 'shiro: only 64-bit Windows is supported'
|
||||
}
|
||||
$asset = 'shiro-windows-x64.exe'
|
||||
|
||||
$base = if ($env:SHIRO_VERSION) {
|
||||
"https://github.com/$repo/releases/download/v$($env:SHIRO_VERSION -replace '^v','')"
|
||||
} else {
|
||||
"https://github.com/$repo/releases/latest/download"
|
||||
}
|
||||
|
||||
$tmp = Join-Path $env:TEMP "shiro-install-$(Get-Random)"
|
||||
New-Item -ItemType Directory -Force -Path $tmp | Out-Null
|
||||
|
||||
try {
|
||||
Write-Host "downloading $asset from $base"
|
||||
$downloaded = Join-Path $tmp $asset
|
||||
Invoke-WebRequest -Uri "$base/$asset" -OutFile $downloaded -UseBasicParsing
|
||||
|
||||
# Verify against the published checksums when they are available; a corrupted
|
||||
# 90 MB download otherwise fails later as an unexplained crash.
|
||||
try {
|
||||
$sums = (Invoke-WebRequest -Uri "$base/SHA256SUMS" -UseBasicParsing).Content
|
||||
$line = ($sums -split "`n" | Where-Object { $_ -match "\s$([regex]::Escape($asset))$" } | Select-Object -First 1)
|
||||
if ($line) {
|
||||
$expected = ($line -split '\s+')[0]
|
||||
$actual = (Get-FileHash -Path $downloaded -Algorithm SHA256).Hash.ToLower()
|
||||
if ($actual -ne $expected.ToLower()) {
|
||||
Write-Error 'shiro: checksum mismatch, refusing to install'
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
Write-Host 'no checksums published for this release, skipping verification'
|
||||
}
|
||||
|
||||
New-Item -ItemType Directory -Force -Path $installDir | Out-Null
|
||||
$target = Join-Path $installDir 'shiro.exe'
|
||||
Move-Item -Force -Path $downloaded -Destination $target
|
||||
|
||||
Write-Host "installed $target"
|
||||
& $target --version
|
||||
|
||||
$onPath = ($env:PATH -split ';' | Where-Object { $_ -and (Join-Path $_ '') -eq (Join-Path $installDir '') })
|
||||
if ($onPath) {
|
||||
Write-Host 'run: shiro'
|
||||
} else {
|
||||
Write-Host ''
|
||||
Write-Host "$installDir is not on PATH. Add it:"
|
||||
Write-Host " [Environment]::SetEnvironmentVariable('PATH', `"`$env:PATH;$installDir`", 'User')"
|
||||
}
|
||||
} finally {
|
||||
Remove-Item -Recurse -Force $tmp -ErrorAction SilentlyContinue
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
#!/usr/bin/env sh
|
||||
# Installs shiro-neko from a GitHub release.
|
||||
#
|
||||
# curl -fsSL https://raw.githubusercontent.com/zakirkun/shiro-neko/main/scripts/install.sh | sh
|
||||
#
|
||||
# Set SHIRO_VERSION to pin a version, SHIRO_INSTALL_DIR to change the target.
|
||||
set -eu
|
||||
|
||||
REPO="${SHIRO_REPO:-zakirkun/shiro-neko}"
|
||||
INSTALL_DIR="${SHIRO_INSTALL_DIR:-$HOME/.local/bin}"
|
||||
|
||||
case "$(uname -s)" in
|
||||
Linux) os=linux ;;
|
||||
Darwin) os=darwin ;;
|
||||
*) echo "shiro: unsupported OS $(uname -s). Windows users: use scripts/install.ps1" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
case "$(uname -m)" in
|
||||
x86_64 | amd64) arch=x64 ;;
|
||||
arm64 | aarch64) arch=arm64 ;;
|
||||
*) echo "shiro: unsupported architecture $(uname -m)" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
asset="shiro-${os}-${arch}"
|
||||
|
||||
if [ -n "${SHIRO_VERSION:-}" ]; then
|
||||
tag="v${SHIRO_VERSION#v}"
|
||||
base="https://github.com/${REPO}/releases/download/${tag}"
|
||||
else
|
||||
base="https://github.com/${REPO}/releases/latest/download"
|
||||
fi
|
||||
|
||||
tmp="$(mktemp -d)"
|
||||
trap 'rm -rf "$tmp"' EXIT
|
||||
|
||||
echo "downloading ${asset} from ${base}"
|
||||
if ! curl -fSL --progress-bar "${base}/${asset}" -o "${tmp}/shiro"; then
|
||||
echo "shiro: download failed. No build for ${os}-${arch} at that version?" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Verify against the published checksums when they are available; a corrupted
|
||||
# 90 MB download otherwise fails later as an unexplained crash.
|
||||
if curl -fsSL "${base}/SHA256SUMS" -o "${tmp}/SHA256SUMS" 2>/dev/null; then
|
||||
expected="$(grep " ${asset}\$" "${tmp}/SHA256SUMS" | cut -d' ' -f1)"
|
||||
if [ -n "$expected" ]; then
|
||||
if command -v sha256sum >/dev/null 2>&1; then
|
||||
actual="$(sha256sum "${tmp}/shiro" | cut -d' ' -f1)"
|
||||
elif command -v shasum >/dev/null 2>&1; then
|
||||
actual="$(shasum -a 256 "${tmp}/shiro" | cut -d' ' -f1)"
|
||||
else
|
||||
actual=""
|
||||
fi
|
||||
if [ -n "$actual" ] && [ "$actual" != "$expected" ]; then
|
||||
echo "shiro: checksum mismatch, refusing to install" >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
mkdir -p "$INSTALL_DIR"
|
||||
chmod +x "${tmp}/shiro"
|
||||
mv "${tmp}/shiro" "${INSTALL_DIR}/shiro"
|
||||
|
||||
echo "installed ${INSTALL_DIR}/shiro"
|
||||
"${INSTALL_DIR}/shiro" --version || true
|
||||
|
||||
case ":${PATH}:" in
|
||||
*":${INSTALL_DIR}:"*) echo "run: shiro" ;;
|
||||
*) echo ""; echo "${INSTALL_DIR} is not on PATH. Add it:"; echo " export PATH=\"${INSTALL_DIR}:\$PATH\"" ;;
|
||||
esac
|
||||
@@ -0,0 +1,58 @@
|
||||
import { chmodSync, mkdirSync } from 'node:fs';
|
||||
import { homedir, platform } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { VERSION } from '../src/version';
|
||||
|
||||
/**
|
||||
* Installs the compiled binary onto PATH.
|
||||
*
|
||||
* `bun link` is not usable here: it writes a shim that re-execs `bun`, so it fails
|
||||
* on any machine where bun is installed without `bun.exe` on PATH (an npm install,
|
||||
* for one). The compiled binary embeds its own runtime and has no such dependency.
|
||||
*/
|
||||
const isWindows = platform() === 'win32';
|
||||
const exe = isWindows ? 'shiro.exe' : 'shiro';
|
||||
const source = join(import.meta.dir, '..', 'dist', exe);
|
||||
|
||||
const target = (() => {
|
||||
const explicit = process.env['SHIRO_INSTALL_DIR'];
|
||||
if (explicit) return explicit;
|
||||
const bunBin = join(homedir(), '.bun', 'bin');
|
||||
return isWindows ? bunBin : join(homedir(), '.local', 'bin');
|
||||
})();
|
||||
|
||||
const built = Bun.file(source);
|
||||
if (!(await built.exists())) {
|
||||
console.error(`shiro: ${source} not found. Run "bun run build" first.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
mkdirSync(target, { recursive: true });
|
||||
const dest = join(target, exe);
|
||||
|
||||
try {
|
||||
await Bun.write(dest, built);
|
||||
} catch (e) {
|
||||
const message = e instanceof Error ? e.message : String(e);
|
||||
console.error(`shiro: could not write ${dest}: ${message}`);
|
||||
if (message.includes('EBUSY') || message.includes('EACCES') || message.includes('EPERM')) {
|
||||
console.error('shiro: a running shiro may be holding the file. Close it and try again.');
|
||||
}
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (!isWindows) chmodSync(dest, 0o755);
|
||||
|
||||
const onPath = (process.env['PATH'] ?? '').split(isWindows ? ';' : ':').some((p) => p && join(p) === join(target));
|
||||
|
||||
console.log(`installed shiro-neko ${VERSION} to ${dest} (${(built.size / 1024 / 1024) | 0} MB)`);
|
||||
if (onPath) {
|
||||
console.log('run: shiro');
|
||||
} else {
|
||||
console.log(`\n${target} is not on PATH. Add it:`);
|
||||
console.log(
|
||||
isWindows
|
||||
? ` [Environment]::SetEnvironmentVariable('PATH', "$env:PATH;${target}", 'User')`
|
||||
: ` export PATH="${target}:$PATH"`,
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,100 @@
|
||||
import { mkdirSync, rmSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { VERSION } from '../src/version';
|
||||
|
||||
export type Target = {
|
||||
/** Bun cross-compilation target. */
|
||||
target: string;
|
||||
/** Suffix in the artifact name, matching what the installers look for. */
|
||||
name: string;
|
||||
windows?: boolean;
|
||||
};
|
||||
|
||||
export const TARGETS: Target[] = [
|
||||
{ target: 'bun-linux-x64', name: 'linux-x64' },
|
||||
{ target: 'bun-linux-arm64', name: 'linux-arm64' },
|
||||
{ target: 'bun-darwin-x64', name: 'darwin-x64' },
|
||||
{ target: 'bun-darwin-arm64', name: 'darwin-arm64' },
|
||||
{ target: 'bun-windows-x64', name: 'windows-x64', windows: true },
|
||||
];
|
||||
|
||||
const OUT = 'dist/release';
|
||||
|
||||
/**
|
||||
* Builds one executable per platform.
|
||||
*
|
||||
* Bun cross-compiles from any host, so a single runner produces every artifact and
|
||||
* no build matrix is needed. The version is compiled into the binary from
|
||||
* src/version.ts; a release tag must agree with it or the build stops, because a
|
||||
* binary reporting the wrong version is worse than a failed release.
|
||||
*/
|
||||
async function main(): Promise<void> {
|
||||
const args = process.argv.slice(2);
|
||||
const only = args.filter((a) => !a.startsWith('-'));
|
||||
const wanted = only.length > 0 ? TARGETS.filter((t) => only.includes(t.name)) : TARGETS;
|
||||
if (wanted.length === 0) {
|
||||
console.error(`no matching target. Available: ${TARGETS.map((t) => t.name).join(', ')}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const pkg = (await Bun.file('package.json').json()) as { version?: string };
|
||||
if (pkg.version !== VERSION) {
|
||||
console.error(`version mismatch: package.json is ${pkg.version}, src/version.ts is ${VERSION}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const tag = (process.env['GITHUB_REF_NAME'] ?? '').replace(/^v/, '');
|
||||
if (tag && tag !== VERSION) {
|
||||
console.error(`tag v${tag} does not match src/version.ts (${VERSION}). Bump the version or retag.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
rmSync(OUT, { recursive: true, force: true });
|
||||
mkdirSync(OUT, { recursive: true });
|
||||
|
||||
const built: { file: string; bytes: number }[] = [];
|
||||
|
||||
for (const t of wanted) {
|
||||
const base = `shiro-${t.name}`;
|
||||
const outfile = join(OUT, base);
|
||||
|
||||
const buildArgs = ['build', '--compile', '--minify', `--target=${t.target}`, 'src/cli.tsx', '--outfile', outfile];
|
||||
if (t.windows) {
|
||||
buildArgs.push(
|
||||
'--windows-title=shiro-neko',
|
||||
'--windows-description=Agentic coding CLI',
|
||||
`--windows-version=${VERSION.split('-')[0]}.0`,
|
||||
);
|
||||
}
|
||||
|
||||
const proc = Bun.spawn(['bun', ...buildArgs], { stdout: 'inherit', stderr: 'inherit' });
|
||||
const code = await proc.exited;
|
||||
if (code !== 0) {
|
||||
console.error(`\nbuild failed for ${t.name} (exit ${code})`);
|
||||
process.exit(code);
|
||||
}
|
||||
|
||||
const file = t.windows ? `${base}.exe` : base;
|
||||
const artifact = Bun.file(join(OUT, file));
|
||||
if (!(await artifact.exists())) {
|
||||
console.error(`\n${file} was not produced`);
|
||||
process.exit(1);
|
||||
}
|
||||
built.push({ file, bytes: artifact.size });
|
||||
}
|
||||
|
||||
// Checksums let an installer verify a download without a second request.
|
||||
const sums: string[] = [];
|
||||
for (const { file } of built) {
|
||||
const bytes = new Uint8Array(await Bun.file(join(OUT, file)).arrayBuffer());
|
||||
sums.push(`${new Bun.CryptoHasher('sha256').update(bytes).digest('hex')} ${file}`);
|
||||
}
|
||||
await Bun.write(join(OUT, 'SHA256SUMS'), `${sums.join('\n')}\n`);
|
||||
|
||||
console.log(`\nshiro-neko ${VERSION}`);
|
||||
for (const { file, bytes } of built) console.log(` ${file.padEnd(26)} ${Math.round(bytes / 1024 / 1024)} MB`);
|
||||
console.log(` ${'SHA256SUMS'.padEnd(26)} ${built.length} entries`);
|
||||
}
|
||||
|
||||
// Guarded so a test can import TARGETS without triggering a five-platform build.
|
||||
if (import.meta.main) await main();
|
||||
+106
@@ -0,0 +1,106 @@
|
||||
export type ThinkingLevel = 'off' | 'low' | 'medium' | 'high' | 'max';
|
||||
|
||||
/** Maps our vocabulary to the SDK's, which each provider then maps to its own knob. */
|
||||
const SDK_REASONING: Record<ThinkingLevel, 'none' | 'low' | 'medium' | 'high' | 'xhigh'> = {
|
||||
off: 'none',
|
||||
low: 'low',
|
||||
medium: 'medium',
|
||||
high: 'high',
|
||||
max: 'xhigh',
|
||||
};
|
||||
|
||||
export const THINKING_LEVELS: ThinkingLevel[] = ['off', 'low', 'medium', 'high', 'max'];
|
||||
|
||||
export const isThinkingLevel = (v: string): v is ThinkingLevel => (THINKING_LEVELS as string[]).includes(v);
|
||||
|
||||
export const sdkReasoning = (level: ThinkingLevel) => SDK_REASONING[level];
|
||||
|
||||
export type AgentVariant = {
|
||||
name: string;
|
||||
summary: string;
|
||||
thinking: ThinkingLevel;
|
||||
/** Appended to the system prompt to shape behaviour. */
|
||||
appendix: string;
|
||||
/** When set, only these tools are offered. Omit to offer everything. */
|
||||
allowTools?: readonly string[];
|
||||
maxSteps?: number;
|
||||
};
|
||||
|
||||
const READ_ONLY = [
|
||||
'read_file',
|
||||
'glob',
|
||||
'grep',
|
||||
'list_dir',
|
||||
'task',
|
||||
'todo_write',
|
||||
'remember',
|
||||
'recall',
|
||||
'skill',
|
||||
] as const;
|
||||
|
||||
export const VARIANTS: AgentVariant[] = [
|
||||
{
|
||||
name: 'default',
|
||||
summary: 'balanced: full tools, medium thinking',
|
||||
thinking: 'medium',
|
||||
appendix: '',
|
||||
},
|
||||
{
|
||||
name: 'quick',
|
||||
summary: 'small edits: no thinking budget, act immediately',
|
||||
thinking: 'off',
|
||||
maxSteps: 12,
|
||||
appendix:
|
||||
'This is a small, well-scoped task. Do not deliberate: locate the code, make the change, verify it. ' +
|
||||
'Do not write a task list. Do not explore beyond what the change requires.',
|
||||
},
|
||||
{
|
||||
name: 'deep',
|
||||
summary: 'hard problems: maximum thinking, more steps',
|
||||
thinking: 'max',
|
||||
maxSteps: 80,
|
||||
appendix:
|
||||
'This task is hard or its cause is unclear. Form more than one hypothesis before you act and say which one ' +
|
||||
'you are testing. Read enough of the code to be sure rather than guessing. Record findings with remember ' +
|
||||
'so they survive compaction. Report what you verified and what you could not.',
|
||||
},
|
||||
{
|
||||
name: 'plan',
|
||||
summary: 'read-only: investigate and propose, never edit',
|
||||
thinking: 'high',
|
||||
allowTools: READ_ONLY,
|
||||
appendix:
|
||||
'You are in planning mode and have no tools that change anything. Investigate, then produce a plan: ' +
|
||||
'the files to touch, the change in each, the order, and how to verify. Flag anything ambiguous instead of ' +
|
||||
'assuming. Do not describe edits as if you had made them.',
|
||||
},
|
||||
{
|
||||
name: 'review',
|
||||
summary: 'read-only: critique a change, find defects',
|
||||
thinking: 'high',
|
||||
allowTools: READ_ONLY,
|
||||
appendix:
|
||||
'You are reviewing code, not writing it. Look for defects in this order: incorrect behaviour, missing error ' +
|
||||
'handling at trust boundaries, security issues, then clarity. For each finding give file, line, why it is ' +
|
||||
'wrong, and the fix. Say plainly when something is fine. Do not invent problems to fill a report.',
|
||||
},
|
||||
];
|
||||
|
||||
export const DEFAULT_VARIANT = VARIANTS[0]!;
|
||||
|
||||
export const variantByName = (name: string) => VARIANTS.find((v) => v.name === name);
|
||||
|
||||
/** Variant with an explicit thinking override applied, for `--agent deep --think low`. */
|
||||
export function resolveAgent(name: string | undefined, thinking: string | undefined): AgentVariant {
|
||||
const base = name ? variantByName(name) : DEFAULT_VARIANT;
|
||||
if (!base) throw new Error(`Unknown agent "${name}". Available: ${VARIANTS.map((v) => v.name).join(', ')}`);
|
||||
if (thinking === undefined) return base;
|
||||
if (!isThinkingLevel(thinking)) {
|
||||
throw new Error(`Unknown thinking level "${thinking}". Available: ${THINKING_LEVELS.join(', ')}`);
|
||||
}
|
||||
return { ...base, thinking };
|
||||
}
|
||||
|
||||
export function renderAgent(variant: AgentVariant): string {
|
||||
return variant.appendix ? `\n${variant.appendix}` : '';
|
||||
}
|
||||
+56
@@ -0,0 +1,56 @@
|
||||
import { tool } from 'ai';
|
||||
import { z } from 'zod';
|
||||
|
||||
export type AskRequest = {
|
||||
question: string;
|
||||
options?: { label: string; detail?: string }[];
|
||||
multiple: boolean;
|
||||
};
|
||||
|
||||
/** Set by the UI. Absent means nothing can answer, so asking is an error. */
|
||||
export type AskFn = (req: AskRequest) => Promise<string[] | undefined>;
|
||||
|
||||
const MAX_OPTIONS = 8;
|
||||
|
||||
/**
|
||||
* Lets the model stop and ask rather than guess.
|
||||
*
|
||||
* Without this a model facing two materially different readings of a request picks
|
||||
* one and writes code for it. The cost of a wrong guess is a whole wasted turn plus
|
||||
* the user's correction, so one question is almost always cheaper.
|
||||
*/
|
||||
export function createAskTool(ask: AskFn | undefined) {
|
||||
return tool({
|
||||
description:
|
||||
'Ask the user a question and wait for the answer. Use it when the request has two or more readings that ' +
|
||||
'lead to materially different work, when a required detail is missing, or to confirm an approach before a ' +
|
||||
'large change. Offer concrete options when you can; omit them for an open question. ' +
|
||||
'Do not use it for things you can determine by reading the code, and do not ask twice about the same thing.',
|
||||
inputSchema: z.object({
|
||||
question: z.string().describe('One specific question. State what you already know, then what you need.'),
|
||||
options: z
|
||||
.array(
|
||||
z.object({
|
||||
label: z.string().describe('Short choice, a few words'),
|
||||
detail: z.string().optional().describe('What choosing this implies, including any tradeoff'),
|
||||
}),
|
||||
)
|
||||
.max(MAX_OPTIONS)
|
||||
.optional()
|
||||
.describe('Concrete choices. Put your recommendation first. Omit for an open question.'),
|
||||
multiple: z.boolean().optional().describe('Allow more than one option to be chosen'),
|
||||
}),
|
||||
execute: async ({ question, options, multiple }) => {
|
||||
if (!ask) {
|
||||
throw new Error(
|
||||
'No one is available to answer: this session is running headless. Decide yourself and state the assumption.',
|
||||
);
|
||||
}
|
||||
const answers = await ask({ question, ...(options ? { options } : {}), multiple: multiple ?? false });
|
||||
if (!answers || answers.length === 0) return 'The user dismissed the question without answering. Proceed with your best judgement and say what you assumed.';
|
||||
return `The user answered: ${answers.join(', ')}`;
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export const ASK_TOOL_NAME = 'ask';
|
||||
+414
@@ -0,0 +1,414 @@
|
||||
#!/usr/bin/env bun
|
||||
import { render } from 'ink';
|
||||
import React from 'react';
|
||||
import type { LanguageModel, ModelMessage } from 'ai';
|
||||
import { resolveAgent, VARIANTS, isThinkingLevel, type AgentVariant } from './agents';
|
||||
import { configPath, loadConfig, missingKeyMessage, resolveModel, writeConfigFile, type Config } from './config';
|
||||
import type { FallbackEvent } from './fallback';
|
||||
import { readStdin, runHeadless } from './headless';
|
||||
import { INIT_PROMPT, loadInstructions } from './instructions';
|
||||
import { connectMcp } from './mcp';
|
||||
import { Memory, KIND_LABEL } from './memory';
|
||||
import { costOf } from './pricing';
|
||||
import { BUILTIN_PLUGINS, DEFAULT_ENABLED } from './plugins-builtin';
|
||||
import { createHost } from './plugins';
|
||||
import { fetchModels, presetById } from './providers';
|
||||
import { Session } from './session';
|
||||
import { loadSkills } from './skills';
|
||||
import * as store from './store';
|
||||
import { createTaskTool } from './subagent';
|
||||
import { VERSION, versionLine } from './version';
|
||||
import { createAskBridge } from './ui/Ask';
|
||||
import { App, createApprovalBridge, createNoticeBus, createSubagentBus, type AppHooks } from './ui/App';
|
||||
|
||||
// SDK warnings go straight to stderr, which tears up the Ink render.
|
||||
(globalThis as { AI_SDK_LOG_WARNINGS?: boolean }).AI_SDK_LOG_WARNINGS = false;
|
||||
|
||||
const HELP = `shiro-neko ${VERSION} - agentic coding CLI
|
||||
|
||||
usage: shiro [options]
|
||||
shiro -p "prompt" headless, prints to stdout
|
||||
cat file | shiro -p prompt read from stdin
|
||||
|
||||
options:
|
||||
-p, --print [prompt] headless mode; requires --yolo for tool use
|
||||
--json with -p, emit one JSON event per line
|
||||
-c, --continue resume the newest session for this directory
|
||||
-r, --resume <id> resume a session by id or id prefix
|
||||
--agent <name> ${VARIANTS.map((v) => v.name).join(' | ')}
|
||||
--think <level> off | low | medium | high | max
|
||||
--provider <anthropic|openai> wire protocol to use (default anthropic)
|
||||
--model <id> model id
|
||||
--base-url <url> OpenAI/Anthropic-compatible endpoint
|
||||
--no-mcp skip MCP servers from the config file
|
||||
--no-subagent omit the task tool
|
||||
--no-instructions ignore AGENTS.md / CLAUDE.md
|
||||
--no-skills ignore builtin and project skills
|
||||
--no-plugins disable all plugins, including the guard
|
||||
--no-memory do not load or write project memory
|
||||
--yolo skip all tool approval prompts
|
||||
-v, --version
|
||||
-h, --help
|
||||
|
||||
first run: start shiro with no key and it opens provider setup, or use /provider anytime.
|
||||
|
||||
config: ${configPath()}
|
||||
{ "provider": "openai", "model": "gpt-5", "apiKey": "...",
|
||||
"agent": "default", "thinking": "medium", "plugins": ["guard", "time"],
|
||||
"mcpServers": { "fs": { "command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", "."] } } }
|
||||
|
||||
env: SHIRO_PROVIDER SHIRO_MODEL SHIRO_BASE_URL SHIRO_API_KEY
|
||||
ANTHROPIC_API_KEY OPENAI_API_KEY
|
||||
|
||||
skills: builtin, plus ~/.shiro-neko/skills/*.md and .shiro/skills/*.md
|
||||
sessions: ${store.sessionsDir()}
|
||||
in-session: /help for the command list`;
|
||||
|
||||
const argv = process.argv.slice(2);
|
||||
|
||||
function flag(...names: string[]): string | undefined {
|
||||
for (const n of names) {
|
||||
const i = argv.indexOf(n);
|
||||
if (i === -1) continue;
|
||||
const next = argv[i + 1];
|
||||
return next && !next.startsWith('-') ? next : '';
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const has = (...names: string[]) => names.some((n) => argv.includes(n));
|
||||
|
||||
if (has('-h', '--help')) {
|
||||
console.log(HELP);
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
if (has('-v', '--version')) {
|
||||
console.log(versionLine());
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const providerFlag = flag('--provider');
|
||||
const modelFlag = flag('--model');
|
||||
const baseUrlFlag = flag('--base-url');
|
||||
if (providerFlag) process.env['SHIRO_PROVIDER'] = providerFlag;
|
||||
if (modelFlag) process.env['SHIRO_MODEL'] = modelFlag;
|
||||
if (baseUrlFlag) process.env['SHIRO_BASE_URL'] = baseUrlFlag;
|
||||
|
||||
let cfg = await loadConfig();
|
||||
const yolo = has('--yolo');
|
||||
const headless = flag('-p', '--print') !== undefined;
|
||||
|
||||
// A missing key is fatal for a pipe, but interactively it just means "not set up yet".
|
||||
if (!cfg.apiKey && headless) {
|
||||
console.error(`shiro: ${missingKeyMessage(cfg.provider)}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const needsProvider = !cfg.apiKey;
|
||||
const notices = createNoticeBus();
|
||||
const subagents = createSubagentBus();
|
||||
const askBridge = createAskBridge();
|
||||
|
||||
function reportFallback(e: FallbackEvent): void {
|
||||
const line = `endpoint fallback: ${e.from} rejected the request, retrying on ${e.to}\n ${e.reason}`;
|
||||
if (headless) process.stderr.write(`shiro: ${line}\n`);
|
||||
else notices.emit(line);
|
||||
}
|
||||
|
||||
let languageModel: LanguageModel | undefined;
|
||||
if (cfg.apiKey) {
|
||||
try {
|
||||
languageModel = resolveModel(cfg, reportFallback);
|
||||
} catch (e) {
|
||||
console.error(`shiro: ${(e as Error).message}`);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
let restored: store.SessionRecord | undefined;
|
||||
const resumeArg = flag('-r', '--resume');
|
||||
if (resumeArg) {
|
||||
const id = await store.resolveId(resumeArg);
|
||||
restored = id ? await store.load(id) : undefined;
|
||||
if (!restored) {
|
||||
console.error(`shiro: no session matching "${resumeArg}"`);
|
||||
process.exit(1);
|
||||
}
|
||||
} else if (has('-c', '--continue')) {
|
||||
restored = await store.latest(process.cwd());
|
||||
if (!restored) {
|
||||
console.error('shiro: no saved session for this directory');
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
|
||||
const instructions = has('--no-instructions') ? [] : await loadInstructions();
|
||||
const skills = has('--no-skills') ? [] : await loadSkills();
|
||||
const promptHistory = await store.loadHistory();
|
||||
|
||||
let agentVariant: AgentVariant;
|
||||
try {
|
||||
agentVariant = resolveAgent(flag('--agent') || cfg.agent, flag('--think') || cfg.thinking);
|
||||
} catch (e) {
|
||||
console.error(`shiro: ${(e as Error).message}`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const enabledPlugins = has('--no-plugins') ? [] : (cfg.plugins ?? DEFAULT_ENABLED);
|
||||
const pluginErrors = enabledPlugins
|
||||
.filter((name) => !BUILTIN_PLUGINS.some((p) => p.name === name))
|
||||
.map((name) => ({ plugin: name, message: 'no such plugin' }));
|
||||
const plugins = createHost(
|
||||
BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)),
|
||||
pluginErrors,
|
||||
);
|
||||
|
||||
const memory = has('--no-memory') ? undefined : new Memory(process.cwd(), languageModel);
|
||||
if (memory) await memory.load();
|
||||
|
||||
/** Placeholder until /provider supplies a key; it never gets called because the UI gates input. */
|
||||
const unconfiguredModel: LanguageModel = {
|
||||
specificationVersion: 'v4',
|
||||
provider: 'unconfigured',
|
||||
modelId: 'unconfigured',
|
||||
supportedUrls: {},
|
||||
doGenerate: () => Promise.reject(new Error('no provider configured - run /provider')),
|
||||
doStream: () => Promise.reject(new Error('no provider configured - run /provider')),
|
||||
};
|
||||
|
||||
const record: store.SessionRecord = restored ?? {
|
||||
id: store.newId(),
|
||||
createdAt: new Date().toISOString(),
|
||||
updatedAt: new Date().toISOString(),
|
||||
cwd: process.cwd(),
|
||||
provider: cfg.provider,
|
||||
model: cfg.model,
|
||||
title: 'untitled',
|
||||
inputTokens: 0,
|
||||
outputTokens: 0,
|
||||
messages: [],
|
||||
};
|
||||
|
||||
let saveTimer: ReturnType<typeof setTimeout> | undefined;
|
||||
|
||||
async function persist(messages: ModelMessage[]): Promise<void> {
|
||||
record.messages = messages;
|
||||
record.title = store.titleOf(messages);
|
||||
record.inputTokens = session.inputTokens;
|
||||
record.outputTokens = session.outputTokens;
|
||||
record.notebook = session.notebook.state();
|
||||
const cost = costOf(record.model, session.inputTokens, session.outputTokens);
|
||||
if (cost !== undefined) record.costUsd = cost;
|
||||
await store.save(record);
|
||||
}
|
||||
|
||||
const bridge = createApprovalBridge();
|
||||
const session = new Session({
|
||||
model: languageModel ?? unconfiguredModel,
|
||||
askApproval: bridge.ask,
|
||||
yolo,
|
||||
instructions,
|
||||
skills,
|
||||
plugins,
|
||||
agent: agentVariant,
|
||||
// Headless has no one to answer, so the tool is withheld rather than left to hang.
|
||||
...(headless ? {} : { ask: askBridge.ask }),
|
||||
...(memory ? { memory } : {}),
|
||||
...(record.notebook ? { notebook: record.notebook } : {}),
|
||||
...(cfg.maxRetries !== undefined ? { maxRetries: cfg.maxRetries } : {}),
|
||||
extraTools: {
|
||||
...(mcp?.tools ?? {}),
|
||||
...(has('--no-subagent')
|
||||
? {}
|
||||
: {
|
||||
task: createTaskTool({
|
||||
model: languageModel ?? unconfiguredModel,
|
||||
...(headless ? {} : { report: subagents.emit }),
|
||||
}),
|
||||
}),
|
||||
},
|
||||
autoApprove: ['task'],
|
||||
messages: [...record.messages],
|
||||
onChange: (messages) => {
|
||||
// Debounced so a long tool loop does not hit the disk on every step.
|
||||
clearTimeout(saveTimer);
|
||||
saveTimer = setTimeout(() => void persist(messages), 400);
|
||||
},
|
||||
});
|
||||
|
||||
async function shutdown(code: number): Promise<never> {
|
||||
clearTimeout(saveTimer);
|
||||
if (session.messages.length > 0) await persist(session.messages);
|
||||
await mcp?.close();
|
||||
process.exit(code);
|
||||
}
|
||||
|
||||
const printArg = flag('-p', '--print');
|
||||
if (printArg !== undefined) {
|
||||
const prompt = printArg || (await readStdin());
|
||||
if (!prompt) {
|
||||
console.error('shiro: -p needs a prompt argument or piped stdin');
|
||||
await shutdown(1);
|
||||
}
|
||||
if (!yolo) {
|
||||
process.stderr.write('shiro: headless denies write_file, edit_file, bash and mcp tools unless --yolo is passed\n');
|
||||
}
|
||||
const code = await runHeadless({ session, prompt, format: has('--json') ? 'json' : 'text' });
|
||||
await shutdown(code);
|
||||
}
|
||||
|
||||
function applyConfig(next: Config): void {
|
||||
cfg = next;
|
||||
record.provider = next.provider;
|
||||
record.model = next.model;
|
||||
session.setModel(resolveModel(next, reportFallback));
|
||||
}
|
||||
|
||||
const hooks: AppHooks = {
|
||||
sessionId: record.id,
|
||||
config: () => cfg,
|
||||
instructionFiles: () => instructions.map((i) => i.path),
|
||||
initPrompt: INIT_PROMPT,
|
||||
history: promptHistory,
|
||||
recordPrompt: (text) => void store.appendHistory(text),
|
||||
agentName: () => session.agent().name,
|
||||
thinkingLevel: () => session.agent().thinking,
|
||||
switchModel: (id) => {
|
||||
applyConfig({ ...cfg, model: id });
|
||||
return `model is now ${id}`;
|
||||
},
|
||||
switchAgent: (name) => {
|
||||
const next = resolveAgent(name, session.agent().thinking);
|
||||
session.setAgent(next);
|
||||
const scope = next.allowTools ? ` (read-only: ${next.allowTools.length} tools)` : '';
|
||||
return `agent is now ${next.name}, thinking ${next.thinking}${scope}`;
|
||||
},
|
||||
switchThinking: (level) => {
|
||||
if (!isThinkingLevel(level)) throw new Error(`Unknown thinking level "${level}"`);
|
||||
session.setAgent({ ...session.agent(), thinking: level });
|
||||
return `thinking is now ${level}`;
|
||||
},
|
||||
listSkills: () => {
|
||||
if (skills.length === 0) return 'no skills loaded';
|
||||
return skills.map((s) => `${s.name.padEnd(10)} ${s.origin.padEnd(8)} ${s.description}`).join('\n');
|
||||
},
|
||||
listPlugins: () => {
|
||||
const active = plugins.plugins.map((p) => `${p.name.padEnd(8)} ${p.description}`);
|
||||
const failed = plugins.errors.map((e) => `${e.plugin.padEnd(8)} ${e.message}`);
|
||||
if (active.length === 0 && failed.length === 0) return 'no plugins active';
|
||||
return [...active, ...failed].join('\n');
|
||||
},
|
||||
listMemory: async () => {
|
||||
if (!memory) return 'memory is disabled (--no-memory)';
|
||||
const all = await memory.load();
|
||||
if (all.length === 0) return 'nothing remembered about this project yet';
|
||||
return all
|
||||
.slice()
|
||||
.reverse()
|
||||
.map((e) => `(${KIND_LABEL[e.kind]}) ${e.text}${e.hits > 0 ? ` [recalled ${e.hits}x]` : ''}`)
|
||||
.join('\n');
|
||||
},
|
||||
summarizeMemory: async () => {
|
||||
if (!memory) return 'memory is disabled (--no-memory)';
|
||||
const { before, after } = await memory.summarize();
|
||||
return before === after
|
||||
? `memory left as is: ${before} entries, too few unused ones to merge`
|
||||
: `memory compacted: ${before} entries into ${after}`;
|
||||
},
|
||||
applyProvider: async (result) => {
|
||||
const next: Config = {
|
||||
...cfg,
|
||||
provider: result.provider,
|
||||
model: result.model,
|
||||
baseURL: result.baseURL,
|
||||
apiKey: result.apiKey,
|
||||
presetId: result.presetId,
|
||||
};
|
||||
applyConfig(next);
|
||||
const path = await writeConfigFile({
|
||||
provider: next.provider,
|
||||
model: next.model,
|
||||
baseURL: next.baseURL,
|
||||
apiKey: next.apiKey,
|
||||
presetId: next.presetId,
|
||||
});
|
||||
const label = presetById(result.presetId)?.label ?? result.presetId;
|
||||
return `${label} configured with ${result.model}\nsaved to ${path}`;
|
||||
},
|
||||
listModels: async () => {
|
||||
if (!cfg.apiKey) return { models: [], warning: 'no API key set - run /provider' };
|
||||
const preset = presetById(cfg.presetId ?? cfg.provider);
|
||||
const { models, warning } = await fetchModels(
|
||||
{
|
||||
kind: cfg.provider,
|
||||
baseURL: cfg.baseURL ?? '',
|
||||
...(preset?.fallbackModels ? { fallbackModels: preset.fallbackModels } : {}),
|
||||
},
|
||||
cfg.apiKey,
|
||||
);
|
||||
return warning ? { models, warning } : { models };
|
||||
},
|
||||
listSessions: async () => {
|
||||
const all = await store.list(15);
|
||||
if (all.length === 0) return 'no saved sessions';
|
||||
return all
|
||||
.map(
|
||||
(r) =>
|
||||
`${r.id.slice(0, 8)} ${r.updatedAt.slice(0, 16).replace('T', ' ')} ${r.messages.length}msg ${r.title}`,
|
||||
)
|
||||
.join('\n');
|
||||
},
|
||||
resumeSession: async (idOrPrefix) => {
|
||||
const id = await store.resolveId(idOrPrefix);
|
||||
const rec = id ? await store.load(id) : undefined;
|
||||
if (!rec) throw new Error(`no session matching "${idOrPrefix}"`);
|
||||
session.replace(rec.messages);
|
||||
record.id = rec.id;
|
||||
record.title = rec.title;
|
||||
hooks.sessionId = rec.id;
|
||||
return `resumed ${rec.id.slice(0, 8)} (${rec.messages.length} messages): ${rec.title}`;
|
||||
},
|
||||
saveSession: async () => {
|
||||
await persist(session.messages);
|
||||
return `saved ${record.id}`;
|
||||
},
|
||||
};
|
||||
|
||||
const header = [
|
||||
needsProvider
|
||||
? `shiro-neko ${VERSION} no provider configured`
|
||||
: `shiro-neko ${VERSION} ${cfg.provider}/${record.model} session ${record.id.slice(0, 8)}`,
|
||||
`agent: ${agentVariant.name} thinking: ${agentVariant.thinking}`,
|
||||
`cwd: ${process.cwd()}`,
|
||||
restored ? `resumed ${record.messages.length} messages` : undefined,
|
||||
instructions.length > 0
|
||||
? `instructions: ${instructions.map((i) => i.path.split(/[\\/]/).at(-1)).join(', ')}`
|
||||
: 'no AGENTS.md found - /init writes one',
|
||||
skills.length > 0 ? `skills: ${skills.map((s) => s.name).join(', ')}` : undefined,
|
||||
plugins.plugins.length > 0 ? `plugins: ${plugins.plugins.map((p) => p.name).join(', ')}` : undefined,
|
||||
...plugins.errors.map((e) => `plugin ${e.plugin}: ${e.message}`),
|
||||
memory && memory.all().length > 0 ? `memory: ${memory.all().length} notes about this project` : undefined,
|
||||
mcp && Object.keys(mcp.tools).length > 0 ? `mcp: ${Object.keys(mcp.tools).length} tools` : undefined,
|
||||
...(mcp?.errors ?? []).map((e) => `mcp ${e.server} failed: ${e.message}`),
|
||||
yolo ? 'approvals: OFF (--yolo)' : 'approvals: on for write_file, edit_file, bash, mcp__*',
|
||||
'/help for commands',
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join('\n');
|
||||
|
||||
const app = render(
|
||||
<App
|
||||
session={session}
|
||||
bridge={bridge}
|
||||
header={header}
|
||||
hooks={hooks}
|
||||
notices={notices}
|
||||
askBridge={askBridge}
|
||||
subagents={subagents}
|
||||
needsProvider={needsProvider}
|
||||
/>,
|
||||
);
|
||||
await app.waitUntilExit();
|
||||
await shutdown(0);
|
||||
+145
@@ -0,0 +1,145 @@
|
||||
export type CommandAction =
|
||||
| { type: 'none' }
|
||||
| { type: 'prompt'; text: string }
|
||||
| { type: 'exit' }
|
||||
| { type: 'clear' }
|
||||
| { type: 'compact' }
|
||||
| { type: 'tools' }
|
||||
| { type: 'cost' }
|
||||
| { type: 'sessions' }
|
||||
| { type: 'save' }
|
||||
| { type: 'provider' }
|
||||
| { type: 'models' }
|
||||
| { type: 'init' }
|
||||
| { type: 'context' }
|
||||
| { type: 'todos' }
|
||||
| { type: 'notes' }
|
||||
| { type: 'skills' }
|
||||
| { type: 'plugins' }
|
||||
| { type: 'memory' }
|
||||
| { type: 'agent'; agent?: string }
|
||||
| { type: 'think'; level?: string }
|
||||
| { type: 'info'; text: string }
|
||||
| { type: 'model'; model: string }
|
||||
| { type: 'resume'; id: string }
|
||||
| { type: 'unknown'; name: string };
|
||||
|
||||
export type CommandSpec = {
|
||||
name: string;
|
||||
/** Extra names that resolve to the same command, hidden from the menu. */
|
||||
aliases?: string[];
|
||||
arg?: string;
|
||||
summary: string;
|
||||
};
|
||||
|
||||
/** Single source of truth for the menu, `/help`, and the parser. */
|
||||
export const COMMANDS: CommandSpec[] = [
|
||||
{ name: 'help', aliases: ['?'], summary: 'list these commands' },
|
||||
{ name: 'agent', arg: '[name]', summary: 'switch agent: default, quick, deep, plan, review' },
|
||||
{ name: 'think', arg: '[level]', summary: 'thinking level: off, low, medium, high, max' },
|
||||
{ name: 'provider', aliases: ['login'], summary: 'set up a provider: pick, paste API key, choose model' },
|
||||
{ name: 'models', summary: 'pick a model from the current provider' },
|
||||
{ name: 'model', arg: '<id>', summary: 'switch model by name' },
|
||||
{ name: 'skills', summary: 'list loaded skills' },
|
||||
{ name: 'plugins', summary: 'list active plugins' },
|
||||
{ name: 'init', summary: 'have the agent write AGENTS.md for this project' },
|
||||
{ name: 'context', summary: 'show which instruction files are loaded' },
|
||||
{ name: 'todos', summary: "show the agent's task list" },
|
||||
{ name: 'notes', summary: 'show what the agent remembers about this project' },
|
||||
{ name: 'memory', summary: 'compact the project memory with the model' },
|
||||
{ name: 'tools', summary: 'list available tools' },
|
||||
{ name: 'compact', summary: 'replace history with a model-written summary' },
|
||||
{ name: 'cost', summary: 'tokens and estimated spend this session' },
|
||||
{ name: 'sessions', summary: 'list saved sessions' },
|
||||
{ name: 'resume', arg: '<id>', summary: 'load a saved session' },
|
||||
{ name: 'save', summary: 'write the session to disk now' },
|
||||
{ name: 'clear', summary: 'clear the transcript and history' },
|
||||
{ name: 'exit', aliases: ['quit'], summary: 'quit' },
|
||||
];
|
||||
|
||||
const usage = (c: CommandSpec) => `/${c.name}${c.arg ? ` ${c.arg}` : ''}`;
|
||||
|
||||
export const HELP = [
|
||||
...COMMANDS.map((c) => `${usage(c).padEnd(18)} ${c.summary}`),
|
||||
'',
|
||||
'esc interrupt the running turn',
|
||||
'tab complete the highlighted command',
|
||||
'up / down recall earlier prompts',
|
||||
].join('\n');
|
||||
|
||||
/**
|
||||
* Commands whose name starts with the typed prefix, for the `/` menu.
|
||||
* An exact name sorts first so pressing enter on `/model` cannot run `/models`.
|
||||
* Aliases stay hidden to keep the list short.
|
||||
*/
|
||||
export function matchCommands(input: string): CommandSpec[] {
|
||||
if (!input.startsWith('/')) return [];
|
||||
const typed = input.slice(1).toLowerCase();
|
||||
if (typed.includes(' ')) return [];
|
||||
const hits = COMMANDS.filter((c) => c.name.startsWith(typed));
|
||||
const exact = hits.findIndex((c) => c.name === typed);
|
||||
return exact > 0 ? [hits[exact]!, ...hits.filter((_, i) => i !== exact)] : hits;
|
||||
}
|
||||
|
||||
/** True while the input is a bare command name being typed, so the menu should show. */
|
||||
export const isMenuOpen = (input: string) => input.startsWith('/') && !input.includes(' ');
|
||||
|
||||
/** Pure parser: no IO, so the TUI and headless mode share one definition. */
|
||||
export function parseCommand(raw: string): CommandAction {
|
||||
const input = raw.trim();
|
||||
if (!input) return { type: 'none' };
|
||||
if (!input.startsWith('/')) return { type: 'prompt', text: input };
|
||||
|
||||
const [name = '', ...rest] = input.slice(1).split(/\s+/);
|
||||
const arg = rest.join(' ').trim();
|
||||
|
||||
switch (name) {
|
||||
case 'help':
|
||||
case '?':
|
||||
return { type: 'info', text: HELP };
|
||||
case 'exit':
|
||||
case 'quit':
|
||||
return { type: 'exit' };
|
||||
case 'clear':
|
||||
return { type: 'clear' };
|
||||
case 'compact':
|
||||
return { type: 'compact' };
|
||||
case 'tools':
|
||||
return { type: 'tools' };
|
||||
case 'cost':
|
||||
return { type: 'cost' };
|
||||
case 'sessions':
|
||||
return { type: 'sessions' };
|
||||
case 'save':
|
||||
return { type: 'save' };
|
||||
case 'provider':
|
||||
case 'login':
|
||||
return { type: 'provider' };
|
||||
case 'models':
|
||||
return { type: 'models' };
|
||||
case 'init':
|
||||
return { type: 'init' };
|
||||
case 'context':
|
||||
return { type: 'context' };
|
||||
case 'todos':
|
||||
return { type: 'todos' };
|
||||
case 'notes':
|
||||
return { type: 'notes' };
|
||||
case 'skills':
|
||||
return { type: 'skills' };
|
||||
case 'plugins':
|
||||
return { type: 'plugins' };
|
||||
case 'memory':
|
||||
return { type: 'memory' };
|
||||
case 'agent':
|
||||
return arg ? { type: 'agent', agent: arg } : { type: 'agent' };
|
||||
case 'think':
|
||||
return arg ? { type: 'think', level: arg } : { type: 'think' };
|
||||
case 'model':
|
||||
return arg ? { type: 'model', model: arg } : { type: 'models' };
|
||||
case 'resume':
|
||||
return arg ? { type: 'resume', id: arg } : { type: 'info', text: 'usage: /resume <session-id>' };
|
||||
default:
|
||||
return { type: 'unknown', name };
|
||||
}
|
||||
}
|
||||
+123
@@ -0,0 +1,123 @@
|
||||
import { createAnthropic } from '@ai-sdk/anthropic';
|
||||
import { createOpenAI } from '@ai-sdk/openai';
|
||||
import { createOpenAICompatible } from '@ai-sdk/openai-compatible';
|
||||
import type { LanguageModel } from 'ai';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { withFallback, type FallbackEvent } from './fallback';
|
||||
import type { McpServerConfig } from './mcp';
|
||||
|
||||
export type ProviderName = 'anthropic' | 'openai';
|
||||
|
||||
export type Config = {
|
||||
provider: ProviderName;
|
||||
model: string;
|
||||
baseURL?: string;
|
||||
apiKey?: string;
|
||||
/** Preset id from providers.ts, kept so /provider can show what is configured. */
|
||||
presetId?: string;
|
||||
/** Retries per model call for transient failures. SDK default is 2. */
|
||||
maxRetries?: number;
|
||||
/** Default agent variant name. */
|
||||
agent?: string;
|
||||
/** Default thinking level. */
|
||||
thinking?: string;
|
||||
/** Plugin names to enable; omit for the default set. */
|
||||
plugins?: string[];
|
||||
mcpServers?: Record<string, McpServerConfig>;
|
||||
};
|
||||
|
||||
const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json');
|
||||
|
||||
const DEFAULT_MODEL: Record<ProviderName, string> = {
|
||||
anthropic: 'claude-sonnet-4-5',
|
||||
openai: 'gpt-5',
|
||||
};
|
||||
|
||||
const DEFAULT_BASE_URL: Record<ProviderName, string> = {
|
||||
anthropic: 'https://api.anthropic.com/v1',
|
||||
openai: 'https://api.openai.com/v1',
|
||||
};
|
||||
|
||||
/** Env key checked per provider when no explicit apiKey is configured. */
|
||||
const ENV_KEY: Record<ProviderName, string> = {
|
||||
anthropic: 'ANTHROPIC_API_KEY',
|
||||
openai: 'OPENAI_API_KEY',
|
||||
};
|
||||
|
||||
function isProvider(v: unknown): v is ProviderName {
|
||||
return v === 'anthropic' || v === 'openai';
|
||||
}
|
||||
|
||||
/** Raw file contents, without env overlay. Used when rewriting the file. */
|
||||
export async function readConfigFile(): Promise<Partial<Config>> {
|
||||
const f = Bun.file(configPath());
|
||||
if (!(await f.exists())) return {};
|
||||
try {
|
||||
const parsed: unknown = await f.json();
|
||||
return parsed && typeof parsed === 'object' ? (parsed as Partial<Config>) : {};
|
||||
} catch {
|
||||
throw new Error(`${configPath()} is not valid JSON`);
|
||||
}
|
||||
}
|
||||
|
||||
/** Merges patch into the config file, preserving unrelated keys such as mcpServers. */
|
||||
export async function writeConfigFile(patch: Partial<Config>): Promise<string> {
|
||||
const merged = { ...(await readConfigFile()), ...patch };
|
||||
await Bun.write(configPath(), `${JSON.stringify(merged, null, 2)}\n`);
|
||||
return configPath();
|
||||
}
|
||||
|
||||
/** File config, then env overrides. Env wins so `SHIRO_MODEL=x shiro` works. */
|
||||
export async function loadConfig(): Promise<Config> {
|
||||
const file = await readConfigFile();
|
||||
|
||||
const envProvider = process.env['SHIRO_PROVIDER'];
|
||||
const provider = isProvider(envProvider) ? envProvider : isProvider(file.provider) ? file.provider : 'anthropic';
|
||||
|
||||
return {
|
||||
provider,
|
||||
model: process.env['SHIRO_MODEL'] ?? file.model ?? DEFAULT_MODEL[provider],
|
||||
baseURL: process.env['SHIRO_BASE_URL'] ?? file.baseURL ?? DEFAULT_BASE_URL[provider],
|
||||
apiKey: process.env['SHIRO_API_KEY'] ?? file.apiKey ?? process.env[ENV_KEY[provider]],
|
||||
...(file.presetId ? { presetId: file.presetId } : {}),
|
||||
...(file.maxRetries !== undefined ? { maxRetries: file.maxRetries } : {}),
|
||||
...(file.agent ? { agent: file.agent } : {}),
|
||||
...(file.thinking ? { thinking: file.thinking } : {}),
|
||||
...(Array.isArray(file.plugins) ? { plugins: file.plugins } : {}),
|
||||
...(file.mcpServers ? { mcpServers: file.mcpServers } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
export function missingKeyMessage(provider: ProviderName): string {
|
||||
return `No API key for provider "${provider}". Run shiro and use /provider to set one, or set ${ENV_KEY[provider]} / SHIRO_API_KEY, or add "apiKey" to ${configPath()}`;
|
||||
}
|
||||
|
||||
const isOfficialOpenAI = (baseURL: string | undefined) =>
|
||||
!baseURL || /^https:\/\/api\.openai\.com(\/|$)/.test(baseURL);
|
||||
|
||||
/**
|
||||
* Newer OpenAI reasoning models refuse function tools on /v1/chat/completions and
|
||||
* demand /v1/responses. Rather than guess per model id, build both and let
|
||||
* withFallback switch when the endpoint rejects the request shape.
|
||||
*/
|
||||
export function resolveModel(cfg: Config, onFallback?: (e: FallbackEvent) => void): LanguageModel {
|
||||
if (!cfg.apiKey) throw new Error(missingKeyMessage(cfg.provider));
|
||||
|
||||
if (cfg.provider === 'anthropic') {
|
||||
return createAnthropic({ apiKey: cfg.apiKey, baseURL: cfg.baseURL })(cfg.model);
|
||||
}
|
||||
|
||||
const chat = createOpenAICompatible({
|
||||
name: 'openai',
|
||||
apiKey: cfg.apiKey,
|
||||
baseURL: cfg.baseURL ?? DEFAULT_BASE_URL.openai,
|
||||
})(cfg.model);
|
||||
|
||||
if (!isOfficialOpenAI(cfg.baseURL)) return chat;
|
||||
|
||||
const openai = createOpenAI({ apiKey: cfg.apiKey, baseURL: cfg.baseURL });
|
||||
return withFallback([chat, openai.responses(cfg.model)], onFallback);
|
||||
}
|
||||
|
||||
export { configPath, ENV_KEY };
|
||||
@@ -0,0 +1,83 @@
|
||||
import { APICallError } from 'ai';
|
||||
import type { LanguageModelV4 } from '@ai-sdk/provider';
|
||||
|
||||
export type FallbackEvent = {
|
||||
from: string;
|
||||
to: string;
|
||||
reason: string;
|
||||
};
|
||||
|
||||
/** Marks a model built by withFallback, so callers can assert the chain is active. */
|
||||
export const FALLBACK_CHAIN = Symbol.for('shiro.fallbackChain');
|
||||
|
||||
export const fallbackChainOf = (model: unknown): string[] | undefined =>
|
||||
(model as Record<symbol, string[] | undefined>)[FALLBACK_CHAIN];
|
||||
|
||||
/**
|
||||
* Status codes that mean "this endpoint cannot serve this request", as opposed to
|
||||
* "try again later". Only these justify switching to a different API shape;
|
||||
* 401/403/429/5xx are either permanent or the SDK's own retry territory.
|
||||
*/
|
||||
const SHAPE_MISMATCH = new Set([400, 404, 405, 415, 422, 501]);
|
||||
|
||||
function shouldFallback(error: unknown): string | undefined {
|
||||
if (!APICallError.isInstance(error)) return undefined;
|
||||
if (error.isRetryable) return undefined;
|
||||
if (error.statusCode === undefined || !SHAPE_MISMATCH.has(error.statusCode)) return undefined;
|
||||
return `${error.statusCode}: ${error.message}`;
|
||||
}
|
||||
|
||||
const label = (m: LanguageModelV4) => `${m.provider}/${m.modelId}`;
|
||||
|
||||
/**
|
||||
* Presents several models as one, walking the list when an endpoint rejects the
|
||||
* request shape. Built for OpenAI models that only accept function tools on
|
||||
* /v1/responses, not /v1/chat/completions.
|
||||
*
|
||||
* ponytail: only a rejected doStream/doGenerate triggers the switch, not an error
|
||||
* emitted mid-stream, since tokens already delivered to the UI cannot be unsent.
|
||||
* Revisit if a provider starts returning 400s inside the stream body.
|
||||
*/
|
||||
export function withFallback(models: LanguageModelV4[], onFallback?: (e: FallbackEvent) => void): LanguageModelV4 {
|
||||
const [primary] = models;
|
||||
if (!primary) throw new Error('withFallback needs at least one model');
|
||||
if (models.length === 1) return primary;
|
||||
|
||||
// Sticky: once an endpoint rejects the request shape it will reject every later
|
||||
// step too, so start from the one that worked instead of re-probing each time.
|
||||
let start = 0;
|
||||
const reported = new Set<string>();
|
||||
|
||||
async function attempt<T>(op: (model: LanguageModelV4) => PromiseLike<T>): Promise<T> {
|
||||
let lastError: unknown;
|
||||
for (let i = start; i < models.length; i++) {
|
||||
const model = models[i]!;
|
||||
try {
|
||||
return await op(model);
|
||||
} catch (error) {
|
||||
const reason = shouldFallback(error);
|
||||
const next = models[i + 1];
|
||||
if (!reason || !next) throw error;
|
||||
lastError = error;
|
||||
start = i + 1;
|
||||
const key = `${label(model)}->${label(next)}`;
|
||||
if (!reported.has(key)) {
|
||||
reported.add(key);
|
||||
onFallback?.({ from: label(model), to: label(next), reason });
|
||||
}
|
||||
}
|
||||
}
|
||||
throw lastError;
|
||||
}
|
||||
|
||||
const wrapped: LanguageModelV4 = {
|
||||
specificationVersion: 'v4',
|
||||
provider: primary.provider,
|
||||
modelId: primary.modelId,
|
||||
supportedUrls: primary.supportedUrls,
|
||||
doGenerate: (options) => attempt((m) => m.doGenerate(options)),
|
||||
doStream: (options) => attempt((m) => m.doStream(options)),
|
||||
};
|
||||
Object.defineProperty(wrapped, FALLBACK_CHAIN, { value: models.map(label), enumerable: false });
|
||||
return wrapped;
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
import type { AgentEvent, Session } from './session';
|
||||
|
||||
export type HeadlessOptions = {
|
||||
session: Session;
|
||||
prompt: string;
|
||||
/** 'text' streams assistant text only; 'json' emits one event object per line. */
|
||||
format?: 'text' | 'json';
|
||||
out?: (chunk: string) => void;
|
||||
};
|
||||
|
||||
const message = (error: unknown) => (error instanceof Error ? error.message : String(error));
|
||||
|
||||
/**
|
||||
* JSON.stringify turns an Error into `{}`, which would make --json useless for
|
||||
* diagnosing a failure, so error payloads are flattened to a message string.
|
||||
*/
|
||||
function serialize(ev: AgentEvent): string {
|
||||
if (ev.type === 'error') return JSON.stringify({ type: 'error', error: message(ev.error) });
|
||||
if (ev.type === 'tool-error') {
|
||||
return JSON.stringify({ type: 'tool-error', id: ev.id, name: ev.name, error: message(ev.error) });
|
||||
}
|
||||
return JSON.stringify(ev);
|
||||
}
|
||||
|
||||
/**
|
||||
* Non-interactive run for pipes and CI. There is no terminal to prompt on, so the
|
||||
* Session must already be constructed with yolo or every mutating call gets denied.
|
||||
* Returns a process exit code.
|
||||
*/
|
||||
export async function runHeadless({ session, prompt, format = 'text', out }: HeadlessOptions): Promise<number> {
|
||||
const write = out ?? ((s: string) => process.stdout.write(s));
|
||||
let failed = false;
|
||||
|
||||
for await (const ev of session.send(prompt)) {
|
||||
if (format === 'json') {
|
||||
write(`${serialize(ev)}\n`);
|
||||
if (ev.type === 'error') failed = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
switch (ev.type) {
|
||||
case 'text':
|
||||
write(ev.text);
|
||||
break;
|
||||
case 'tool-call':
|
||||
process.stderr.write(`[tool] ${ev.name} ${JSON.stringify(ev.input)}\n`);
|
||||
break;
|
||||
case 'tool-denied':
|
||||
process.stderr.write(`[denied] ${ev.name} (run with --yolo to allow tool use in headless mode)\n`);
|
||||
break;
|
||||
case 'tool-error':
|
||||
process.stderr.write(`[tool-error] ${ev.name}: ${message(ev.error)}\n`);
|
||||
break;
|
||||
case 'notice':
|
||||
process.stderr.write(`[notice] ${ev.text}\n`);
|
||||
break;
|
||||
case 'compacted':
|
||||
process.stderr.write(`[compacted] ${ev.before} messages pruned to ${ev.after}\n`);
|
||||
break;
|
||||
case 'error':
|
||||
process.stderr.write(`[error] ${message(ev.error)}\n`);
|
||||
failed = true;
|
||||
break;
|
||||
case 'done':
|
||||
write('\n');
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return failed ? 1 : 0;
|
||||
}
|
||||
|
||||
export async function readStdin(): Promise<string> {
|
||||
if (process.stdin.isTTY) return '';
|
||||
return (await Bun.stdin.text()).trim();
|
||||
}
|
||||
+149
@@ -0,0 +1,149 @@
|
||||
import { isAbsolute, join, relative, resolve } from 'node:path';
|
||||
import { readdir as readdirFs } from 'node:fs/promises';
|
||||
|
||||
const ALWAYS_SKIP = ['.git', 'node_modules'];
|
||||
|
||||
type Rule = {
|
||||
/** Directory the rule was declared in, relative and posix-separated. */
|
||||
base: string;
|
||||
negated: boolean;
|
||||
dirOnly: boolean;
|
||||
re: RegExp;
|
||||
};
|
||||
|
||||
const posix = (p: string) => p.replaceAll('\\', '/');
|
||||
|
||||
/**
|
||||
* Translates one gitignore pattern into a regex over posix-relative paths.
|
||||
* Supports `!` negation, trailing `/`, leading `/` anchoring, `*`, `?`, and `**`.
|
||||
*/
|
||||
function compile(pattern: string, base: string): Rule | undefined {
|
||||
let body = pattern.trim();
|
||||
if (!body || body.startsWith('#')) return undefined;
|
||||
|
||||
const negated = body.startsWith('!');
|
||||
if (negated) body = body.slice(1);
|
||||
|
||||
const dirOnly = body.endsWith('/');
|
||||
if (dirOnly) body = body.slice(0, -1);
|
||||
|
||||
const anchored = body.startsWith('/') || body.slice(0, -1).includes('/');
|
||||
if (body.startsWith('/')) body = body.slice(1);
|
||||
if (!body) return undefined;
|
||||
|
||||
let re = '';
|
||||
for (let i = 0; i < body.length; i++) {
|
||||
const ch = body[i]!;
|
||||
if (ch === '*') {
|
||||
if (body[i + 1] === '*') {
|
||||
// `**/` spans any number of directories, bare `**` spans anything.
|
||||
if (body[i + 2] === '/') {
|
||||
re += '(?:.*/)?';
|
||||
i += 2;
|
||||
} else {
|
||||
re += '.*';
|
||||
i += 1;
|
||||
}
|
||||
} else {
|
||||
re += '[^/]*';
|
||||
}
|
||||
} else if (ch === '?') re += '[^/]';
|
||||
else if ('.+^${}()|[]\\'.includes(ch)) re += `\\${ch}`;
|
||||
else re += ch;
|
||||
}
|
||||
|
||||
// An unanchored pattern matches at any depth; both forms also match everything
|
||||
// beneath a matched directory.
|
||||
const prefix = anchored ? '' : '(?:.*/)?';
|
||||
return { base, negated, dirOnly, re: new RegExp(`^${prefix}${re}(?:/.*)?$`) };
|
||||
}
|
||||
|
||||
async function rulesIn(root: string, dir: string): Promise<Rule[]> {
|
||||
const base = posix(relative(root, dir));
|
||||
const out: Rule[] = [];
|
||||
for (const name of ['.gitignore', '.shiroignore']) {
|
||||
const file = Bun.file(join(dir, name));
|
||||
if (!(await file.exists())) continue;
|
||||
for (const line of (await file.text()).split('\n')) {
|
||||
const rule = compile(line, base);
|
||||
if (rule) out.push(rule);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function ignored(relPath: string, isDir: boolean, rules: Rule[]): boolean {
|
||||
let hit = false;
|
||||
for (const rule of rules) {
|
||||
if (rule.dirOnly && !isDir) continue;
|
||||
const scoped = rule.base ? (relPath.startsWith(`${rule.base}/`) ? relPath.slice(rule.base.length + 1) : undefined) : relPath;
|
||||
if (scoped === undefined) continue;
|
||||
// Later rules win, which is how git resolves a negation after an ignore.
|
||||
if (rule.re.test(scoped)) hit = !rule.negated;
|
||||
}
|
||||
return hit;
|
||||
}
|
||||
|
||||
export type WalkOptions = {
|
||||
root?: string;
|
||||
/** Include files git would ignore. */
|
||||
noIgnore?: boolean;
|
||||
limit?: number;
|
||||
};
|
||||
|
||||
/**
|
||||
* Yields workspace-relative posix paths, skipping .git, node_modules, and anything
|
||||
* .gitignore or .shiroignore excludes. Nested ignore files are honoured, so a
|
||||
* `dist/` rule in a subpackage only applies inside it.
|
||||
*/
|
||||
export async function* walk(options: WalkOptions = {}): AsyncGenerator<string> {
|
||||
const root = resolve(options.root ?? process.cwd());
|
||||
const limit = options.limit ?? Infinity;
|
||||
let yielded = 0;
|
||||
|
||||
const queue: { dir: string; rules: Rule[] }[] = [
|
||||
{ dir: root, rules: options.noIgnore ? [] : await rulesIn(root, root) },
|
||||
];
|
||||
|
||||
while (queue.length > 0) {
|
||||
const { dir, rules } = queue.shift()!;
|
||||
let entries: Entry[];
|
||||
try {
|
||||
entries = (await readdirFs(dir, { withFileTypes: true })).map((d) => ({
|
||||
name: d.name,
|
||||
isDirectory: d.isDirectory(),
|
||||
}));
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (const entry of entries) {
|
||||
if (ALWAYS_SKIP.includes(entry.name)) continue;
|
||||
const full = join(dir, entry.name);
|
||||
const rel = posix(relative(root, full));
|
||||
if (!options.noIgnore && ignored(rel, entry.isDirectory, rules)) continue;
|
||||
|
||||
if (entry.isDirectory) {
|
||||
const nested = options.noIgnore ? rules : [...rules, ...(await rulesIn(root, full))];
|
||||
queue.push({ dir: full, rules: nested });
|
||||
} else {
|
||||
yield rel;
|
||||
if (++yielded >= limit) return;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
type Entry = { name: string; isDirectory: boolean };
|
||||
|
||||
/** Resolves a model-supplied path inside the workspace, rejecting escapes. */
|
||||
export function jail(p: string, root = process.cwd()): string {
|
||||
const abs = isAbsolute(p) ? resolve(p) : resolve(root, p);
|
||||
const rel = relative(resolve(root), abs);
|
||||
if (rel.startsWith('..') || isAbsolute(rel)) {
|
||||
throw new Error(`Path escapes workspace: ${p}`);
|
||||
}
|
||||
return abs;
|
||||
}
|
||||
|
||||
export { posix };
|
||||
@@ -0,0 +1,71 @@
|
||||
import { dirname, join, resolve } from 'node:path';
|
||||
|
||||
const NAMES = ['AGENTS.md', 'CLAUDE.md', '.shiro.md'];
|
||||
/** Cap per file so one huge doc cannot crowd out the conversation. */
|
||||
const MAX_CHARS = 12_000;
|
||||
|
||||
export type Instructions = { path: string; text: string }[];
|
||||
|
||||
/**
|
||||
* Collects project instruction files from the git root down to cwd, outermost
|
||||
* first so a nested file's rules read as refinements of the ones above it.
|
||||
* Stops at the git root, or the filesystem root when there is no repo.
|
||||
*/
|
||||
export async function loadInstructions(cwd = process.cwd()): Promise<Instructions> {
|
||||
const dirs: string[] = [];
|
||||
let dir = resolve(cwd);
|
||||
while (true) {
|
||||
dirs.unshift(dir);
|
||||
if (await Bun.file(join(dir, '.git', 'HEAD')).exists()) break;
|
||||
const parent = dirname(dir);
|
||||
if (parent === dir) break;
|
||||
dir = parent;
|
||||
}
|
||||
|
||||
const found: Instructions = [];
|
||||
const seen = new Set<string>();
|
||||
for (const d of dirs) {
|
||||
for (const name of NAMES) {
|
||||
const path = join(d, name);
|
||||
if (seen.has(path)) continue;
|
||||
const file = Bun.file(path);
|
||||
if (!(await file.exists())) continue;
|
||||
seen.add(path);
|
||||
const text = (await file.text()).trim();
|
||||
if (text) found.push({ path, text: text.slice(0, MAX_CHARS) });
|
||||
}
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
export function formatInstructions(instructions: Instructions, cwd = process.cwd()): string {
|
||||
if (instructions.length === 0) return '';
|
||||
const blocks = instructions.map(({ path, text }) => {
|
||||
const label = path.startsWith(cwd) ? path.slice(cwd.length + 1) || path : path;
|
||||
return `--- ${label} ---\n${text}`;
|
||||
});
|
||||
return [
|
||||
'',
|
||||
'Project instructions (from the files below). Treat these as standing orders from the user;',
|
||||
'they override your defaults but never your safety rules.',
|
||||
'',
|
||||
...blocks,
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
export const INSTRUCTION_NAMES = NAMES;
|
||||
|
||||
export const INIT_PROMPT = `Write an AGENTS.md at the workspace root that will orient a coding agent joining this project cold.
|
||||
|
||||
Investigate first: read the manifest, the config files, the entry points, and a couple of representative
|
||||
source files. Run the test and build commands if that is the only way to learn how they are invoked.
|
||||
|
||||
Then write AGENTS.md covering only what you actually verified:
|
||||
- what this project is, in two or three sentences
|
||||
- the exact commands for install, build, test, typecheck, lint
|
||||
- the layout: which directory holds what
|
||||
- conventions a newcomer would otherwise get wrong: naming, error handling, module boundaries, test style
|
||||
- anything surprising or easy to break
|
||||
|
||||
Keep it under 100 lines. No filler sections, no "best practices" boilerplate, nothing you did not confirm
|
||||
by reading the code. If a section would be guesswork, leave it out.`;
|
||||
+147
@@ -0,0 +1,147 @@
|
||||
export type Span = { text: string; bold?: boolean; italic?: boolean; code?: boolean; strike?: boolean; link?: boolean };
|
||||
|
||||
export type Block =
|
||||
| { kind: 'heading'; level: number; spans: Span[] }
|
||||
| { kind: 'paragraph'; spans: Span[] }
|
||||
| { kind: 'bullet'; indent: number; marker: string; spans: Span[] }
|
||||
| { kind: 'quote'; spans: Span[] }
|
||||
| { kind: 'code'; language: string; lines: string[] }
|
||||
| { kind: 'rule' }
|
||||
| { kind: 'blank' };
|
||||
|
||||
const INLINE =
|
||||
/(`+)([\s\S]*?)\1|\*\*([\s\S]+?)\*\*|__([\s\S]+?)__|~~([\s\S]+?)~~|(?<![A-Za-z0-9_])_([^\s_][\s\S]*?)_(?![A-Za-z0-9_])|\*([^\s*][\s\S]*?)\*|\[([^\]]+)\]\(([^)]+)\)/;
|
||||
|
||||
/**
|
||||
* Inline markdown to styled spans.
|
||||
*
|
||||
* Code spans are matched first and their contents are never re-scanned, so
|
||||
* `` `**not bold**` `` stays literal — the mistake a naive replace-based
|
||||
* renderer makes on every code sample an agent prints. Underscore emphasis is
|
||||
* also required to sit at a word boundary, so `snake_case_name` survives.
|
||||
*/
|
||||
export function parseInline(input: string): Span[] {
|
||||
const spans: Span[] = [];
|
||||
let rest = input;
|
||||
|
||||
while (rest.length > 0) {
|
||||
const m = INLINE.exec(rest);
|
||||
if (!m || m.index === undefined) {
|
||||
spans.push({ text: rest });
|
||||
break;
|
||||
}
|
||||
|
||||
if (m.index > 0) spans.push({ text: rest.slice(0, m.index) });
|
||||
|
||||
if (m[2] !== undefined) spans.push({ text: m[2].trim(), code: true });
|
||||
else if (m[3] !== undefined) spans.push(...parseInline(m[3]).map((s) => ({ ...s, bold: true })));
|
||||
else if (m[4] !== undefined) spans.push(...parseInline(m[4]).map((s) => ({ ...s, bold: true })));
|
||||
else if (m[5] !== undefined) spans.push(...parseInline(m[5]).map((s) => ({ ...s, strike: true })));
|
||||
else if (m[6] !== undefined) spans.push(...parseInline(m[6]).map((s) => ({ ...s, italic: true })));
|
||||
else if (m[7] !== undefined) spans.push(...parseInline(m[7]).map((s) => ({ ...s, italic: true })));
|
||||
else if (m[8] !== undefined) spans.push({ text: m[8], link: true });
|
||||
|
||||
rest = rest.slice(m.index + m[0].length);
|
||||
}
|
||||
|
||||
return spans.filter((s) => s.text.length > 0);
|
||||
}
|
||||
|
||||
const FENCE = /^\s*(```+|~~~+)\s*([\w+-]*)\s*$/;
|
||||
const HEADING = /^(#{1,6})\s+(.*)$/;
|
||||
const BULLET = /^(\s*)([-*+]|\d+[.)])\s+(.*)$/;
|
||||
const QUOTE = /^\s*>\s?(.*)$/;
|
||||
const RULE = /^\s*([-*_])(\s*\1){2,}\s*$/;
|
||||
|
||||
/**
|
||||
* Line-based markdown parser covering what an agent actually emits: headings,
|
||||
* fences, lists, quotes, rules, and inline styling. Not CommonMark — no nested
|
||||
* blocks, tables, or reference links, none of which appear in agent replies.
|
||||
*/
|
||||
export function parseMarkdown(input: string): Block[] {
|
||||
const blocks: Block[] = [];
|
||||
const lines = input.replace(/\r\n/g, '\n').split('\n');
|
||||
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const line = lines[i]!;
|
||||
|
||||
const fence = FENCE.exec(line);
|
||||
if (fence) {
|
||||
const closer = fence[1]!;
|
||||
const body: string[] = [];
|
||||
i++;
|
||||
while (i < lines.length && !new RegExp(`^\\s*${closer[0]}{${closer.length},}\\s*$`).test(lines[i]!)) {
|
||||
body.push(lines[i]!);
|
||||
i++;
|
||||
}
|
||||
blocks.push({ kind: 'code', language: fence[2] ?? '', lines: body });
|
||||
continue;
|
||||
}
|
||||
|
||||
if (line.trim().length === 0) {
|
||||
if (blocks.at(-1)?.kind !== 'blank') blocks.push({ kind: 'blank' });
|
||||
continue;
|
||||
}
|
||||
|
||||
if (RULE.test(line)) {
|
||||
blocks.push({ kind: 'rule' });
|
||||
continue;
|
||||
}
|
||||
|
||||
const heading = HEADING.exec(line);
|
||||
if (heading) {
|
||||
blocks.push({ kind: 'heading', level: heading[1]!.length, spans: parseInline(heading[2]!) });
|
||||
continue;
|
||||
}
|
||||
|
||||
const bullet = BULLET.exec(line);
|
||||
if (bullet) {
|
||||
blocks.push({
|
||||
kind: 'bullet',
|
||||
indent: Math.floor(bullet[1]!.length / 2),
|
||||
marker: /\d/.test(bullet[2]!) ? bullet[2]! : '-',
|
||||
spans: parseInline(bullet[3]!),
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
const quote = QUOTE.exec(line);
|
||||
if (quote) {
|
||||
blocks.push({ kind: 'quote', spans: parseInline(quote[1]!) });
|
||||
continue;
|
||||
}
|
||||
|
||||
// Consecutive plain lines join into one paragraph so wrapping is the terminal's job.
|
||||
const previous = blocks.at(-1);
|
||||
if (previous?.kind === 'paragraph') {
|
||||
previous.spans.push({ text: ' ' }, ...parseInline(line.trim()));
|
||||
} else {
|
||||
blocks.push({ kind: 'paragraph', spans: parseInline(line.trim()) });
|
||||
}
|
||||
}
|
||||
|
||||
while (blocks.at(-1)?.kind === 'blank') blocks.pop();
|
||||
return blocks;
|
||||
}
|
||||
|
||||
/** Plain text with the markup removed, for widths and non-styled surfaces. */
|
||||
export function toPlainText(blocks: Block[]): string {
|
||||
return blocks
|
||||
.map((b) => {
|
||||
switch (b.kind) {
|
||||
case 'code':
|
||||
return b.lines.join('\n');
|
||||
case 'rule':
|
||||
return '---';
|
||||
case 'blank':
|
||||
return '';
|
||||
case 'bullet':
|
||||
return `${' '.repeat(b.indent)}${b.marker} ${b.spans.map((s) => s.text).join('')}`;
|
||||
case 'heading':
|
||||
return `${'#'.repeat(b.level)} ${b.spans.map((s) => s.text).join('')}`;
|
||||
default:
|
||||
return b.spans.map((s) => s.text).join('');
|
||||
}
|
||||
})
|
||||
.join('\n');
|
||||
}
|
||||
+57
@@ -0,0 +1,57 @@
|
||||
import { createMCPClient, type MCPClient } from '@ai-sdk/mcp';
|
||||
import { Experimental_StdioMCPTransport } from '@ai-sdk/mcp/mcp-stdio';
|
||||
import type { ToolSet } from 'ai';
|
||||
|
||||
export type McpServerConfig =
|
||||
| { command: string; args?: string[]; env?: Record<string, string>; cwd?: string }
|
||||
| { url: string; type?: 'http' | 'sse'; headers?: Record<string, string> };
|
||||
|
||||
export type McpHandle = {
|
||||
tools: ToolSet;
|
||||
errors: { server: string; message: string }[];
|
||||
close: () => Promise<void>;
|
||||
};
|
||||
|
||||
const isRemote = (c: McpServerConfig): c is Extract<McpServerConfig, { url: string }> => 'url' in c;
|
||||
|
||||
/**
|
||||
* Connects every configured server and namespaces its tools as `mcp__<server>__<tool>`
|
||||
* so two servers exposing `search` cannot silently shadow each other.
|
||||
* A server that fails to start is reported, never fatal.
|
||||
*/
|
||||
export async function connectMcp(servers: Record<string, McpServerConfig>): Promise<McpHandle> {
|
||||
const clients: MCPClient[] = [];
|
||||
const tools: ToolSet = {};
|
||||
const errors: McpHandle['errors'] = [];
|
||||
|
||||
await Promise.all(
|
||||
Object.entries(servers).map(async ([name, cfg]) => {
|
||||
try {
|
||||
const client = await createMCPClient({
|
||||
transport: isRemote(cfg)
|
||||
? { type: cfg.type ?? 'http', url: cfg.url, ...(cfg.headers ? { headers: cfg.headers } : {}) }
|
||||
: new Experimental_StdioMCPTransport({
|
||||
command: cfg.command,
|
||||
...(cfg.args ? { args: cfg.args } : {}),
|
||||
...(cfg.env ? { env: cfg.env } : {}),
|
||||
...(cfg.cwd ? { cwd: cfg.cwd } : {}),
|
||||
}),
|
||||
});
|
||||
clients.push(client);
|
||||
for (const [toolName, tool] of Object.entries(await client.tools())) {
|
||||
tools[`mcp__${name}__${toolName}`] = tool;
|
||||
}
|
||||
} catch (e) {
|
||||
errors.push({ server: name, message: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}),
|
||||
);
|
||||
|
||||
return {
|
||||
tools,
|
||||
errors,
|
||||
close: async () => {
|
||||
await Promise.all(clients.map((c) => c.close().catch(() => {})));
|
||||
},
|
||||
};
|
||||
}
|
||||
+253
@@ -0,0 +1,253 @@
|
||||
import { tool, type LanguageModel } from 'ai';
|
||||
import { generateText } from 'ai';
|
||||
import { createHash } from 'node:crypto';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
|
||||
export type MemoryKind = 'fact' | 'decision' | 'gotcha' | 'command';
|
||||
|
||||
export type MemoryEntry = {
|
||||
id: string;
|
||||
kind: MemoryKind;
|
||||
text: string;
|
||||
createdAt: string;
|
||||
/** Bumped on each recall so summarisation can keep what gets used. */
|
||||
hits: number;
|
||||
};
|
||||
|
||||
const MAX_ENTRIES = 300;
|
||||
const MAX_TEXT = 400;
|
||||
const BOOT_ENTRIES = 20;
|
||||
const SEARCH_HITS = 15;
|
||||
/** Summarise once the store passes this, so the boot block stays small. */
|
||||
const SUMMARISE_AT = 60;
|
||||
|
||||
const root = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'memory');
|
||||
|
||||
/** One file per project directory; the path is hashed because it is not filename-safe. */
|
||||
const fileFor = (cwd: string) => join(root(), `${createHash('sha256').update(cwd).digest('hex').slice(0, 16)}.json`);
|
||||
|
||||
const KIND_LABEL: Record<MemoryKind, string> = {
|
||||
fact: 'fact',
|
||||
decision: 'decision',
|
||||
gotcha: 'gotcha',
|
||||
command: 'command',
|
||||
};
|
||||
|
||||
/**
|
||||
* Durable per-project memory, separate from the session transcript.
|
||||
*
|
||||
* The transcript is destroyed by compaction and discarded when a session ends.
|
||||
* Anything worth knowing on the next run has to live here instead.
|
||||
*/
|
||||
export class Memory {
|
||||
private entries: MemoryEntry[] = [];
|
||||
private loaded = false;
|
||||
|
||||
constructor(
|
||||
private readonly cwd = process.cwd(),
|
||||
private readonly model?: LanguageModel,
|
||||
) {}
|
||||
|
||||
async load(): Promise<MemoryEntry[]> {
|
||||
if (this.loaded) return this.entries;
|
||||
this.loaded = true;
|
||||
const f = Bun.file(fileFor(this.cwd));
|
||||
if (await f.exists()) {
|
||||
try {
|
||||
const parsed: unknown = await f.json();
|
||||
if (Array.isArray(parsed)) this.entries = parsed.filter(isEntry);
|
||||
} catch {
|
||||
this.entries = [];
|
||||
}
|
||||
}
|
||||
return this.entries;
|
||||
}
|
||||
|
||||
all(): MemoryEntry[] {
|
||||
return [...this.entries];
|
||||
}
|
||||
|
||||
private async persist(): Promise<void> {
|
||||
this.entries = this.entries.slice(-MAX_ENTRIES);
|
||||
await Bun.write(fileFor(this.cwd), JSON.stringify(this.entries, null, 2));
|
||||
}
|
||||
|
||||
async add(kind: MemoryKind, text: string): Promise<MemoryEntry | undefined> {
|
||||
await this.load();
|
||||
const clean = text.trim().slice(0, MAX_TEXT);
|
||||
if (!clean) throw new Error('memory text is empty');
|
||||
if (this.entries.some((e) => e.text === clean)) return undefined;
|
||||
|
||||
const entry: MemoryEntry = {
|
||||
id: Bun.randomUUIDv7(),
|
||||
kind,
|
||||
text: clean,
|
||||
createdAt: new Date().toISOString(),
|
||||
hits: 0,
|
||||
};
|
||||
this.entries.push(entry);
|
||||
await this.persist();
|
||||
return entry;
|
||||
}
|
||||
|
||||
async forget(idOrPrefix: string): Promise<number> {
|
||||
await this.load();
|
||||
const before = this.entries.length;
|
||||
this.entries = this.entries.filter((e) => !e.id.startsWith(idOrPrefix));
|
||||
if (this.entries.length !== before) await this.persist();
|
||||
return before - this.entries.length;
|
||||
}
|
||||
|
||||
async clear(): Promise<void> {
|
||||
await this.load();
|
||||
this.entries = [];
|
||||
await this.persist();
|
||||
}
|
||||
|
||||
/** Every term must appear. Matching entries get a hit, which protects them from summarisation. */
|
||||
async search(query: string): Promise<MemoryEntry[]> {
|
||||
await this.load();
|
||||
const terms = query.toLowerCase().split(/\s+/).filter(Boolean);
|
||||
if (terms.length === 0) throw new Error('query is empty');
|
||||
|
||||
const found = this.entries.filter((e) => {
|
||||
const lower = e.text.toLowerCase();
|
||||
return terms.every((t) => lower.includes(t));
|
||||
});
|
||||
for (const e of found) e.hits += 1;
|
||||
if (found.length > 0) await this.persist();
|
||||
return found.slice(-SEARCH_HITS).reverse();
|
||||
}
|
||||
|
||||
/** The block injected at boot: most-used first, then most recent. */
|
||||
render(limit = BOOT_ENTRIES): string {
|
||||
if (this.entries.length === 0) return '';
|
||||
const ranked = [...this.entries]
|
||||
.sort((a, b) => b.hits - a.hits || b.createdAt.localeCompare(a.createdAt))
|
||||
.slice(0, limit);
|
||||
return [
|
||||
'',
|
||||
'What you learned about this project in earlier sessions. Trust it, but verify anything',
|
||||
'that contradicts what you can see in the code now:',
|
||||
...ranked.map((e) => `- (${KIND_LABEL[e.kind]}) ${e.text}`),
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
needsSummary(): boolean {
|
||||
return this.entries.length >= SUMMARISE_AT;
|
||||
}
|
||||
|
||||
/**
|
||||
* Collapses the store into fewer, denser entries using the model. Unused entries
|
||||
* are the ones that get merged away; anything recalled at least once is kept verbatim.
|
||||
*/
|
||||
async summarize(): Promise<{ before: number; after: number }> {
|
||||
await this.load();
|
||||
const before = this.entries.length;
|
||||
if (!this.model) throw new Error('no model available to summarize memory');
|
||||
if (before === 0) return { before, after: 0 };
|
||||
|
||||
const used = this.entries.filter((e) => e.hits > 0);
|
||||
const unused = this.entries.filter((e) => e.hits === 0);
|
||||
if (unused.length < 2) return { before, after: before };
|
||||
|
||||
const { text } = await generateText({
|
||||
model: this.model,
|
||||
system:
|
||||
'You are compacting an agent\'s notes about one codebase. Merge duplicates and near-duplicates, ' +
|
||||
'drop anything that is no longer useful or was only true of one past task, and keep the rest verbatim ' +
|
||||
'where you can. Output one note per line, each prefixed with its kind in brackets: ' +
|
||||
'[fact], [decision], [gotcha], or [command]. No preamble, no numbering, no blank lines.',
|
||||
prompt: unused.map((e) => `[${e.kind}] ${e.text}`).join('\n'),
|
||||
maxRetries: 2,
|
||||
});
|
||||
|
||||
const merged = text
|
||||
.split('\n')
|
||||
.map((line) => /^\s*\[(fact|decision|gotcha|command)\]\s*(.+?)\s*$/i.exec(line))
|
||||
.filter((m): m is RegExpExecArray => m !== null)
|
||||
.map((m) => ({
|
||||
id: Bun.randomUUIDv7(),
|
||||
kind: m[1]!.toLowerCase() as MemoryKind,
|
||||
text: m[2]!.slice(0, MAX_TEXT),
|
||||
createdAt: new Date().toISOString(),
|
||||
hits: 0,
|
||||
}));
|
||||
|
||||
// A model that returned nothing parseable must not wipe the store.
|
||||
if (merged.length === 0) return { before, after: before };
|
||||
|
||||
this.entries = [...used, ...merged];
|
||||
await this.persist();
|
||||
return { before, after: this.entries.length };
|
||||
}
|
||||
|
||||
tools() {
|
||||
return {
|
||||
remember: tool({
|
||||
description:
|
||||
'Record something about this project that will still be true next session: a decision and its reason, ' +
|
||||
'a command that works, a constraint, a trap you hit. Persisted across sessions and shown to you at start. ' +
|
||||
'Do not use it for narration or for anything specific to the current task only.',
|
||||
inputSchema: z.object({
|
||||
kind: z
|
||||
.enum(['fact', 'decision', 'gotcha', 'command'])
|
||||
.describe('fact: how it is. decision: what was chosen and why. gotcha: a trap. command: an invocation that works'),
|
||||
text: z.string().describe('One self-contained line, understandable with no other context'),
|
||||
}),
|
||||
execute: async ({ kind, text }) => {
|
||||
const entry = await this.add(kind, text);
|
||||
if (!entry) return `Already recorded: ${text.trim()}`;
|
||||
return `Remembered as ${entry.kind} (${this.entries.length} stored): ${entry.text}`;
|
||||
},
|
||||
}),
|
||||
|
||||
recall: tool({
|
||||
description:
|
||||
'Search what you recorded about this project in earlier sessions. Use it before investigating anything ' +
|
||||
'that might already be known, and when the user refers to past work.',
|
||||
inputSchema: z.object({
|
||||
query: z.string().describe('Words that would appear in the note'),
|
||||
}),
|
||||
execute: async ({ query }) => {
|
||||
const found = await this.search(query);
|
||||
if (found.length === 0) return `Nothing recorded about "${query}".`;
|
||||
return found.map((e) => `(${e.kind}) ${e.text}`).join('\n');
|
||||
},
|
||||
}),
|
||||
|
||||
forget: tool({
|
||||
description:
|
||||
'Remove a memory that turned out to be wrong or is now obsolete. Search with recall first to get its text.',
|
||||
inputSchema: z.object({
|
||||
text: z.string().describe('Exact text of the memory to remove, or a distinctive part of it'),
|
||||
}),
|
||||
execute: async ({ text }) => {
|
||||
await this.load();
|
||||
const needle = text.trim().toLowerCase();
|
||||
const before = this.entries.length;
|
||||
this.entries = this.entries.filter((e) => !e.text.toLowerCase().includes(needle));
|
||||
const removed = before - this.entries.length;
|
||||
if (removed > 0) await this.persist();
|
||||
return removed > 0 ? `Forgot ${removed} memor${removed === 1 ? 'y' : 'ies'}.` : `No memory matches "${text}".`;
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export { fileFor as memoryFileFor, root as memoryDir, KIND_LABEL };
|
||||
|
||||
function isEntry(value: unknown): value is MemoryEntry {
|
||||
if (!value || typeof value !== 'object') return false;
|
||||
const v = value as Record<string, unknown>;
|
||||
return (
|
||||
typeof v['id'] === 'string' &&
|
||||
typeof v['text'] === 'string' &&
|
||||
typeof v['createdAt'] === 'string' &&
|
||||
typeof v['hits'] === 'number' &&
|
||||
['fact', 'decision', 'gotcha', 'command'].includes(String(v['kind']))
|
||||
);
|
||||
}
|
||||
+120
@@ -0,0 +1,120 @@
|
||||
import { tool } from 'ai';
|
||||
import { z } from 'zod';
|
||||
|
||||
export type TodoStatus = 'pending' | 'in_progress' | 'done' | 'blocked';
|
||||
export type Todo = { content: string; status: TodoStatus; note?: string };
|
||||
|
||||
export type NotebookState = { todos: Todo[] };
|
||||
|
||||
const MARK: Record<TodoStatus, string> = {
|
||||
pending: '[ ]',
|
||||
in_progress: '[~]',
|
||||
done: '[x]',
|
||||
blocked: '[!]',
|
||||
};
|
||||
|
||||
const STATUSES = ['pending', 'in_progress', 'done', 'blocked'] as const;
|
||||
|
||||
const renderTodo = (t: Todo) => `${MARK[t.status]} ${t.content}${t.note?.trim() ? ` (${t.note.trim()})` : ''}`;
|
||||
|
||||
function isTodo(value: unknown): value is Todo {
|
||||
if (!value || typeof value !== 'object') return false;
|
||||
const v = value as Record<string, unknown>;
|
||||
return typeof v['content'] === 'string' && (STATUSES as readonly string[]).includes(String(v['status']));
|
||||
}
|
||||
|
||||
/**
|
||||
* The task list for the current session.
|
||||
*
|
||||
* Both `pruneMessages` and `/compact` destroy tool results and older turns, so a plan
|
||||
* recorded only in the transcript is lost exactly when a long task needs it. This is
|
||||
* re-rendered into the system prompt on every step instead, so it survives both.
|
||||
* Anything that should outlive the session belongs in Memory, not here.
|
||||
*/
|
||||
export class Notebook {
|
||||
private todos: Todo[] = [];
|
||||
|
||||
constructor(private readonly onChange?: (state: NotebookState) => void) {}
|
||||
|
||||
state(): NotebookState {
|
||||
return { todos: this.todos.map((t) => ({ ...t })) };
|
||||
}
|
||||
|
||||
restore(state: Partial<NotebookState> | undefined): void {
|
||||
if (Array.isArray(state?.todos)) this.todos = state.todos.filter(isTodo);
|
||||
}
|
||||
|
||||
clear(): void {
|
||||
this.todos = [];
|
||||
this.onChange?.(this.state());
|
||||
}
|
||||
|
||||
progress(): { done: number; total: number; blocked: number; current?: Todo } {
|
||||
const current = this.todos.find((t) => t.status === 'in_progress');
|
||||
return {
|
||||
done: this.todos.filter((t) => t.status === 'done').length,
|
||||
total: this.todos.length,
|
||||
blocked: this.todos.filter((t) => t.status === 'blocked').length,
|
||||
...(current ? { current } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
render(): string {
|
||||
if (this.todos.length === 0) return '';
|
||||
const { done, total, blocked } = this.progress();
|
||||
const header = `\nYour task list (${done}/${total} done${blocked > 0 ? `, ${blocked} blocked` : ''}). Keep it current with todo_write:`;
|
||||
return `${header}\n${this.todos.map(renderTodo).join('\n')}`;
|
||||
}
|
||||
|
||||
tools() {
|
||||
return {
|
||||
todo_write: tool({
|
||||
description:
|
||||
'Record or update your task list for a multi-step job. Send the whole list every time; it replaces the ' +
|
||||
'previous one. Exactly one task should be in_progress. Mark a task done the moment it is finished, not in ' +
|
||||
'a batch at the end. Use blocked with a note when something outside your control stops you. The list is ' +
|
||||
'shown to the user and survives context compaction, so it is where your plan lives. ' +
|
||||
'Skip it entirely for single-step work.',
|
||||
inputSchema: z.object({
|
||||
todos: z
|
||||
.array(
|
||||
z.object({
|
||||
content: z
|
||||
.string()
|
||||
.describe('One concrete action with a verifiable outcome, e.g. "add limit/offset to listUsers()"'),
|
||||
status: z.enum(STATUSES),
|
||||
note: z
|
||||
.string()
|
||||
.optional()
|
||||
.describe('Required for blocked: what is blocking it. Otherwise a short finding worth keeping.'),
|
||||
}),
|
||||
)
|
||||
.describe('The complete list, in the order you will do them'),
|
||||
}),
|
||||
execute: async ({ todos }) => {
|
||||
this.todos = todos;
|
||||
this.onChange?.(this.state());
|
||||
|
||||
const active = todos.filter((t) => t.status === 'in_progress');
|
||||
const blocked = todos.filter((t) => t.status === 'blocked');
|
||||
const { done, total } = this.progress();
|
||||
|
||||
const warnings: string[] = [];
|
||||
if (active.length > 1) warnings.push(`${active.length} tasks are in_progress; keep it to one.`);
|
||||
if (active.length === 0 && done < total && blocked.length < total - done) {
|
||||
warnings.push('nothing is in_progress; mark what you are working on.');
|
||||
}
|
||||
for (const t of blocked) {
|
||||
if (!t.note?.trim()) warnings.push(`"${t.content}" is blocked with no note saying why.`);
|
||||
}
|
||||
|
||||
const lines = [`Task list updated: ${done}/${total} done.`, ...todos.map(renderTodo)];
|
||||
if (warnings.length > 0) lines.push(`Warning: ${warnings.join(' ')}`);
|
||||
return lines.join('\n');
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export { MARK as TODO_MARK, STATUSES };
|
||||
@@ -0,0 +1,74 @@
|
||||
import { tool } from 'ai';
|
||||
import { z } from 'zod';
|
||||
import type { Plugin } from './plugins';
|
||||
|
||||
/**
|
||||
* Commands that destroy work irreversibly. Approval alone is a weak defence here:
|
||||
* a user holding `a` for a batch of edits will approve one of these without reading it,
|
||||
* so they are refused outright and the user has to run them by hand.
|
||||
*/
|
||||
const DESTRUCTIVE: { re: RegExp; why: string }[] = [
|
||||
{ re: /\brm\s+(-[a-zA-Z]*\s+)*-[a-zA-Z]*[rf]/, why: 'recursive or forced delete' },
|
||||
{ re: /\bgit\s+reset\s+--hard\b/, why: 'discards uncommitted work' },
|
||||
{ re: /\bgit\s+clean\s+-[a-zA-Z]*f/, why: 'deletes untracked files' },
|
||||
{ re: /\bgit\s+push\b.*(--force\b|--force-with-lease\b|\s-f\b)/, why: 'rewrites remote history' },
|
||||
{ re: /\bgit\s+branch\s+-D\b/, why: 'deletes a branch without a merge check' },
|
||||
{ re: /\b(DROP|TRUNCATE)\s+(TABLE|DATABASE|SCHEMA)\b/i, why: 'destroys database data' },
|
||||
{ re: /\bmkfs(\.\w+)?\b|\bdd\s+[^|]*of=\/dev\//, why: 'writes to a raw device' },
|
||||
{ re: />\s*\/dev\/(sd|nvme|disk)/, why: 'writes to a raw device' },
|
||||
{ re: /\bchmod\s+(-[a-zA-Z]*\s+)*777\b/, why: 'makes files world-writable' },
|
||||
{ re: /\b(shutdown|reboot|halt)\b/, why: 'affects the whole machine' },
|
||||
{ re: /:\(\)\s*\{.*\}\s*;\s*:/, why: 'fork bomb' },
|
||||
{ re: /\bcurl\b[^|]*\|\s*(ba|z|k)?sh\b|\bwget\b[^|]*\|\s*(ba|z|k)?sh\b/, why: 'pipes a download straight into a shell' },
|
||||
];
|
||||
|
||||
export const guardPlugin: Plugin = {
|
||||
name: 'guard',
|
||||
description: 'refuses irreversible shell commands outright',
|
||||
appendix:
|
||||
'The guard plugin refuses irreversible shell commands (recursive deletes, hard resets, force pushes, ' +
|
||||
'DROP TABLE, piping downloads into a shell). If one is refused, do not work around it: tell the user ' +
|
||||
'what needs running and let them do it themselves.',
|
||||
beforeToolCall: ({ toolName, input }) => {
|
||||
if (toolName !== 'bash') return undefined;
|
||||
const command = String((input as { command?: unknown } | null)?.command ?? '');
|
||||
if (!command) return undefined;
|
||||
for (const { re, why } of DESTRUCTIVE) {
|
||||
if (re.test(command)) {
|
||||
return `refusing "${command.slice(0, 120)}" (${why}). Ask the user to run it themselves if it is really needed.`;
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
};
|
||||
|
||||
export const bellPlugin: Plugin = {
|
||||
name: 'bell',
|
||||
description: 'rings the terminal bell when a turn ends',
|
||||
afterTurn: () => {
|
||||
process.stderr.write('\u0007');
|
||||
},
|
||||
};
|
||||
|
||||
export const timePlugin: Plugin = {
|
||||
name: 'time',
|
||||
description: 'adds a current_time tool',
|
||||
autoApprove: ['current_time'],
|
||||
tools: {
|
||||
current_time: tool({
|
||||
description: 'Current date and time in ISO 8601, with the local timezone. Use it when the date matters.',
|
||||
inputSchema: z.object({}),
|
||||
execute: async () => {
|
||||
const now = new Date();
|
||||
return `${now.toISOString()} (local: ${now.toString()})`;
|
||||
},
|
||||
}),
|
||||
},
|
||||
};
|
||||
|
||||
export const BUILTIN_PLUGINS: Plugin[] = [guardPlugin, bellPlugin, timePlugin];
|
||||
|
||||
/** Enabled unless the config turns them off. bell is opt-in; a bell per turn is intrusive. */
|
||||
export const DEFAULT_ENABLED = ['guard', 'time'];
|
||||
|
||||
export { DESTRUCTIVE };
|
||||
@@ -0,0 +1,77 @@
|
||||
import type { ToolSet } from 'ai';
|
||||
|
||||
export type ToolCallContext = {
|
||||
toolName: string;
|
||||
input: unknown;
|
||||
cwd: string;
|
||||
};
|
||||
|
||||
/** Returning a string blocks the call; the string is handed to the model as the reason. */
|
||||
export type BeforeToolCall = (ctx: ToolCallContext) => string | undefined | Promise<string | undefined>;
|
||||
|
||||
export type Plugin = {
|
||||
name: string;
|
||||
description: string;
|
||||
/** Extra tools contributed by this plugin. */
|
||||
tools?: ToolSet;
|
||||
/** Tool names that should never prompt for approval. */
|
||||
autoApprove?: readonly string[];
|
||||
beforeToolCall?: BeforeToolCall;
|
||||
afterTurn?: () => void | Promise<void>;
|
||||
/** Text appended to the system prompt. */
|
||||
appendix?: string;
|
||||
};
|
||||
|
||||
export type PluginHost = {
|
||||
plugins: Plugin[];
|
||||
tools: ToolSet;
|
||||
autoApprove: string[];
|
||||
appendix: string;
|
||||
/** Runs every beforeToolCall hook; the first block wins. */
|
||||
guard: BeforeToolCall;
|
||||
afterTurn: () => Promise<void>;
|
||||
errors: { plugin: string; message: string }[];
|
||||
};
|
||||
|
||||
export function createHost(plugins: Plugin[], errors: PluginHost['errors'] = []): PluginHost {
|
||||
const tools: ToolSet = {};
|
||||
const autoApprove: string[] = [];
|
||||
const appendices: string[] = [];
|
||||
|
||||
for (const p of plugins) {
|
||||
for (const [name, t] of Object.entries(p.tools ?? {})) tools[name] = t;
|
||||
autoApprove.push(...(p.autoApprove ?? []));
|
||||
if (p.appendix) appendices.push(p.appendix);
|
||||
}
|
||||
|
||||
return {
|
||||
plugins,
|
||||
tools,
|
||||
autoApprove,
|
||||
appendix: appendices.length > 0 ? `\n${appendices.join('\n')}` : '',
|
||||
errors,
|
||||
guard: async (ctx) => {
|
||||
for (const p of plugins) {
|
||||
if (!p.beforeToolCall) continue;
|
||||
try {
|
||||
const blocked = await p.beforeToolCall(ctx);
|
||||
if (blocked) return `Blocked by the ${p.name} plugin: ${blocked}`;
|
||||
} catch (e) {
|
||||
// A broken hook must not take the agent down, but it must not silently
|
||||
// allow the call either: treat a throwing guard as a block.
|
||||
return `The ${p.name} plugin failed while checking this call: ${e instanceof Error ? e.message : String(e)}`;
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
afterTurn: async () => {
|
||||
for (const p of plugins) {
|
||||
try {
|
||||
await p.afterTurn?.();
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
export type Rate = { inputPerMTok: number; outputPerMTok: number };
|
||||
|
||||
/**
|
||||
* USD per million tokens. Prefix match on the model id, longest first, so
|
||||
* `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates
|
||||
* drift, so this is a best-effort estimate rather than a billing source.
|
||||
*/
|
||||
const RATES: Record<string, Rate> = {
|
||||
'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 },
|
||||
'claude-sonnet-4': { inputPerMTok: 3, outputPerMTok: 15 },
|
||||
'claude-haiku-4': { inputPerMTok: 1, outputPerMTok: 5 },
|
||||
'claude-3-5-haiku': { inputPerMTok: 0.8, outputPerMTok: 4 },
|
||||
'gpt-5-mini': { inputPerMTok: 0.25, outputPerMTok: 2 },
|
||||
'gpt-5-nano': { inputPerMTok: 0.05, outputPerMTok: 0.4 },
|
||||
'gpt-5': { inputPerMTok: 1.25, outputPerMTok: 10 },
|
||||
'gpt-4o-mini': { inputPerMTok: 0.15, outputPerMTok: 0.6 },
|
||||
'gpt-4o': { inputPerMTok: 2.5, outputPerMTok: 10 },
|
||||
'o4-mini': { inputPerMTok: 1.1, outputPerMTok: 4.4 },
|
||||
'deepseek-chat': { inputPerMTok: 0.27, outputPerMTok: 1.1 },
|
||||
'deepseek-reasoner': { inputPerMTok: 0.55, outputPerMTok: 2.19 },
|
||||
'grok-4': { inputPerMTok: 3, outputPerMTok: 15 },
|
||||
};
|
||||
|
||||
/** Strips a provider prefix such as `anthropic/` that OpenRouter-style ids carry. */
|
||||
const bare = (modelId: string) => modelId.toLowerCase().split('/').at(-1) ?? modelId.toLowerCase();
|
||||
|
||||
export function rateFor(modelId: string): Rate | undefined {
|
||||
const id = bare(modelId);
|
||||
const key = Object.keys(RATES)
|
||||
.filter((k) => id.startsWith(k))
|
||||
.sort((a, b) => b.length - a.length)[0];
|
||||
return key ? RATES[key] : undefined;
|
||||
}
|
||||
|
||||
export function costOf(modelId: string, inputTokens: number, outputTokens: number): number | undefined {
|
||||
const rate = rateFor(modelId);
|
||||
if (!rate) return undefined;
|
||||
return (inputTokens / 1_000_000) * rate.inputPerMTok + (outputTokens / 1_000_000) * rate.outputPerMTok;
|
||||
}
|
||||
|
||||
export function formatUsd(amount: number): string {
|
||||
if (amount === 0) return '$0.00';
|
||||
if (amount < 0.01) return `$${amount.toFixed(4)}`;
|
||||
return `$${amount.toFixed(2)}`;
|
||||
}
|
||||
|
||||
/** One-line token and cost summary, omitting the cost when the model is unpriced. */
|
||||
export function usageLine(modelId: string, inputTokens: number, outputTokens: number): string {
|
||||
const tokens = `${inputTokens} in / ${outputTokens} out tokens`;
|
||||
const cost = costOf(modelId, inputTokens, outputTokens);
|
||||
return cost === undefined ? `${tokens} (${modelId} is unpriced)` : `${tokens} - ${formatUsd(cost)}`;
|
||||
}
|
||||
+141
@@ -0,0 +1,141 @@
|
||||
import { formatInstructions, type Instructions } from './instructions';
|
||||
|
||||
export type PromptParts = {
|
||||
cwd: string;
|
||||
instructions?: Instructions;
|
||||
/** Session task list from the Notebook. */
|
||||
notebook?: string;
|
||||
/** Durable project memory. */
|
||||
memory?: string;
|
||||
/** Skill catalogue: names and descriptions only. */
|
||||
skills?: string;
|
||||
/** Behaviour appendix from the selected agent variant. */
|
||||
agent?: string;
|
||||
/** Appendices contributed by plugins. */
|
||||
plugins?: string;
|
||||
/** Tool names actually offered this turn, so the prompt cannot describe a tool that is absent. */
|
||||
availableTools?: readonly string[];
|
||||
/** True when the ask tool has somewhere to send a question. */
|
||||
canAsk?: boolean;
|
||||
};
|
||||
|
||||
type ToolDoc = { name: string; line: string };
|
||||
|
||||
/**
|
||||
* Guidance per tool, beyond the schema description the model already receives.
|
||||
*
|
||||
* The schema says what a tool takes; this says when to reach for it and what goes
|
||||
* wrong. Only tools actually offered are described, because a prompt that mentions
|
||||
* a withheld tool teaches the model to attempt calls that cannot succeed.
|
||||
*/
|
||||
const TOOL_DOCS: ToolDoc[] = [
|
||||
{ name: 'read_file', line: 'read before you edit. Never describe code you have not opened.' },
|
||||
{
|
||||
name: 'glob',
|
||||
line: 'find files by pattern. Skips binaries and .gitignore; pass includeIgnored to look anyway.',
|
||||
},
|
||||
{
|
||||
name: 'grep',
|
||||
line: 'search contents. Prefer it over reading many files; scope with include to keep results small.',
|
||||
},
|
||||
{
|
||||
name: 'edit_file',
|
||||
line: 'oldString must match byte-for-byte including indentation, and be unique. Include surrounding lines to disambiguate. Prefer several small edits over one large rewrite.',
|
||||
},
|
||||
{ name: 'write_file', line: 'new files and full rewrites only. Reach for edit_file on anything that exists.' },
|
||||
{
|
||||
name: 'bash',
|
||||
line: 'builds, tests, git, package managers. Output streams live. Long-running commands are fine; interactive ones are not.',
|
||||
},
|
||||
{
|
||||
name: 'task',
|
||||
line: 'delegate a read-only search to a subagent. Its prompt must be self-contained; it sees none of this conversation. Worth it when a search would span many files, wasteful for a single grep.',
|
||||
},
|
||||
{
|
||||
name: 'ask',
|
||||
line: 'stop and ask the user. Cheaper than a wrong guess when a request has two readings that lead to different work.',
|
||||
},
|
||||
{
|
||||
name: 'todo_write',
|
||||
line: 'your plan for a multi-step job. Send the whole list each time. One task in_progress. Mark done immediately, not in a batch.',
|
||||
},
|
||||
{ name: 'remember', line: 'record something still true next session: a decision, a working command, a trap.' },
|
||||
{ name: 'recall', line: 'search what you recorded before. Try it before investigating something possibly known.' },
|
||||
{ name: 'forget', line: 'remove a memory that turned out wrong.' },
|
||||
{ name: 'skill', line: 'load detailed instructions for a kind of task. Call it before starting, not after.' },
|
||||
{ name: 'current_time', line: 'the current date and time, when it matters.' },
|
||||
];
|
||||
|
||||
function renderTools(available: readonly string[]): string {
|
||||
const known = TOOL_DOCS.filter((d) => available.includes(d.name));
|
||||
const extra = available.filter((name) => !TOOL_DOCS.some((d) => d.name === name)).sort();
|
||||
|
||||
const lines = known.map((d) => `- ${d.name}: ${d.line}`);
|
||||
|
||||
const mcp = extra.filter((n) => n.startsWith('mcp__'));
|
||||
const other = extra.filter((n) => !n.startsWith('mcp__'));
|
||||
if (mcp.length > 0) {
|
||||
lines.push(
|
||||
`- ${mcp.join(', ')}: from MCP servers, named mcp__<server>__<tool>. Each needs approval; read its own description before calling.`,
|
||||
);
|
||||
}
|
||||
for (const name of other) lines.push(`- ${name}: see its own description.`);
|
||||
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
export function systemPrompt(parts: PromptParts): string {
|
||||
const {
|
||||
cwd,
|
||||
instructions = [],
|
||||
notebook = '',
|
||||
memory = '',
|
||||
skills = '',
|
||||
agent = '',
|
||||
plugins = '',
|
||||
availableTools,
|
||||
canAsk = false,
|
||||
} = parts;
|
||||
|
||||
const toolNames = availableTools ?? TOOL_DOCS.map((d) => d.name);
|
||||
const canEdit = toolNames.includes('edit_file') || toolNames.includes('write_file');
|
||||
const canRun = toolNames.includes('bash');
|
||||
|
||||
const workflow = [
|
||||
'- Read before you write. Ground every claim about the code in something you actually opened.',
|
||||
'- Make the smallest change that solves the task. A bugfix diff contains only the bug.',
|
||||
'- Match the existing style, libraries, and conventions. Sample a neighbouring file before inventing a pattern.',
|
||||
canEdit
|
||||
? '- write_file, edit_file, and bash need the user to approve each call. If one is denied, stop and ask what to do instead of working around it.'
|
||||
: '- You have no tools that change anything this turn. Investigate and report; do not describe edits as if you had made them.',
|
||||
canRun
|
||||
? "- After changing code, verify it: run the project's build or tests. \"Should work\" is not verification."
|
||||
: '- You cannot run commands this turn, so say what should be run to verify rather than claiming it passes.',
|
||||
'- When something fails twice, stop and re-read the error literally. Check that the code you think is running is the code that is running.',
|
||||
canAsk
|
||||
? '- Ask rather than guess when two readings of the request lead to different work. Decide small things yourself and say what you assumed.'
|
||||
: '- No one can answer a question this run. Decide yourself and state the assumption plainly.',
|
||||
].join('\n');
|
||||
|
||||
return `You are Shiro Neko, a coding agent working in the user's terminal.
|
||||
|
||||
Environment
|
||||
- Workspace root: ${cwd}
|
||||
- Platform: ${process.platform}
|
||||
- Paths are resolved inside the workspace. Anything outside it is refused.
|
||||
|
||||
Tools available to you now
|
||||
${renderTools(toolNames)}
|
||||
|
||||
How to work
|
||||
${workflow}
|
||||
|
||||
How to reply
|
||||
- Lead with the outcome. The user wants to know what happened, not what you are about to do.
|
||||
- No preamble, no restating the task, no summary of your own summary.
|
||||
- Markdown is rendered: use fenced code blocks for code, backticks for identifiers and paths.
|
||||
- Report failures with their actual output. Never imply a command passed when you did not run it.
|
||||
${formatInstructions(instructions, cwd)}${memory}${skills}${agent}${plugins}${notebook}`;
|
||||
}
|
||||
|
||||
export { TOOL_DOCS, renderTools };
|
||||
@@ -0,0 +1,141 @@
|
||||
import type { ProviderName } from './config';
|
||||
|
||||
export type ProviderPreset = {
|
||||
id: string;
|
||||
label: string;
|
||||
/** Which wire protocol to speak. */
|
||||
kind: ProviderName;
|
||||
baseURL: string;
|
||||
/** Env var checked before asking for a key. */
|
||||
envKey?: string;
|
||||
/** Servers that ignore auth, e.g. a local Ollama. */
|
||||
keyless?: boolean;
|
||||
keyHint?: string;
|
||||
fallbackModels?: string[];
|
||||
};
|
||||
|
||||
export const PRESETS: ProviderPreset[] = [
|
||||
{
|
||||
id: 'anthropic',
|
||||
label: 'Anthropic',
|
||||
kind: 'anthropic',
|
||||
baseURL: 'https://api.anthropic.com/v1',
|
||||
envKey: 'ANTHROPIC_API_KEY',
|
||||
keyHint: 'sk-ant-...',
|
||||
fallbackModels: ['claude-sonnet-4-5', 'claude-opus-4-1', 'claude-haiku-4-5'],
|
||||
},
|
||||
{
|
||||
id: 'openai',
|
||||
label: 'OpenAI',
|
||||
kind: 'openai',
|
||||
baseURL: 'https://api.openai.com/v1',
|
||||
envKey: 'OPENAI_API_KEY',
|
||||
keyHint: 'sk-...',
|
||||
fallbackModels: ['gpt-5', 'gpt-5-mini', 'o4-mini'],
|
||||
},
|
||||
{
|
||||
id: 'openrouter',
|
||||
label: 'OpenRouter (many models, one key)',
|
||||
kind: 'openai',
|
||||
baseURL: 'https://openrouter.ai/api/v1',
|
||||
envKey: 'OPENROUTER_API_KEY',
|
||||
keyHint: 'sk-or-...',
|
||||
},
|
||||
{
|
||||
id: 'groq',
|
||||
label: 'Groq',
|
||||
kind: 'openai',
|
||||
baseURL: 'https://api.groq.com/openai/v1',
|
||||
envKey: 'GROQ_API_KEY',
|
||||
keyHint: 'gsk_...',
|
||||
},
|
||||
{
|
||||
id: 'deepseek',
|
||||
label: 'DeepSeek',
|
||||
kind: 'openai',
|
||||
baseURL: 'https://api.deepseek.com/v1',
|
||||
envKey: 'DEEPSEEK_API_KEY',
|
||||
keyHint: 'sk-...',
|
||||
},
|
||||
{
|
||||
id: 'xai',
|
||||
label: 'xAI (Grok)',
|
||||
kind: 'openai',
|
||||
baseURL: 'https://api.x.ai/v1',
|
||||
envKey: 'XAI_API_KEY',
|
||||
keyHint: 'xai-...',
|
||||
},
|
||||
{
|
||||
id: 'ollama',
|
||||
label: 'Ollama (local)',
|
||||
kind: 'openai',
|
||||
baseURL: 'http://localhost:11434/v1',
|
||||
keyless: true,
|
||||
},
|
||||
{
|
||||
id: 'lmstudio',
|
||||
label: 'LM Studio (local)',
|
||||
kind: 'openai',
|
||||
baseURL: 'http://localhost:1234/v1',
|
||||
keyless: true,
|
||||
},
|
||||
{
|
||||
id: 'custom-openai',
|
||||
label: 'Custom OpenAI-compatible endpoint',
|
||||
kind: 'openai',
|
||||
baseURL: '',
|
||||
},
|
||||
{
|
||||
id: 'custom-anthropic',
|
||||
label: 'Custom Anthropic-compatible endpoint',
|
||||
kind: 'anthropic',
|
||||
baseURL: '',
|
||||
},
|
||||
];
|
||||
|
||||
export const presetById = (id: string) => PRESETS.find((p) => p.id === id);
|
||||
|
||||
export type ModelListResult = { models: string[]; source: 'api' | 'fallback'; warning?: string };
|
||||
|
||||
type ModelsResponse = { data?: unknown };
|
||||
|
||||
/**
|
||||
* Both OpenAI- and Anthropic-compatible servers expose `GET /v1/models` with a
|
||||
* `data[].id` shape, only the auth header differs. A server that does not
|
||||
* implement it is not fatal: the caller can still type a model id by hand.
|
||||
*/
|
||||
export async function fetchModels(
|
||||
preset: Pick<ProviderPreset, 'kind' | 'baseURL' | 'fallbackModels'>,
|
||||
apiKey: string,
|
||||
timeoutMs = 15_000,
|
||||
): Promise<ModelListResult> {
|
||||
const url = `${preset.baseURL.replace(/\/+$/, '')}/models`;
|
||||
const headers: Record<string, string> =
|
||||
preset.kind === 'anthropic'
|
||||
? { 'x-api-key': apiKey, 'anthropic-version': '2023-06-01' }
|
||||
: { authorization: `Bearer ${apiKey}` };
|
||||
|
||||
const fallback = (warning: string): ModelListResult => ({
|
||||
models: preset.fallbackModels ?? [],
|
||||
source: 'fallback',
|
||||
warning,
|
||||
});
|
||||
|
||||
try {
|
||||
const res = await fetch(url, { headers, signal: AbortSignal.timeout(timeoutMs) });
|
||||
if (!res.ok) {
|
||||
const body = (await res.text()).slice(0, 200);
|
||||
return fallback(`${url} returned ${res.status}. ${body}`.trim());
|
||||
}
|
||||
const json = (await res.json()) as ModelsResponse;
|
||||
const models = Array.isArray(json.data)
|
||||
? json.data
|
||||
.map((m) => (m && typeof m === 'object' ? (m as { id?: unknown }).id : undefined))
|
||||
.filter((id): id is string => typeof id === 'string')
|
||||
: [];
|
||||
if (models.length === 0) return fallback(`${url} listed no models.`);
|
||||
return { models: models.sort(), source: 'api' };
|
||||
} catch (e) {
|
||||
return fallback(e instanceof Error ? e.message : String(e));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
import { pruneMessages, type ModelMessage } from 'ai';
|
||||
|
||||
type Part = { type: string; providerOptions?: Record<string, Record<string, unknown>> };
|
||||
|
||||
/** Parts the OpenAI responses API refuses to accept without their reasoning item. */
|
||||
const DEPENDENT = new Set(['text', 'tool-call']);
|
||||
|
||||
function itemId(part: Part): string | undefined {
|
||||
for (const options of Object.values(part.providerOptions ?? {})) {
|
||||
const id = options['itemId'];
|
||||
if (typeof id === 'string') return id;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const partsOf = (message: ModelMessage): Part[] =>
|
||||
message.role === 'assistant' && Array.isArray(message.content) ? (message.content as Part[]) : [];
|
||||
|
||||
/**
|
||||
* Drops assistant parts left orphaned by reasoning removal.
|
||||
*
|
||||
* The OpenAI responses API treats a `message` item as a dependent of the `reasoning`
|
||||
* item from the same response: send the message without its reasoning and the request
|
||||
* is rejected with 400 "was provided without its required 'reasoning' item".
|
||||
* `pruneMessages({ reasoning: 'all' })` strips the reasoning and keeps the message,
|
||||
* producing exactly that request.
|
||||
*
|
||||
* The two carry different item ids, so they cannot be matched by id. What links them
|
||||
* is the assistant message they arrived in: one message is one response, and its
|
||||
* reasoning item covers every other item in it.
|
||||
*
|
||||
* Reasoning only disappears from turns pruning is already discarding, so dropping the
|
||||
* orphaned text costs nothing pruning was not already spending.
|
||||
*/
|
||||
export function dropOrphanedItems(before: ModelMessage[], after: ModelMessage[]): ModelMessage[] {
|
||||
const survivingReasoning = new Set<string>();
|
||||
for (const message of after) {
|
||||
for (const part of partsOf(message)) {
|
||||
if (part.type !== 'reasoning') continue;
|
||||
const id = itemId(part);
|
||||
if (id) survivingReasoning.add(id);
|
||||
}
|
||||
}
|
||||
|
||||
const orphaned = new Set<string>();
|
||||
for (const message of before) {
|
||||
const parts = partsOf(message);
|
||||
const reasoning = parts.filter((p) => p.type === 'reasoning').map(itemId);
|
||||
if (reasoning.length === 0) continue;
|
||||
if (reasoning.some((id) => id !== undefined && survivingReasoning.has(id))) continue;
|
||||
|
||||
for (const part of parts) {
|
||||
if (!DEPENDENT.has(part.type)) continue;
|
||||
const id = itemId(part);
|
||||
if (id) orphaned.add(id);
|
||||
}
|
||||
}
|
||||
|
||||
if (orphaned.size === 0) return after;
|
||||
|
||||
const cleaned: ModelMessage[] = [];
|
||||
for (const message of after) {
|
||||
const parts = partsOf(message);
|
||||
if (parts.length === 0) {
|
||||
cleaned.push(message);
|
||||
continue;
|
||||
}
|
||||
|
||||
const kept = parts.filter((part) => {
|
||||
const id = itemId(part);
|
||||
return id === undefined || !orphaned.has(id);
|
||||
});
|
||||
|
||||
if (kept.length > 0) cleaned.push({ ...message, content: kept } as ModelMessage);
|
||||
}
|
||||
|
||||
return cleaned;
|
||||
}
|
||||
|
||||
export type PruneOptions = Parameters<typeof pruneMessages>[0];
|
||||
|
||||
/** pruneMessages, then repair the provider-item dependencies it breaks. */
|
||||
export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
|
||||
const pruned = pruneMessages(options);
|
||||
return dropOrphanedItems(options.messages, pruned);
|
||||
}
|
||||
+381
@@ -0,0 +1,381 @@
|
||||
import {
|
||||
isStepCount,
|
||||
generateText,
|
||||
streamText,
|
||||
type LanguageModel,
|
||||
type ModelMessage,
|
||||
type ToolApprovalResponse,
|
||||
type ToolSet,
|
||||
} from 'ai';
|
||||
import { DEFAULT_VARIANT, sdkReasoning, renderAgent, type AgentVariant } from './agents';
|
||||
import { createAskTool, type AskFn } from './ask';
|
||||
import type { Instructions } from './instructions';
|
||||
import type { Memory } from './memory';
|
||||
import { Notebook, type NotebookState } from './notebook';
|
||||
import type { PluginHost } from './plugins';
|
||||
import { systemPrompt } from './prompt';
|
||||
import { prunePreservingItems } from './prune';
|
||||
import { createSkillTool, renderSkills, type Skill } from './skills';
|
||||
import { MUTATING_TOOLS, onBashOutput, tools as builtinTools } from './tools';
|
||||
|
||||
export type ApprovalRequest = {
|
||||
approvalId: string;
|
||||
toolName: string;
|
||||
input: unknown;
|
||||
};
|
||||
|
||||
/** 'once' runs this call only; 'always' whitelists the tool for the rest of the session. */
|
||||
export type ApprovalDecision = 'once' | 'always' | 'deny';
|
||||
|
||||
export type AgentEvent =
|
||||
| { type: 'text'; text: string }
|
||||
| { type: 'reasoning'; text: string }
|
||||
| { type: 'tool-call'; id: string; name: string; input: unknown }
|
||||
| { type: 'tool-output'; id: string; chunk: string }
|
||||
| { type: 'tool-result'; id: string; name: string; output: unknown }
|
||||
| { type: 'tool-error'; id: string; name: string; error: unknown }
|
||||
| { type: 'tool-denied'; name: string }
|
||||
| { type: 'compacted'; before: number; after: number }
|
||||
| { type: 'notice'; text: string }
|
||||
| { type: 'error'; error: unknown }
|
||||
| { type: 'done'; inputTokens?: number; outputTokens?: number };
|
||||
|
||||
export type SessionOptions = {
|
||||
model: LanguageModel;
|
||||
askApproval: (req: ApprovalRequest) => Promise<ApprovalDecision>;
|
||||
yolo?: boolean;
|
||||
cwd?: string;
|
||||
maxSteps?: number;
|
||||
/** MCP and subagent tools merged on top of the built-ins. */
|
||||
extraTools?: ToolSet;
|
||||
/** Tool names that never prompt, e.g. the read-only subagent tool. */
|
||||
autoApprove?: readonly string[];
|
||||
/** Prune the history once the estimated token count crosses this. */
|
||||
compactThreshold?: number;
|
||||
/** Retries per model call for transient failures. */
|
||||
maxRetries?: number;
|
||||
/** AGENTS.md-style files appended to the system prompt. */
|
||||
instructions?: Instructions;
|
||||
/** Task list restored from a resumed session. */
|
||||
notebook?: NotebookState;
|
||||
/** Thinking level, tool restrictions, and behaviour appendix. */
|
||||
agent?: AgentVariant;
|
||||
skills?: Skill[];
|
||||
memory?: Memory;
|
||||
plugins?: PluginHost;
|
||||
/** Where an `ask` tool call goes. Omit in headless runs. */
|
||||
ask?: AskFn;
|
||||
messages?: ModelMessage[];
|
||||
onChange?: (messages: ModelMessage[]) => void;
|
||||
/** Live stdout/stderr from bash, for a UI that wants progress. */
|
||||
onToolOutput?: (id: string, chunk: string) => void;
|
||||
onNotebookChange?: (state: NotebookState) => void;
|
||||
};
|
||||
|
||||
const estimateTokens = (messages: ModelMessage[]) => Math.round(JSON.stringify(messages).length / 4);
|
||||
|
||||
export class Session {
|
||||
readonly messages: ModelMessage[];
|
||||
readonly tools: ToolSet;
|
||||
readonly notebook: Notebook;
|
||||
inputTokens = 0;
|
||||
outputTokens = 0;
|
||||
private model: LanguageModel;
|
||||
private variant: AgentVariant;
|
||||
private readonly alwaysAllow = new Set<string>();
|
||||
private controller: AbortController | undefined;
|
||||
|
||||
constructor(private readonly opts: SessionOptions) {
|
||||
this.messages = opts.messages ?? [];
|
||||
this.notebook = new Notebook(opts.onNotebookChange);
|
||||
this.notebook.restore(opts.notebook);
|
||||
this.model = opts.model;
|
||||
this.variant = opts.agent ?? DEFAULT_VARIANT;
|
||||
|
||||
const sessionTools = {
|
||||
...this.notebook.tools(),
|
||||
...(opts.memory ? opts.memory.tools() : {}),
|
||||
...(opts.skills && opts.skills.length > 0 ? { skill: createSkillTool(opts.skills) } : {}),
|
||||
...(opts.ask ? { ask: createAskTool(opts.ask) } : {}),
|
||||
};
|
||||
this.tools = { ...builtinTools, ...sessionTools, ...(opts.plugins?.tools ?? {}), ...(opts.extraTools ?? {}) };
|
||||
|
||||
for (const name of [
|
||||
...(opts.autoApprove ?? []),
|
||||
...(opts.plugins?.autoApprove ?? []),
|
||||
...Object.keys(sessionTools),
|
||||
]) {
|
||||
this.alwaysAllow.add(name);
|
||||
}
|
||||
}
|
||||
|
||||
setModel(model: LanguageModel): void {
|
||||
this.model = model;
|
||||
}
|
||||
|
||||
setAgent(variant: AgentVariant): void {
|
||||
this.variant = variant;
|
||||
}
|
||||
|
||||
agent(): AgentVariant {
|
||||
return this.variant;
|
||||
}
|
||||
|
||||
/** Tool names offered this turn; a read-only variant hides the rest. */
|
||||
activeTools(): string[] {
|
||||
const all = Object.keys(this.tools);
|
||||
if (!this.variant.allowTools) return all;
|
||||
return all.filter((name) => this.variant.allowTools!.includes(name));
|
||||
}
|
||||
|
||||
reset(): void {
|
||||
this.messages.length = 0;
|
||||
this.inputTokens = 0;
|
||||
this.outputTokens = 0;
|
||||
this.notebook.clear();
|
||||
this.opts.onChange?.(this.messages);
|
||||
}
|
||||
|
||||
replace(messages: ModelMessage[]): void {
|
||||
this.messages.length = 0;
|
||||
this.messages.push(...messages);
|
||||
this.opts.onChange?.(this.messages);
|
||||
}
|
||||
|
||||
abort(): void {
|
||||
this.controller?.abort();
|
||||
}
|
||||
|
||||
estimatedTokens(): number {
|
||||
return estimateTokens(this.messages);
|
||||
}
|
||||
|
||||
private systemFor(): string {
|
||||
return systemPrompt({
|
||||
cwd: this.opts.cwd ?? process.cwd(),
|
||||
instructions: this.opts.instructions ?? [],
|
||||
notebook: this.notebook.render(),
|
||||
memory: this.opts.memory?.render() ?? '',
|
||||
skills: renderSkills(this.opts.skills ?? []),
|
||||
agent: renderAgent(this.variant),
|
||||
plugins: this.opts.plugins?.appendix ?? '',
|
||||
availableTools: this.activeTools(),
|
||||
canAsk: this.opts.ask !== undefined && this.activeTools().includes('ask'),
|
||||
});
|
||||
}
|
||||
|
||||
/** Tools that mutate the workspace, plus every externally provided MCP tool. */
|
||||
private needsApproval(name: string): boolean {
|
||||
return (MUTATING_TOOLS as readonly string[]).includes(name) || name.startsWith('mcp__');
|
||||
}
|
||||
|
||||
/**
|
||||
* Approval decisions, evaluated per call by the SDK.
|
||||
*
|
||||
* A plugin guard denies outright and is checked before anything else, so `--yolo`
|
||||
* cannot bypass it. Only after the guard passes does yolo or the mutating-tool
|
||||
* rule decide whether the user is asked.
|
||||
*/
|
||||
private toolApproval(notices: string[]) {
|
||||
return async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
|
||||
const blocked = await this.opts.plugins?.guard({
|
||||
toolName: toolCall.toolName,
|
||||
input: toolCall.input,
|
||||
cwd: this.opts.cwd ?? process.cwd(),
|
||||
});
|
||||
if (blocked) {
|
||||
notices.push(blocked);
|
||||
return { type: 'denied' as const, reason: blocked };
|
||||
}
|
||||
if (this.opts.yolo) return undefined;
|
||||
if (!this.needsApproval(toolCall.toolName)) return undefined;
|
||||
if (this.alwaysAllow.has(toolCall.toolName)) return undefined;
|
||||
return 'user-approval' as const;
|
||||
};
|
||||
}
|
||||
|
||||
/** Replaces the history with a model-written summary. Backs the /compact command. */
|
||||
async summarize(): Promise<{ before: number; after: number }> {
|
||||
const before = this.messages.length;
|
||||
if (before === 0) return { before, after: 0 };
|
||||
|
||||
const { text } = await generateText({
|
||||
model: this.model,
|
||||
system:
|
||||
'Summarize this coding session for use as the sole context of a fresh session. ' +
|
||||
'Keep: the user goal, files touched with paths, decisions made, commands run and their outcome, ' +
|
||||
'and what remains to be done. Drop pleasantries and full file contents. Write it as notes, not prose.',
|
||||
messages: this.messages,
|
||||
maxRetries: this.opts.maxRetries ?? 3,
|
||||
});
|
||||
|
||||
this.messages.length = 0;
|
||||
this.messages.push({ role: 'user', content: `Summary of the session so far:\n\n${text}` });
|
||||
this.opts.onChange?.(this.messages);
|
||||
return { before, after: this.messages.length };
|
||||
}
|
||||
|
||||
async *send(userText: string): AsyncGenerator<AgentEvent> {
|
||||
this.messages.push({ role: 'user', content: userText });
|
||||
this.opts.onChange?.(this.messages);
|
||||
this.controller = new AbortController();
|
||||
const signal = this.controller.signal;
|
||||
const threshold = this.opts.compactThreshold ?? 120_000;
|
||||
|
||||
const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = [];
|
||||
onBashOutput(({ toolCallId, chunk }) => {
|
||||
outputs.push({ type: 'tool-output', id: toolCallId, chunk });
|
||||
this.opts.onToolOutput?.(toolCallId, chunk);
|
||||
});
|
||||
|
||||
try {
|
||||
yield* this.run(signal, threshold, outputs);
|
||||
} finally {
|
||||
onBashOutput(undefined);
|
||||
await this.opts.plugins?.afterTurn();
|
||||
}
|
||||
}
|
||||
|
||||
private async *run(
|
||||
signal: AbortSignal,
|
||||
threshold: number,
|
||||
outputs: Extract<AgentEvent, { type: 'tool-output' }>[],
|
||||
): AsyncGenerator<AgentEvent> {
|
||||
// Each iteration is one model run. A run ends either finished, or suspended
|
||||
// on tool approvals, in which case we collect decisions and run again.
|
||||
while (true) {
|
||||
const pending: ApprovalRequest[] = [];
|
||||
const compactions: Extract<AgentEvent, { type: 'compacted' }>[] = [];
|
||||
const guardNotices: string[] = [];
|
||||
let sawError = false;
|
||||
|
||||
const result = streamText({
|
||||
model: this.model,
|
||||
system: this.systemFor(),
|
||||
messages: this.messages,
|
||||
tools: this.tools,
|
||||
activeTools: this.activeTools(),
|
||||
reasoning: sdkReasoning(this.variant.thinking),
|
||||
toolApproval: this.toolApproval(guardNotices),
|
||||
stopWhen: isStepCount(this.variant.maxSteps ?? this.opts.maxSteps ?? 50),
|
||||
maxRetries: this.opts.maxRetries ?? 3,
|
||||
abortSignal: signal,
|
||||
prepareStep: ({ messages }) => {
|
||||
// Rebuilt every step: a todo_write earlier in this same run must be
|
||||
// visible to the steps that follow it, not only to the next turn.
|
||||
const instructions = this.systemFor();
|
||||
if (estimateTokens(messages) <= threshold) return { instructions };
|
||||
const pruned = prunePreservingItems({
|
||||
messages,
|
||||
reasoning: 'all',
|
||||
toolCalls: 'before-last-3-messages',
|
||||
emptyMessages: 'remove',
|
||||
});
|
||||
// prepareStep cannot yield, so queue the notice and drain it in the loop.
|
||||
compactions.push({ type: 'compacted', before: messages.length, after: pruned.length });
|
||||
return { instructions, messages: pruned };
|
||||
},
|
||||
});
|
||||
|
||||
// Every promise-shaped accessor settles independently of the stream. Any one
|
||||
// left without a rejection sink surfaces as an unhandled rejection on abort
|
||||
// or API failure, which scribbles over the Ink render.
|
||||
const sink = () => {};
|
||||
void result.responseMessages.then(undefined, sink);
|
||||
void result.usage.then(undefined, sink);
|
||||
void result.steps.then(undefined, sink);
|
||||
void result.finalStep.then(undefined, sink);
|
||||
void result.text.then(undefined, sink);
|
||||
void result.finishReason.then(undefined, sink);
|
||||
|
||||
try {
|
||||
for await (const part of result.stream) {
|
||||
while (compactions.length > 0) yield compactions.shift()!;
|
||||
while (outputs.length > 0) yield outputs.shift()!;
|
||||
while (guardNotices.length > 0) yield { type: 'notice', text: guardNotices.shift()! };
|
||||
switch (part.type) {
|
||||
case 'text-delta':
|
||||
yield { type: 'text', text: part.text };
|
||||
break;
|
||||
case 'reasoning-delta':
|
||||
yield { type: 'reasoning', text: part.text };
|
||||
break;
|
||||
case 'tool-call':
|
||||
yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input };
|
||||
break;
|
||||
case 'tool-result':
|
||||
yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output };
|
||||
break;
|
||||
case 'tool-error':
|
||||
yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error };
|
||||
break;
|
||||
case 'tool-approval-request':
|
||||
// A guard denial is answered by the SDK itself and arrives flagged
|
||||
// automatic; queueing it would prompt the user for a settled call.
|
||||
if (part.isAutomatic) break;
|
||||
pending.push({
|
||||
approvalId: part.approvalId,
|
||||
toolName: part.toolCall.toolName,
|
||||
input: part.toolCall.input,
|
||||
});
|
||||
break;
|
||||
case 'tool-approval-response':
|
||||
if (!part.approved) yield { type: 'tool-denied', name: part.toolCall.toolName };
|
||||
break;
|
||||
case 'tool-output-denied':
|
||||
yield { type: 'tool-denied', name: part.toolName };
|
||||
break;
|
||||
case 'abort':
|
||||
yield { type: 'done' };
|
||||
return;
|
||||
case 'error':
|
||||
sawError = true;
|
||||
yield { type: 'error', error: part.error };
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
if (signal.aborted) {
|
||||
yield { type: 'done' };
|
||||
return;
|
||||
}
|
||||
yield { type: 'error', error };
|
||||
return;
|
||||
}
|
||||
|
||||
// A stream that ended in an error has no response messages or usage to
|
||||
// await; touching them would throw NoOutputGeneratedError.
|
||||
if (sawError) return;
|
||||
|
||||
while (compactions.length > 0) yield compactions.shift()!;
|
||||
while (outputs.length > 0) yield outputs.shift()!;
|
||||
while (guardNotices.length > 0) yield { type: 'notice', text: guardNotices.shift()! };
|
||||
|
||||
this.messages.push(...(await result.responseMessages));
|
||||
this.opts.onChange?.(this.messages);
|
||||
|
||||
if (pending.length === 0) {
|
||||
const usage = await result.usage;
|
||||
this.inputTokens += usage.inputTokens ?? 0;
|
||||
this.outputTokens += usage.outputTokens ?? 0;
|
||||
yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens };
|
||||
return;
|
||||
}
|
||||
|
||||
const responses: ToolApprovalResponse[] = [];
|
||||
for (const req of pending) {
|
||||
const decision = this.alwaysAllow.has(req.toolName) ? 'always' : await this.opts.askApproval(req);
|
||||
if (decision === 'always') this.alwaysAllow.add(req.toolName);
|
||||
responses.push({
|
||||
type: 'tool-approval-response',
|
||||
approvalId: req.approvalId,
|
||||
approved: decision !== 'deny',
|
||||
...(decision === 'deny' ? { reason: 'User denied this tool call.' } : {}),
|
||||
});
|
||||
}
|
||||
this.messages.push({ role: 'tool', content: responses });
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
/**
|
||||
* Skills bundled with the binary.
|
||||
*
|
||||
* These are string constants rather than files on disk because `bun build --compile`
|
||||
* only embeds modules reachable through imports; a directory of .md files would be
|
||||
* missing from the shipped binary.
|
||||
*/
|
||||
export const BUILTIN_SKILLS: { name: string; source: string }[] = [
|
||||
{
|
||||
name: 'debug',
|
||||
source: `---
|
||||
name: debug
|
||||
description: Track down a bug whose cause is not obvious. Use when a test fails for unclear reasons, behaviour differs between environments, or an earlier fix did not hold.
|
||||
---
|
||||
|
||||
# Debugging
|
||||
|
||||
Do not guess. A guess that happens to work leaves the real cause in place.
|
||||
|
||||
## Reproduce first
|
||||
|
||||
Find the smallest command that shows the failure and record it with \`remember\`. If you
|
||||
cannot reproduce it, say so and ask what the user did differently — do not proceed on a
|
||||
hypothesis you cannot test.
|
||||
|
||||
## Three hypotheses, then evidence
|
||||
|
||||
Write down at least three causes that would produce this exact symptom. Rank them by how
|
||||
cheap they are to disprove, then disprove them in that order. State which one you are
|
||||
testing before you test it.
|
||||
|
||||
Evidence means observed output: a log line, a failing assertion, a value printed at the
|
||||
point of failure. "It should be X" is not evidence.
|
||||
|
||||
## Bisect when the space is large
|
||||
|
||||
- Recent regression: check what changed last.
|
||||
- Unclear layer: assert the value at each boundary until one is wrong.
|
||||
- Intermittent: run it in a loop and capture the failing case, do not reason about it abstractly.
|
||||
|
||||
## Fix the cause
|
||||
|
||||
Once you know the cause, fix that and nothing else. Do not tidy surrounding code in the
|
||||
same change — a bugfix diff should contain only the bug.
|
||||
|
||||
Write a test that fails before the fix and passes after. If you cannot express the bug as
|
||||
a test, say why.
|
||||
|
||||
## After two failed attempts
|
||||
|
||||
Stop. Re-read the error text literally, character by character. Check your assumption
|
||||
about which code is actually running: the wrong file, a stale build, a shadowed import,
|
||||
or a cached dependency accounts for most "impossible" bugs.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'review',
|
||||
source: `---
|
||||
name: review
|
||||
description: Review a diff or a file for defects. Use when asked to review, critique, or check code before it ships.
|
||||
---
|
||||
|
||||
# Code review
|
||||
|
||||
Severity order. Do not lead with style.
|
||||
|
||||
1. **Incorrect behaviour** — wrong result, wrong edge case, wrong state after failure.
|
||||
2. **Missing validation at trust boundaries** — user input, network responses, file contents,
|
||||
anything crossing a process line. Internal calls need no defensive checks.
|
||||
3. **Security** — injection, path traversal, secrets in logs or errors, missing authz.
|
||||
4. **Resource handling** — unclosed handles, unbounded growth, unawaited promises.
|
||||
5. **Clarity** — only when it will cause a future defect.
|
||||
|
||||
## For each finding
|
||||
|
||||
State file and line, what breaks, and the change. Show the fix as code when it is short.
|
||||
|
||||
Skip anything a formatter would fix. Skip preference. If a choice is defensible, leave it.
|
||||
|
||||
## Say when it is fine
|
||||
|
||||
A review that invents problems to look thorough is worse than a short one. If the change
|
||||
is correct, say so and stop.
|
||||
|
||||
## Verify, do not assume
|
||||
|
||||
Read the surrounding code before calling something a bug. A "missing" null check often
|
||||
exists one level up. Run the tests if that is what settles it.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'refactor',
|
||||
source: `---
|
||||
name: refactor
|
||||
description: Restructure code without changing behaviour. Use when asked to refactor, clean up, extract, or reorganise.
|
||||
---
|
||||
|
||||
# Refactoring
|
||||
|
||||
Behaviour must not change. That is the whole constraint.
|
||||
|
||||
## Establish the safety net first
|
||||
|
||||
Run the existing tests and record that they pass. If the code has no tests, write one that
|
||||
pins current behaviour — including the ugly parts — before touching anything. Refactoring
|
||||
untested code is rewriting it.
|
||||
|
||||
## Then move in small steps
|
||||
|
||||
One transformation at a time, tests green between each. Rename, then extract, then move —
|
||||
not all three in one edit. A large refactor that fails leaves you unable to tell which step
|
||||
broke it.
|
||||
|
||||
## What not to do
|
||||
|
||||
- Do not fix bugs while refactoring. Note them, finish, fix separately.
|
||||
- Do not add abstraction for a single caller. Duplication beats a premature interface.
|
||||
- Do not widen the scope. The request was this code, not its neighbours.
|
||||
- Do not change public API unless asked; if it must change, say so first.
|
||||
|
||||
## Done means
|
||||
|
||||
Tests pass, behaviour is identical, and the diff is smaller than the reader feared.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'test',
|
||||
source: `---
|
||||
name: test
|
||||
description: Write or repair tests. Use when adding coverage, fixing a flaky test, or asked how something should be tested.
|
||||
---
|
||||
|
||||
# Testing
|
||||
|
||||
A test earns its place by failing when the code is wrong.
|
||||
|
||||
## Match the project
|
||||
|
||||
Read two existing test files first. Use their runner, their assertion style, their file
|
||||
layout, their naming. A test that looks foreign is a test nobody maintains.
|
||||
|
||||
## Test behaviour, not implementation
|
||||
|
||||
Assert on what a caller observes. A test that reaches into private state breaks on every
|
||||
refactor and catches nothing.
|
||||
|
||||
Cover: the normal case, the boundaries, and the failure. Failure cases catch more real
|
||||
defects than happy paths.
|
||||
|
||||
## Never do this
|
||||
|
||||
- Do not assert what the code currently returns without knowing it is correct — that pins
|
||||
the bug.
|
||||
- Do not weaken an assertion to make a test pass. If it fails, either the code or the
|
||||
expectation is wrong; find out which.
|
||||
- Do not delete a failing test. It is telling you something.
|
||||
|
||||
## Flaky tests
|
||||
|
||||
A test that passes alone and fails in a suite is a shared-state problem: a global, a
|
||||
temp directory, a port, an unawaited promise, or ordering. Find which, do not add a retry.
|
||||
|
||||
## Verify
|
||||
|
||||
Run the test and watch it fail before the fix, pass after. A test you never saw fail is
|
||||
not known to work.
|
||||
`,
|
||||
},
|
||||
];
|
||||
+111
@@ -0,0 +1,111 @@
|
||||
import { tool } from 'ai';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { BUILTIN_SKILLS } from './skills-builtin';
|
||||
|
||||
export type SkillOrigin = 'builtin' | 'user' | 'project';
|
||||
|
||||
export type Skill = {
|
||||
name: string;
|
||||
description: string;
|
||||
origin: SkillOrigin;
|
||||
path?: string;
|
||||
body: string;
|
||||
};
|
||||
|
||||
const MAX_BODY = 20_000;
|
||||
|
||||
/**
|
||||
* Minimal YAML frontmatter reader: `name` and `description` only.
|
||||
* A real YAML parser would be a dependency for two string fields.
|
||||
*/
|
||||
export function parseSkill(source: string, origin: SkillOrigin, path?: string): Skill | undefined {
|
||||
const match = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/.exec(source.trimStart());
|
||||
if (!match) return undefined;
|
||||
|
||||
const meta: Record<string, string> = {};
|
||||
for (const line of match[1]!.split(/\r?\n/)) {
|
||||
const kv = /^([A-Za-z_-]+)\s*:\s*(.*)$/.exec(line.trim());
|
||||
if (kv) meta[kv[1]!.toLowerCase()] = kv[2]!.replace(/^["']|["']$/g, '').trim();
|
||||
}
|
||||
|
||||
const name = meta['name'];
|
||||
const description = meta['description'];
|
||||
if (!name || !description) return undefined;
|
||||
|
||||
return { name, description, origin, ...(path ? { path } : {}), body: match[2]!.trim().slice(0, MAX_BODY) };
|
||||
}
|
||||
|
||||
const skillDirs = (cwd: string) => [
|
||||
{ dir: join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'skills'), origin: 'user' as const },
|
||||
{ dir: join(cwd, '.shiro', 'skills'), origin: 'project' as const },
|
||||
];
|
||||
|
||||
/**
|
||||
* Builtin, then user, then project. Later wins, so a project can override a
|
||||
* bundled skill by using the same name.
|
||||
*/
|
||||
export async function loadSkills(cwd = process.cwd()): Promise<Skill[]> {
|
||||
const byName = new Map<string, Skill>();
|
||||
|
||||
for (const { name, source } of BUILTIN_SKILLS) {
|
||||
const skill = parseSkill(source, 'builtin');
|
||||
if (skill) byName.set(skill.name, skill);
|
||||
else byName.delete(name);
|
||||
}
|
||||
|
||||
for (const { dir, origin } of skillDirs(cwd)) {
|
||||
let files: string[] = [];
|
||||
try {
|
||||
for await (const f of new Bun.Glob('*.md').scan({ cwd: dir, onlyFiles: true })) files.push(f);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const file of files.sort()) {
|
||||
const path = join(dir, file);
|
||||
try {
|
||||
const skill = parseSkill(await Bun.file(path).text(), origin, path);
|
||||
if (skill) byName.set(skill.name, skill);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name));
|
||||
}
|
||||
|
||||
/**
|
||||
* Catalogue for the system prompt: names and one-line descriptions only.
|
||||
* Bodies stay out of context until the model asks, which is the point.
|
||||
*/
|
||||
export function renderSkills(skills: Skill[]): string {
|
||||
if (skills.length === 0) return '';
|
||||
const lines = skills.map((s) => `- ${s.name}: ${s.description}`);
|
||||
return [
|
||||
'',
|
||||
'Skills available through the skill tool. Load one when its description matches the task,',
|
||||
'before you start working, and follow it as if the user had written it:',
|
||||
...lines,
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
export function createSkillTool(skills: Skill[]) {
|
||||
const names = skills.map((s) => s.name);
|
||||
return tool({
|
||||
description:
|
||||
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
|
||||
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`,
|
||||
inputSchema: z.object({
|
||||
name: z.string().describe('Skill name from the list in your instructions'),
|
||||
}),
|
||||
execute: async ({ name }) => {
|
||||
const skill = skills.find((s) => s.name === name.trim().toLowerCase());
|
||||
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`);
|
||||
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export { skillDirs };
|
||||
+109
@@ -0,0 +1,109 @@
|
||||
import type { ModelMessage } from 'ai';
|
||||
import { createHash } from 'node:crypto';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import type { NotebookState } from './notebook';
|
||||
|
||||
export type SessionRecord = {
|
||||
id: string;
|
||||
createdAt: string;
|
||||
updatedAt: string;
|
||||
cwd: string;
|
||||
provider: string;
|
||||
model: string;
|
||||
title: string;
|
||||
inputTokens: number;
|
||||
outputTokens: number;
|
||||
/** Estimated USD, absent when the model has no known rate. */
|
||||
costUsd?: number;
|
||||
/** Task list and notes, so a resumed session keeps its plan. */
|
||||
notebook?: NotebookState;
|
||||
messages: ModelMessage[];
|
||||
};
|
||||
|
||||
/** Resolved per call so tests can point SHIRO_HOME at a temp directory. */
|
||||
const root = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko');
|
||||
const dir = () => join(root(), 'sessions');
|
||||
|
||||
const file = (id: string) => join(dir(), `${id}.json`);
|
||||
|
||||
export function newId(): string {
|
||||
return Bun.randomUUIDv7();
|
||||
}
|
||||
|
||||
export async function save(rec: SessionRecord): Promise<void> {
|
||||
await Bun.write(file(rec.id), JSON.stringify({ ...rec, updatedAt: new Date().toISOString() }, null, 2));
|
||||
}
|
||||
|
||||
export async function load(id: string): Promise<SessionRecord | undefined> {
|
||||
const f = Bun.file(file(id));
|
||||
if (!(await f.exists())) return undefined;
|
||||
try {
|
||||
return (await f.json()) as SessionRecord;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
export async function list(limit = 20): Promise<SessionRecord[]> {
|
||||
const found: SessionRecord[] = [];
|
||||
// Bun.Glob throws ENOENT on a directory that does not exist yet, which is the
|
||||
// normal state on a fresh install.
|
||||
try {
|
||||
for await (const name of new Bun.Glob('*.json').scan({ cwd: dir(), onlyFiles: true })) {
|
||||
const rec = await load(name.replace(/\.json$/, ''));
|
||||
if (rec) found.push(rec);
|
||||
}
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
return found.sort((a, b) => b.updatedAt.localeCompare(a.updatedAt)).slice(0, limit);
|
||||
}
|
||||
|
||||
export async function latest(cwd?: string): Promise<SessionRecord | undefined> {
|
||||
const all = await list(100);
|
||||
return cwd ? all.find((r) => r.cwd === cwd) : all[0];
|
||||
}
|
||||
|
||||
/** Resolves a full id or a unique prefix, so users can type the first few chars. */
|
||||
export async function resolveId(prefix: string): Promise<string | undefined> {
|
||||
if (await Bun.file(file(prefix)).exists()) return prefix;
|
||||
const matches = (await list(100)).filter((r) => r.id.startsWith(prefix));
|
||||
return matches.length === 1 ? matches[0]!.id : undefined;
|
||||
}
|
||||
|
||||
export function titleOf(messages: ModelMessage[]): string {
|
||||
const first = messages.find((m) => m.role === 'user');
|
||||
const text = typeof first?.content === 'string' ? first.content : '';
|
||||
return text.length > 60 ? `${text.slice(0, 60)}...` : text || 'untitled';
|
||||
}
|
||||
|
||||
const MAX_HISTORY = 200;
|
||||
|
||||
/** Per-directory file, hashed because a path is not a safe filename. */
|
||||
const historyFile = (cwd: string) =>
|
||||
join(root(), 'history', `${createHash('sha256').update(cwd).digest('hex').slice(0, 16)}.json`);
|
||||
|
||||
export async function loadHistory(cwd = process.cwd()): Promise<string[]> {
|
||||
const f = Bun.file(historyFile(cwd));
|
||||
if (!(await f.exists())) return [];
|
||||
try {
|
||||
const parsed: unknown = await f.json();
|
||||
return Array.isArray(parsed) ? parsed.filter((x): x is string => typeof x === 'string') : [];
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
/** Appends unless it repeats the previous entry, keeping the newest MAX_HISTORY. */
|
||||
export async function appendHistory(prompt: string, cwd = process.cwd()): Promise<string[]> {
|
||||
const text = prompt.trim();
|
||||
if (!text) return loadHistory(cwd);
|
||||
const existing = await loadHistory(cwd);
|
||||
if (existing.at(-1) === text) return existing;
|
||||
const next = [...existing, text].slice(-MAX_HISTORY);
|
||||
await Bun.write(historyFile(cwd), JSON.stringify(next, null, 2));
|
||||
return next;
|
||||
}
|
||||
|
||||
export { dir as sessionsDir, root as shiroHome };
|
||||
+127
@@ -0,0 +1,127 @@
|
||||
import { isStepCount, streamText, tool, type LanguageModel, type ToolSet } from 'ai';
|
||||
import { z } from 'zod';
|
||||
import { globTool, grepTool, readFileTool } from './tools';
|
||||
|
||||
export type SubagentKind = 'explore' | 'review';
|
||||
|
||||
export type SubagentEvent =
|
||||
| { type: 'start'; id: string; kind: SubagentKind; description: string }
|
||||
| { type: 'step'; id: string; tool: string; summary: string }
|
||||
| { type: 'end'; id: string; ok: boolean; steps: number }
|
||||
| { type: 'error'; id: string; message: string };
|
||||
|
||||
export type SubagentReporter = (event: SubagentEvent) => void;
|
||||
|
||||
const READ_ONLY: ToolSet = { read_file: readFileTool, glob: globTool, grep: grepTool };
|
||||
|
||||
const PROMPTS: Record<SubagentKind, (cwd: string) => string> = {
|
||||
explore: (cwd) => `You are a research subagent inside a coding agent.
|
||||
|
||||
Workspace root: ${cwd}
|
||||
Tools: read_file, glob, grep. You cannot write files, run commands, or ask questions.
|
||||
|
||||
Find what was asked and report once. Rules:
|
||||
- Give file paths with line numbers, plus a short quote where the quote is the answer.
|
||||
- Report what you actually read. If you could not determine something, say so; do not fill the gap.
|
||||
- No preamble, no restating the task, no offers of further help.
|
||||
- Aim for under 30 lines. The parent agent pays for every line you write.`,
|
||||
|
||||
review: (cwd) => `You are a review subagent inside a coding agent.
|
||||
|
||||
Workspace root: ${cwd}
|
||||
Tools: read_file, glob, grep. You cannot write files, run commands, or ask questions.
|
||||
|
||||
Review what was asked and report once. Severity order: incorrect behaviour, missing validation at
|
||||
trust boundaries, security, resource handling, then clarity. For each finding give file, line, what
|
||||
breaks, and the fix. Say plainly when something is correct. Do not invent findings to look thorough.`,
|
||||
};
|
||||
|
||||
const summarize = (input: unknown): string => {
|
||||
if (input === null || typeof input !== 'object') return String(input);
|
||||
const o = input as Record<string, unknown>;
|
||||
const first = o['pattern'] ?? o['path'] ?? o['include'];
|
||||
return typeof first === 'string' ? first : JSON.stringify(o).slice(0, 80);
|
||||
};
|
||||
|
||||
let counter = 0;
|
||||
|
||||
/**
|
||||
* Read-only child agent.
|
||||
*
|
||||
* It runs its own tool loop and returns one message, so the parent pays for the
|
||||
* findings rather than the whole search transcript. No write, bash, or ask tool is
|
||||
* passed in, which is also why a subagent can never trigger an approval prompt.
|
||||
*/
|
||||
export function createTaskTool(opts: {
|
||||
model: LanguageModel;
|
||||
cwd?: string;
|
||||
maxSteps?: number;
|
||||
report?: SubagentReporter;
|
||||
}) {
|
||||
return tool({
|
||||
description:
|
||||
'Delegate a read-only investigation to a subagent that can read, glob, and grep. Use it for questions ' +
|
||||
'spanning many files ("where is auth handled", "every caller of X") and to keep a long search out of your ' +
|
||||
'own context. The subagent sees none of this conversation, so its prompt must be self-contained. ' +
|
||||
'It returns one text report. Do not delegate something you can answer with a single grep.',
|
||||
inputSchema: z.object({
|
||||
description: z.string().describe('Short label shown to the user, 3-6 words'),
|
||||
prompt: z.string().describe('Self-contained instructions: what to find, where to look, what to return'),
|
||||
kind: z
|
||||
.enum(['explore', 'review'])
|
||||
.optional()
|
||||
.describe('explore: find and report. review: critique code for defects. Default explore.'),
|
||||
}),
|
||||
execute: async ({ description, prompt, kind }, { abortSignal }) => {
|
||||
const id = `sub${++counter}`;
|
||||
const flavour: SubagentKind = kind ?? 'explore';
|
||||
const report = opts.report;
|
||||
report?.({ type: 'start', id, kind: flavour, description });
|
||||
|
||||
let steps = 0;
|
||||
let text = '';
|
||||
|
||||
try {
|
||||
const result = streamText({
|
||||
model: opts.model,
|
||||
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
|
||||
messages: [{ role: 'user', content: prompt }],
|
||||
tools: READ_ONLY,
|
||||
stopWhen: isStepCount(opts.maxSteps ?? 20),
|
||||
...(abortSignal ? { abortSignal } : {}),
|
||||
});
|
||||
|
||||
const sink = () => {};
|
||||
void result.responseMessages.then(undefined, sink);
|
||||
void result.usage.then(undefined, sink);
|
||||
void result.steps.then(undefined, sink);
|
||||
void result.finalStep.then(undefined, sink);
|
||||
void result.finishReason.then(undefined, sink);
|
||||
|
||||
for await (const part of result.stream) {
|
||||
if (part.type === 'tool-call') {
|
||||
steps++;
|
||||
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
|
||||
} else if (part.type === 'text-delta') {
|
||||
text += part.text;
|
||||
} else if (part.type === 'error') {
|
||||
// A provider failure arrives as a stream part, not a throw, so it has to
|
||||
// be rethrown here or the subagent silently returns nothing.
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
throw part.error instanceof Error ? part.error : new Error(message);
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
const message = e instanceof Error ? e.message : String(e);
|
||||
report?.({ type: 'error', id, message });
|
||||
throw e;
|
||||
}
|
||||
|
||||
const trimmed = text.trim();
|
||||
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
|
||||
return trimmed || 'Subagent returned no findings.';
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export const TASK_TOOL_NAME = 'task';
|
||||
+281
@@ -0,0 +1,281 @@
|
||||
import { tool } from 'ai';
|
||||
import { resolve } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { jail, posix, walk } from './ignore';
|
||||
|
||||
/** Max chars returned by any single tool. Beyond this the output is truncated. */
|
||||
const MAX_OUTPUT = 30_000;
|
||||
const MAX_GREP_HITS = 200;
|
||||
/** Bytes sniffed for a NUL to decide a file is not text. */
|
||||
const SNIFF_BYTES = 8192;
|
||||
|
||||
function cap(s: string): string {
|
||||
return s.length <= MAX_OUTPUT ? s : `${s.slice(0, MAX_OUTPUT)}\n... [truncated ${s.length - MAX_OUTPUT} chars]`;
|
||||
}
|
||||
|
||||
/**
|
||||
* A NUL byte in the first few KB means this is not text. Cheap, and the same
|
||||
* heuristic git and ripgrep use; without it a model can burn its whole context
|
||||
* on one accidental `read_file dist/binary`.
|
||||
*/
|
||||
async function isBinary(abs: string): Promise<boolean> {
|
||||
const bytes = new Uint8Array(await Bun.file(abs).slice(0, SNIFF_BYTES).arrayBuffer());
|
||||
return bytes.includes(0);
|
||||
}
|
||||
|
||||
export const readFileTool = tool({
|
||||
description: 'Read a UTF-8 text file. Returns contents with 1-based line numbers.',
|
||||
inputSchema: z.object({
|
||||
path: z.string().describe('File path relative to the workspace root'),
|
||||
offset: z.number().int().min(1).optional().describe('First line to return (1-based)'),
|
||||
limit: z.number().int().min(1).optional().describe('Max lines to return, default 2000'),
|
||||
}),
|
||||
execute: async ({ path, offset = 1, limit = 2000 }) => {
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
if (await isBinary(abs)) throw new Error(`${path} is a binary file, not text. Use bash if you need to inspect it.`);
|
||||
const lines = (await file.text()).split('\n');
|
||||
const slice = lines.slice(offset - 1, offset - 1 + limit);
|
||||
return cap(slice.map((l, i) => `${offset + i}: ${l}`).join('\n'));
|
||||
},
|
||||
});
|
||||
|
||||
export const writeFileTool = tool({
|
||||
description: 'Create a file or overwrite it completely. Prefer edit_file for existing files.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
content: z.string(),
|
||||
}),
|
||||
execute: async ({ path, content }) => {
|
||||
const abs = jail(path);
|
||||
await Bun.write(abs, content);
|
||||
return `Wrote ${content.length} chars to ${path}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const editFileTool = tool({
|
||||
description:
|
||||
'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
|
||||
}),
|
||||
execute: async ({ path, oldString, newString, replaceAll = false }) => {
|
||||
if (oldString === newString) throw new Error('oldString and newString are identical');
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
const before = await file.text();
|
||||
|
||||
const count = before.split(oldString).length - 1;
|
||||
if (count === 0) throw new Error(`oldString not found in ${path}`);
|
||||
if (count > 1 && !replaceAll) {
|
||||
throw new Error(`oldString appears ${count} times in ${path}. Add surrounding context or set replaceAll.`);
|
||||
}
|
||||
|
||||
const after = replaceAll ? before.split(oldString).join(newString) : before.replace(oldString, newString);
|
||||
await Bun.write(abs, after);
|
||||
return `Replaced ${replaceAll ? count : 1} occurrence(s) in ${path}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const globTool = tool({
|
||||
description:
|
||||
'Find files by glob pattern, e.g. "src/**/*.ts". Skips anything .gitignore excludes. Returns paths relative to the workspace root.',
|
||||
inputSchema: z.object({
|
||||
pattern: z.string(),
|
||||
limit: z.number().int().min(1).optional().describe('Max paths to return, default 200'),
|
||||
includeIgnored: z.boolean().optional().describe('Also search files git ignores'),
|
||||
}),
|
||||
execute: async ({ pattern, limit = 200, includeIgnored = false }) => {
|
||||
const glob = new Bun.Glob(pattern);
|
||||
const hits: string[] = [];
|
||||
for await (const rel of walk({ noIgnore: includeIgnored })) {
|
||||
if (!glob.match(rel)) continue;
|
||||
hits.push(rel);
|
||||
if (hits.length >= limit) break;
|
||||
}
|
||||
return hits.length ? hits.join('\n') : 'No files matched.';
|
||||
},
|
||||
});
|
||||
|
||||
type GrepArgs = { pattern: string; include?: string; ignoreCase?: boolean; includeIgnored?: boolean };
|
||||
|
||||
/**
|
||||
* ripgrep is 10-100x faster than walking in JS and already understands
|
||||
* .gitignore and binary detection, so use it whenever it is installed.
|
||||
* Output shape stays identical to the fallback so the model sees one format.
|
||||
*/
|
||||
async function grepWithRipgrep({ pattern, include, ignoreCase, includeIgnored }: GrepArgs): Promise<string | undefined> {
|
||||
// --no-require-git: rg skips .gitignore outside a repo by default, but the JS
|
||||
// fallback always honours it, and the two paths must agree.
|
||||
const args = [
|
||||
'--line-number',
|
||||
'--no-heading',
|
||||
'--color',
|
||||
'never',
|
||||
'--no-require-git',
|
||||
'--max-count',
|
||||
String(MAX_GREP_HITS),
|
||||
];
|
||||
if (ignoreCase) args.push('--ignore-case');
|
||||
if (includeIgnored) args.push('--no-ignore');
|
||||
if (include) args.push('--glob', include);
|
||||
args.push('--regexp', pattern, '.');
|
||||
|
||||
let proc: Bun.Subprocess<'ignore', 'pipe', 'pipe'>;
|
||||
try {
|
||||
proc = Bun.spawn(['rg', ...args], { cwd: process.cwd(), stdout: 'pipe', stderr: 'pipe', timeout: 60_000 });
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const [stdout, stderr, code] = await Promise.all([
|
||||
new Response(proc.stdout).text(),
|
||||
new Response(proc.stderr).text(),
|
||||
proc.exited,
|
||||
]);
|
||||
|
||||
// 0 = matches, 1 = no matches. Anything else means rg could not run the search.
|
||||
if (code > 1) {
|
||||
if (/regex parse error|error parsing/i.test(stderr)) throw new Error(`Invalid regex: ${stderr.trim()}`);
|
||||
return undefined;
|
||||
}
|
||||
if (code === 1) return 'No matches.';
|
||||
|
||||
const hits = stdout
|
||||
.split('\n')
|
||||
.map((line) => line.replace(/\r$/, ''))
|
||||
.filter(Boolean)
|
||||
.map((line) => {
|
||||
const m = /^(.*?):(\d+):(.*)$/.exec(line);
|
||||
if (!m) return line;
|
||||
// rg prefixes every path with the search root and uses native separators.
|
||||
const rel = posix(m[1]!).replace(/^\.\//, '');
|
||||
return `${rel}:${m[2]}: ${m[3]!.slice(0, 300)}`;
|
||||
})
|
||||
.slice(0, MAX_GREP_HITS);
|
||||
|
||||
return cap(hits.join('\n'));
|
||||
}
|
||||
|
||||
async function grepInJs({ pattern, include = '**/*', ignoreCase, includeIgnored }: GrepArgs): Promise<string> {
|
||||
let re: RegExp;
|
||||
try {
|
||||
re = new RegExp(pattern, ignoreCase ? 'i' : '');
|
||||
} catch (e) {
|
||||
throw new Error(`Invalid regex: ${(e as Error).message}`);
|
||||
}
|
||||
|
||||
const glob = new Bun.Glob(include);
|
||||
const hits: string[] = [];
|
||||
for await (const rel of walk({ noIgnore: includeIgnored })) {
|
||||
if (!glob.match(rel)) continue;
|
||||
const abs = resolve(process.cwd(), rel);
|
||||
let text: string;
|
||||
try {
|
||||
if (await isBinary(abs)) continue;
|
||||
text = await Bun.file(abs).text();
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
const lines = text.split('\n');
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const line = lines[i] ?? '';
|
||||
if (re.test(line)) hits.push(`${rel}:${i + 1}: ${line.slice(0, 300)}`);
|
||||
if (hits.length >= MAX_GREP_HITS) return cap(`${hits.join('\n')}\n... [hit limit ${MAX_GREP_HITS}]`);
|
||||
}
|
||||
}
|
||||
return hits.length ? cap(hits.join('\n')) : 'No matches.';
|
||||
}
|
||||
|
||||
export const grepTool = tool({
|
||||
description:
|
||||
'Search file contents with a regular expression. Skips binaries and anything .gitignore excludes. Returns path:line:text hits.',
|
||||
inputSchema: z.object({
|
||||
pattern: z.string().describe('Regex source. ripgrep syntax when available, otherwise JavaScript'),
|
||||
include: z.string().optional().describe('Glob limiting which files are searched, default "**/*"'),
|
||||
ignoreCase: z.boolean().optional(),
|
||||
includeIgnored: z.boolean().optional().describe('Also search files git ignores'),
|
||||
}),
|
||||
execute: async (args) => (await grepWithRipgrep(args)) ?? (await grepInJs(args)),
|
||||
});
|
||||
|
||||
export type BashOutput = { toolCallId: string; chunk: string };
|
||||
|
||||
/** Set by Session so long-running commands can report progress before exiting. */
|
||||
let bashListener: ((out: BashOutput) => void) | undefined;
|
||||
|
||||
export function onBashOutput(fn: ((out: BashOutput) => void) | undefined): void {
|
||||
bashListener = fn;
|
||||
}
|
||||
|
||||
async function pump(
|
||||
stream: ReadableStream<Uint8Array> | undefined,
|
||||
toolCallId: string,
|
||||
): Promise<string> {
|
||||
if (!stream) return '';
|
||||
const decoder = new TextDecoder();
|
||||
let all = '';
|
||||
for await (const chunk of stream) {
|
||||
const text = decoder.decode(chunk, { stream: true });
|
||||
if (!text) continue;
|
||||
all += text;
|
||||
bashListener?.({ toolCallId, chunk: text });
|
||||
}
|
||||
return all;
|
||||
}
|
||||
|
||||
export const bashTool = tool({
|
||||
description: 'Run a shell command in the workspace root. Use for builds, tests, git, and package managers.',
|
||||
inputSchema: z.object({
|
||||
command: z.string(),
|
||||
timeout: z.number().int().min(1000).max(600_000).optional().describe('Timeout in ms, default 120000'),
|
||||
}),
|
||||
execute: async ({ command, timeout = 120_000 }, { toolCallId, abortSignal }) => {
|
||||
const shell = process.platform === 'win32' ? ['cmd', '/c', command] : ['bash', '-lc', command];
|
||||
const proc = Bun.spawn(shell, {
|
||||
cwd: process.cwd(),
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
timeout,
|
||||
...(abortSignal ? { signal: abortSignal } : {}),
|
||||
});
|
||||
|
||||
// Drained concurrently: a command that fills one pipe while we block on the
|
||||
// other would deadlock, and buffering both hides progress for minutes.
|
||||
const [stdout, stderr, exitCode] = await Promise.all([
|
||||
pump(proc.stdout as ReadableStream<Uint8Array>, toolCallId),
|
||||
pump(proc.stderr as ReadableStream<Uint8Array>, toolCallId),
|
||||
proc.exited,
|
||||
]);
|
||||
|
||||
return cap(
|
||||
[
|
||||
`exit: ${exitCode}`,
|
||||
proc.signalCode && `(killed by ${proc.signalCode}; timeout is ${timeout}ms)`,
|
||||
stdout.trim() && `stdout:\n${stdout.trim()}`,
|
||||
stderr.trim() && `stderr:\n${stderr.trim()}`,
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join('\n\n'),
|
||||
);
|
||||
},
|
||||
});
|
||||
|
||||
export const tools = {
|
||||
read_file: readFileTool,
|
||||
write_file: writeFileTool,
|
||||
edit_file: editFileTool,
|
||||
glob: globTool,
|
||||
grep: grepTool,
|
||||
bash: bashTool,
|
||||
};
|
||||
|
||||
/** Tools that mutate the workspace or run arbitrary code always ask the user first. */
|
||||
export const MUTATING_TOOLS = ['write_file', 'edit_file', 'bash'] as const;
|
||||
|
||||
export { jail };
|
||||
+792
@@ -0,0 +1,792 @@
|
||||
import { Box, Static, Text, useApp, useInput, useStdout } from 'ink';
|
||||
import SelectInput from 'ink-select-input';
|
||||
import Spinner from 'ink-spinner';
|
||||
import React, { useCallback, useEffect, useRef, useState } from 'react';
|
||||
import { parseCommand, matchCommands, type CommandSpec } from '../commands';
|
||||
import { THINKING_LEVELS, VARIANTS } from '../agents';
|
||||
import type { Config } from '../config';
|
||||
import { TODO_MARK, type NotebookState } from '../notebook';
|
||||
import { costOf, formatUsd, usageLine } from '../pricing';
|
||||
import type { ApprovalDecision, ApprovalRequest, Session } from '../session';
|
||||
import type { SubagentEvent } from '../subagent';
|
||||
import { AskPanel, type AskBridge, type AskPending } from './Ask';
|
||||
import { Diff } from './Diff';
|
||||
import { Markdown } from './Markdown';
|
||||
import { Onboard, type OnboardResult } from './Onboard';
|
||||
import { InfoPanel, OutputPanel, StatusBar, SubagentPanel, TodoPanel, type SubagentView } from './Panels';
|
||||
import { PromptInput } from './PromptInput';
|
||||
|
||||
type Line =
|
||||
| { key: string; kind: 'user'; text: string }
|
||||
| { key: string; kind: 'assistant'; text: string }
|
||||
| { key: string; kind: 'tool'; name: string; summary: string; ok: boolean }
|
||||
| { key: string; kind: 'info'; text: string }
|
||||
| { key: string; kind: 'error'; text: string };
|
||||
|
||||
type NewLine = Line extends infer T ? (T extends Line ? Omit<T, 'key'> : never) : never;
|
||||
|
||||
type Pending = { req: ApprovalRequest; resolve: (d: ApprovalDecision) => void };
|
||||
|
||||
/** Bridges Session's promise-based approval callback into React state. */
|
||||
export type ApprovalBridge = {
|
||||
bind: (fn: (p: Pending | undefined) => void) => void;
|
||||
ask: (req: ApprovalRequest) => Promise<ApprovalDecision>;
|
||||
};
|
||||
|
||||
export function createApprovalBridge(): ApprovalBridge {
|
||||
let setter: ((p: Pending | undefined) => void) | undefined;
|
||||
return {
|
||||
bind(fn) {
|
||||
setter = fn;
|
||||
},
|
||||
ask(req) {
|
||||
return new Promise((resolve) => {
|
||||
if (!setter) return resolve('deny'); // UI not mounted: fail closed
|
||||
setter({
|
||||
req,
|
||||
resolve: (d) => {
|
||||
setter?.(undefined);
|
||||
resolve(d);
|
||||
},
|
||||
});
|
||||
});
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** One-way channel for out-of-band notices, e.g. an endpoint fallback. */
|
||||
export type NoticeBus = {
|
||||
bind: (fn: (text: string) => void) => void;
|
||||
emit: (text: string) => void;
|
||||
};
|
||||
|
||||
export function createNoticeBus(): NoticeBus {
|
||||
const queued: string[] = [];
|
||||
let sink: ((text: string) => void) | undefined;
|
||||
return {
|
||||
bind(fn) {
|
||||
sink = fn;
|
||||
for (const text of queued.splice(0)) fn(text);
|
||||
},
|
||||
emit(text) {
|
||||
if (sink) sink(text);
|
||||
else queued.push(text);
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** Subagent progress, from the task tool to the panel. */
|
||||
export type SubagentBus = {
|
||||
bind: (fn: (event: SubagentEvent) => void) => void;
|
||||
emit: (event: SubagentEvent) => void;
|
||||
};
|
||||
|
||||
export function createSubagentBus(): SubagentBus {
|
||||
const queued: SubagentEvent[] = [];
|
||||
let sink: ((event: SubagentEvent) => void) | undefined;
|
||||
return {
|
||||
bind(fn) {
|
||||
sink = fn;
|
||||
for (const event of queued.splice(0)) fn(event);
|
||||
},
|
||||
emit(event) {
|
||||
if (sink) sink(event);
|
||||
else queued.push(event);
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** Folds a subagent event into the panel's view, keeping finished agents visible. */
|
||||
export function applySubagentEvent(current: SubagentView[], event: SubagentEvent): SubagentView[] {
|
||||
switch (event.type) {
|
||||
case 'start':
|
||||
return [
|
||||
...current,
|
||||
{ id: event.id, kind: event.kind, description: event.description, steps: [], status: 'running' },
|
||||
];
|
||||
case 'step':
|
||||
return current.map((a) =>
|
||||
a.id === event.id ? { ...a, steps: [...a.steps, { tool: event.tool, summary: event.summary }] } : a,
|
||||
);
|
||||
case 'end':
|
||||
return current.map((a) => (a.id === event.id ? { ...a, status: event.ok ? 'done' : 'failed' } : a));
|
||||
case 'error':
|
||||
return current.map((a) => (a.id === event.id ? { ...a, status: 'failed', error: event.message } : a));
|
||||
}
|
||||
}
|
||||
|
||||
/** Everything the slash commands need from the outside world. */
|
||||
export type AppHooks = {
|
||||
sessionId: string;
|
||||
config: () => Config;
|
||||
switchModel: (id: string) => string;
|
||||
switchAgent: (name: string) => string;
|
||||
switchThinking: (level: string) => string;
|
||||
agentName: () => string;
|
||||
thinkingLevel: () => string;
|
||||
applyProvider: (result: OnboardResult) => Promise<string>;
|
||||
listModels: () => Promise<{ models: string[]; warning?: string }>;
|
||||
listSessions: () => Promise<string>;
|
||||
listSkills: () => string;
|
||||
listPlugins: () => string;
|
||||
listMemory: () => Promise<string>;
|
||||
summarizeMemory: () => Promise<string>;
|
||||
resumeSession: (idOrPrefix: string) => Promise<string>;
|
||||
saveSession: () => Promise<string>;
|
||||
/** Loaded AGENTS.md-style files, for /context. */
|
||||
instructionFiles: () => string[];
|
||||
/** Prompt to hand the model for /init. */
|
||||
initPrompt: string;
|
||||
history: string[];
|
||||
recordPrompt: (text: string) => void;
|
||||
};
|
||||
|
||||
let seq = 0;
|
||||
const nextKey = () => `l${seq++}`;
|
||||
|
||||
function preview(input: unknown): string {
|
||||
if (input === null || typeof input !== 'object') return String(input);
|
||||
const o = input as Record<string, unknown>;
|
||||
const first = o['command'] ?? o['path'] ?? o['pattern'] ?? o['description'] ?? o['question'] ?? o['name'];
|
||||
if (typeof first === 'string') return first.length > 90 ? `${first.slice(0, 90)}...` : first;
|
||||
|
||||
// A tool with no obvious label, e.g. todo_write, gets a shape rather than a
|
||||
// JSON dump; the panels below already show the content.
|
||||
const todos = o['todos'];
|
||||
if (Array.isArray(todos)) return `${todos.length} task${todos.length === 1 ? '' : 's'}`;
|
||||
const keys = Object.keys(o);
|
||||
return keys.length === 0 ? '' : keys.slice(0, 3).join(', ');
|
||||
}
|
||||
|
||||
function ApprovalDetail({ name, input }: { name: string; input: unknown }) {
|
||||
const o = (input ?? {}) as Record<string, unknown>;
|
||||
if (name === 'bash') return <Text dimColor>{String(o['command'] ?? '')}</Text>;
|
||||
if (name === 'write_file') {
|
||||
const content = String(o['content'] ?? '');
|
||||
return <Diff before="" after={content} path={`${String(o['path'])} (new content)`} />;
|
||||
}
|
||||
if (name === 'edit_file') {
|
||||
return <Diff before={String(o['oldString'] ?? '')} after={String(o['newString'] ?? '')} path={String(o['path'])} />;
|
||||
}
|
||||
return <Text dimColor>{JSON.stringify(input, null, 2)}</Text>;
|
||||
}
|
||||
|
||||
function Approval({ pending }: { pending: Pending }) {
|
||||
useInput((input, key) => {
|
||||
const c = input.toLowerCase();
|
||||
if (c === 'y' || key.return) pending.resolve('once');
|
||||
else if (c === 'a') pending.resolve('always');
|
||||
else if (c === 'n' || key.escape) pending.resolve('deny');
|
||||
});
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="yellow" paddingX={1}>
|
||||
<Text color="yellow" bold>
|
||||
{pending.req.toolName} wants to run
|
||||
</Text>
|
||||
<ApprovalDetail name={pending.req.toolName} input={pending.req.input} />
|
||||
<Text>
|
||||
<Text color="green">y</Text> allow once | <Text color="green">a</Text> always allow {pending.req.toolName} |{' '}
|
||||
<Text color="red">n</Text> deny
|
||||
</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
function CommandMenu({ matches, index }: { matches: CommandSpec[]; index: number }) {
|
||||
return (
|
||||
<Box flexDirection="column" marginTop={1}>
|
||||
{matches.map((c, i) => (
|
||||
<Text key={c.name} color={i === index ? 'cyan' : undefined} dimColor={i !== index}>
|
||||
{i === index ? '> ' : ' '}
|
||||
{`/${c.name}${c.arg ? ` ${c.arg}` : ''}`.padEnd(18)} {c.summary}
|
||||
</Text>
|
||||
))}
|
||||
<Text dimColor>up/down move | tab complete | enter run | esc dismiss</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
export function App({
|
||||
session,
|
||||
bridge,
|
||||
header,
|
||||
hooks,
|
||||
notices,
|
||||
askBridge,
|
||||
subagents,
|
||||
needsProvider = false,
|
||||
}: {
|
||||
session: Session;
|
||||
bridge: ApprovalBridge;
|
||||
header: string;
|
||||
hooks: AppHooks;
|
||||
notices?: NoticeBus;
|
||||
askBridge?: AskBridge;
|
||||
subagents?: SubagentBus;
|
||||
needsProvider?: boolean;
|
||||
}) {
|
||||
const { exit } = useApp();
|
||||
const { write } = useStdout();
|
||||
const [history, setHistory] = useState<Line[]>([]);
|
||||
const [draft, setDraft] = useState('');
|
||||
const [live, setLive] = useState('');
|
||||
const [busy, setBusy] = useState(false);
|
||||
const [pending, setPending] = useState<Pending | undefined>();
|
||||
const [asking, setAsking] = useState<AskPending | undefined>();
|
||||
const [onboarding, setOnboarding] = useState(needsProvider);
|
||||
const [unconfigured, setUnconfigured] = useState(needsProvider);
|
||||
const [modelPicker, setModelPicker] = useState<string[] | undefined>();
|
||||
const [agentPicker, setAgentPicker] = useState(false);
|
||||
const [thinkPicker, setThinkPicker] = useState(false);
|
||||
const [menuIndex, setMenuIndex] = useState(0);
|
||||
const [menuDismissed, setMenuDismissed] = useState(false);
|
||||
const [inputGeneration, setInputGeneration] = useState(0);
|
||||
const [toolOutput, setToolOutput] = useState('');
|
||||
const [recall, setRecall] = useState<string[]>(hooks.history);
|
||||
const [notebook, setNotebook] = useState<NotebookState>(session.notebook.state());
|
||||
const [agents, setAgents] = useState<SubagentView[]>([]);
|
||||
const [panel, setPanel] = useState<{ title: string; hint?: string; body: string } | undefined>();
|
||||
|
||||
const modal = pending !== undefined || asking !== undefined || onboarding;
|
||||
const anyPicker = modelPicker !== undefined || agentPicker || thinkPicker;
|
||||
const matches = matchCommands(draft);
|
||||
const menuOpen = matches.length > 0 && !menuDismissed && !busy && !modal && !anyPicker && !panel;
|
||||
const highlighted = matches[Math.min(menuIndex, matches.length - 1)];
|
||||
|
||||
useEffect(() => bridge.bind(setPending), [bridge]);
|
||||
useEffect(() => askBridge?.bind(setAsking), [askBridge]);
|
||||
|
||||
useEffect(
|
||||
() =>
|
||||
subagents?.bind((event) => {
|
||||
setAgents((current) => applySubagentEvent(current, event));
|
||||
}),
|
||||
[subagents],
|
||||
);
|
||||
|
||||
// Ink re-renders the whole tree per setState, so deltas accumulate in a ref
|
||||
// and are flushed on a timer instead of once per token.
|
||||
const text = useRef('');
|
||||
useEffect(() => {
|
||||
const t = setInterval(() => {
|
||||
setLive((s) => (s === text.current ? s : text.current));
|
||||
}, 60);
|
||||
return () => clearInterval(t);
|
||||
}, []);
|
||||
|
||||
const push = useCallback((line: NewLine) => {
|
||||
setHistory((h) => [...h, { ...line, key: nextKey() }]);
|
||||
}, []);
|
||||
|
||||
useEffect(() => notices?.bind((text) => push({ kind: 'info', text })), [notices, push]);
|
||||
|
||||
useInput(
|
||||
(_input, key) => {
|
||||
if (key.escape) session.abort();
|
||||
},
|
||||
{ isActive: busy && !modal },
|
||||
);
|
||||
|
||||
useInput(
|
||||
(_input, key) => {
|
||||
if (!key.escape) return;
|
||||
setModelPicker(undefined);
|
||||
setAgentPicker(false);
|
||||
setThinkPicker(false);
|
||||
},
|
||||
{ isActive: anyPicker },
|
||||
);
|
||||
|
||||
// PromptInput hands up/down/tab/esc to us first, so the menu and any open panel
|
||||
// can claim them before the input treats them as editing keys.
|
||||
const handleInputKey = useCallback(
|
||||
(_input: string, key: { upArrow: boolean; downArrow: boolean; tab: boolean; escape: boolean }) => {
|
||||
if (key.escape && panel) {
|
||||
setPanel(undefined);
|
||||
return true;
|
||||
}
|
||||
if (!menuOpen) return false;
|
||||
if (key.escape) {
|
||||
setMenuDismissed(true);
|
||||
return true;
|
||||
}
|
||||
if (key.upArrow) {
|
||||
setMenuIndex((i) => (i - 1 + matches.length) % matches.length);
|
||||
return true;
|
||||
}
|
||||
if (key.downArrow) {
|
||||
setMenuIndex((i) => (i + 1) % matches.length);
|
||||
return true;
|
||||
}
|
||||
if (key.tab && highlighted) {
|
||||
setDraft(highlighted.arg ? `/${highlighted.name} ` : `/${highlighted.name}`);
|
||||
setMenuIndex(0);
|
||||
setMenuDismissed(true);
|
||||
setInputGeneration((g) => g + 1);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
},
|
||||
[highlighted, matches.length, menuOpen, panel],
|
||||
);
|
||||
|
||||
const onDraftChange = useCallback((value: string) => {
|
||||
setDraft(value);
|
||||
setMenuIndex(0);
|
||||
setMenuDismissed(false);
|
||||
}, []);
|
||||
|
||||
const runTurn = useCallback(
|
||||
async (value: string) => {
|
||||
setBusy(true);
|
||||
text.current = '';
|
||||
|
||||
for await (const ev of session.send(value)) {
|
||||
switch (ev.type) {
|
||||
case 'text':
|
||||
text.current += ev.text;
|
||||
break;
|
||||
case 'tool-call':
|
||||
push({ kind: 'tool', name: ev.name, summary: preview(ev.input), ok: true });
|
||||
break;
|
||||
case 'tool-output':
|
||||
setToolOutput((s) => `${s}${ev.chunk}`.slice(-2000));
|
||||
break;
|
||||
case 'tool-error':
|
||||
push({ kind: 'tool', name: ev.name, summary: String(ev.error), ok: false });
|
||||
break;
|
||||
case 'tool-result':
|
||||
setToolOutput('');
|
||||
setNotebook(session.notebook.state());
|
||||
break;
|
||||
case 'tool-denied':
|
||||
push({ kind: 'info', text: `denied ${ev.name}` });
|
||||
break;
|
||||
case 'notice':
|
||||
push({ kind: 'info', text: ev.text });
|
||||
break;
|
||||
case 'compacted':
|
||||
push({ kind: 'info', text: `context compacted: ${ev.before} messages pruned to ${ev.after} on the wire` });
|
||||
break;
|
||||
case 'error':
|
||||
push({ kind: 'error', text: ev.error instanceof Error ? ev.error.message : String(ev.error) });
|
||||
break;
|
||||
case 'done': {
|
||||
const full = text.current.trim();
|
||||
text.current = '';
|
||||
setLive('');
|
||||
setToolOutput('');
|
||||
setAgents([]);
|
||||
setHistory((h) => {
|
||||
const merged: Line[] = [...h];
|
||||
if (full) merged.push({ kind: 'assistant', text: full, key: nextKey() });
|
||||
if (ev.inputTokens !== undefined) {
|
||||
merged.push({
|
||||
kind: 'info',
|
||||
text: `${usageLine(hooks.config().model, ev.inputTokens, ev.outputTokens ?? 0)} (~${session.estimatedTokens()} in context)`,
|
||||
key: nextKey(),
|
||||
});
|
||||
}
|
||||
return merged;
|
||||
});
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
setBusy(false);
|
||||
},
|
||||
[hooks, push, session],
|
||||
);
|
||||
|
||||
const submit = useCallback(
|
||||
async (raw: string) => {
|
||||
setDraft('');
|
||||
setMenuIndex(0);
|
||||
setMenuDismissed(false);
|
||||
setPanel(undefined);
|
||||
|
||||
// Enter on an open menu runs the highlighted entry, so `/mo` + enter works.
|
||||
const chosen = menuOpen && highlighted ? `/${highlighted.name}` : raw;
|
||||
const action = parseCommand(chosen);
|
||||
|
||||
switch (action.type) {
|
||||
case 'none':
|
||||
return;
|
||||
case 'exit':
|
||||
return exit();
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
// Nothing can reach the model until a provider is configured.
|
||||
if (unconfigured && action.type !== 'provider' && action.type !== 'info') {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
push({ kind: 'error', text: 'no provider configured yet - run /provider' });
|
||||
return;
|
||||
}
|
||||
|
||||
switch (action.type) {
|
||||
case 'clear':
|
||||
session.reset();
|
||||
setHistory([]);
|
||||
setNotebook(session.notebook.state());
|
||||
// <Static> lines are already committed to the scrollback, so clearing
|
||||
// React state alone leaves them on screen. Wipe screen + scrollback.
|
||||
write('\u001B[2J\u001B[3J\u001B[H');
|
||||
return;
|
||||
case 'info':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setPanel({ title: 'commands', hint: 'type / for the menu', body: action.text });
|
||||
return;
|
||||
case 'unknown':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
push({ kind: 'error', text: `unknown command /${action.name} - try /help` });
|
||||
return;
|
||||
case 'tools':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setPanel({
|
||||
title: 'tools',
|
||||
hint: `${session.activeTools().length} offered this turn of ${Object.keys(session.tools).length} registered`,
|
||||
body: session
|
||||
.activeTools()
|
||||
.sort()
|
||||
.map((t) => `- \`${t}\``)
|
||||
.join('\n'),
|
||||
});
|
||||
return;
|
||||
case 'cost': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
const model = hooks.config().model;
|
||||
const spend = costOf(model, session.inputTokens, session.outputTokens);
|
||||
setPanel({
|
||||
title: 'cost',
|
||||
hint: `session ${hooks.sessionId}`,
|
||||
body: [
|
||||
`- model: \`${model}\``,
|
||||
`- billed: ${session.inputTokens} in / ${session.outputTokens} out`,
|
||||
`- spend: ${spend === undefined ? 'unpriced model' : formatUsd(spend)}`,
|
||||
`- context: ~${session.estimatedTokens()} tokens`,
|
||||
`- agent: \`${hooks.agentName()}\` thinking \`${hooks.thinkingLevel()}\``,
|
||||
].join('\n'),
|
||||
});
|
||||
return;
|
||||
}
|
||||
case 'context': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
const files = hooks.instructionFiles();
|
||||
setPanel({
|
||||
title: 'project instructions',
|
||||
body: files.length
|
||||
? files.map((f) => `- \`${f}\``).join('\n')
|
||||
: 'No `AGENTS.md`, `CLAUDE.md`, or `.shiro.md` found. Run `/init` to write one.',
|
||||
});
|
||||
return;
|
||||
}
|
||||
case 'todos': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
const { todos } = session.notebook.state();
|
||||
setPanel({
|
||||
title: 'task list',
|
||||
body: todos.length
|
||||
? todos.map((t) => `- ${TODO_MARK[t.status]} ${t.content}${t.note ? ` (${t.note})` : ''}`).join('\n')
|
||||
: 'No task list yet.',
|
||||
});
|
||||
return;
|
||||
}
|
||||
case 'notes': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setPanel({ title: 'project memory', body: await hooks.listMemory() });
|
||||
return;
|
||||
}
|
||||
case 'agent': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
if (action.agent) {
|
||||
try {
|
||||
push({ kind: 'info', text: hooks.switchAgent(action.agent) });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
return;
|
||||
}
|
||||
setAgentPicker(true);
|
||||
return;
|
||||
}
|
||||
case 'think': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
if (action.level) {
|
||||
try {
|
||||
push({ kind: 'info', text: hooks.switchThinking(action.level) });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
return;
|
||||
}
|
||||
setThinkPicker(true);
|
||||
return;
|
||||
}
|
||||
case 'skills':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setPanel({ title: 'skills', hint: 'the agent loads one with the skill tool', body: hooks.listSkills() });
|
||||
return;
|
||||
case 'plugins':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setPanel({ title: 'plugins', body: hooks.listPlugins() });
|
||||
return;
|
||||
case 'memory': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setBusy(true);
|
||||
try {
|
||||
push({ kind: 'info', text: await hooks.summarizeMemory() });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setBusy(false);
|
||||
return;
|
||||
}
|
||||
case 'init':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
await runTurn(hooks.initPrompt);
|
||||
return;
|
||||
case 'model':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
try {
|
||||
push({ kind: 'info', text: hooks.switchModel(action.model) });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
return;
|
||||
case 'sessions':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
push({ kind: 'info', text: await hooks.listSessions() });
|
||||
return;
|
||||
case 'save':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
push({ kind: 'info', text: await hooks.saveSession() });
|
||||
return;
|
||||
case 'resume':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
try {
|
||||
const msg = await hooks.resumeSession(action.id);
|
||||
setHistory([]);
|
||||
push({ kind: 'info', text: msg });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
return;
|
||||
case 'provider':
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setOnboarding(true);
|
||||
return;
|
||||
case 'models': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setBusy(true);
|
||||
const { models, warning } = await hooks.listModels();
|
||||
setBusy(false);
|
||||
if (warning) push({ kind: 'info', text: `could not list models: ${warning}` });
|
||||
if (models.length === 0) {
|
||||
push({ kind: 'error', text: 'no models to choose from - use /model <id> or /provider' });
|
||||
return;
|
||||
}
|
||||
setModelPicker(models);
|
||||
return;
|
||||
}
|
||||
case 'compact': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setBusy(true);
|
||||
try {
|
||||
const { before, after } = await session.summarize();
|
||||
push({ kind: 'info', text: `compacted ${before} messages into ${after}` });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setBusy(false);
|
||||
return;
|
||||
}
|
||||
case 'prompt':
|
||||
push({ kind: 'user', text: action.text });
|
||||
hooks.recordPrompt(action.text);
|
||||
setRecall((h) => (h.at(-1) === action.text ? h : [...h, action.text]));
|
||||
await runTurn(action.text);
|
||||
return;
|
||||
}
|
||||
},
|
||||
[exit, highlighted, hooks, menuOpen, push, runTurn, session, unconfigured, write],
|
||||
);
|
||||
|
||||
return (
|
||||
<Box flexDirection="column">
|
||||
<Static items={history}>
|
||||
{(line) => (
|
||||
<Box key={line.key} flexDirection="column" marginBottom={1}>
|
||||
{line.kind === 'user' && <Text color="cyan">{`> ${line.text}`}</Text>}
|
||||
{line.kind === 'assistant' && <Markdown text={line.text} />}
|
||||
{line.kind === 'tool' && (
|
||||
<Text color={line.ok ? 'magenta' : 'red'}>
|
||||
{line.ok ? '*' : 'x'} {line.name}({line.summary})
|
||||
</Text>
|
||||
)}
|
||||
{line.kind === 'info' && <Text dimColor>{line.text}</Text>}
|
||||
{line.kind === 'error' && <Text color="red">error: {line.text}</Text>}
|
||||
</Box>
|
||||
)}
|
||||
</Static>
|
||||
|
||||
{history.length === 0 && (
|
||||
<Box marginBottom={1}>
|
||||
<Text dimColor>{header}</Text>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
{agents.length > 0 && <SubagentPanel agents={agents} />}
|
||||
|
||||
{notebook.todos.length > 0 && <TodoPanel todos={notebook.todos} />}
|
||||
|
||||
{live.length > 0 && (
|
||||
<Box marginBottom={1}>
|
||||
<Markdown text={live} />
|
||||
</Box>
|
||||
)}
|
||||
|
||||
{panel && (
|
||||
<InfoPanel title={panel.title} {...(panel.hint ? { hint: panel.hint } : {})} lines={panel.body} />
|
||||
)}
|
||||
|
||||
{asking && <AskPanel pending={asking} />}
|
||||
|
||||
{pending && <Approval pending={pending} />}
|
||||
|
||||
{onboarding && (
|
||||
<Onboard
|
||||
current={hooks.config()}
|
||||
onCancel={() => {
|
||||
setOnboarding(false);
|
||||
push({ kind: 'info', text: 'provider setup cancelled' });
|
||||
}}
|
||||
onDone={async (result) => {
|
||||
setOnboarding(false);
|
||||
try {
|
||||
push({ kind: 'info', text: await hooks.applyProvider(result) });
|
||||
setUnconfigured(false);
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
|
||||
{modelPicker && (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
|
||||
<Text color="cyan" bold>
|
||||
Choose a model ({modelPicker.length} available)
|
||||
</Text>
|
||||
<Text dimColor>enter to select, esc to cancel</Text>
|
||||
<SelectInput
|
||||
items={modelPicker.map((m) => ({ key: m, label: m, value: m }))}
|
||||
limit={10}
|
||||
initialIndex={Math.max(0, modelPicker.indexOf(hooks.config().model))}
|
||||
onSelect={(item) => {
|
||||
setModelPicker(undefined);
|
||||
try {
|
||||
push({ kind: 'info', text: hooks.switchModel(item.value) });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}}
|
||||
/>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
{agentPicker && (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
|
||||
<Text color="cyan" bold>
|
||||
Choose an agent
|
||||
</Text>
|
||||
<Text dimColor>enter to select, esc to cancel</Text>
|
||||
<SelectInput
|
||||
items={VARIANTS.map((v) => ({
|
||||
key: v.name,
|
||||
label: `${v.name.padEnd(8)} ${v.summary}`,
|
||||
value: v.name,
|
||||
}))}
|
||||
limit={8}
|
||||
initialIndex={Math.max(
|
||||
0,
|
||||
VARIANTS.findIndex((v) => v.name === hooks.agentName()),
|
||||
)}
|
||||
onSelect={(item) => {
|
||||
setAgentPicker(false);
|
||||
try {
|
||||
push({ kind: 'info', text: hooks.switchAgent(item.value) });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}}
|
||||
/>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
{thinkPicker && (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
|
||||
<Text color="cyan" bold>
|
||||
Thinking level
|
||||
</Text>
|
||||
<Text dimColor>higher costs more and is slower; enter to select, esc to cancel</Text>
|
||||
<SelectInput
|
||||
items={THINKING_LEVELS.map((l) => ({ key: l, label: l, value: l }))}
|
||||
limit={8}
|
||||
initialIndex={Math.max(0, THINKING_LEVELS.indexOf(hooks.thinkingLevel() as (typeof THINKING_LEVELS)[number]))}
|
||||
onSelect={(item) => {
|
||||
setThinkPicker(false);
|
||||
try {
|
||||
push({ kind: 'info', text: hooks.switchThinking(item.value) });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}}
|
||||
/>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
{busy && !modal && (
|
||||
<Box flexDirection="column">
|
||||
<OutputPanel text={toolOutput} />
|
||||
<Text color="yellow">
|
||||
<Spinner type="dots" /> <Text dimColor>working... esc to interrupt</Text>
|
||||
</Text>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
{!busy && !modal && !anyPicker && (
|
||||
<Box flexDirection="column">
|
||||
<Box>
|
||||
<Text color="cyan">{'> '}</Text>
|
||||
<PromptInput
|
||||
key={inputGeneration}
|
||||
value={draft}
|
||||
onChange={onDraftChange}
|
||||
onSubmit={submit}
|
||||
history={recall}
|
||||
onKey={handleInputKey}
|
||||
placeholder="ask shiro-neko... (/ for commands)"
|
||||
/>
|
||||
</Box>
|
||||
{menuOpen && <CommandMenu matches={matches} index={Math.min(menuIndex, matches.length - 1)} />}
|
||||
<StatusBar
|
||||
model={hooks.config().model}
|
||||
agent={hooks.agentName()}
|
||||
thinking={hooks.thinkingLevel()}
|
||||
contextTokens={session.estimatedTokens()}
|
||||
cost={(() => {
|
||||
const spend = costOf(hooks.config().model, session.inputTokens, session.outputTokens);
|
||||
return spend === undefined ? 'unpriced' : formatUsd(spend);
|
||||
})()}
|
||||
toolCount={session.activeTools().length}
|
||||
/>
|
||||
</Box>
|
||||
)}
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
+116
@@ -0,0 +1,116 @@
|
||||
import { Box, Text, useInput } from 'ink';
|
||||
import SelectInput from 'ink-select-input';
|
||||
import React, { useState } from 'react';
|
||||
import type { AskRequest } from '../ask';
|
||||
import { InlineMarkdown } from './Markdown';
|
||||
import { PromptInput } from './PromptInput';
|
||||
|
||||
export type AskPending = { req: AskRequest; resolve: (answers: string[] | undefined) => void };
|
||||
|
||||
/** Bridges the ask tool's promise into React state, the same shape as the approval bridge. */
|
||||
export type AskBridge = {
|
||||
bind: (fn: (p: AskPending | undefined) => void) => void;
|
||||
ask: (req: AskRequest) => Promise<string[] | undefined>;
|
||||
};
|
||||
|
||||
export function createAskBridge(): AskBridge {
|
||||
let setter: ((p: AskPending | undefined) => void) | undefined;
|
||||
return {
|
||||
bind(fn) {
|
||||
setter = fn;
|
||||
},
|
||||
ask(req) {
|
||||
return new Promise((resolve) => {
|
||||
// No UI mounted means no one can answer; resolving undefined lets the tool
|
||||
// tell the model to decide for itself rather than hanging forever.
|
||||
if (!setter) return resolve(undefined);
|
||||
setter({
|
||||
req,
|
||||
resolve: (answers) => {
|
||||
setter?.(undefined);
|
||||
resolve(answers);
|
||||
},
|
||||
});
|
||||
});
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
const TYPE_YOUR_OWN = '__own__';
|
||||
|
||||
/**
|
||||
* The question popup.
|
||||
*
|
||||
* Options become a picker; no options, or "type your own", falls back to free text.
|
||||
* Escape resolves undefined rather than leaving the tool waiting.
|
||||
*/
|
||||
export function AskPanel({ pending }: { pending: AskPending }) {
|
||||
const { question, options, multiple } = pending.req;
|
||||
const [chosen, setChosen] = useState<string[]>([]);
|
||||
const [typing, setTyping] = useState(!options || options.length === 0);
|
||||
const [draft, setDraft] = useState('');
|
||||
|
||||
useInput(
|
||||
(_input, key) => {
|
||||
if (key.escape) pending.resolve(undefined);
|
||||
},
|
||||
{ isActive: !typing },
|
||||
);
|
||||
|
||||
const items = [
|
||||
...(options ?? []).map((o) => ({
|
||||
key: o.label,
|
||||
label: chosen.includes(o.label) ? `[x] ${o.label}` : multiple ? `[ ] ${o.label}` : o.label,
|
||||
value: o.label,
|
||||
})),
|
||||
...(multiple && chosen.length > 0 ? [{ key: '__done__', label: `-- submit ${chosen.length} --`, value: '__done__' }] : []),
|
||||
{ key: TYPE_YOUR_OWN, label: 'type your own answer...', value: TYPE_YOUR_OWN },
|
||||
];
|
||||
|
||||
const detailOf = (label: string) => (options ?? []).find((o) => o.label === label)?.detail;
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="double" borderColor="yellow" paddingX={1}>
|
||||
<Text color="yellow" bold>
|
||||
shiro is asking
|
||||
</Text>
|
||||
<Box marginBottom={1}>
|
||||
<InlineMarkdown text={question} />
|
||||
</Box>
|
||||
|
||||
{typing ? (
|
||||
<Box>
|
||||
<Text color="yellow">{'> '}</Text>
|
||||
<PromptInput
|
||||
value={draft}
|
||||
onChange={setDraft}
|
||||
onSubmit={(v) => pending.resolve(v.trim() ? [v.trim()] : undefined)}
|
||||
placeholder="type your answer, enter to send"
|
||||
/>
|
||||
</Box>
|
||||
) : (
|
||||
<Box flexDirection="column">
|
||||
<SelectInput
|
||||
items={items}
|
||||
limit={10}
|
||||
onSelect={(item) => {
|
||||
if (item.value === TYPE_YOUR_OWN) return setTyping(true);
|
||||
if (item.value === '__done__') return pending.resolve(chosen);
|
||||
if (!multiple) return pending.resolve([item.value]);
|
||||
setChosen((c) => (c.includes(item.value) ? c.filter((x) => x !== item.value) : [...c, item.value]));
|
||||
}}
|
||||
onHighlight={(item) => {
|
||||
const detail = detailOf(item.value);
|
||||
if (detail) setDraft(detail);
|
||||
else setDraft('');
|
||||
}}
|
||||
/>
|
||||
{draft.length > 0 && <Text dimColor>{draft}</Text>}
|
||||
<Text dimColor>
|
||||
{multiple ? 'space/enter toggles, pick submit when done' : 'enter to choose'} | esc to skip
|
||||
</Text>
|
||||
</Box>
|
||||
)}
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user