release 1.0.0: cost control, 41 tools, 29 skills, custom commands, auto-load
This commit is contained in:
@@ -0,0 +1,132 @@
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { loadExternalPlugins, loadExternalTools, parseToolManifest, manifestToTool } from '../src/autoload';
|
||||
|
||||
let cwd: string;
|
||||
let home: string;
|
||||
let origCwd: string;
|
||||
let origHome: string | undefined;
|
||||
|
||||
beforeEach(() => {
|
||||
origCwd = process.cwd();
|
||||
origHome = process.env['SHIRO_HOME'];
|
||||
cwd = mkdtempSync(join(tmpdir(), 'shiro-auto-'));
|
||||
home = mkdtempSync(join(tmpdir(), 'shiro-autohome-'));
|
||||
process.chdir(cwd);
|
||||
process.env['SHIRO_HOME'] = home;
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
process.chdir(origCwd);
|
||||
if (origHome === undefined) delete process.env['SHIRO_HOME'];
|
||||
else process.env['SHIRO_HOME'] = origHome;
|
||||
rmSync(cwd, { recursive: true, force: true });
|
||||
rmSync(home, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
const noGuard = async () => undefined;
|
||||
const denyAll = async () => 'not allowed';
|
||||
|
||||
test('a tool manifest validates its kind-specific field', () => {
|
||||
expect(() => parseToolManifest('{"name":"x","description":"d","kind":"shell"}')).toThrow(/needs a command/);
|
||||
expect(() => parseToolManifest('{"name":"x","description":"d","kind":"http"}')).toThrow(/needs a url/);
|
||||
expect(() => parseToolManifest('{"name":"x","description":"d","kind":"read"}')).toThrow(/needs a path/);
|
||||
expect(parseToolManifest('{"name":"x","description":"d","kind":"shell","command":"echo hi"}').name).toBe('x');
|
||||
});
|
||||
|
||||
test('a malformed manifest is a descriptive error, not a crash', () => {
|
||||
expect(() => parseToolManifest('not json')).toThrow(/not valid JSON/);
|
||||
expect(() => parseToolManifest('{"name":"x"}')).toThrow(/malformed/);
|
||||
});
|
||||
|
||||
test('an external read tool returns a workspace file, jailed', async () => {
|
||||
await Bun.write('note.txt', 'hello workspace');
|
||||
const t = manifestToTool(parseToolManifest('{"name":"rd","description":"d","kind":"read","path":"note.txt"}'), noGuard);
|
||||
const out = (await t.execute!({ arg: '' }, { toolCallId: 't', messages: [] } as never)) as string;
|
||||
expect(out).toBe('hello workspace');
|
||||
});
|
||||
|
||||
test('a read tool cannot escape the workspace', async () => {
|
||||
const t = manifestToTool(
|
||||
parseToolManifest('{"name":"rd","description":"d","kind":"read","path":"../../secret"}'),
|
||||
noGuard,
|
||||
);
|
||||
await expect(t.execute!({ arg: '' }, { toolCallId: 't', messages: [] } as never)).rejects.toThrow();
|
||||
});
|
||||
|
||||
test('an external shell tool substitutes {arg} and runs it', async () => {
|
||||
const t = manifestToTool(
|
||||
parseToolManifest('{"name":"hi","description":"d","kind":"shell","command":"echo got-{arg}"}'),
|
||||
noGuard,
|
||||
);
|
||||
const out = (await t.execute!({ arg: 'there' }, { toolCallId: 't', messages: [] } as never)) as string;
|
||||
expect(out).toBe('got-there');
|
||||
});
|
||||
|
||||
test('a shell tool is stopped by the guard before it runs', async () => {
|
||||
const t = manifestToTool(
|
||||
parseToolManifest('{"name":"bad","description":"d","kind":"shell","command":"echo should-not-run"}'),
|
||||
denyAll,
|
||||
);
|
||||
await expect(t.execute!({ arg: '' }, { toolCallId: 't', messages: [] } as never)).rejects.toThrow(/refused/);
|
||||
});
|
||||
|
||||
test('an http tool refuses a non-https URL', async () => {
|
||||
const t = manifestToTool(
|
||||
parseToolManifest('{"name":"w","description":"d","kind":"http","url":"http://insecure.example/x"}'),
|
||||
noGuard,
|
||||
);
|
||||
await expect(t.execute!({ arg: '' }, { toolCallId: 't', messages: [] } as never)).rejects.toThrow(/https/);
|
||||
});
|
||||
|
||||
test('loadExternalTools auto-registers a project tool and marks it auto-approved', async () => {
|
||||
await Bun.write(join('.shiro', 'tools', 'greet.json'), '{"name":"greet","description":"d","kind":"shell","command":"echo hi"}');
|
||||
const { tools, autoApprove, errors } = await loadExternalTools(cwd, noGuard);
|
||||
expect(Object.keys(tools)).toContain('greet');
|
||||
expect(autoApprove).toContain('greet');
|
||||
expect(errors).toHaveLength(0);
|
||||
});
|
||||
|
||||
test('a tool with autoApprove:false is registered but not auto-approved', async () => {
|
||||
await Bun.write(
|
||||
join('.shiro', 'tools', 'deploy.json'),
|
||||
'{"name":"deploy","description":"d","kind":"shell","command":"echo deploy","autoApprove":false}',
|
||||
);
|
||||
const { tools, autoApprove } = await loadExternalTools(cwd, noGuard);
|
||||
expect(Object.keys(tools)).toContain('deploy');
|
||||
expect(autoApprove).not.toContain('deploy');
|
||||
});
|
||||
|
||||
test('a bad tool file is reported and skipped, never fatal', async () => {
|
||||
await Bun.write(join('.shiro', 'tools', 'broken.json'), '{ not json');
|
||||
await Bun.write(join('.shiro', 'tools', 'good.json'), '{"name":"good","description":"d","kind":"read","path":"x.txt"}');
|
||||
const { tools, errors } = await loadExternalTools(cwd, noGuard);
|
||||
expect(Object.keys(tools)).toContain('good');
|
||||
expect(errors).toHaveLength(1);
|
||||
expect(errors[0]!.name).toBe('broken');
|
||||
});
|
||||
|
||||
test('an external plugin manifest auto-loads and blocks a matching call', async () => {
|
||||
await Bun.write(
|
||||
join('.shiro', 'plugins', 'no-drop.json'),
|
||||
JSON.stringify({
|
||||
name: 'no-drop',
|
||||
description: 'never drop a table',
|
||||
deny: [{ tools: ['bash'], commandPattern: 'DROP TABLE', reason: 'drops data' }],
|
||||
}),
|
||||
);
|
||||
const { plugins, errors } = await loadExternalPlugins(cwd);
|
||||
expect(errors).toHaveLength(0);
|
||||
expect(plugins.map((p) => p.name)).toContain('no-drop');
|
||||
const blocked = await plugins[0]!.beforeToolCall!({ toolName: 'bash', input: { command: 'DROP TABLE users' }, cwd });
|
||||
expect(blocked).toBe('drops data');
|
||||
});
|
||||
|
||||
test('a bad plugin file is reported and skipped', async () => {
|
||||
await Bun.write(join('.shiro', 'plugins', 'bad.json'), '{"name":"bad"}');
|
||||
const { plugins, errors } = await loadExternalPlugins(cwd);
|
||||
expect(plugins).toHaveLength(0);
|
||||
expect(errors).toHaveLength(1);
|
||||
});
|
||||
+9
-5
@@ -33,16 +33,20 @@ test('release verifies before it builds, and builds before it publishes', async
|
||||
expect(release.indexOf('bun test')).toBeLessThan(release.indexOf('bun run release'));
|
||||
});
|
||||
|
||||
test('publishing is gated on a tag, so a manual run cannot release by accident', async () => {
|
||||
test('publishing is gated on a tag or an explicit dry_run=false dispatch', async () => {
|
||||
const release = await read('.github/workflows/release.yml');
|
||||
expect(release).toContain("if: startsWith(github.ref, 'refs/tags/v')");
|
||||
// Tag pushes always publish; a manual run publishes only when dry_run is
|
||||
// explicitly unchecked, so the default (true) can never release by accident.
|
||||
expect(release).toContain("startsWith(github.ref, 'refs/tags/v')");
|
||||
expect(release).toContain("github.event_name == 'workflow_dispatch' && inputs.dry_run == false");
|
||||
});
|
||||
|
||||
test('a prerelease tag is marked as a prerelease', async () => {
|
||||
test('the workflow can mark a prerelease when the tag carries a suffix', async () => {
|
||||
const release = await read('.github/workflows/release.yml');
|
||||
// The `--prerelease` flag fires only when the tag name contains a `-`, so a stable
|
||||
// release like the current one publishes normally and a `vX.Y.Z-rc.1` marks itself.
|
||||
expect(release).toContain('--prerelease');
|
||||
// The current version is a prerelease, so the marker has to be reachable.
|
||||
expect(VERSION).toContain('-');
|
||||
expect(release).toContain('== *-* ]] && echo --prerelease');
|
||||
});
|
||||
|
||||
test('the release pins the bun version rather than tracking latest', async () => {
|
||||
|
||||
+2
-4
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -12,10 +13,7 @@ import { Session } from '../src/session';
|
||||
import { disabledToolNames, toolSetOf } from '../src/tools';
|
||||
import { GIT_TOOL_NAMES } from '../src/tools-git';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 5 },
|
||||
} as any;
|
||||
const usage = usageOf(10);
|
||||
|
||||
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
||||
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -7,10 +8,7 @@ import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { Session, type AgentEvent } from '../src/session';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 5 },
|
||||
} as any;
|
||||
const usage = usageOf(10);
|
||||
|
||||
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
||||
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
||||
|
||||
@@ -4,9 +4,9 @@ import React from 'react';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge } from '../src/ui/App';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = { inputTokens: { total: 3, noCache: 3, cacheRead: 0, cacheWrite: 0 }, outputTokens: { total: 1 } } as any;
|
||||
const usage = usageOf(3);
|
||||
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { parseCommand } from '../src/commands';
|
||||
import { expandCommand, loadCustomCommands, type CustomCommand } from '../src/custom-commands';
|
||||
|
||||
let dir: string;
|
||||
let origCwd: string;
|
||||
let origHome: string | undefined;
|
||||
|
||||
beforeEach(() => {
|
||||
origCwd = process.cwd();
|
||||
origHome = process.env['SHIRO_HOME'];
|
||||
dir = mkdtempSync(join(tmpdir(), 'shiro-cmd-'));
|
||||
process.chdir(dir);
|
||||
// An isolated home so a real ~/.shiro-neko/commands cannot leak into a test.
|
||||
const home = mkdtempSync(join(tmpdir(), 'shiro-home-'));
|
||||
process.env['SHIRO_HOME'] = home;
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
process.chdir(origCwd);
|
||||
if (origHome === undefined) delete process.env['SHIRO_HOME'];
|
||||
else process.env['SHIRO_HOME'] = origHome;
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
const cmd = (over: Partial<CustomCommand>): CustomCommand => ({
|
||||
name: 'review-diff',
|
||||
description: 'review the current diff',
|
||||
body: 'Review the diff.',
|
||||
origin: 'project',
|
||||
path: '/x/review-diff.md',
|
||||
...over,
|
||||
});
|
||||
|
||||
test('a project command is discovered from .shiro/commands', async () => {
|
||||
await Bun.write(join('.shiro', 'commands', 'deploy.md'), '---\ndescription: ship it\n---\nDeploy the app.');
|
||||
const cmds = await loadCustomCommands(dir);
|
||||
expect(cmds.map((c) => c.name)).toContain('deploy');
|
||||
expect(cmds.find((c) => c.name === 'deploy')?.origin).toBe('project');
|
||||
});
|
||||
|
||||
test('frontmatter supplies description and agent; the body is the prompt', async () => {
|
||||
await Bun.write(
|
||||
join('.shiro', 'commands', 'fix.md'),
|
||||
'---\ndescription: fix a test\nagent: deep\n---\nFix the failing test.',
|
||||
);
|
||||
const cmds = await loadCustomCommands(dir);
|
||||
const fix = cmds.find((c) => c.name === 'fix');
|
||||
expect(fix?.description).toBe('fix a test');
|
||||
expect(fix?.agent).toBe('deep');
|
||||
expect(fix?.body).toBe('Fix the failing test.');
|
||||
});
|
||||
|
||||
test('a project command shadows a user command of the same name', async () => {
|
||||
const home = process.env['SHIRO_HOME']!;
|
||||
await Bun.write(join(home, '.shiro-neko', 'commands', 'go.md'), 'user version');
|
||||
await Bun.write(join('.shiro', 'commands', 'go.md'), 'project version');
|
||||
const cmds = await loadCustomCommands(dir);
|
||||
const go = cmds.find((c) => c.name === 'go');
|
||||
expect(go?.origin).toBe('project');
|
||||
expect(go?.body).toBe('project version');
|
||||
});
|
||||
|
||||
test('an empty body or a bad filename is skipped', async () => {
|
||||
await Bun.write(join('.shiro', 'commands', 'empty.md'), '---\ndescription: nothing\n---\n ');
|
||||
await Bun.write(join('.shiro', 'commands', 'BAD NAME.md'), 'not a valid command name');
|
||||
const cmds = await loadCustomCommands(dir);
|
||||
expect(cmds.map((c) => c.name)).not.toContain('empty');
|
||||
expect(cmds).toHaveLength(0);
|
||||
});
|
||||
|
||||
test('the parser resolves a custom command and splits its arguments', () => {
|
||||
const action = parseCommand('/review-diff src/auth.ts', [cmd({})]);
|
||||
expect(action).toEqual({ type: 'custom', command: expect.objectContaining({ name: 'review-diff' }), args: ['src/auth.ts'] });
|
||||
});
|
||||
|
||||
test('a custom command never shadows a built-in', () => {
|
||||
const action = parseCommand('/cost', [cmd({ name: 'cost', body: 'hijack' })]);
|
||||
expect(action.type).toBe('cost');
|
||||
});
|
||||
|
||||
test('$ARGUMENTS and positionals expand', async () => {
|
||||
const c = cmd({ body: 'Review $1 against $2. All: $ARGUMENTS' });
|
||||
const out = await expandCommand(c, ['a.ts', 'b.ts']);
|
||||
expect(out).toBe('Review a.ts against b.ts. All: a.ts b.ts');
|
||||
});
|
||||
|
||||
test('a missing positional expands to nothing', async () => {
|
||||
const c = cmd({ body: 'one $1 two $2 end' });
|
||||
expect(await expandCommand(c, ['only'])).toBe('one only two end');
|
||||
});
|
||||
|
||||
test('a shell substitution inlines its output', async () => {
|
||||
const c = cmd({ body: 'The branch is !`echo feature-x`.' });
|
||||
const out = await expandCommand(c, []);
|
||||
expect(out).toBe('The branch is feature-x.');
|
||||
});
|
||||
|
||||
test('a destructive shell substitution is refused by the guard', async () => {
|
||||
const c = cmd({ body: 'run !`rm -rf /`' });
|
||||
await expect(expandCommand(c, [])).rejects.toThrow(/refused/i);
|
||||
});
|
||||
|
||||
test('a failing substitution reports the command and exit', async () => {
|
||||
const c = cmd({ body: 'value: !`exit 3`' });
|
||||
await expect(expandCommand(c, [])).rejects.toThrow(/exited 3/);
|
||||
});
|
||||
@@ -1,13 +1,11 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { APICallError } from 'ai';
|
||||
import type { LanguageModelV4, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
import { simulateReadableStream } from 'ai/test';
|
||||
import { withFallback, type FallbackEvent } from '../src/fallback';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 1 },
|
||||
} as any;
|
||||
const usage = usageOf(1);
|
||||
|
||||
const okStream = (body: string) => ({
|
||||
stream: simulateReadableStream<LanguageModelV4StreamPart>({
|
||||
|
||||
@@ -4,12 +4,9 @@ import React from 'react';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 3, noCache: 3, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 1 },
|
||||
} as any;
|
||||
const usage = usageOf(3);
|
||||
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -7,10 +8,7 @@ import { join } from 'node:path';
|
||||
import { runHeadless } from '../src/headless';
|
||||
import { Session } from '../src/session';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 9, noCache: 9, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 4 },
|
||||
} as any;
|
||||
const usage = usageOf(9, 4);
|
||||
|
||||
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
||||
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
||||
|
||||
@@ -1,5 +1,20 @@
|
||||
import type { LanguageModelV4Usage } from '@ai-sdk/provider';
|
||||
import type { AppHooks } from '../src/ui/App';
|
||||
|
||||
/**
|
||||
* A `usage` chunk for `simulateReadableStream`, typed so the cast goes away.
|
||||
*
|
||||
* The same object was copied verbatim into 17 test files with `as any`, one per
|
||||
* stream finish. This builds the real `LanguageModelV4Usage` shape with the
|
||||
* numbers a test cares about and the cache fields zeroed.
|
||||
*/
|
||||
export function usageOf(input: number, output = 1): LanguageModelV4Usage {
|
||||
return {
|
||||
inputTokens: { total: input, noCache: input, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: output, text: output, reasoning: 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/** Default AppHooks for UI tests; override only what a test cares about. */
|
||||
export function testHooks(over: Partial<AppHooks> = {}): AppHooks {
|
||||
return {
|
||||
|
||||
+2
-5
@@ -4,12 +4,9 @@ import React from 'react';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 1000, noCache: 1000, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 500 },
|
||||
} as any;
|
||||
const usage = usageOf(1000, 500);
|
||||
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
|
||||
@@ -6,12 +6,9 @@ import { parseCommand, COMMANDS, HELP } from '../src/commands';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
|
||||
import { invalidName, parseHeaders, splitArgs, McpAdd } from '../src/ui/McpAdd';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 3, noCache: 3, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 1 },
|
||||
} as any;
|
||||
const usage = usageOf(3);
|
||||
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
|
||||
+2
-4
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4 } from 'ai/test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
@@ -28,10 +29,7 @@ const call = (tools: ToolSet, name: string, input: Record<string, unknown>) => {
|
||||
return Promise.resolve(t.execute(input as never, { toolCallId: 'x', messages: [] } as never)) as Promise<string>;
|
||||
};
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 1 },
|
||||
} as any;
|
||||
const usage = usageOf(1);
|
||||
|
||||
const summarizer = (text: string) =>
|
||||
new MockLanguageModelV4({
|
||||
|
||||
+2
-5
@@ -4,12 +4,9 @@ import React from 'react';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 1 },
|
||||
} as any;
|
||||
const usage = usageOf(2);
|
||||
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
|
||||
@@ -3,7 +3,20 @@ import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { createHost } from '../src/plugins';
|
||||
import { BUILTIN_PLUGINS, DEFAULT_ENABLED, formatPlugin, protectPlugin, secretsPlugin } from '../src/plugins-builtin';
|
||||
import {
|
||||
BUILTIN_PLUGINS,
|
||||
DEFAULT_ENABLED,
|
||||
confirmDeletePlugin,
|
||||
formatPlugin,
|
||||
noEnvWritePlugin,
|
||||
noForcePushPlugin,
|
||||
noGitConfigPlugin,
|
||||
noMainCommitPlugin,
|
||||
noNetPipePlugin,
|
||||
noRootPlugin,
|
||||
protectPlugin,
|
||||
secretsPlugin,
|
||||
} from '../src/plugins-builtin';
|
||||
|
||||
const cwd = process.cwd();
|
||||
const check = (toolName: string, input: unknown) => secretsPlugin.beforeToolCall!({ toolName, input, cwd });
|
||||
@@ -195,3 +208,62 @@ test('a throwing afterTurn does not stop the other plugins', async () => {
|
||||
await host.afterTurn();
|
||||
expect(ran).toBe(1);
|
||||
});
|
||||
|
||||
// --- The ten additional plugins ---
|
||||
|
||||
const bash = (plugin: (typeof BUILTIN_PLUGINS)[number], command: string) =>
|
||||
plugin.beforeToolCall!({ toolName: 'bash', input: { command }, cwd });
|
||||
|
||||
test('the narrow safety refusals are on by default; opinionated ones are opt-in', () => {
|
||||
for (const on of ['no-force-push', 'no-net-pipe', 'no-root', 'no-env-write']) {
|
||||
expect(DEFAULT_ENABLED, on).toContain(on);
|
||||
}
|
||||
for (const off of ['no-main-commit', 'conventional-commit', 'tests-first', 'small-diffs', 'confirm-delete', 'no-git-config']) {
|
||||
expect(DEFAULT_ENABLED, off).not.toContain(off);
|
||||
}
|
||||
});
|
||||
|
||||
test('no-force-push refuses a force push but allows a normal one', async () => {
|
||||
expect(await bash(noForcePushPlugin, 'git push --force origin main')).toContain('refusing');
|
||||
expect(await bash(noForcePushPlugin, 'git push -f')).toContain('refusing');
|
||||
expect(await bash(noForcePushPlugin, 'git push origin feature')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('no-main-commit refuses committing on the default branch', async () => {
|
||||
expect(await bash(noMainCommitPlugin, 'git commit -m "x" main')).toContain('refusing');
|
||||
expect(await bash(noMainCommitPlugin, 'git commit -m "x"')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('no-root refuses sudo and elevation', async () => {
|
||||
expect(await bash(noRootPlugin, 'sudo rm -rf /tmp/x')).toContain('refusing');
|
||||
expect(await bash(noRootPlugin, 'npm test')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('no-net-pipe refuses executing a download into a shell or runtime', async () => {
|
||||
expect(await bash(noNetPipePlugin, 'curl https://x.sh | bash')).toContain('refusing');
|
||||
expect(await bash(noNetPipePlugin, 'curl https://x.js | node')).toContain('refusing');
|
||||
expect(await bash(noNetPipePlugin, 'curl -o setup.sh https://x.sh')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('no-git-config refuses changing global git configuration', async () => {
|
||||
expect(await bash(noGitConfigPlugin, 'git config --global user.name "x"')).toContain('refusing');
|
||||
expect(await bash(noGitConfigPlugin, 'git config --local core.autocrlf true')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('no-env-write refuses exporting a credential into the environment', async () => {
|
||||
expect(await bash(noEnvWritePlugin, 'export OPENAI_API_KEY=sk-abc')).toContain('refusing');
|
||||
expect(await bash(noEnvWritePlugin, 'export NODE_ENV=production')).toBeUndefined();
|
||||
});
|
||||
|
||||
test('confirm-delete refuses broad deletes but allows one explicit file', async () => {
|
||||
expect(await confirmDeletePlugin.beforeToolCall!({ toolName: 'delete_file', input: { path: 'src/*' }, cwd })).toContain('refusing');
|
||||
expect(await confirmDeletePlugin.beforeToolCall!({ toolName: 'delete_file', input: { path: 'build/' }, cwd })).toContain('refusing');
|
||||
expect(await confirmDeletePlugin.beforeToolCall!({ toolName: 'delete_file', input: { path: 'tmp/old.log' }, cwd })).toBeUndefined();
|
||||
});
|
||||
|
||||
test('advisory plugins carry an appendix and no blocking hook', () => {
|
||||
for (const p of BUILTIN_PLUGINS.filter((x) => ['conventional-commit', 'tests-first', 'small-diffs'].includes(x.name))) {
|
||||
expect(p.appendix, p.name).toBeDefined();
|
||||
expect(p.beforeToolCall, p.name).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
+3
-2
@@ -104,6 +104,7 @@ test('omitting every section leaves no dangling markers', () => {
|
||||
|
||||
test('the prompt stays a reasonable size with everything on', () => {
|
||||
const prompt = systemPrompt({ cwd: '/repo', availableTools: ALL, canAsk: true });
|
||||
// Sent on every request, so a runaway prompt is a direct cost.
|
||||
expect(prompt.length).toBeLessThan(5000);
|
||||
// Sent on every request, so a runaway prompt is a direct cost. The budget scales
|
||||
// with the documented-tool count: a new tool earns its own line, the rest must not.
|
||||
expect(prompt.length).toBeLessThan(5000 + (TOOL_DOCS.length - 22) * 120);
|
||||
});
|
||||
|
||||
@@ -60,6 +60,19 @@ test('writeConfigFile merges instead of clobbering unrelated keys', async () =>
|
||||
expect(file.model).toBe('gpt-5');
|
||||
});
|
||||
|
||||
test('maxSpendUsd and subagentModel load from the config file', async () => {
|
||||
await writeConfigFile({ provider: 'openai', model: 'gpt-5', apiKey: 'sk-1', maxSpendUsd: 5, subagentModel: 'gpt-5-nano' });
|
||||
const cfg = await loadConfig();
|
||||
expect(cfg.maxSpendUsd).toBe(5);
|
||||
expect(cfg.subagentModel).toBe('gpt-5-nano');
|
||||
});
|
||||
|
||||
test('a non-positive maxSpendUsd is ignored rather than enforced', async () => {
|
||||
await writeConfigFile({ provider: 'openai', model: 'gpt-5', apiKey: 'sk-1', maxSpendUsd: 0 });
|
||||
const cfg = await loadConfig();
|
||||
expect(cfg.maxSpendUsd).toBeUndefined();
|
||||
});
|
||||
|
||||
/**
|
||||
* What `/mcp add` and `/mcp remove` do to the file, exercised through the real
|
||||
* persistence path. The hooks themselves live inline in cli.tsx, which a test
|
||||
|
||||
+4
-7
@@ -5,12 +5,9 @@ import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge } from '../src/ui/App';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 4, noCache: 4, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 2 },
|
||||
} as any;
|
||||
const usage = usageOf(4);
|
||||
|
||||
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
@@ -64,7 +61,7 @@ test('the input stays live while a turn runs, and a submission is queued', async
|
||||
// Mid-turn: the spinner and the input coexist rather than swapping. The
|
||||
// placeholder's first character is inverted for the cursor, hence the offset.
|
||||
const midTurn = app.lastFrame() ?? '';
|
||||
expect(midTurn).toContain('working...');
|
||||
expect(midTurn).toContain('esc to interrupt');
|
||||
expect(midTurn).toContain('ype to queue');
|
||||
|
||||
await type(app, 'second');
|
||||
@@ -145,7 +142,7 @@ test('reasoning shows as a collapsed line before any text arrives, then leaves w
|
||||
await wait(700);
|
||||
|
||||
const thinking = app.lastFrame() ?? '';
|
||||
expect(thinking).toContain('thinking...');
|
||||
expect(thinking).toContain('thinking');
|
||||
expect(thinking).toContain('tokens');
|
||||
expect(thinking).not.toContain('weighing the options');
|
||||
|
||||
|
||||
@@ -5,9 +5,9 @@ import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
|
||||
import { InstallPrompt, RegistryPanel, type RegistryRow } from '../src/ui/Panels';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = { inputTokens: { total: 3, noCache: 3, cacheRead: 0, cacheWrite: 0 }, outputTokens: { total: 1 } } as any;
|
||||
const usage = usageOf(3);
|
||||
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -24,10 +25,7 @@ function inTempDir<T>(fn: () => Promise<T>): Promise<T> {
|
||||
});
|
||||
}
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 5, noCache: 5, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 2 },
|
||||
} as any;
|
||||
const usage = usageOf(5);
|
||||
|
||||
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
||||
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -9,10 +10,7 @@ import { join } from 'node:path';
|
||||
import { Session } from '../src/session';
|
||||
import { interruptBash } from '../src/tools';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 5, total_text: 5 },
|
||||
} as any;
|
||||
const usage = usageOf(10, 5);
|
||||
|
||||
function stream(parts: LanguageModelV4StreamPart[]) {
|
||||
return { stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }) };
|
||||
|
||||
+33
-11
@@ -5,6 +5,8 @@ import { join } from 'node:path';
|
||||
import { createSkillTool, loadSkills, parseSkill, renderSkills } from '../src/skills';
|
||||
import { BUILTIN_SKILLS } from '../src/skills-builtin';
|
||||
|
||||
const SKILLS_MD_DIR = join(import.meta.dir, '..', 'src', 'skills-md');
|
||||
|
||||
let home: string;
|
||||
let work: string;
|
||||
let savedHome: string | undefined;
|
||||
@@ -59,20 +61,40 @@ test('every bundled skill parses and has a usable description', () => {
|
||||
}
|
||||
});
|
||||
|
||||
test('every builtin skill is a Markdown file on disk with valid frontmatter and a real body', async () => {
|
||||
for (const { name, source } of BUILTIN_SKILLS) {
|
||||
// The .md file is the source of truth and must exist beside the others.
|
||||
const file = Bun.file(join(SKILLS_MD_DIR, `${name}.md`));
|
||||
expect(await file.exists(), `${name}.md missing`).toBe(true);
|
||||
|
||||
// Proper skill-Markdown: a --- frontmatter fence with name + description, then a body.
|
||||
expect(source.trimStart().startsWith('---'), `${name} has no frontmatter fence`).toBe(true);
|
||||
const parsed = parseSkill(source, 'builtin');
|
||||
expect(parsed, `${name} failed to parse`).toBeDefined();
|
||||
expect(parsed!.name, name).toBe(name);
|
||||
expect(parsed!.description.length, `${name} description too short`).toBeGreaterThan(20);
|
||||
expect(parsed!.body.length, `${name} body too short`).toBeGreaterThan(200);
|
||||
|
||||
// The body is real Markdown, not an escaped blob: it has a heading and structure.
|
||||
expect(parsed!.body, `${name} has no Markdown heading`).toMatch(/^#\s/m);
|
||||
}
|
||||
});
|
||||
|
||||
test('the builtin skills load with no files on disk', async () => {
|
||||
const skills = await loadSkills(work);
|
||||
expect(skills.map((s) => s.name)).toEqual([
|
||||
'commit',
|
||||
'debug',
|
||||
'migrate',
|
||||
'perf',
|
||||
'refactor',
|
||||
'review',
|
||||
'security',
|
||||
'test',
|
||||
'verify',
|
||||
]);
|
||||
// The catalogue is the builtin set and nothing else: every entry parses, is a
|
||||
// builtin, and the list is sorted and de-duplicated. Counted rather than named so
|
||||
// adding a skill does not mean editing this test.
|
||||
expect(skills.length).toBe(BUILTIN_SKILLS.length);
|
||||
expect(skills.every((s) => s.origin === 'builtin')).toBe(true);
|
||||
expect(skills.every((s) => s.name.length > 0 && s.description.length > 10 && s.body.length > 100)).toBe(true);
|
||||
const names = skills.map((s) => s.name);
|
||||
expect(new Set(names).size).toBe(names.length);
|
||||
expect([...names].sort((a, b) => a.localeCompare(b))).toEqual(names);
|
||||
// The long-standing skills must still be present even as the set grows.
|
||||
for (const name of ['commit', 'debug', 'migrate', 'perf', 'plan', 'refactor', 'review', 'security', 'test', 'verify', 'docs']) {
|
||||
expect(names).toContain(name);
|
||||
}
|
||||
});
|
||||
|
||||
test('a project skill is discovered and reported as project origin', async () => {
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
import { Session, type AgentEvent } from '../src/session';
|
||||
|
||||
function stream(parts: LanguageModelV4StreamPart[]) {
|
||||
return { stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }) };
|
||||
}
|
||||
|
||||
/** A priced model whose single run bills `usage`, so the ceiling arithmetic is exact. */
|
||||
function modelWith(usage: ReturnType<typeof usageOf>) {
|
||||
return new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
stream([
|
||||
{ type: 'text-start', id: '0' },
|
||||
{ type: 'text-delta', id: '0', delta: 'ok' },
|
||||
{ type: 'text-end', id: '0' },
|
||||
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
|
||||
]),
|
||||
});
|
||||
}
|
||||
|
||||
async function drain(session: Session): Promise<AgentEvent[]> {
|
||||
const events: AgentEvent[] = [];
|
||||
for await (const ev of session.send('go')) events.push(ev);
|
||||
return events;
|
||||
}
|
||||
|
||||
const noop = async () => 'once' as const;
|
||||
|
||||
test('a session past its ceiling refuses the turn and names the ceiling', async () => {
|
||||
// gpt-5 is $1.25/M in, $10/M out. 800k in / 200k out = $1.00 + $2.00 = $3.00.
|
||||
const session = new Session({
|
||||
model: modelWith(usageOf(800_000, 200_000)),
|
||||
modelId: 'gpt-5',
|
||||
askApproval: noop,
|
||||
maxSpendUsd: 2,
|
||||
});
|
||||
|
||||
const first = await drain(session);
|
||||
expect(first.some((e) => e.type === 'done')).toBe(true);
|
||||
|
||||
// The first turn put spend over the $2 ceiling; the next is refused before the model runs.
|
||||
const second = await drain(session);
|
||||
const refusal = second.find((e) => e.type === 'error');
|
||||
expect(refusal).toBeDefined();
|
||||
expect(refusal!.type === 'error' && String(refusal!.error)).toContain('spend ceiling reached');
|
||||
expect(String(refusal!.type === 'error' && refusal!.error)).toContain('$2.00');
|
||||
});
|
||||
|
||||
test('crossing 80% warns once, then stays quiet', async () => {
|
||||
// 200k in / 100k out = $0.25 + $1.00 = $1.25, which is 83% of a $1.50 ceiling.
|
||||
const session = new Session({
|
||||
model: modelWith(usageOf(200_000, 100_000)),
|
||||
modelId: 'gpt-5',
|
||||
askApproval: noop,
|
||||
maxSpendUsd: 1.5,
|
||||
});
|
||||
|
||||
const first = await drain(session);
|
||||
const warnings = first.filter((e) => e.type === 'notice' && e.text.includes('approaching spend ceiling'));
|
||||
expect(warnings).toHaveLength(1);
|
||||
expect(warnings[0]!.type === 'notice' && warnings[0]!.text).toContain('$1.50');
|
||||
|
||||
// A second turn at the same level must not repeat the warning.
|
||||
const second = await drain(session);
|
||||
expect(second.filter((e) => e.type === 'notice' && e.text.includes('approaching spend ceiling'))).toHaveLength(0);
|
||||
});
|
||||
|
||||
test('an unpriced model is never refused, because the ceiling cannot see it', async () => {
|
||||
const session = new Session({
|
||||
model: modelWith(usageOf(9_000_000, 9_000_000)),
|
||||
modelId: 'some-local-model',
|
||||
askApproval: noop,
|
||||
maxSpendUsd: 0.01,
|
||||
});
|
||||
const events = await drain(session);
|
||||
expect(events.some((e) => e.type === 'done')).toBe(true);
|
||||
expect(events.some((e) => e.type === 'error')).toBe(false);
|
||||
});
|
||||
|
||||
test('with no ceiling configured a session spends freely', async () => {
|
||||
const session = new Session({ model: modelWith(usageOf(9_000_000, 9_000_000)), modelId: 'gpt-5', askApproval: noop });
|
||||
const events = await drain(session);
|
||||
expect(events.some((e) => e.type === 'done')).toBe(true);
|
||||
});
|
||||
|
||||
test('spend() reports the ceiling state for the status surface', () => {
|
||||
const session = new Session({ model: modelWith(usageOf(1)), modelId: 'gpt-5', askApproval: noop, maxSpendUsd: 4 });
|
||||
session.inputTokens = 800_000;
|
||||
session.outputTokens = 200_000; // $3.00 of $4.00
|
||||
const s = session.spend();
|
||||
expect(s.usd).toBeCloseTo(3, 5);
|
||||
expect(s.ceiling).toBe(4);
|
||||
expect(s.overWarn).toBe(false);
|
||||
expect(s.overLimit).toBe(false);
|
||||
});
|
||||
|
||||
test('subagent spend folds into the ceiling and resets with the session', async () => {
|
||||
const session = new Session({
|
||||
model: modelWith(usageOf(1)),
|
||||
modelId: 'gpt-5',
|
||||
subagentModelId: 'gpt-5-nano',
|
||||
askApproval: noop,
|
||||
maxSpendUsd: 4,
|
||||
});
|
||||
session.recordSubagentUsage({ inputTokens: 1_000_000, outputTokens: 100_000 });
|
||||
expect(session.subagentInputTokens).toBe(1_000_000);
|
||||
// gpt-5-nano: $0.05/M in, $0.40/M out -> $0.05 + $0.04 = $0.09
|
||||
expect(session.spend().usd).toBeCloseTo(0.09, 5);
|
||||
|
||||
session.reset();
|
||||
expect(session.subagentInputTokens).toBe(0);
|
||||
expect(session.spend().usd).toBeCloseTo(0, 5);
|
||||
});
|
||||
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -5,10 +6,7 @@ import { z } from 'zod';
|
||||
import { Session } from '../src/session';
|
||||
import { createTaskTool, type SubagentEvent } from '../src/subagent';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 5, noCache: 5, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 2 },
|
||||
} as any;
|
||||
const usage = usageOf(5);
|
||||
|
||||
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
||||
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
||||
|
||||
+60
-4
@@ -1,3 +1,4 @@
|
||||
import { usageOf } from './helpers';
|
||||
import { expect, test } from 'bun:test';
|
||||
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
||||
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
||||
@@ -9,10 +10,7 @@ import { guardPlugin } from '../src/plugins-builtin';
|
||||
import { Session } from '../src/session';
|
||||
import { createTaskTool } from '../src/subagent';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 5 },
|
||||
} as any;
|
||||
const usage = usageOf(10);
|
||||
|
||||
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
||||
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
||||
@@ -268,3 +266,61 @@ test('subagent does not see the parent conversation', async () =>
|
||||
expect(JSON.stringify(seen[1]?.prompt)).not.toContain('MY-SECRET-PARENT-CONTEXT');
|
||||
expect(JSON.stringify(seen[1]?.prompt)).toContain('Look at glob src/*.');
|
||||
}));
|
||||
|
||||
test('an explore subagent runs on the cheaper model, not the parent model', async () =>
|
||||
inTempDir(async () => {
|
||||
// Two separate models: the parent's and the cheaper explore model. Whichever
|
||||
// one the subagent loop hits tells us which it was handed.
|
||||
let parentCalls = 0;
|
||||
const parent = new MockLanguageModelV4({
|
||||
doStream: async () => {
|
||||
if (parentCalls++ === 0)
|
||||
return stream(toolCall('c1', 'task', { description: 'search', prompt: 'find x', kind: 'explore' }));
|
||||
return stream(text('parent answer'));
|
||||
},
|
||||
});
|
||||
let cheapCalls = 0;
|
||||
const cheap = new MockLanguageModelV4({
|
||||
doStream: async () => {
|
||||
cheapCalls++;
|
||||
return stream(text('explore found nothing'));
|
||||
},
|
||||
});
|
||||
|
||||
const session = new Session({
|
||||
model: parent,
|
||||
askApproval: async () => 'deny',
|
||||
extraTools: { task: createTaskTool({ model: parent, subagentModel: cheap }) },
|
||||
autoApprove: ['task'],
|
||||
});
|
||||
|
||||
for await (const _ of session.send('search for x')) void _;
|
||||
|
||||
expect(cheapCalls).toBeGreaterThan(0);
|
||||
}));
|
||||
|
||||
test('a finished subagent reports its token use to the parent', async () =>
|
||||
inTempDir(async () => {
|
||||
let calls = 0;
|
||||
const model = new MockLanguageModelV4({
|
||||
doStream: async () => {
|
||||
if (calls++ === 0)
|
||||
return stream(toolCall('c1', 'task', { description: 'search', prompt: 'find x', kind: 'explore' }));
|
||||
if (calls === 2) return stream(text('findings'));
|
||||
return stream(text('done'));
|
||||
},
|
||||
});
|
||||
const usageEvents: { kind: string; inputTokens: number; outputTokens: number }[] = [];
|
||||
|
||||
const session = new Session({
|
||||
model,
|
||||
askApproval: async () => 'deny',
|
||||
extraTools: { task: createTaskTool({ model, onUsage: (u) => usageEvents.push(u) }) },
|
||||
autoApprove: ['task'],
|
||||
});
|
||||
|
||||
for await (const _ of session.send('go')) void _;
|
||||
|
||||
expect(usageEvents).toHaveLength(1);
|
||||
expect(usageEvents[0]).toMatchObject({ kind: 'explore' });
|
||||
}));
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { tools, toolSetOf } from '../src/tools';
|
||||
|
||||
let dir: string;
|
||||
let origCwd: string;
|
||||
|
||||
beforeEach(() => {
|
||||
origCwd = process.cwd();
|
||||
dir = mkdtempSync(join(tmpdir(), 'shiro-extra-'));
|
||||
process.chdir(dir);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
process.chdir(origCwd);
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
const run = <T>(name: string, input: T) =>
|
||||
Promise.resolve(
|
||||
(tools as Record<string, { execute?: (i: T, o: unknown) => unknown }>)[name]!.execute!(input, {
|
||||
toolCallId: 't1',
|
||||
messages: [],
|
||||
}),
|
||||
) as Promise<string>;
|
||||
|
||||
const write = (path: string, text: string) => Bun.write(path, text);
|
||||
|
||||
test('all 20 extra tools are registered in the extra set', () => {
|
||||
for (const n of [
|
||||
'insert_lines', 'delete_lines', 'replace_lines', 'append_file', 'prepend_file', 'count_lines',
|
||||
'tree', 'file_info', 'find_files', 'recent_files', 'changed_files',
|
||||
'git_log_file', 'git_diff_commits', 'git_show_file', 'git_current_branch', 'git_changed_in_ref',
|
||||
'outline', 'read_symbol', 'env_info', 'count_tokens',
|
||||
]) {
|
||||
expect(tools[n as keyof typeof tools], n).toBeDefined();
|
||||
expect(toolSetOf(n), n).toBe('extra');
|
||||
}
|
||||
});
|
||||
|
||||
// --- edit ---
|
||||
|
||||
test('insert_lines adds a block at a position, pushing the rest down', async () => {
|
||||
await write('a.txt', 'one\ntwo\nthree');
|
||||
await run('insert_lines', { path: 'a.txt', line: 2, text: 'inserted' });
|
||||
expect(await Bun.file('a.txt').text()).toBe('one\ninserted\ntwo\nthree');
|
||||
});
|
||||
|
||||
test('insert_lines refuses a position past the end', async () => {
|
||||
await write('a.txt', 'one\ntwo');
|
||||
await expect(run('insert_lines', { path: 'a.txt', line: 99, text: 'x' })).rejects.toThrow(/past the end/);
|
||||
});
|
||||
|
||||
test('delete_lines removes an inclusive range', async () => {
|
||||
await write('a.txt', 'one\ntwo\nthree\nfour');
|
||||
await run('delete_lines', { path: 'a.txt', start: 2, end: 3 });
|
||||
expect(await Bun.file('a.txt').text()).toBe('one\nfour');
|
||||
});
|
||||
|
||||
test('delete_lines refuses to delete the whole file', async () => {
|
||||
await write('a.txt', 'one\ntwo');
|
||||
await expect(run('delete_lines', { path: 'a.txt', start: 1, end: 2 })).rejects.toThrow(/delete_file/);
|
||||
});
|
||||
|
||||
test('replace_lines swaps a range for new text', async () => {
|
||||
await write('a.txt', 'one\ntwo\nthree');
|
||||
await run('replace_lines', { path: 'a.txt', start: 2, end: 2, text: 'TWO\nTWO2' });
|
||||
expect(await Bun.file('a.txt').text()).toBe('one\nTWO\nTWO2\nthree');
|
||||
});
|
||||
|
||||
test('append_file and prepend_file add at the ends', async () => {
|
||||
await write('a.txt', 'middle\n');
|
||||
await run('append_file', { path: 'a.txt', text: 'end' });
|
||||
await run('prepend_file', { path: 'a.txt', text: 'start' });
|
||||
expect(await Bun.file('a.txt').text()).toBe('start\nmiddle\nend\n');
|
||||
});
|
||||
|
||||
test('count_lines counts one file and a glob', async () => {
|
||||
await write('a.ts', '1\n2\n3');
|
||||
await Bun.write(join('sub', 'b.ts'), '1\n2');
|
||||
const one = await run('count_lines', { path: 'a.ts' });
|
||||
expect(one).toContain('3');
|
||||
expect(one).toContain('a.ts');
|
||||
const many = await run('count_lines', { pattern: '**/*.ts' });
|
||||
expect(many).toContain('a.ts');
|
||||
expect(many).toContain('b.ts');
|
||||
});
|
||||
|
||||
// --- inspect ---
|
||||
|
||||
test('tree shows an indented, directories-first shape', async () => {
|
||||
await Bun.write(join('src', 'app', 'index.ts'), 'x');
|
||||
await Bun.write(join('src', 'util.ts'), 'x');
|
||||
const out = await run('tree', { path: 'src', depth: 3 });
|
||||
expect(out).toContain('app/');
|
||||
expect(out).toContain('index.ts');
|
||||
expect(out).toContain('util.ts');
|
||||
});
|
||||
|
||||
test('file_info reports size, lines, kind, and mtime', async () => {
|
||||
await write('a.txt', 'one\ntwo');
|
||||
const out = await run('file_info', { path: 'a.txt' });
|
||||
expect(out).toContain('2 lines');
|
||||
expect(out).toContain('text');
|
||||
expect(out).toContain('bytes');
|
||||
});
|
||||
|
||||
test('find_files matches a filename substring', async () => {
|
||||
await Bun.write(join('src', 'auth.ts'), 'x');
|
||||
await Bun.write(join('src', 'auth.test.ts'), 'x');
|
||||
await Bun.write(join('src', 'other.ts'), 'x');
|
||||
const out = await run('find_files', { name: 'auth' });
|
||||
expect(out).toContain('auth.ts');
|
||||
expect(out).toContain('auth.test.ts');
|
||||
expect(out).not.toContain('other.ts');
|
||||
});
|
||||
|
||||
test('recent_files lists newest first', async () => {
|
||||
await write('old.txt', 'x');
|
||||
await new Promise((r) => setTimeout(r, 20));
|
||||
await write('new.txt', 'x');
|
||||
const out = await run('recent_files', { limit: 2 });
|
||||
expect(out.indexOf('new.txt')).toBeLessThan(out.indexOf('old.txt'));
|
||||
});
|
||||
|
||||
test('changed_files reports the working-tree delta in a repo, or refuses cleanly outside one', async () => {
|
||||
try {
|
||||
const out = await run('changed_files', {});
|
||||
expect(typeof out).toBe('string');
|
||||
} catch (e) {
|
||||
// A temp dir is not a repository, so the tool must refuse with a clear message.
|
||||
expect(String(e)).toMatch(/not a git repository|not installed/i);
|
||||
}
|
||||
});
|
||||
|
||||
// --- code ---
|
||||
|
||||
test('outline lists top-level declarations', async () => {
|
||||
await write('m.ts', 'import x from "y";\nexport function build() {}\nclass Thing {}\nconst helper = () => {};\n');
|
||||
const out = await run('outline', { path: 'm.ts' });
|
||||
expect(out).toContain('build');
|
||||
expect(out).toContain('Thing');
|
||||
expect(out).toContain('helper');
|
||||
expect(out).not.toContain('import x');
|
||||
});
|
||||
|
||||
test('read_symbol extracts one definition body', async () => {
|
||||
await write('m.ts', 'function alpha() {\n return 1;\n}\n\nfunction beta() {\n return 2;\n}\n');
|
||||
const out = await run('read_symbol', { path: 'm.ts', name: 'alpha' });
|
||||
expect(out).toContain('function alpha');
|
||||
expect(out).toContain('return 1;');
|
||||
expect(out).not.toContain('function beta');
|
||||
});
|
||||
|
||||
test('read_symbol errors plainly on a miss', async () => {
|
||||
await write('m.ts', 'const x = 1;\n');
|
||||
await expect(run('read_symbol', { path: 'm.ts', name: 'nope' })).rejects.toThrow(/No definition/);
|
||||
});
|
||||
|
||||
test('env_info reports the platform and cwd', async () => {
|
||||
const out = await run('env_info', {});
|
||||
expect(out).toContain('platform:');
|
||||
expect(out).toContain('cwd:');
|
||||
});
|
||||
|
||||
test('count_tokens estimates a file and a string', async () => {
|
||||
await write('a.txt', 'x'.repeat(400));
|
||||
const fileOut = await run('count_tokens', { path: 'a.txt' });
|
||||
expect(fileOut).toContain('~100 tokens');
|
||||
const strOut = await run('count_tokens', { text: 'abcd'.repeat(25) });
|
||||
expect(strOut).toContain('~25 tokens');
|
||||
});
|
||||
@@ -0,0 +1,96 @@
|
||||
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { findSymbolTool, jsonQueryTool, tools, toolSetOf } from '../src/tools';
|
||||
|
||||
let dir: string;
|
||||
let origCwd: string;
|
||||
|
||||
beforeEach(() => {
|
||||
origCwd = process.cwd();
|
||||
dir = mkdtempSync(join(tmpdir(), 'shiro-nav-'));
|
||||
process.chdir(dir);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
process.chdir(origCwd);
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
const run = <T>(t: { execute?: (input: T, opts: any) => unknown }, input: T) =>
|
||||
Promise.resolve(t.execute!(input, { toolCallId: 't1', messages: [] })) as Promise<string>;
|
||||
|
||||
test('the new tools are registered and assigned to the nav set', () => {
|
||||
expect(tools['find_symbol']).toBeDefined();
|
||||
expect(tools['json_query']).toBeDefined();
|
||||
expect(toolSetOf('find_symbol')).toBe('nav');
|
||||
expect(toolSetOf('json_query')).toBe('nav');
|
||||
});
|
||||
|
||||
test('find_symbol jumps to a function declaration, not its uses', async () => {
|
||||
await Bun.write(
|
||||
'src/util.ts',
|
||||
'export function parseConfig() {\n return {};\n}\n\nconst x = parseConfig();\nparseConfig();\n',
|
||||
);
|
||||
const out = await run(findSymbolTool, { name: 'parseConfig' });
|
||||
// Exactly one hit, at the declaration on line 1, not the two calls below it.
|
||||
expect(out).toContain('src/util.ts:1');
|
||||
expect(out).toContain('function parseConfig');
|
||||
expect(out.split('\n')).toHaveLength(1);
|
||||
});
|
||||
|
||||
test('find_symbol finds a Python def and a class', async () => {
|
||||
await Bun.write('app.py', 'class Server:\n pass\n\ndef serve():\n pass\n');
|
||||
const cls = await run(findSymbolTool, { name: 'Server' });
|
||||
expect(cls).toContain('app.py:1');
|
||||
const fn = await run(findSymbolTool, { name: 'serve' });
|
||||
expect(fn).toContain('app.py:4');
|
||||
});
|
||||
|
||||
test('find_symbol reports a miss plainly and escapes regex metacharacters', async () => {
|
||||
await Bun.write('a.ts', 'const other = 1;\n');
|
||||
expect(await run(findSymbolTool, { name: 'missing' })).toBe('No definition of "missing" found.');
|
||||
// A name with regex metacharacters must not throw or match wildly.
|
||||
expect(await run(findSymbolTool, { name: 'a.b(c)' })).toContain('No definition');
|
||||
});
|
||||
|
||||
test('find_symbol skips comments', async () => {
|
||||
await Bun.write('c.ts', '// function fake() {}\nfunction real() {}\n');
|
||||
const out = await run(findSymbolTool, { name: 'fake' });
|
||||
expect(out).toContain('No definition');
|
||||
});
|
||||
|
||||
test('json_query reads a nested value by dotted path', async () => {
|
||||
await Bun.write('package.json', JSON.stringify({ scripts: { build: 'tsc', test: 'vitest' }, name: 'demo' }));
|
||||
expect(await run(jsonQueryTool, { path: 'package.json', query: 'scripts.build' })).toBe('scripts.build = tsc');
|
||||
});
|
||||
|
||||
test('json_query walks arrays with numeric segments', async () => {
|
||||
await Bun.write('data.json', JSON.stringify({ items: [{ id: 'a' }, { id: 'b' }] }));
|
||||
expect(await run(jsonQueryTool, { path: 'data.json', query: 'items.1.id' })).toBe('items.1.id = b');
|
||||
});
|
||||
|
||||
test('json_query renders an object value as JSON', async () => {
|
||||
await Bun.write('c.json', JSON.stringify({ deps: { react: '^19' } }));
|
||||
const out = await run(jsonQueryTool, { path: 'c.json', query: 'deps' });
|
||||
expect(out).toContain('react');
|
||||
expect(out).toContain('^19');
|
||||
});
|
||||
|
||||
test('json_query names the keys present when a segment misses', async () => {
|
||||
await Bun.write('c.json', JSON.stringify({ scripts: { build: 'x' } }));
|
||||
const out = run(jsonQueryTool, { path: 'c.json', query: 'scripts.deploy' });
|
||||
await expect(out).rejects.toThrow(/no key "deploy"/);
|
||||
await expect(out).rejects.toThrow(/build/);
|
||||
});
|
||||
|
||||
test('json_query refuses a missing file and invalid JSON', async () => {
|
||||
await expect(run(jsonQueryTool, { path: 'nope.json', query: 'a' })).rejects.toThrow(/No such file/);
|
||||
await Bun.write('bad.json', '{ not json');
|
||||
await expect(run(jsonQueryTool, { path: 'bad.json', query: 'a' })).rejects.toThrow(/not valid JSON/);
|
||||
});
|
||||
|
||||
test('json_query stays inside the workspace', async () => {
|
||||
await expect(run(jsonQueryTool, { path: '../../etc/passwd', query: 'a' })).rejects.toThrow();
|
||||
});
|
||||
@@ -129,7 +129,7 @@ test('the status bar warns before compaction rather than after', () => {
|
||||
const fine = render(
|
||||
<StatusBar model="m" agent="default" thinking="medium" contextTokens={10} contextLimit={100} cost="$0" toolCount={1} />,
|
||||
);
|
||||
expect(fine.lastFrame()).toContain('10% ctx');
|
||||
expect(fine.lastFrame()).toContain('10%');
|
||||
fine.unmount();
|
||||
|
||||
// At 90% the next turn may lose history, so the bar says so in words rather
|
||||
@@ -137,7 +137,7 @@ test('the status bar warns before compaction rather than after', () => {
|
||||
const late = render(
|
||||
<StatusBar model="m" agent="default" thinking="medium" contextTokens={95} contextLimit={100} cost="$0" toolCount={1} />,
|
||||
);
|
||||
expect(late.lastFrame()).toContain('95% ctx');
|
||||
expect(late.lastFrame()).toContain('95%');
|
||||
expect(late.lastFrame()).toContain('compacting soon');
|
||||
late.unmount();
|
||||
});
|
||||
|
||||
@@ -137,7 +137,7 @@ test('a context limit turns the raw token count into a percentage', () => {
|
||||
/>,
|
||||
);
|
||||
const frame = app.lastFrame() ?? '';
|
||||
expect(frame).toContain('50% ctx');
|
||||
expect(frame).toContain('50%');
|
||||
expect(frame).not.toContain('60000');
|
||||
app.unmount();
|
||||
});
|
||||
@@ -154,7 +154,7 @@ test('the context percentage is capped at 100 rather than running over', () => {
|
||||
toolCount={1}
|
||||
/>,
|
||||
);
|
||||
expect(app.lastFrame()).toContain('100% ctx');
|
||||
expect(app.lastFrame()).toContain('100%');
|
||||
app.unmount();
|
||||
});
|
||||
|
||||
|
||||
+2
-5
@@ -7,12 +7,9 @@ import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { Session } from '../src/session';
|
||||
import { App, createApprovalBridge } from '../src/ui/App';
|
||||
import { testHooks } from './helpers';
|
||||
import { testHooks, usageOf } from './helpers';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 12, noCache: 12, cacheRead: 0, cacheWrite: 0 },
|
||||
outputTokens: { total: 7 },
|
||||
} as any;
|
||||
const usage = usageOf(12, 7);
|
||||
|
||||
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
|
||||
Reference in New Issue
Block a user