Add batch reads, @file completion, interruptible commands, tool sets

Tools, six built-in to fourteen:
- read_many_files: up to 20 paths read concurrently, each with its own window.
  An unreadable path is reported in its own block instead of throwing.
- multi_edit: several edits to one file, validated in memory first so a late
  failure cannot leave the file half-written.
- list_dir: ignore-aware depth-limited tree.
- git_status/diff/log/show/blame: read-only, spawned with a fixed argv rather
  than a shell string, which is what makes them safe to auto-approve.

toolSets gates them. core is always on; edit-plus and git are optional. A
disabled set reaches neither the wire nor the system prompt, since a prompt
naming an absent tool teaches calls that cannot succeed.

Interface:
- Reasoning streams to a collapsed panel, ctrl-r expands, dropped when the turn
  ends: it is progress, not the answer.
- The tool in flight is named from tool-input-start, before its arguments finish
  streaming, and cleared on its result.
- Prompts typed mid-turn queue and drain in order. esc clears the queue as well
  as aborting.
- @ opens a path picker fed by the ignore-aware walker. Prefix matches rank
  above substring matches, so @src/ means "under src/". The walk runs on the
  first @, not at startup.

ctrl-c kills the command in flight and keeps the turn. The call throws rather
than returning, so the model cannot read a killed command as one that ran and
failed on its own terms. The kill takes the whole process tree: killing cmd /c
alone left the real command holding both pipes open, so the read never returned
and the interrupt did nothing for 19 seconds.

Two pruning fixes:
- A tool result whose tool call was pruned is now dropped with it. Pruning
  counts messages, so the cut landed between an assistant tool-call and the tool
  message answering it, producing 400 "No tool call found for function call
  output with call_id ...". The reverse pairing is left alone: a call awaiting
  its result is what a suspended approval looks like.
- ignore.ts called statFs without importing it, so walk() crashed on the first
  symlink.

482 tests, up from 404. Docs synced across README, ROADMAP, TODO, and all of
docs/: tool sets, the new tools, ctrl-c semantics, the tool-start event, and the
two hand-maintained tool-name lists recorded as a known weakness.
This commit is contained in:
Muhammad Zakir Ramadhan
2026-09-03 01:37:48 +07:00
parent a5ace7a23f
commit 2fa6ee247b
36 changed files with 2541 additions and 215 deletions
+222
View File
@@ -0,0 +1,222 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { Session } from '../src/session';
import { App, createApprovalBridge } from '../src/ui/App';
import { testHooks } from './helpers';
const usage = { inputTokens: { total: 3, noCache: 3, cacheRead: 0, cacheWrite: 0 }, outputTokens: { total: 1 } } as any;
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
});
const paths = ['README.md', 'src/app.ts', 'src/session.ts', 'src/ui/App.tsx', 'test/session.test.ts'];
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
const DOWN = '\u001B[B';
const TAB = '\t';
function mount(listPaths = async () => paths) {
const sent: string[] = [];
const bridge = createApprovalBridge();
const session = new Session({ model, askApproval: bridge.ask });
const app = render(
<App
session={session}
bridge={bridge}
header="hdr"
hooks={testHooks({ listPaths, recordPrompt: (t) => sent.push(t) })}
/>,
);
return { app, sent };
}
async function type(app: ReturnType<typeof render>, s: string) {
for (const ch of s) {
app.stdin.write(ch);
await wait(30);
}
}
test('@ opens the file picker and typing narrows it', async () => {
const { app } = mount();
await wait(150);
await type(app, '@');
await wait(200);
expect(app.lastFrame()).toContain('README.md');
await type(app, 'src/ui/');
await wait(200);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('src/ui/App.tsx');
expect(frame).not.toContain('README.md');
expect(frame).not.toContain('test/session.test.ts');
app.unmount();
}, 25_000);
test('tab inserts the highlighted path with no @', async () => {
const { app } = mount();
await wait(150);
await type(app, 'explain @src/ses');
await wait(200);
expect(app.lastFrame()).toContain('src/session.ts');
app.stdin.write(TAB);
await wait(250);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('explain src/session.ts');
expect(frame).not.toContain('@src');
expect(frame).not.toContain('up/down move | tab or enter insert');
app.unmount();
}, 25_000);
test('down then tab inserts the second match', async () => {
const { app } = mount();
await wait(150);
await type(app, '@src/');
await wait(200);
app.stdin.write(DOWN);
await wait(150);
app.stdin.write(TAB);
await wait(250);
expect(app.lastFrame()).toContain('src/session.ts');
app.unmount();
}, 25_000);
test('a completed prompt submits as a plain path', async () => {
const { app, sent } = mount();
await wait(150);
await type(app, '@src/app');
await wait(200);
app.stdin.write(TAB);
await wait(250);
app.stdin.write('\r');
await wait(400);
expect(sent).toHaveLength(1);
expect(sent[0]).toBe('src/app.ts');
app.unmount();
}, 25_000);
test('esc dismisses the picker and leaves the text alone', async () => {
const { app } = mount();
await wait(150);
await type(app, '@src/');
await wait(200);
expect(app.lastFrame()).toContain('src/app.ts');
app.stdin.write('\u001B');
await wait(250);
const frame = app.lastFrame() ?? '';
expect(frame).not.toContain('tab or enter insert');
expect(frame).toContain('@src/');
app.unmount();
}, 25_000);
test('a query matching nothing says so instead of showing a stale list', async () => {
const { app } = mount();
await wait(150);
await type(app, '@zzzz');
await wait(250);
expect(app.lastFrame()).toContain('no file matches zzzz');
app.unmount();
}, 25_000);
test('an @ inside a word is not a completion', async () => {
const { app } = mount();
await wait(150);
await type(app, 'mail me@example');
await wait(250);
const frame = app.lastFrame() ?? '';
expect(frame).not.toContain('tab or enter insert');
expect(frame).not.toContain('no file matches');
app.unmount();
}, 25_000);
test('the walk is reported as loading and only runs once', async () => {
let calls = 0;
const { app } = mount(async () => {
calls++;
await wait(400);
return paths;
});
await wait(150);
await type(app, '@');
await wait(80);
expect(app.lastFrame()).toContain('indexing files');
await wait(600);
expect(app.lastFrame()).toContain('README.md');
// Dismiss, reopen: the list is cached rather than walked again.
app.stdin.write('\u001B');
await wait(150);
await type(app, ' @src/');
await wait(300);
expect(app.lastFrame()).toContain('src/app.ts');
expect(calls).toBe(1);
app.unmount();
}, 25_000);
test('the command menu still works and does not fight the file picker', async () => {
const { app } = mount();
await wait(150);
await type(app, '/mo');
await wait(200);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('models');
expect(frame).not.toContain('tab or enter insert');
app.unmount();
}, 25_000);
test('ctrl-c with nothing running does not kill a command that is not there', async () => {
const { app } = mount();
await wait(150);
app.stdin.write('\u0003');
await wait(250);
expect(app.lastFrame() ?? '').not.toContain('interrupted:');
app.unmount();
}, 25_000);
+89
View File
@@ -0,0 +1,89 @@
import { expect, test } from 'bun:test';
import { completePath, matchPaths, pathToken } from '../src/complete';
const paths = [
'README.md',
'src/app.ts',
'src/session.ts',
'src/ui/App.tsx',
'src/ui/Panels.tsx',
'test/session.test.ts',
'vendor/src/legacy.ts',
];
test('a bare @ opens the token with an empty query', () => {
expect(pathToken('@', 1)).toEqual({ start: 0, end: 1, query: '' });
});
test('the token is the text between @ and the cursor', () => {
expect(pathToken('look at @src/ses', 16)).toEqual({ start: 8, end: 16, query: 'src/ses' });
});
test('@ mid-word is not a completion, so an email is left alone', () => {
expect(pathToken('mail me@example.com', 19)).toBeUndefined();
expect(pathToken('user@host', 9)).toBeUndefined();
});
test('a space ends the token', () => {
expect(pathToken('@src/app.ts and then', 20)).toBeUndefined();
});
test('the cursor before the @ sees no token', () => {
expect(pathToken('@src', 0)).toBeUndefined();
});
test('the nearest @ wins when there are two', () => {
const token = pathToken('@first then @sec', 16);
expect(token?.query).toBe('sec');
expect(token?.start).toBe(12);
});
test('an empty query offers the shallowest paths first', () => {
expect(matchPaths(paths, '', 3)).toEqual(['README.md', 'src/app.ts', 'src/session.ts']);
});
test('a directory prefix narrows to what is under it', () => {
const hits = matchPaths(paths, 'src/ui/');
expect(hits).toEqual(['src/ui/App.tsx', 'src/ui/Panels.tsx']);
});
test('prefix matches rank above substring matches', () => {
const hits = matchPaths(paths, 'src/');
expect(hits[0]).toBe('src/app.ts');
// vendor/src/legacy.ts contains "src/" but is not under src/, so it comes last.
expect(hits.at(-1)).toBe('vendor/src/legacy.ts');
expect(hits.indexOf('src/session.ts')).toBeLessThan(hits.indexOf('vendor/src/legacy.ts'));
});
test('matching ignores case', () => {
expect(matchPaths(paths, 'readme')).toEqual(['README.md']);
});
test('a query matching nothing yields nothing', () => {
expect(matchPaths(paths, 'zzz')).toEqual([]);
});
test('the limit is honoured', () => {
expect(matchPaths(paths, 's', 2)).toHaveLength(2);
});
test('completion replaces the token with a plain relative path', () => {
const value = 'look at @src/ses';
const token = pathToken(value, value.length)!;
expect(completePath(value, token, 'src/session.ts')).toEqual({
value: 'look at src/session.ts ',
cursor: 23,
});
});
test('completion keeps whatever followed the cursor', () => {
const value = '@src/ses and explain';
const token = pathToken(value, 8)!;
const { value: next } = completePath(value, token, 'src/session.ts');
expect(next).toBe('src/session.ts and explain');
});
test('the inserted path carries no @', () => {
const token = pathToken('@READ', 5)!;
expect(completePath('@READ', token, 'README.md').value).not.toContain('@');
});
+1
View File
@@ -20,6 +20,7 @@ export function testHooks(over: Partial<AppHooks> = {}): AppHooks {
resumeSession: async () => 'resumed',
saveSession: async () => 'saved',
instructionFiles: () => [],
listPaths: async () => [],
initPrompt: 'write AGENTS.md',
history: [],
recordPrompt: () => {},
+97 -1
View File
@@ -1,6 +1,6 @@
import { expect, test } from 'bun:test';
import type { ModelMessage } from 'ai';
import { dropOrphanedItems, prunePreservingItems } from '../src/prune';
import { dropOrphanedItems, dropOrphanedResults, prunePreservingItems } from '../src/prune';
const kinds = (messages: ModelMessage[]) =>
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
@@ -152,3 +152,99 @@ test('a provider other than openai is handled the same way', () => {
];
expect(dropOrphanedItems(before, after)).toEqual([]);
});
/** The assistant tool-call plus the tool message answering it, as one exchange. */
const callAndResult = (call: string, rs?: string): ModelMessage[] => [
{
role: 'assistant',
content: [
...(rs ? [{ type: 'reasoning' as const, text: 'deciding', providerOptions: { openai: { itemId: rs } } }] : []),
{ type: 'tool-call', toolCallId: call, toolName: 'grep', input: { pattern: 'x' } },
],
},
{
role: 'tool',
content: [{ type: 'tool-result', toolCallId: call, toolName: 'grep', output: { type: 'text', value: 'hit' } }],
},
];
test('a tool result left without its tool call is dropped', () => {
const [, resultMessage] = callAndResult('call_A');
const cleaned = dropOrphanedResults([{ role: 'user', content: 'q' }, resultMessage!]);
expect(JSON.stringify(cleaned)).not.toContain('call_A');
expect(kinds(cleaned)).toEqual(['user']);
});
test('a tool result keeps its place while the call is still there', () => {
const messages: ModelMessage[] = [{ role: 'user', content: 'q' }, ...callAndResult('call_A')];
expect(dropOrphanedResults(messages)).toEqual(messages);
});
test('a tool call awaiting its result survives, since that is a suspended approval', () => {
const [callMessage] = callAndResult('call_A');
const messages: ModelMessage[] = [{ role: 'user', content: 'q' }, callMessage!];
expect(dropOrphanedResults(messages)).toEqual(messages);
});
test('a tool-error is treated as a result and dropped with its call', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'q' },
{
role: 'tool',
content: [{ type: 'tool-error', toolCallId: 'call_A', toolName: 'grep', error: 'boom' } as never],
},
];
expect(JSON.stringify(dropOrphanedResults(messages))).not.toContain('call_A');
});
test('only the orphaned result is dropped, not a healthy one beside it', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'q' },
...callAndResult('call_LIVE'),
{
role: 'tool',
content: [
{ type: 'tool-result', toolCallId: 'call_LIVE', toolName: 'grep', output: { type: 'text', value: 'a' } },
{ type: 'tool-result', toolCallId: 'call_GONE', toolName: 'grep', output: { type: 'text', value: 'b' } },
],
},
];
const json = JSON.stringify(dropOrphanedResults(messages));
expect(json).toContain('call_LIVE');
expect(json).not.toContain('call_GONE');
});
/**
* The 400 this guards against: "No tool call found for function call output with
* call_id ...". Pruning counts messages, so its cut lands between the assistant
* tool-call and the tool message answering it, stranding the result on the wire.
*/
test('prunePreservingItems never strands a tool result on the wire', () => {
const messages: ModelMessage[] = [{ role: 'user', content: `q ${'x'.repeat(4000)}` }];
for (let i = 0; i < 5; i++) {
messages.push(...callAndResult(`call_${i}`, `rs_${i}`));
messages.push({ role: 'user', content: `follow up ${i} ${'y'.repeat(4000)}` });
}
const pruned = prunePreservingItems({
messages,
reasoning: 'all',
toolCalls: 'before-last-3-messages',
emptyMessages: 'remove',
});
const calls = new Set<string>();
for (const m of pruned) {
if (!Array.isArray(m.content)) continue;
for (const p of m.content as { type: string; toolCallId?: string }[]) {
if (p.type === 'tool-call' && p.toolCallId) calls.add(p.toolCallId);
}
}
for (const m of pruned) {
if (!Array.isArray(m.content)) continue;
for (const p of m.content as { type: string; toolCallId?: string }[]) {
if (p.type === 'tool-result' || p.type === 'tool-error') expect(calls.has(p.toolCallId!)).toBe(true);
}
}
});
+207
View File
@@ -0,0 +1,207 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import type { LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { Session } from '../src/session';
import { App, createApprovalBridge } from '../src/ui/App';
import { testHooks } from './helpers';
const usage = {
inputTokens: { total: 4, noCache: 4, cacheRead: 0, cacheWrite: 0 },
outputTokens: { total: 2 },
} as any;
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
/** One assistant reply, delivered slowly enough to type during. */
const slowReply = (body: string): LanguageModelV4StreamPart[] => [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
function mount(chunkDelayInMs: number) {
const prompts: string[] = [];
const model = new MockLanguageModelV4({
doStream: async (opts) => {
const last = opts.prompt.at(-1);
const content = last?.content;
prompts.push(typeof content === 'string' ? content : JSON.stringify(content));
return {
stream: simulateReadableStream({
chunks: slowReply(`reply ${prompts.length}`),
chunkDelayInMs,
initialDelayInMs: null,
}),
};
},
});
const bridge = createApprovalBridge();
const session = new Session({ model, askApproval: bridge.ask });
const app = render(<App session={session} bridge={bridge} header="hdr" hooks={testHooks()} />);
return { app, prompts, session };
}
async function type(app: ReturnType<typeof render>, s: string) {
for (const ch of s) {
app.stdin.write(ch);
await wait(25);
}
app.stdin.write('\r');
await wait(60);
}
test('the input stays live while a turn runs, and a submission is queued', async () => {
const { app, prompts } = mount(600);
await wait(150);
await type(app, 'first');
await wait(200);
// Mid-turn: the spinner and the input coexist rather than swapping. The
// placeholder's first character is inverted for the cursor, hence the offset.
const midTurn = app.lastFrame() ?? '';
expect(midTurn).toContain('working...');
expect(midTurn).toContain('ype to queue');
await type(app, 'second');
await wait(150);
expect(app.lastFrame()).toContain('queued: 1');
expect(prompts).toHaveLength(1);
app.unmount();
}, 20_000);
test('two prompts typed during a turn run in order afterwards', async () => {
const { app, prompts } = mount(400);
await wait(150);
await type(app, 'first');
await wait(120);
await type(app, 'second');
await type(app, 'third');
expect(app.lastFrame()).toContain('queued: 2');
await wait(3000);
expect(prompts).toHaveLength(3);
expect(prompts[0]).toContain('first');
expect(prompts[1]).toContain('second');
expect(prompts[2]).toContain('third');
expect(app.lastFrame()).not.toContain('queued:');
app.unmount();
}, 25_000);
test('esc clears the queue as well as aborting the turn', async () => {
const { app, prompts } = mount(800);
await wait(150);
await type(app, 'first');
await wait(150);
await type(app, 'queued one');
await type(app, 'queued two');
expect(app.lastFrame()).toContain('queued: 2');
app.stdin.write('\u001B');
await wait(1200);
expect(app.lastFrame()).not.toContain('queued:');
expect(prompts).toHaveLength(1);
app.unmount();
}, 25_000);
test('reasoning shows as a collapsed line before any text arrives, then leaves with the turn', async () => {
const model = new MockLanguageModelV4({
doStream: async () => ({
stream: simulateReadableStream({
chunks: [
{ type: 'reasoning-start', id: 'r' },
{ type: 'reasoning-delta', id: 'r', delta: 'weighing the options at some length' },
{ type: 'reasoning-end', id: 'r' },
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'the answer' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
] as LanguageModelV4StreamPart[],
chunkDelayInMs: 250,
initialDelayInMs: null,
}),
}),
});
const bridge = createApprovalBridge();
const session = new Session({ model, askApproval: bridge.ask });
const app = render(<App session={session} bridge={bridge} header="hdr" hooks={testHooks()} />);
await wait(150);
await type(app, 'think about it');
await wait(700);
const thinking = app.lastFrame() ?? '';
expect(thinking).toContain('thinking...');
expect(thinking).toContain('tokens');
expect(thinking).not.toContain('weighing the options');
await wait(2500);
const done = app.lastFrame() ?? '';
expect(done).toContain('the answer');
expect(done).not.toContain('thinking');
app.unmount();
}, 25_000);
test('the tool in flight is named on screen and cleared when it returns', async () => {
const orig = process.cwd();
let n = 0;
const model = new MockLanguageModelV4({
doStream: async () => {
const chunks: LanguageModelV4StreamPart[] =
n++ === 0
? [
{ type: 'tool-input-start', id: 'c1', toolName: 'read_file' },
{ type: 'tool-input-end', id: 'c1' },
{
type: 'tool-call',
toolCallId: 'c1',
toolName: 'read_file',
input: JSON.stringify({ path: 'src/session.ts' }),
},
{ type: 'finish', finishReason: { unified: 'tool-calls', raw: 'tool_use' }, usage },
]
: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'read it' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
return { stream: simulateReadableStream({ chunks, chunkDelayInMs: 200, initialDelayInMs: null }) };
},
});
try {
const bridge = createApprovalBridge();
const session = new Session({ model, askApproval: bridge.ask });
const app = render(<App session={session} bridge={bridge} header="hdr" hooks={testHooks()} />);
await wait(150);
await type(app, 'read the session file');
await wait(500);
expect(app.lastFrame()).toContain('read_file');
await wait(2500);
expect(app.lastFrame()).toContain('read it');
app.unmount();
} finally {
process.chdir(orig);
}
}, 25_000);
+70
View File
@@ -7,6 +7,8 @@ import { createHost } from '../src/plugins';
import { guardPlugin, timePlugin } from '../src/plugins-builtin';
import { Session } from '../src/session';
import { loadSkills } from '../src/skills';
import { MUTATING_TOOLS, TOOL_SETS, TOOL_SET_NAMES, isToolSetName, toolSetOf } from '../src/tools';
import { GIT_TOOL_NAMES } from '../src/tools-git';
const usage = {
inputTokens: { total: 5, noCache: 5, cacheRead: 0, cacheWrite: 0 },
@@ -226,3 +228,71 @@ test('afterTurn fires once the turn ends', async () => {
for await (const _ of session.send('hi')) void _;
expect(fired).toBe(1);
});
test('the git tools are offered by default and never prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny' });
for await (const _ of session.send('what changed')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
for (const name of GIT_TOOL_NAMES) expect(offered).toContain(name);
for (const name of GIT_TOOL_NAMES) expect(MUTATING_TOOLS as readonly string[]).not.toContain(name);
});
test('a disabled tool set reaches neither the wire nor the prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: [] });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('read_file');
expect(offered).toContain('bash');
expect(offered).not.toContain('git_status');
expect(offered).not.toContain('multi_edit');
expect(offered).not.toContain('list_dir');
const system = JSON.stringify(seen[0]?.prompt.find((m) => m.role === 'system'));
expect(system).not.toContain('git_status');
expect(system).not.toContain('multi_edit');
});
test('an enabled set is offered while the others stay withheld', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: ['git'] });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('git_diff');
expect(offered).not.toContain('multi_edit');
});
test('core is never withheld, whatever the config says', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: ['git'] });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
for (const name of TOOL_SETS.core) expect(offered).toContain(name);
});
test('session tools survive tool-set gating, since they are not part of that budget', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: [], memory: new Memory('/repo-test') });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('todo_write');
expect(offered).toContain('remember');
});
test('toolSetOf names the set a tool came from, and nothing for a session tool', () => {
expect(toolSetOf('read_file')).toBe('core');
expect(toolSetOf('multi_edit')).toBe('edit-plus');
expect(toolSetOf('git_log')).toBe('git');
expect(toolSetOf('todo_write')).toBeUndefined();
});
test('isToolSetName accepts the real sets only, so a typo in config is ignored', () => {
for (const name of TOOL_SET_NAMES) expect(isToolSetName(name)).toBe(true);
expect(isToolSetName('gti')).toBe(false);
});
+41 -4
View File
@@ -7,6 +7,7 @@ import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { Session } from '../src/session';
import { interruptBash } from '../src/tools';
const usage = {
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
@@ -62,7 +63,7 @@ test('read-only tool runs without approval and the loop terminates', async () =>
const kinds: string[] = [];
for await (const ev of session.send('read note.txt')) kinds.push(ev.type);
expect(kinds).toEqual(['tool-call', 'tool-result', 'text', 'done']);
expect(kinds).toEqual(['tool-start', 'tool-call', 'tool-result', 'text', 'done']);
expect(call).toBe(2);
}));
@@ -264,7 +265,7 @@ test('a read-only built-in stays free even when mcp tools are present', async ()
const kinds: string[] = [];
for await (const ev of session.send('read note.txt')) kinds.push(ev.type);
expect(kinds).toEqual(['tool-call', 'tool-result', 'text', 'done']);
expect(kinds).toEqual(['tool-start', 'tool-call', 'tool-result', 'text', 'done']);
}));
test('setModel swaps the model used by the next turn', async () => {
@@ -302,8 +303,7 @@ test('reset clears history and token counters; replace swaps history in', async
expect(session.messages).toEqual([{ role: 'user', content: 'restored' }]);
});
test('abort mid-stream ends the turn with done, keeping the text already delivered', async () => {
const session = new Session({
test('abort mid-stream ends the turn with done, keeping the text already delivered', async () => { const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => ({
stream: simulateReadableStream({
@@ -338,3 +338,40 @@ test('abort mid-stream ends the turn with done, keeping the text already deliver
expect(kinds.at(-1)).toBe('done');
expect(kinds).not.toContain('error');
}, 15_000);
test('an interrupted command becomes a tool error and the turn carries on', async () =>
inTempDir(async () => {
const sleeper = process.platform === 'win32' ? 'ping -n 20 127.0.0.1 > nul' : 'sleep 20';
let call = 0;
const session = new Session({
yolo: true,
model: new MockLanguageModelV4({
doStream: async () =>
stream(call++ === 0 ? toolCall('c1', 'bash', { command: sleeper }) : text('I stopped there.')),
}),
askApproval: async () => {
throw new Error('yolo must not ask');
},
});
const kinds: string[] = [];
let toolError = '';
const turn = (async () => {
for await (const ev of session.send('run the long thing')) {
kinds.push(ev.type);
if (ev.type === 'tool-error') toolError = String((ev.error as Error).message ?? ev.error);
}
})();
// Interrupt once the command is actually running.
await Bun.sleep(700);
expect(interruptBash()).toEqual([sleeper]);
await turn;
expect(kinds).toContain('tool-error');
expect(toolError).toMatch(/user interrupted this command/i);
// The turn survived: the model was asked again and its reply arrived.
expect(kinds).toContain('text');
expect(kinds.at(-1)).toBe('done');
expect(call).toBe(2);
}), 30_000);
+1 -1
View File
@@ -73,7 +73,7 @@ test('subagent greps the workspace and returns text to the parent', async () =>
const events: string[] = [];
for await (const ev of session.send('where is login defined?')) events.push(ev.type);
expect(events).toEqual(['tool-call', 'tool-result', 'text', 'done']);
expect(events).toEqual(['tool-start', 'tool-call', 'tool-result', 'text', 'done']);
// The subagent gets only read tools, so it can never trigger an approval prompt.
const subagentTools = (seen[1]?.tools ?? []).map((t) => t.name).sort();
+150
View File
@@ -0,0 +1,150 @@
import { afterEach, beforeEach, expect, test } from 'bun:test';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import {
GIT_TOOL_NAMES,
gitBlameTool,
gitDiffTool,
gitLogTool,
gitShowTool,
gitStatusTool,
gitTools,
} from '../src/tools-git';
let dir: string;
let origCwd: string;
const run = <T>(t: { execute?: (input: T, opts: any) => unknown }, input: T) =>
Promise.resolve(t.execute!(input, { toolCallId: 't1', messages: [] })) as Promise<string>;
async function git(...args: string[]): Promise<void> {
const proc = Bun.spawn(['git', ...args], { cwd: dir, stdout: 'pipe', stderr: 'pipe' });
const code = await proc.exited;
if (code !== 0) throw new Error(`git ${args.join(' ')} failed: ${await new Response(proc.stderr).text()}`);
}
beforeEach(() => {
origCwd = process.cwd();
dir = mkdtempSync(join(tmpdir(), 'shiro-git-'));
process.chdir(dir);
});
afterEach(() => {
process.chdir(origCwd);
rmSync(dir, { recursive: true, force: true });
});
async function repoWithOneCommit(): Promise<void> {
await git('init', '-b', 'main');
await git('config', 'user.email', 'test@example.com');
await git('config', 'user.name', 'Test');
await Bun.write(join(dir, 'app.ts'), 'export const port = 8080;\n');
await git('add', '.');
await git('commit', '-m', 'add the server port');
}
test('every git tool is registered and named consistently', () => {
expect(GIT_TOOL_NAMES.sort()).toEqual(['git_blame', 'git_diff', 'git_log', 'git_show', 'git_status']);
expect(Object.keys(gitTools).sort()).toEqual(GIT_TOOL_NAMES.sort());
});
test('git_status names the branch and describes each change', async () => {
await repoWithOneCommit();
expect(await run(gitStatusTool, {})).toContain('working tree clean');
await Bun.write(join(dir, 'app.ts'), 'export const port = 9090;\n');
await Bun.write(join(dir, 'new.ts'), 'x\n');
const out = await run(gitStatusTool, {});
expect(out).toContain('On main');
expect(out).toContain('app.ts');
expect(out).toContain('modified');
expect(out).toContain('new.ts');
expect(out).toContain('untracked');
});
test('git_status separates staged from unstaged', async () => {
await repoWithOneCommit();
await Bun.write(join(dir, 'app.ts'), 'export const port = 9090;\n');
await git('add', 'app.ts');
expect(await run(gitStatusTool, {})).toContain('staged modified');
});
test('git_diff shows the working tree, and staged on request', async () => {
await repoWithOneCommit();
expect(await run(gitDiffTool, {})).toBe('No uncommitted changes.');
await Bun.write(join(dir, 'app.ts'), 'export const port = 9090;\n');
const unstaged = await run(gitDiffTool, {});
expect(unstaged).toContain('-export const port = 8080;');
expect(unstaged).toContain('+export const port = 9090;');
expect(await run(gitDiffTool, { staged: true })).toBe('Nothing staged.');
await git('add', 'app.ts');
expect(await run(gitDiffTool, { staged: true })).toContain('9090');
});
test('git_diff narrows to a path', async () => {
await repoWithOneCommit();
await Bun.write(join(dir, 'app.ts'), 'changed\n');
await Bun.write(join(dir, 'other.ts'), 'also changed\n');
await git('add', 'other.ts');
await git('commit', '-m', 'add other');
await Bun.write(join(dir, 'other.ts'), 'changed again\n');
const out = await run(gitDiffTool, { path: 'app.ts' });
expect(out).toContain('app.ts');
expect(out).not.toContain('other.ts');
});
test('git_log lists commits newest first and honours the limit', async () => {
await repoWithOneCommit();
await Bun.write(join(dir, 'app.ts'), 'export const port = 9090;\n');
await git('commit', '-am', 'bump the port');
const out = await run(gitLogTool, {});
expect(out.split('\n')[0]).toContain('bump the port');
expect(out).toContain('add the server port');
expect(out).toContain('Test');
expect((await run(gitLogTool, { limit: 1 })).split('\n')).toHaveLength(1);
});
test('git_show renders one commit with its diff', async () => {
await repoWithOneCommit();
const out = await run(gitShowTool, { ref: 'HEAD' });
expect(out).toContain('add the server port');
expect(out).toContain('+export const port = 8080;');
});
test('git_show reports a bad ref rather than returning nothing', async () => {
await repoWithOneCommit();
expect(run(gitShowTool, { ref: 'no-such-ref' })).rejects.toThrow();
});
test('git_blame attributes each line and narrows by range', async () => {
await repoWithOneCommit();
const out = await run(gitBlameTool, { path: 'app.ts' });
expect(out).toContain('Test');
expect(out).toContain('export const port = 8080;');
expect(await run(gitBlameTool, { path: 'app.ts', startLine: 1, endLine: 1 })).toContain('8080');
});
test('outside a repository every tool fails with a clear message, not git porcelain', async () => {
for (const [name, t] of Object.entries(gitTools)) {
const input =
name === 'git_show' ? { ref: 'HEAD' } : name === 'git_blame' ? { path: 'nothing.ts' } : ({} as never);
expect(run(t as never, input as never), name).rejects.toThrow(/not a git repository/i);
}
});
test('an argument that looks like a shell injection is passed through as one argument', async () => {
await repoWithOneCommit();
// argv spawning, not a shell string, so this can only ever be a pathspec.
const out = await run(gitLogTool, { path: '; touch pwned.txt' }).catch((e: Error) => e.message);
expect(await Bun.file(join(dir, 'pwned.txt')).exists()).toBe(false);
expect(out).toBeTruthy();
});
+181
View File
@@ -7,9 +7,13 @@ import {
editFileTool,
globTool,
grepTool,
interruptBash,
jail,
listDirTool,
multiEditTool,
onBashOutput,
readFileTool,
readManyFilesTool,
writeFileTool,
} from '../src/tools';
@@ -58,6 +62,52 @@ test('read_file still accepts UTF-8 with high codepoints', async () => {
expect(await run(readFileTool, { path: 'u.txt' })).toContain('hello -> world');
});
test('read_many_files returns one labelled block per file', async () => {
await Bun.write(join(dir, 'a.ts'), 'const a = 1;\n');
await Bun.write(join(dir, 'b.ts'), 'const b = 2;\n');
const out = await run(readManyFilesTool, { files: [{ path: 'a.ts' }, { path: 'b.ts' }] });
expect(out).toContain('===== a.ts =====');
expect(out).toContain('1: const a = 1;');
expect(out).toContain('===== b.ts =====');
expect(out).toContain('1: const b = 2;');
});
test('read_many_files keeps the order it was given', async () => {
await Bun.write(join(dir, 'first.ts'), 'x\n');
await Bun.write(join(dir, 'second.ts'), 'y\n');
const out = await run(readManyFilesTool, { files: [{ path: 'second.ts' }, { path: 'first.ts' }] });
expect(out.indexOf('second.ts')).toBeLessThan(out.indexOf('first.ts'));
});
test('read_many_files honours a per-file offset and limit', async () => {
await Bun.write(join(dir, 'long.ts'), 'one\ntwo\nthree\nfour\n');
const out = await run(readManyFilesTool, { files: [{ path: 'long.ts', offset: 2, limit: 2 }] });
expect(out).toContain('2: two');
expect(out).toContain('3: three');
expect(out).not.toContain('1: one');
expect(out).not.toContain('4: four');
});
test('an unreadable path is named in its own block without aborting the rest', async () => {
await Bun.write(join(dir, 'good.ts'), 'fine\n');
await Bun.write(join(dir, 'blob.bin'), new Uint8Array([0x00, 0x01, 0x02]));
const out = await run(readManyFilesTool, {
files: [{ path: 'good.ts' }, { path: 'gone.ts' }, { path: 'blob.bin' }],
});
expect(out).toContain('1: fine');
expect(out).toContain('===== gone.ts =====');
expect(out).toContain('No such file: gone.ts');
expect(out).toContain('binary file');
});
test('read_many_files refuses a path outside the workspace', async () => {
expect(run(readManyFilesTool, { files: [{ path: '../escape.ts' }] })).resolves.toContain('escapes workspace');
});
test('edit_file replaces a unique occurrence', async () => {
await Bun.write(join(dir, 'x.ts'), 'const a = 1;\nconst b = 2;\n');
await run(editFileTool, { path: 'x.ts', oldString: 'const b = 2;', newString: 'const b = 3;' });
@@ -85,6 +135,100 @@ test('write_file then glob and grep find the content', async () => {
);
});
test('multi_edit applies every edit in order, each seeing the last', async () => {
await Bun.write(join(dir, 'm.ts'), 'const a = 1;\nconst b = 2;\n');
const out = await run(multiEditTool, {
path: 'm.ts',
edits: [
{ oldString: 'const a = 1;', newString: 'const a = 10;' },
{ oldString: 'const a = 10;\nconst b = 2;', newString: 'const a = 10;\nconst b = 20;' },
],
});
expect(out).toContain('2 edit(s)');
expect(await Bun.file(join(dir, 'm.ts')).text()).toBe('const a = 10;\nconst b = 20;\n');
});
test('a failing second edit leaves the file exactly as it was', async () => {
const before = 'const a = 1;\nconst b = 2;\n';
await Bun.write(join(dir, 'm.ts'), before);
expect(
run(multiEditTool, {
path: 'm.ts',
edits: [
{ oldString: 'const a = 1;', newString: 'const a = 10;' },
{ oldString: 'const NOPE = 0;', newString: 'x' },
],
}),
).rejects.toThrow(/edit 2: oldString not found/);
await Bun.sleep(20);
expect(await Bun.file(join(dir, 'm.ts')).text()).toBe(before);
});
test('multi_edit refuses an ambiguous match unless replaceAll, writing nothing', async () => {
const before = 'x\nx\n';
await Bun.write(join(dir, 'a.ts'), before);
expect(
run(multiEditTool, { path: 'a.ts', edits: [{ oldString: 'x', newString: 'y' }] }),
).rejects.toThrow(/appears 2 times/);
await Bun.sleep(20);
expect(await Bun.file(join(dir, 'a.ts')).text()).toBe(before);
await run(multiEditTool, { path: 'a.ts', edits: [{ oldString: 'x', newString: 'y', replaceAll: true }] });
expect(await Bun.file(join(dir, 'a.ts')).text()).toBe('y\ny\n');
});
test('multi_edit reports a missing file rather than creating one', async () => {
expect(
run(multiEditTool, { path: 'gone.ts', edits: [{ oldString: 'a', newString: 'b' }] }),
).rejects.toThrow(/No such file/);
expect(await Bun.file(join(dir, 'gone.ts')).exists()).toBe(false);
});
test('list_dir shows a tree with sizes and marks directories', async () => {
await Bun.write(join(dir, 'src/app.ts'), 'x'.repeat(2048));
await Bun.write(join(dir, 'readme.md'), 'hi');
const out = await run(listDirTool, {});
expect(out).toContain('src/');
expect(out).toContain('app.ts');
expect(out).toContain('2K');
expect(out).toContain('readme.md 2B');
});
test('list_dir stops at the depth limit, still naming the directory', async () => {
await Bun.write(join(dir, 'a/b/c/deep.ts'), 'x');
const shallow = await run(listDirTool, { depth: 1 });
expect(shallow).toContain('a/');
expect(shallow).not.toContain('deep.ts');
expect(await run(listDirTool, { depth: 4 })).toContain('deep.ts');
});
test('list_dir honours .gitignore and includeIgnored', async () => {
await Bun.write(join(dir, '.gitignore'), 'dist/\n');
await Bun.write(join(dir, 'dist/bundle.js'), 'x');
await Bun.write(join(dir, 'src/app.ts'), 'x');
expect(await run(listDirTool, {})).not.toContain('bundle.js');
expect(await run(listDirTool, { includeIgnored: true })).toContain('bundle.js');
});
test('list_dir scopes to a subdirectory and refuses a file', async () => {
await Bun.write(join(dir, 'src/app.ts'), 'x');
await Bun.write(join(dir, 'other.ts'), 'x');
const out = await run(listDirTool, { path: 'src' });
expect(out).toContain('app.ts');
expect(out).not.toContain('other.ts');
expect(run(listDirTool, { path: 'other.ts' })).rejects.toThrow(/Not a directory/);
});
test('glob skips gitignored paths and honours includeIgnored', async () => {
await Bun.write(join(dir, '.gitignore'), 'dist/\n');
await Bun.write(join(dir, 'dist/app.js'), 'x');
@@ -170,3 +314,40 @@ test('the bash listener is cleared when unset', async () => {
await run(bashTool, { command: 'echo two' });
expect(chunks.length).toBe(afterFirst);
}, 20_000);
const sleeper = process.platform === 'win32' ? 'ping -n 20 127.0.0.1 > nul' : 'sleep 20';
test('interruptBash kills the command in flight and names it', async () => {
const started = Date.now();
const call = run(bashTool, { command: sleeper, timeout: 30_000 });
await Bun.sleep(400);
expect(interruptBash()).toEqual([sleeper]);
expect(call).rejects.toThrow(/user interrupted this command/i);
await call.catch(() => {});
// Killed, not waited out: the 20s command must not have run to completion.
expect(Date.now() - started).toBeLessThan(10_000);
}, 30_000);
test('the interrupt error carries whatever the command printed first', async () => {
const script =
process.platform === 'win32' ? 'echo before && ping -n 20 127.0.0.1 > nul' : 'echo before; sleep 20';
const call = run(bashTool, { command: script, timeout: 30_000 });
await Bun.sleep(600);
interruptBash();
const message = await call.then(() => '', (e: Error) => e.message);
expect(message).toContain('before');
expect(message).toContain('effects are unknown');
}, 30_000);
test('interruptBash with nothing running is a no-op', () => {
expect(interruptBash()).toEqual([]);
});
test('a command that finished is no longer interruptible', async () => {
await run(bashTool, { command: 'echo done' });
expect(interruptBash()).toEqual([]);
}, 20_000);
+50 -1
View File
@@ -5,7 +5,7 @@ import { createAskTool } from '../src/ask';
import { applySubagentEvent } from '../src/ui/App';
import { AskPanel, createAskBridge } from '../src/ui/Ask';
import { Markdown } from '../src/ui/Markdown';
import { SubagentPanel, TodoPanel, StatusBar, InfoPanel, type SubagentView } from '../src/ui/Panels';
import { SubagentPanel, TodoPanel, StatusBar, InfoPanel, ActiveTool, QueuePanel, ThinkingPanel, type SubagentView } from '../src/ui/Panels';
import type { SubagentEvent } from '../src/subagent';
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
@@ -116,6 +116,55 @@ test('the info panel renders a markdown body', () => {
app.unmount();
});
test('the active tool line names the tool and the file it is touching', () => {
const app = render(<ActiveTool name="read_file" summary="src/session.ts" />);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('read_file');
expect(frame).toContain('src/session.ts');
app.unmount();
});
test('the active tool line renders before the arguments have arrived', () => {
const app = render(<ActiveTool name="grep" />);
expect(app.lastFrame()).toContain('grep');
app.unmount();
});
test('thinking collapses to a token count, and expands on request', () => {
const text = 'x'.repeat(1648);
const collapsed = render(<ThinkingPanel text={text} />);
expect(collapsed.lastFrame()).toContain('~412 tokens');
expect(collapsed.lastFrame()).not.toContain('xxxx');
collapsed.unmount();
const open = render(<ThinkingPanel text={'first line\nsecond line'} expanded />);
const frame = open.lastFrame() ?? '';
expect(frame).toContain('second line');
expect(frame).toContain('collapse');
open.unmount();
});
test('no reasoning renders nothing at all', () => {
const app = render(<ThinkingPanel text="" />);
expect(app.lastFrame() ?? '').toBe('');
app.unmount();
});
test('the queue panel counts what is waiting and lists it in order', () => {
const app = render(<QueuePanel prompts={['first thing', 'second thing']} />);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('queued: 2');
expect(frame).toContain('1. first thing');
expect(frame).toContain('2. second thing');
app.unmount();
});
test('an empty queue renders nothing', () => {
const app = render(<QueuePanel prompts={[]} />);
expect(app.lastFrame() ?? '').toBe('');
app.unmount();
});
const events: SubagentEvent[] = [
{ type: 'start', id: 'a', kind: 'explore', description: 'first task' },
{ type: 'step', id: 'a', tool: 'grep', summary: 'needle' },