Tools, six built-in to fourteen: - read_many_files: up to 20 paths read concurrently, each with its own window. An unreadable path is reported in its own block instead of throwing. - multi_edit: several edits to one file, validated in memory first so a late failure cannot leave the file half-written. - list_dir: ignore-aware depth-limited tree. - git_status/diff/log/show/blame: read-only, spawned with a fixed argv rather than a shell string, which is what makes them safe to auto-approve. toolSets gates them. core is always on; edit-plus and git are optional. A disabled set reaches neither the wire nor the system prompt, since a prompt naming an absent tool teaches calls that cannot succeed. Interface: - Reasoning streams to a collapsed panel, ctrl-r expands, dropped when the turn ends: it is progress, not the answer. - The tool in flight is named from tool-input-start, before its arguments finish streaming, and cleared on its result. - Prompts typed mid-turn queue and drain in order. esc clears the queue as well as aborting. - @ opens a path picker fed by the ignore-aware walker. Prefix matches rank above substring matches, so @src/ means "under src/". The walk runs on the first @, not at startup. ctrl-c kills the command in flight and keeps the turn. The call throws rather than returning, so the model cannot read a killed command as one that ran and failed on its own terms. The kill takes the whole process tree: killing cmd /c alone left the real command holding both pipes open, so the read never returned and the interrupt did nothing for 19 seconds. Two pruning fixes: - A tool result whose tool call was pruned is now dropped with it. Pruning counts messages, so the cut landed between an assistant tool-call and the tool message answering it, producing 400 "No tool call found for function call output with call_id ...". The reverse pairing is left alone: a call awaiting its result is what a suspended approval looks like. - ignore.ts called statFs without importing it, so walk() crashed on the first symlink. 482 tests, up from 404. Docs synced across README, ROADMAP, TODO, and all of docs/: tool sets, the new tools, ctrl-c semantics, the tool-start event, and the two hand-maintained tool-name lists recorded as a known weakness.
112 lines
4.3 KiB
TypeScript
112 lines
4.3 KiB
TypeScript
import { expect, test } from 'bun:test';
|
|
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
|
|
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
|
|
import { mkdtempSync, rmSync } from 'node:fs';
|
|
import { tmpdir } from 'node:os';
|
|
import { join } from 'node:path';
|
|
import { Session } from '../src/session';
|
|
import { createTaskTool } from '../src/subagent';
|
|
|
|
const usage = {
|
|
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
|
|
outputTokens: { total: 5 },
|
|
} as any;
|
|
|
|
const stream = (parts: LanguageModelV4StreamPart[]) => ({
|
|
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
|
|
});
|
|
|
|
const toolCall = (id: string, toolName: string, input: unknown): LanguageModelV4StreamPart[] => [
|
|
{ type: 'tool-input-start', id, toolName },
|
|
{ type: 'tool-input-end', id },
|
|
{ type: 'tool-call', toolCallId: id, toolName, input: JSON.stringify(input) },
|
|
{ type: 'finish', finishReason: { unified: 'tool-calls', raw: 'tool_use' }, usage },
|
|
];
|
|
|
|
const text = (body: string): LanguageModelV4StreamPart[] => [
|
|
{ type: 'text-start', id: '0' },
|
|
{ type: 'text-delta', id: '0', delta: body },
|
|
{ type: 'text-end', id: '0' },
|
|
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
|
|
];
|
|
|
|
function inTempDir<T>(fn: (dir: string) => Promise<T>): Promise<T> {
|
|
const orig = process.cwd();
|
|
const dir = mkdtempSync(join(tmpdir(), 'shiro-sub-'));
|
|
process.chdir(dir);
|
|
return fn(dir).finally(() => {
|
|
process.chdir(orig);
|
|
rmSync(dir, { recursive: true, force: true });
|
|
});
|
|
}
|
|
|
|
test('subagent greps the workspace and returns text to the parent', async () =>
|
|
inTempDir(async () => {
|
|
await Bun.write('src/auth.ts', 'export function login() {}\n');
|
|
|
|
// Two independent loops share this model: the parent, then the subagent.
|
|
const seen: LanguageModelV4CallOptions[] = [];
|
|
const model = new MockLanguageModelV4({
|
|
doStream: async (opts) => {
|
|
const call = seen.length;
|
|
seen.push(opts);
|
|
if (call === 0) {
|
|
return stream(
|
|
toolCall('c1', 'task', { description: 'find auth', prompt: 'Find where login is defined under src/.' }),
|
|
);
|
|
}
|
|
if (call === 1) return stream(toolCall('s1', 'grep', { pattern: 'login', include: '**/*.ts' }));
|
|
if (call === 2) return stream(text('login() is defined at src/auth.ts:1'));
|
|
return stream(text('The subagent found it in src/auth.ts.'));
|
|
},
|
|
});
|
|
|
|
const session = new Session({
|
|
model,
|
|
askApproval: async () => {
|
|
throw new Error('the task tool must never require approval');
|
|
},
|
|
extraTools: { task: createTaskTool({ model }) },
|
|
autoApprove: ['task'],
|
|
});
|
|
|
|
const events: string[] = [];
|
|
for await (const ev of session.send('where is login defined?')) events.push(ev.type);
|
|
|
|
expect(events).toEqual(['tool-start', 'tool-call', 'tool-result', 'text', 'done']);
|
|
|
|
// The subagent gets only read tools, so it can never trigger an approval prompt.
|
|
const subagentTools = (seen[1]?.tools ?? []).map((t) => t.name).sort();
|
|
expect(subagentTools).toEqual(['glob', 'grep', 'read_file']);
|
|
|
|
// Its findings reach the parent as a tool result, not as raw transcript.
|
|
const toolMessage = session.messages.find((m) => m.role === 'tool');
|
|
expect(JSON.stringify(toolMessage)).toContain('src/auth.ts:1');
|
|
}));
|
|
|
|
test('subagent does not see the parent conversation', async () =>
|
|
inTempDir(async () => {
|
|
const seen: LanguageModelV4CallOptions[] = [];
|
|
const model = new MockLanguageModelV4({
|
|
doStream: async (opts) => {
|
|
const call = seen.length;
|
|
seen.push(opts);
|
|
if (call === 0) return stream(toolCall('c1', 'task', { description: 'probe', prompt: 'Look at glob src/*.' }));
|
|
if (call === 1) return stream(text('nothing notable'));
|
|
return stream(text('done'));
|
|
},
|
|
});
|
|
|
|
const session = new Session({
|
|
model,
|
|
askApproval: async () => 'deny',
|
|
extraTools: { task: createTaskTool({ model }) },
|
|
autoApprove: ['task'],
|
|
});
|
|
|
|
for await (const _ of session.send('MY-SECRET-PARENT-CONTEXT')) void _;
|
|
|
|
expect(JSON.stringify(seen[1]?.prompt)).not.toContain('MY-SECRET-PARENT-CONTEXT');
|
|
expect(JSON.stringify(seen[1]?.prompt)).toContain('Look at glob src/*.');
|
|
}));
|