Add batch reads, @file completion, interruptible commands, tool sets
Tools, six built-in to fourteen: - read_many_files: up to 20 paths read concurrently, each with its own window. An unreadable path is reported in its own block instead of throwing. - multi_edit: several edits to one file, validated in memory first so a late failure cannot leave the file half-written. - list_dir: ignore-aware depth-limited tree. - git_status/diff/log/show/blame: read-only, spawned with a fixed argv rather than a shell string, which is what makes them safe to auto-approve. toolSets gates them. core is always on; edit-plus and git are optional. A disabled set reaches neither the wire nor the system prompt, since a prompt naming an absent tool teaches calls that cannot succeed. Interface: - Reasoning streams to a collapsed panel, ctrl-r expands, dropped when the turn ends: it is progress, not the answer. - The tool in flight is named from tool-input-start, before its arguments finish streaming, and cleared on its result. - Prompts typed mid-turn queue and drain in order. esc clears the queue as well as aborting. - @ opens a path picker fed by the ignore-aware walker. Prefix matches rank above substring matches, so @src/ means "under src/". The walk runs on the first @, not at startup. ctrl-c kills the command in flight and keeps the turn. The call throws rather than returning, so the model cannot read a killed command as one that ran and failed on its own terms. The kill takes the whole process tree: killing cmd /c alone left the real command holding both pipes open, so the read never returned and the interrupt did nothing for 19 seconds. Two pruning fixes: - A tool result whose tool call was pruned is now dropped with it. Pruning counts messages, so the cut landed between an assistant tool-call and the tool message answering it, producing 400 "No tool call found for function call output with call_id ...". The reverse pairing is left alone: a call awaiting its result is what a suspended approval looks like. - ignore.ts called statFs without importing it, so walk() crashed on the first symlink. 482 tests, up from 404. Docs synced across README, ROADMAP, TODO, and all of docs/: tool sets, the new tools, ctrl-c semantics, the tool-start event, and the two hand-maintained tool-name lists recorded as a known weakness.
This commit is contained in:
+41
-4
@@ -7,6 +7,7 @@ import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { Session } from '../src/session';
|
||||
import { interruptBash } from '../src/tools';
|
||||
|
||||
const usage = {
|
||||
inputTokens: { total: 10, noCache: 10, cacheRead: 0, cacheWrite: 0 },
|
||||
@@ -62,7 +63,7 @@ test('read-only tool runs without approval and the loop terminates', async () =>
|
||||
const kinds: string[] = [];
|
||||
for await (const ev of session.send('read note.txt')) kinds.push(ev.type);
|
||||
|
||||
expect(kinds).toEqual(['tool-call', 'tool-result', 'text', 'done']);
|
||||
expect(kinds).toEqual(['tool-start', 'tool-call', 'tool-result', 'text', 'done']);
|
||||
expect(call).toBe(2);
|
||||
}));
|
||||
|
||||
@@ -264,7 +265,7 @@ test('a read-only built-in stays free even when mcp tools are present', async ()
|
||||
|
||||
const kinds: string[] = [];
|
||||
for await (const ev of session.send('read note.txt')) kinds.push(ev.type);
|
||||
expect(kinds).toEqual(['tool-call', 'tool-result', 'text', 'done']);
|
||||
expect(kinds).toEqual(['tool-start', 'tool-call', 'tool-result', 'text', 'done']);
|
||||
}));
|
||||
|
||||
test('setModel swaps the model used by the next turn', async () => {
|
||||
@@ -302,8 +303,7 @@ test('reset clears history and token counters; replace swaps history in', async
|
||||
expect(session.messages).toEqual([{ role: 'user', content: 'restored' }]);
|
||||
});
|
||||
|
||||
test('abort mid-stream ends the turn with done, keeping the text already delivered', async () => {
|
||||
const session = new Session({
|
||||
test('abort mid-stream ends the turn with done, keeping the text already delivered', async () => { const session = new Session({
|
||||
model: new MockLanguageModelV4({
|
||||
doStream: async () => ({
|
||||
stream: simulateReadableStream({
|
||||
@@ -338,3 +338,40 @@ test('abort mid-stream ends the turn with done, keeping the text already deliver
|
||||
expect(kinds.at(-1)).toBe('done');
|
||||
expect(kinds).not.toContain('error');
|
||||
}, 15_000);
|
||||
|
||||
test('an interrupted command becomes a tool error and the turn carries on', async () =>
|
||||
inTempDir(async () => {
|
||||
const sleeper = process.platform === 'win32' ? 'ping -n 20 127.0.0.1 > nul' : 'sleep 20';
|
||||
let call = 0;
|
||||
const session = new Session({
|
||||
yolo: true,
|
||||
model: new MockLanguageModelV4({
|
||||
doStream: async () =>
|
||||
stream(call++ === 0 ? toolCall('c1', 'bash', { command: sleeper }) : text('I stopped there.')),
|
||||
}),
|
||||
askApproval: async () => {
|
||||
throw new Error('yolo must not ask');
|
||||
},
|
||||
});
|
||||
|
||||
const kinds: string[] = [];
|
||||
let toolError = '';
|
||||
const turn = (async () => {
|
||||
for await (const ev of session.send('run the long thing')) {
|
||||
kinds.push(ev.type);
|
||||
if (ev.type === 'tool-error') toolError = String((ev.error as Error).message ?? ev.error);
|
||||
}
|
||||
})();
|
||||
|
||||
// Interrupt once the command is actually running.
|
||||
await Bun.sleep(700);
|
||||
expect(interruptBash()).toEqual([sleeper]);
|
||||
await turn;
|
||||
|
||||
expect(kinds).toContain('tool-error');
|
||||
expect(toolError).toMatch(/user interrupted this command/i);
|
||||
// The turn survived: the model was asked again and its reply arrived.
|
||||
expect(kinds).toContain('text');
|
||||
expect(kinds.at(-1)).toBe('done');
|
||||
expect(call).toBe(2);
|
||||
}), 30_000);
|
||||
|
||||
Reference in New Issue
Block a user