Add batch reads, @file completion, interruptible commands, tool sets
Tools, six built-in to fourteen: - read_many_files: up to 20 paths read concurrently, each with its own window. An unreadable path is reported in its own block instead of throwing. - multi_edit: several edits to one file, validated in memory first so a late failure cannot leave the file half-written. - list_dir: ignore-aware depth-limited tree. - git_status/diff/log/show/blame: read-only, spawned with a fixed argv rather than a shell string, which is what makes them safe to auto-approve. toolSets gates them. core is always on; edit-plus and git are optional. A disabled set reaches neither the wire nor the system prompt, since a prompt naming an absent tool teaches calls that cannot succeed. Interface: - Reasoning streams to a collapsed panel, ctrl-r expands, dropped when the turn ends: it is progress, not the answer. - The tool in flight is named from tool-input-start, before its arguments finish streaming, and cleared on its result. - Prompts typed mid-turn queue and drain in order. esc clears the queue as well as aborting. - @ opens a path picker fed by the ignore-aware walker. Prefix matches rank above substring matches, so @src/ means "under src/". The walk runs on the first @, not at startup. ctrl-c kills the command in flight and keeps the turn. The call throws rather than returning, so the model cannot read a killed command as one that ran and failed on its own terms. The kill takes the whole process tree: killing cmd /c alone left the real command holding both pipes open, so the read never returned and the interrupt did nothing for 19 seconds. Two pruning fixes: - A tool result whose tool call was pruned is now dropped with it. Pruning counts messages, so the cut landed between an assistant tool-call and the tool message answering it, producing 400 "No tool call found for function call output with call_id ...". The reverse pairing is left alone: a call awaiting its result is what a suspended approval looks like. - ignore.ts called statFs without importing it, so walk() crashed on the first symlink. 482 tests, up from 404. Docs synced across README, ROADMAP, TODO, and all of docs/: tool sets, the new tools, ctrl-c semantics, the tool-start event, and the two hand-maintained tool-name lists recorded as a known weakness.
This commit is contained in:
+97
-1
@@ -1,6 +1,6 @@
|
||||
import { expect, test } from 'bun:test';
|
||||
import type { ModelMessage } from 'ai';
|
||||
import { dropOrphanedItems, prunePreservingItems } from '../src/prune';
|
||||
import { dropOrphanedItems, dropOrphanedResults, prunePreservingItems } from '../src/prune';
|
||||
|
||||
const kinds = (messages: ModelMessage[]) =>
|
||||
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
|
||||
@@ -152,3 +152,99 @@ test('a provider other than openai is handled the same way', () => {
|
||||
];
|
||||
expect(dropOrphanedItems(before, after)).toEqual([]);
|
||||
});
|
||||
|
||||
/** The assistant tool-call plus the tool message answering it, as one exchange. */
|
||||
const callAndResult = (call: string, rs?: string): ModelMessage[] => [
|
||||
{
|
||||
role: 'assistant',
|
||||
content: [
|
||||
...(rs ? [{ type: 'reasoning' as const, text: 'deciding', providerOptions: { openai: { itemId: rs } } }] : []),
|
||||
{ type: 'tool-call', toolCallId: call, toolName: 'grep', input: { pattern: 'x' } },
|
||||
],
|
||||
},
|
||||
{
|
||||
role: 'tool',
|
||||
content: [{ type: 'tool-result', toolCallId: call, toolName: 'grep', output: { type: 'text', value: 'hit' } }],
|
||||
},
|
||||
];
|
||||
|
||||
test('a tool result left without its tool call is dropped', () => {
|
||||
const [, resultMessage] = callAndResult('call_A');
|
||||
const cleaned = dropOrphanedResults([{ role: 'user', content: 'q' }, resultMessage!]);
|
||||
expect(JSON.stringify(cleaned)).not.toContain('call_A');
|
||||
expect(kinds(cleaned)).toEqual(['user']);
|
||||
});
|
||||
|
||||
test('a tool result keeps its place while the call is still there', () => {
|
||||
const messages: ModelMessage[] = [{ role: 'user', content: 'q' }, ...callAndResult('call_A')];
|
||||
expect(dropOrphanedResults(messages)).toEqual(messages);
|
||||
});
|
||||
|
||||
test('a tool call awaiting its result survives, since that is a suspended approval', () => {
|
||||
const [callMessage] = callAndResult('call_A');
|
||||
const messages: ModelMessage[] = [{ role: 'user', content: 'q' }, callMessage!];
|
||||
expect(dropOrphanedResults(messages)).toEqual(messages);
|
||||
});
|
||||
|
||||
test('a tool-error is treated as a result and dropped with its call', () => {
|
||||
const messages: ModelMessage[] = [
|
||||
{ role: 'user', content: 'q' },
|
||||
{
|
||||
role: 'tool',
|
||||
content: [{ type: 'tool-error', toolCallId: 'call_A', toolName: 'grep', error: 'boom' } as never],
|
||||
},
|
||||
];
|
||||
expect(JSON.stringify(dropOrphanedResults(messages))).not.toContain('call_A');
|
||||
});
|
||||
|
||||
test('only the orphaned result is dropped, not a healthy one beside it', () => {
|
||||
const messages: ModelMessage[] = [
|
||||
{ role: 'user', content: 'q' },
|
||||
...callAndResult('call_LIVE'),
|
||||
{
|
||||
role: 'tool',
|
||||
content: [
|
||||
{ type: 'tool-result', toolCallId: 'call_LIVE', toolName: 'grep', output: { type: 'text', value: 'a' } },
|
||||
{ type: 'tool-result', toolCallId: 'call_GONE', toolName: 'grep', output: { type: 'text', value: 'b' } },
|
||||
],
|
||||
},
|
||||
];
|
||||
|
||||
const json = JSON.stringify(dropOrphanedResults(messages));
|
||||
expect(json).toContain('call_LIVE');
|
||||
expect(json).not.toContain('call_GONE');
|
||||
});
|
||||
|
||||
/**
|
||||
* The 400 this guards against: "No tool call found for function call output with
|
||||
* call_id ...". Pruning counts messages, so its cut lands between the assistant
|
||||
* tool-call and the tool message answering it, stranding the result on the wire.
|
||||
*/
|
||||
test('prunePreservingItems never strands a tool result on the wire', () => {
|
||||
const messages: ModelMessage[] = [{ role: 'user', content: `q ${'x'.repeat(4000)}` }];
|
||||
for (let i = 0; i < 5; i++) {
|
||||
messages.push(...callAndResult(`call_${i}`, `rs_${i}`));
|
||||
messages.push({ role: 'user', content: `follow up ${i} ${'y'.repeat(4000)}` });
|
||||
}
|
||||
|
||||
const pruned = prunePreservingItems({
|
||||
messages,
|
||||
reasoning: 'all',
|
||||
toolCalls: 'before-last-3-messages',
|
||||
emptyMessages: 'remove',
|
||||
});
|
||||
|
||||
const calls = new Set<string>();
|
||||
for (const m of pruned) {
|
||||
if (!Array.isArray(m.content)) continue;
|
||||
for (const p of m.content as { type: string; toolCallId?: string }[]) {
|
||||
if (p.type === 'tool-call' && p.toolCallId) calls.add(p.toolCallId);
|
||||
}
|
||||
}
|
||||
for (const m of pruned) {
|
||||
if (!Array.isArray(m.content)) continue;
|
||||
for (const p of m.content as { type: string; toolCallId?: string }[]) {
|
||||
if (p.type === 'tool-result' || p.type === 'tool-error') expect(calls.has(p.toolCallId!)).toBe(true);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user