Files
shiro-neko/test/prune.test.ts
T
Muhammad Zakir Ramadhan 2fa6ee247b Add batch reads, @file completion, interruptible commands, tool sets
Tools, six built-in to fourteen:
- read_many_files: up to 20 paths read concurrently, each with its own window.
  An unreadable path is reported in its own block instead of throwing.
- multi_edit: several edits to one file, validated in memory first so a late
  failure cannot leave the file half-written.
- list_dir: ignore-aware depth-limited tree.
- git_status/diff/log/show/blame: read-only, spawned with a fixed argv rather
  than a shell string, which is what makes them safe to auto-approve.

toolSets gates them. core is always on; edit-plus and git are optional. A
disabled set reaches neither the wire nor the system prompt, since a prompt
naming an absent tool teaches calls that cannot succeed.

Interface:
- Reasoning streams to a collapsed panel, ctrl-r expands, dropped when the turn
  ends: it is progress, not the answer.
- The tool in flight is named from tool-input-start, before its arguments finish
  streaming, and cleared on its result.
- Prompts typed mid-turn queue and drain in order. esc clears the queue as well
  as aborting.
- @ opens a path picker fed by the ignore-aware walker. Prefix matches rank
  above substring matches, so @src/ means "under src/". The walk runs on the
  first @, not at startup.

ctrl-c kills the command in flight and keeps the turn. The call throws rather
than returning, so the model cannot read a killed command as one that ran and
failed on its own terms. The kill takes the whole process tree: killing cmd /c
alone left the real command holding both pipes open, so the read never returned
and the interrupt did nothing for 19 seconds.

Two pruning fixes:
- A tool result whose tool call was pruned is now dropped with it. Pruning
  counts messages, so the cut landed between an assistant tool-call and the tool
  message answering it, producing 400 "No tool call found for function call
  output with call_id ...". The reverse pairing is left alone: a call awaiting
  its result is what a suspended approval looks like.
- ignore.ts called statFs without importing it, so walk() crashed on the first
  symlink.

482 tests, up from 404. Docs synced across README, ROADMAP, TODO, and all of
docs/: tool sets, the new tools, ctrl-c semantics, the tool-start event, and the
two hand-maintained tool-name lists recorded as a known weakness.
2026-09-03 01:37:48 +07:00

251 lines
9.4 KiB
TypeScript

import { expect, test } from 'bun:test';
import type { ModelMessage } from 'ai';
import { dropOrphanedItems, dropOrphanedResults, prunePreservingItems } from '../src/prune';
const kinds = (messages: ModelMessage[]) =>
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
/** An assistant turn as the OpenAI responses API returns it. */
const reasoningTurn = (rs: string, msg: string, text = 'answer'): ModelMessage => ({
role: 'assistant',
content: [
{ type: 'reasoning', text: 'thinking', providerOptions: { openai: { itemId: rs } } },
{ type: 'text', text, providerOptions: { openai: { itemId: msg } } },
],
});
const toolTurn = (rs: string, call: string): ModelMessage => ({
role: 'assistant',
content: [
{ type: 'reasoning', text: 'deciding', providerOptions: { openai: { itemId: rs } } },
{
type: 'tool-call',
toolCallId: 'tc1',
toolName: 'grep',
input: { pattern: 'x' },
providerOptions: { openai: { itemId: call } },
},
],
});
test('a message left without its reasoning item is dropped', () => {
const before = [{ role: 'user' as const, content: 'q' }, reasoningTurn('rs_1', 'msg_1')];
const after = [{ role: 'user' as const, content: 'q' }, { role: 'assistant' as const, content: [{ type: 'text' as const, text: 'answer', providerOptions: { openai: { itemId: 'msg_1' } } }] }];
const cleaned = dropOrphanedItems(before, after);
expect(JSON.stringify(cleaned)).not.toContain('msg_1');
expect(kinds(cleaned)).toEqual(['user']);
});
test('a tool call left without its reasoning item is dropped too', () => {
const before = [{ role: 'user' as const, content: 'q' }, toolTurn('rs_1', 'fc_1')];
const after = [
{ role: 'user' as const, content: 'q' },
{
role: 'assistant' as const,
content: [
{
type: 'tool-call' as const,
toolCallId: 'tc1',
toolName: 'grep',
input: { pattern: 'x' },
providerOptions: { openai: { itemId: 'fc_1' } },
},
],
},
];
expect(JSON.stringify(dropOrphanedItems(before, after))).not.toContain('fc_1');
});
test('a turn whose reasoning survived is left alone', () => {
const messages = [{ role: 'user' as const, content: 'q' }, reasoningTurn('rs_1', 'msg_1')];
expect(dropOrphanedItems(messages, messages)).toEqual(messages);
});
test('nothing is touched when no reasoning was removed', () => {
const before: ModelMessage[] = [
{ role: 'user', content: 'q' },
{ role: 'assistant', content: 'plain answer' },
];
expect(dropOrphanedItems(before, before)).toEqual(before);
});
test('parts with no provider itemId are always kept', () => {
const before = [reasoningTurn('rs_1', 'msg_1')];
const after: ModelMessage[] = [{ role: 'assistant', content: [{ type: 'text', text: 'no item id here' }] }];
expect(dropOrphanedItems(before, after)).toEqual(after);
});
test('user and tool messages are never affected', () => {
const before = [{ role: 'user' as const, content: 'q' }, reasoningTurn('rs_1', 'msg_1')];
const after: ModelMessage[] = [
{ role: 'user', content: 'q' },
{ role: 'tool', content: [{ type: 'tool-result', toolCallId: 't1', toolName: 'grep', output: { type: 'text', value: 'hit' } }] },
];
expect(dropOrphanedItems(before, after)).toEqual(after);
});
test('one orphaned turn does not take a healthy one with it', () => {
const before = [
{ role: 'user' as const, content: 'q1' },
reasoningTurn('rs_1', 'msg_1', 'old answer'),
{ role: 'user' as const, content: 'q2' },
reasoningTurn('rs_2', 'msg_2', 'new answer'),
];
const after: ModelMessage[] = [
{ role: 'user', content: 'q1' },
{ role: 'assistant', content: [{ type: 'text', text: 'old answer', providerOptions: { openai: { itemId: 'msg_1' } } }] },
{ role: 'user', content: 'q2' },
reasoningTurn('rs_2', 'msg_2', 'new answer'),
];
const cleaned = dropOrphanedItems(before, after);
const json = JSON.stringify(cleaned);
expect(json).not.toContain('msg_1');
expect(json).toContain('msg_2');
expect(json).toContain('rs_2');
});
test('prunePreservingItems leaves no orphan behind on a real prune', () => {
const messages: ModelMessage[] = [];
for (let i = 0; i < 6; i++) {
messages.push({ role: 'user', content: `question ${i} ${'x'.repeat(3000)}` });
messages.push(reasoningTurn(`rs_${i}`, `msg_${i}`, `answer ${i}`));
}
const pruned = prunePreservingItems({
messages,
reasoning: 'all',
toolCalls: 'before-last-3-messages',
emptyMessages: 'remove',
});
// Every surviving text part must either have no item id or belong to a turn
// whose reasoning also survived. Since reasoning: 'all' removes them all, no
// itemId-bearing assistant part may remain.
const survivingIds = JSON.stringify(pruned);
for (let i = 0; i < 6; i++) expect(survivingIds).not.toContain(`msg_${i}`);
expect(pruned.filter((m) => m.role === 'user')).toHaveLength(6);
});
test('prunePreservingItems is a no-op when nothing needs pruning', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'small' },
{ role: 'assistant', content: 'reply' },
];
expect(prunePreservingItems({ messages, reasoning: 'none', emptyMessages: 'keep' })).toEqual(messages);
});
test('a provider other than openai is handled the same way', () => {
const before: ModelMessage[] = [
{
role: 'assistant',
content: [
{ type: 'reasoning', text: 't', providerOptions: { someProvider: { itemId: 'r1' } } },
{ type: 'text', text: 'a', providerOptions: { someProvider: { itemId: 'm1' } } },
],
},
];
const after: ModelMessage[] = [
{ role: 'assistant', content: [{ type: 'text', text: 'a', providerOptions: { someProvider: { itemId: 'm1' } } }] },
];
expect(dropOrphanedItems(before, after)).toEqual([]);
});
/** The assistant tool-call plus the tool message answering it, as one exchange. */
const callAndResult = (call: string, rs?: string): ModelMessage[] => [
{
role: 'assistant',
content: [
...(rs ? [{ type: 'reasoning' as const, text: 'deciding', providerOptions: { openai: { itemId: rs } } }] : []),
{ type: 'tool-call', toolCallId: call, toolName: 'grep', input: { pattern: 'x' } },
],
},
{
role: 'tool',
content: [{ type: 'tool-result', toolCallId: call, toolName: 'grep', output: { type: 'text', value: 'hit' } }],
},
];
test('a tool result left without its tool call is dropped', () => {
const [, resultMessage] = callAndResult('call_A');
const cleaned = dropOrphanedResults([{ role: 'user', content: 'q' }, resultMessage!]);
expect(JSON.stringify(cleaned)).not.toContain('call_A');
expect(kinds(cleaned)).toEqual(['user']);
});
test('a tool result keeps its place while the call is still there', () => {
const messages: ModelMessage[] = [{ role: 'user', content: 'q' }, ...callAndResult('call_A')];
expect(dropOrphanedResults(messages)).toEqual(messages);
});
test('a tool call awaiting its result survives, since that is a suspended approval', () => {
const [callMessage] = callAndResult('call_A');
const messages: ModelMessage[] = [{ role: 'user', content: 'q' }, callMessage!];
expect(dropOrphanedResults(messages)).toEqual(messages);
});
test('a tool-error is treated as a result and dropped with its call', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'q' },
{
role: 'tool',
content: [{ type: 'tool-error', toolCallId: 'call_A', toolName: 'grep', error: 'boom' } as never],
},
];
expect(JSON.stringify(dropOrphanedResults(messages))).not.toContain('call_A');
});
test('only the orphaned result is dropped, not a healthy one beside it', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'q' },
...callAndResult('call_LIVE'),
{
role: 'tool',
content: [
{ type: 'tool-result', toolCallId: 'call_LIVE', toolName: 'grep', output: { type: 'text', value: 'a' } },
{ type: 'tool-result', toolCallId: 'call_GONE', toolName: 'grep', output: { type: 'text', value: 'b' } },
],
},
];
const json = JSON.stringify(dropOrphanedResults(messages));
expect(json).toContain('call_LIVE');
expect(json).not.toContain('call_GONE');
});
/**
* The 400 this guards against: "No tool call found for function call output with
* call_id ...". Pruning counts messages, so its cut lands between the assistant
* tool-call and the tool message answering it, stranding the result on the wire.
*/
test('prunePreservingItems never strands a tool result on the wire', () => {
const messages: ModelMessage[] = [{ role: 'user', content: `q ${'x'.repeat(4000)}` }];
for (let i = 0; i < 5; i++) {
messages.push(...callAndResult(`call_${i}`, `rs_${i}`));
messages.push({ role: 'user', content: `follow up ${i} ${'y'.repeat(4000)}` });
}
const pruned = prunePreservingItems({
messages,
reasoning: 'all',
toolCalls: 'before-last-3-messages',
emptyMessages: 'remove',
});
const calls = new Set<string>();
for (const m of pruned) {
if (!Array.isArray(m.content)) continue;
for (const p of m.content as { type: string; toolCallId?: string }[]) {
if (p.type === 'tool-call' && p.toolCallId) calls.add(p.toolCallId);
}
}
for (const m of pruned) {
if (!Array.isArray(m.content)) continue;
for (const p of m.content as { type: string; toolCallId?: string }[]) {
if (p.type === 'tool-result' || p.type === 'tool-error') expect(calls.has(p.toolCallId!)).toBe(true);
}
}
});