Files
shiro-neko/test/session-features.test.ts
T
Muhammad Zakir Ramadhan 2fa6ee247b Add batch reads, @file completion, interruptible commands, tool sets
Tools, six built-in to fourteen:
- read_many_files: up to 20 paths read concurrently, each with its own window.
  An unreadable path is reported in its own block instead of throwing.
- multi_edit: several edits to one file, validated in memory first so a late
  failure cannot leave the file half-written.
- list_dir: ignore-aware depth-limited tree.
- git_status/diff/log/show/blame: read-only, spawned with a fixed argv rather
  than a shell string, which is what makes them safe to auto-approve.

toolSets gates them. core is always on; edit-plus and git are optional. A
disabled set reaches neither the wire nor the system prompt, since a prompt
naming an absent tool teaches calls that cannot succeed.

Interface:
- Reasoning streams to a collapsed panel, ctrl-r expands, dropped when the turn
  ends: it is progress, not the answer.
- The tool in flight is named from tool-input-start, before its arguments finish
  streaming, and cleared on its result.
- Prompts typed mid-turn queue and drain in order. esc clears the queue as well
  as aborting.
- @ opens a path picker fed by the ignore-aware walker. Prefix matches rank
  above substring matches, so @src/ means "under src/". The walk runs on the
  first @, not at startup.

ctrl-c kills the command in flight and keeps the turn. The call throws rather
than returning, so the model cannot read a killed command as one that ran and
failed on its own terms. The kill takes the whole process tree: killing cmd /c
alone left the real command holding both pipes open, so the read never returned
and the interrupt did nothing for 19 seconds.

Two pruning fixes:
- A tool result whose tool call was pruned is now dropped with it. Pruning
  counts messages, so the cut landed between an assistant tool-call and the tool
  message answering it, producing 400 "No tool call found for function call
  output with call_id ...". The reverse pairing is left alone: a call awaiting
  its result is what a suspended approval looks like.
- ignore.ts called statFs without importing it, so walk() crashed on the first
  symlink.

482 tests, up from 404. Docs synced across README, ROADMAP, TODO, and all of
docs/: tool sets, the new tools, ctrl-c semantics, the tool-start event, and the
two hand-maintained tool-name lists recorded as a known weakness.
2026-09-03 01:37:48 +07:00

299 lines
12 KiB
TypeScript

import { expect, test } from 'bun:test';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { variantByName } from '../src/agents';
import { Memory } from '../src/memory';
import { createHost } from '../src/plugins';
import { guardPlugin, timePlugin } from '../src/plugins-builtin';
import { Session } from '../src/session';
import { loadSkills } from '../src/skills';
import { MUTATING_TOOLS, TOOL_SETS, TOOL_SET_NAMES, isToolSetName, toolSetOf } from '../src/tools';
import { GIT_TOOL_NAMES } from '../src/tools-git';
const usage = {
inputTokens: { total: 5, noCache: 5, cacheRead: 0, cacheWrite: 0 },
outputTokens: { total: 2 },
} as any;
const stream = (parts: LanguageModelV4StreamPart[]) => ({
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
});
const toolCall = (id: string, toolName: string, input: unknown): LanguageModelV4StreamPart[] => [
{ type: 'tool-input-start', id, toolName },
{ type: 'tool-input-end', id },
{ type: 'tool-call', toolCallId: id, toolName, input: JSON.stringify(input) },
{ type: 'finish', finishReason: { unified: 'tool-calls', raw: 'tool_use' }, usage },
];
const text = (body: string): LanguageModelV4StreamPart[] => [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
function recorder() {
const seen: LanguageModelV4CallOptions[] = [];
const model = new MockLanguageModelV4({
doStream: async (o) => {
seen.push(o);
return stream(text('ok'));
},
});
return { seen, model };
}
test('the thinking level reaches the provider call', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', agent: variantByName('deep') });
for await (const _ of session.send('hi')) void _;
expect(seen[0]?.reasoning).toBe('xhigh');
});
test('the quick variant asks for no thinking at all', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', agent: variantByName('quick') });
for await (const _ of session.send('hi')) void _;
expect(seen[0]?.reasoning).toBe('none');
});
test('switching the agent mid-session changes the next call', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny' });
for await (const _ of session.send('first')) void _;
expect(seen[0]?.reasoning).toBe('medium');
session.setAgent(variantByName('deep')!);
for await (const _ of session.send('second')) void _;
expect(seen[1]?.reasoning).toBe('xhigh');
});
test('a read-only variant hides the mutating tools from the model', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', agent: variantByName('plan') });
for await (const _ of session.send('investigate')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('read_file');
expect(offered).toContain('grep');
expect(offered).not.toContain('write_file');
expect(offered).not.toContain('edit_file');
expect(offered).not.toContain('bash');
});
test('the default variant offers everything', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny' });
for await (const _ of session.send('go')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('bash');
expect(offered).toContain('write_file');
});
test('the variant appendix reaches the system prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', agent: variantByName('review') });
for await (const _ of session.send('review it')) void _;
expect(JSON.stringify(seen[0]?.prompt.find((m) => m.role === 'system'))).toContain('reviewing code');
});
test('a variant maxSteps overrides the session default', async () => {
const { model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', agent: variantByName('quick'), maxSteps: 99 });
expect(session.agent().maxSteps).toBe(12);
});
test('the skill catalogue and skill tool are offered when skills are loaded', async () => {
const { seen, model } = recorder();
const skills = await loadSkills(process.cwd());
const session = new Session({ model, askApproval: async () => 'deny', skills });
for await (const _ of session.send('hi')) void _;
expect((seen[0]?.tools ?? []).map((t) => t.name)).toContain('skill');
const system = JSON.stringify(seen[0]?.prompt.find((m) => m.role === 'system'));
expect(system).toContain('debug');
expect(system).toContain('Skills available');
});
test('no skill tool is offered when there are no skills', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', skills: [] });
for await (const _ of session.send('hi')) void _;
expect((seen[0]?.tools ?? []).map((t) => t.name)).not.toContain('skill');
});
test('the skill tool never needs approval', async () => {
let n = 0;
const skills = await loadSkills(process.cwd());
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => (n++ === 0 ? stream(toolCall('c1', 'skill', { name: 'debug' })) : stream(text('loaded'))),
}),
askApproval: async () => {
throw new Error('skill must not prompt');
},
skills,
});
const kinds: string[] = [];
for await (const ev of session.send('debug this')) kinds.push(ev.type);
expect(kinds).toContain('tool-result');
expect(kinds).not.toContain('tool-denied');
});
test('memory tools are offered and never prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', memory: new Memory('/repo-test') });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('remember');
expect(offered).toContain('recall');
expect(offered).toContain('forget');
});
test('plugin tools reach the model and are auto-approved', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', plugins: createHost([timePlugin]) });
for await (const _ of session.send('what time is it')) void _;
expect((seen[0]?.tools ?? []).map((t) => t.name)).toContain('current_time');
});
test('the plugin appendix reaches the system prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', plugins: createHost([guardPlugin]) });
for await (const _ of session.send('hi')) void _;
expect(JSON.stringify(seen[0]?.prompt.find((m) => m.role === 'system'))).toContain('refuses irreversible');
});
test('the guard blocks a destructive bash call without asking the user', async () => {
let asked = 0;
let n = 0;
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () =>
n++ === 0 ? stream(toolCall('c1', 'bash', { command: 'rm -rf /' })) : stream(text('understood')),
}),
askApproval: async () => {
asked++;
return 'once';
},
plugins: createHost([guardPlugin]),
});
const events: string[] = [];
const notices: string[] = [];
for await (const ev of session.send('clean up')) {
events.push(ev.type);
if (ev.type === 'notice') notices.push(ev.text);
}
expect(asked).toBe(0);
expect(events).toContain('notice');
expect(events).toContain('tool-denied');
expect(notices.join()).toContain('recursive or forced delete');
});
test('a safe bash call still reaches the approval prompt', async () => {
let asked = 0;
let n = 0;
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => (n++ === 0 ? stream(toolCall('c1', 'bash', { command: 'echo hi' })) : stream(text('done'))),
}),
askApproval: async () => {
asked++;
return 'once';
},
plugins: createHost([guardPlugin]),
});
for await (const _ of session.send('say hi')) void _;
expect(asked).toBe(1);
});
test('afterTurn fires once the turn ends', async () => {
let fired = 0;
const { model } = recorder();
const session = new Session({
model,
askApproval: async () => 'deny',
plugins: createHost([{ name: 'counter', description: '', afterTurn: () => void fired++ }]),
});
for await (const _ of session.send('hi')) void _;
expect(fired).toBe(1);
});
test('the git tools are offered by default and never prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny' });
for await (const _ of session.send('what changed')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
for (const name of GIT_TOOL_NAMES) expect(offered).toContain(name);
for (const name of GIT_TOOL_NAMES) expect(MUTATING_TOOLS as readonly string[]).not.toContain(name);
});
test('a disabled tool set reaches neither the wire nor the prompt', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: [] });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('read_file');
expect(offered).toContain('bash');
expect(offered).not.toContain('git_status');
expect(offered).not.toContain('multi_edit');
expect(offered).not.toContain('list_dir');
const system = JSON.stringify(seen[0]?.prompt.find((m) => m.role === 'system'));
expect(system).not.toContain('git_status');
expect(system).not.toContain('multi_edit');
});
test('an enabled set is offered while the others stay withheld', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: ['git'] });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('git_diff');
expect(offered).not.toContain('multi_edit');
});
test('core is never withheld, whatever the config says', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: ['git'] });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
for (const name of TOOL_SETS.core) expect(offered).toContain(name);
});
test('session tools survive tool-set gating, since they are not part of that budget', async () => {
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', toolSets: [], memory: new Memory('/repo-test') });
for await (const _ of session.send('hi')) void _;
const offered = (seen[0]?.tools ?? []).map((t) => t.name);
expect(offered).toContain('todo_write');
expect(offered).toContain('remember');
});
test('toolSetOf names the set a tool came from, and nothing for a session tool', () => {
expect(toolSetOf('read_file')).toBe('core');
expect(toolSetOf('multi_edit')).toBe('edit-plus');
expect(toolSetOf('git_log')).toBe('git');
expect(toolSetOf('todo_write')).toBeUndefined();
});
test('isToolSetName accepts the real sets only, so a typo in config is ignored', () => {
for (const name of TOOL_SET_NAMES) expect(isToolSetName(name)).toBe(true);
expect(isToolSetName('gti')).toBe(false);
});