Merge remote-tracking branch 'refs/remotes/upstream/main'
ci / check (macos-latest) (push) Canceled after 0s
ci / check (ubuntu-latest) (push) Canceled after 0s
ci / check (windows-latest) (push) Canceled after 0s

# Conflicts:
#	ROADMAP.md
#	TODO.md
#	src/cli.tsx
#	src/config.ts
#	src/mcp.ts
#	src/permission.ts
#	src/prompt.ts
#	src/session.ts
#	src/snapshot.ts
#	src/subagent.ts
#	src/tools-extra.ts
#	src/tools.ts
#	src/ui/App.tsx
#	test/mcp.test.ts
#	test/prune.test.ts
#	test/session.test.ts
#	test/tools.test.ts
This commit is contained in:
asepharyana
2026-09-21 20:43:35 +07:00
59 changed files with 3481 additions and 463 deletions
+156
View File
@@ -0,0 +1,156 @@
import { afterAll, beforeAll, expect, test } from 'bun:test';
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { VERSION } from '../src/version';
/**
* The CLI entry point is an executable, not a module: importing it parses argv,
* loads config, connects MCP, and renders Ink. Nothing in `test/` imports it for
* that reason, and its flag surface was previously exercised only by hand.
*
* So it is tested the way a user meets it — by running the real entry point in a
* child process — with two things held constant. `SHIRO_HOME` points at a scratch
* directory, and every API-key variable is stripped, so the "no key" branches are
* deterministic rather than accidentally satisfied by the developer's shell. The
* entry is invoked by absolute path from a scratch cwd, so the repository's own
* `config.json`, `AGENTS.md`, and `.shiro/` entries stay out of the run.
*
* Only paths that exit before `render()` are covered; anything past it needs a
* TTY. That still reaches every argument-parsing decision, which is the point.
*/
const ROOT = join(import.meta.dir, '..');
const ENTRY = join(ROOT, 'src', 'cli.tsx');
/** Credential variables that would otherwise decide `cfg.apiKey` for us. */
const KEY_VARS = [
'SHIRO_API_KEY',
'ANTHROPIC_API_KEY',
'OPENAI_API_KEY',
'SHIRO_PROVIDER',
'SHIRO_MODEL',
'SHIRO_BASE_URL',
];
let home: string;
let scratch: string;
beforeAll(() => {
home = mkdtempSync(join(tmpdir(), 'shiro-cli-home-'));
scratch = mkdtempSync(join(tmpdir(), 'shiro-cli-cwd-'));
});
afterAll(() => {
rmSync(home, { recursive: true, force: true });
rmSync(scratch, { recursive: true, force: true });
});
function cleanEnv(shiroHome: string): Record<string, string> {
// Bun.spawn's `env` replaces the environment, so start from a copy of ours and
// remove the credentials rather than hand-assembling a minimal one.
const env: Record<string, string> = {};
for (const [k, v] of Object.entries(process.env)) if (v !== undefined) env[k] = v;
for (const k of KEY_VARS) delete env[k];
env['SHIRO_HOME'] = shiroHome;
return env;
}
async function cli(args: string[], shiroHome = home): Promise<{ stdout: string; stderr: string; code: number }> {
const proc = Bun.spawn([process.execPath, ENTRY, ...args], {
cwd: scratch,
env: cleanEnv(shiroHome),
// /dev/null, so `readStdin` sees EOF instead of blocking on a pipe nobody closes.
stdin: 'ignore',
stdout: 'pipe',
stderr: 'pipe',
});
const [stdout, stderr, code] = await Promise.all([
new Response(proc.stdout).text(),
new Response(proc.stderr).text(),
proc.exited,
]);
return { stdout, stderr, code };
}
test('--help prints usage and exits 0', async () => {
const { stdout, code } = await cli(['--help']);
expect(code).toBe(0);
expect(stdout).toContain('usage: shiro');
expect(stdout).toContain('--yolo');
expect(stdout).toContain('--print');
expect(stdout).toContain('--agent');
});
test('-h is the same as --help', async () => {
const { stdout, code } = await cli(['-h']);
expect(code).toBe(0);
expect(stdout).toContain('usage: shiro');
});
test('--version prints the version line and exits 0', async () => {
const { stdout, code } = await cli(['--version']);
expect(code).toBe(0);
expect(stdout).toContain(`shiro-neko ${VERSION}`);
expect(stdout).toContain(process.platform);
});
test('-v is the same as --version', async () => {
const { stdout, code } = await cli(['-v']);
expect(code).toBe(0);
expect(stdout).toContain(`shiro-neko ${VERSION}`);
});
test('an unknown --agent fails with the valid list', async () => {
const { stderr, code } = await cli(['--agent', 'turbo']);
expect(code).toBe(1);
expect(stderr).toContain('Unknown agent "turbo"');
expect(stderr).toContain('default, quick, deep, plan, review');
});
test('an unknown --think fails with the valid list', async () => {
const { stderr, code } = await cli(['--think', 'turbo']);
expect(code).toBe(1);
expect(stderr).toContain('Unknown thinking level "turbo"');
expect(stderr).toContain('off, low, medium, high, max');
});
test('-p without a key refuses to run headless', async () => {
const { stderr, code } = await cli(['-p', 'hello']);
expect(code).toBe(1);
expect(stderr).toContain('No API key for provider "anthropic"');
});
test('--resume with no matching session fails', async () => {
const { stderr, code } = await cli(['-r', 'nosuchid']);
expect(code).toBe(1);
expect(stderr).toContain('no session matching "nosuchid"');
});
test('--continue with no saved session fails', async () => {
const { stderr, code } = await cli(['-c']);
expect(code).toBe(1);
expect(stderr).toContain('no saved session for this directory');
});
test('a configured key with no prompt and no stdin reports the missing prompt', async () => {
// A second home, so the key file this writes does not leak into the tests above.
const keyed = mkdtempSync(join(tmpdir(), 'shiro-cli-keyed-'));
try {
mkdirSync(join(keyed, '.shiro-neko'), { recursive: true });
await Bun.write(
join(keyed, '.shiro-neko', 'config.json'),
`${JSON.stringify({ provider: 'anthropic', model: 'claude-sonnet-4-5', apiKey: 'sk-test' }, null, 2)}\n`,
);
// The full isolation flag set is passed so this also proves they all parse and
// the module boots with them; with an empty stdin, `-p` must report the prompt.
const { stderr, code } = await cli(
['-p', '--no-plugins', '--no-skills', '--no-memory', '--no-instructions', '--no-mcp', '--no-subagent'],
keyed,
);
expect(code).toBe(1);
expect(stderr).toContain('needs a prompt');
} finally {
rmSync(keyed, { recursive: true, force: true });
}
});
+6 -20
View File
@@ -1,7 +1,7 @@
import { usageOf } from './helpers';
import { generateResult, streamOf, textChunks, usageOf } from './helpers';
import { afterEach, beforeEach, expect, test } from 'bun:test';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { MockLanguageModelV4 } from 'ai/test';
import type { LanguageModelV4CallOptions } from '@ai-sdk/provider';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
@@ -15,16 +15,7 @@ import { GIT_TOOL_NAMES } from '../src/tools-git';
const usage = usageOf(10);
const stream = (parts: LanguageModelV4StreamPart[]) => ({
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
});
const text = (body: string): LanguageModelV4StreamPart[] => [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
const text = (body: string) => textChunks(body, usage);
let dir: string;
let origCwd: string;
@@ -69,16 +60,11 @@ function recordingModel(
const model = new MockLanguageModelV4({
doStream: async (o) => {
seen.push(o);
return stream(text(reply));
return streamOf(text(reply));
},
doGenerate: async (o) => {
seen.push(o);
return {
content: [{ type: 'text', text: reply }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
} as any;
return generateResult(reply, usage);
},
});
return { model, seen };
+104 -22
View File
@@ -1,25 +1,19 @@
import { usageOf } from './helpers';
import { generateResult, streamOf, textChunks, usageOf } from './helpers';
import { expect, test } from 'bun:test';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import type { LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { APICallError, type ModelMessage } from 'ai';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { Session, type AgentEvent } from '../src/session';
import { isPrunedSpanSummary, PRUNED_SPAN_PREFIX } from '../src/prune';
const usage = usageOf(10);
const stream = (parts: LanguageModelV4StreamPart[]) => ({
stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }),
});
const stream = (parts: LanguageModelV4StreamPart[]) => streamOf(parts);
const text = (body: string): LanguageModelV4StreamPart[] => [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
const text = (body: string) => textChunks(body, usage);
/** One assistant turn carrying a bulky tool call plus its result. */
function bulkyExchange(i: number): ModelMessage[] {
@@ -86,7 +80,9 @@ test('history over the threshold is pruned before reaching the model', async ()
const sentSize = JSON.stringify(seen[0]?.prompt).length;
expect(sentSize).toBeLessThan(JSON.stringify(messages).length);
// Pruning is for the wire only; the local history keeps every message.
// Pruning is for the wire only; the local history keeps every message. A
// history left at its full size pins the context meter, but re-pruning from
// scratch every turn is the cost of keeping a complete local record.
expect(session.messages.length).toBeGreaterThan(messages.length);
expect(events).toContain('compacted');
});
@@ -132,12 +128,7 @@ test('summarize replaces the whole history with one summary message', async () =
model: new MockLanguageModelV4({
doGenerate: async () => {
generateCalls++;
return {
content: [{ type: 'text', text: '- goal: add pagination\n- touched: src/users.ts\n- todo: add tests' }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
} as any;
return generateResult('- goal: add pagination\n- touched: src/users.ts\n- todo: add tests', usage);
},
}),
askApproval: async () => 'deny',
@@ -156,7 +147,7 @@ test('summarize replaces the whole history with one summary message', async () =
test('summarize on an empty session is a no-op', async () => {
const session = new Session({
model: new MockLanguageModelV4({ doGenerate: async () => ({}) as any }),
model: new MockLanguageModelV4({ doGenerate: async () => generateResult('') }),
askApproval: async () => 'deny',
});
expect(await session.summarize()).toEqual({ before: 0, after: 0 });
@@ -379,7 +370,7 @@ test('lossless compaction appends a retained note when tool content was dropped'
for await (const ev of session.send('next')) events.push(ev);
expect(generateCalls).toBe(1);
expect(events.some((e) => e.type === 'compacted')).toBe(true);
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(true);
expect(session.messages.some((m) => m.role === 'user' && String(m.content).startsWith('Earlier in this session, now compacted away:'))).toBe(true);
});
test('no retained note when history fits', async () => {
@@ -399,7 +390,7 @@ test('no retained note when history fits', async () => {
});
for await (const _ of session.send('next')) void _;
expect(generateCalls).toBe(0);
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(false);
expect(session.messages.some((m) => m.role === 'user' && String(m.content).startsWith('Earlier in this session, now compacted away:'))).toBe(false);
});
test('a failing retained-note model does not break the turn', async () => {
@@ -418,7 +409,8 @@ test('a failing retained-note model does not break the turn', async () => {
for await (const ev of session.send('next')) events.push(ev);
expect(events.map((e) => e.type)).toContain('done');
expect(events.map((e) => e.type)).not.toContain('error');
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(false);
// A failing summarizer still leaves the digest fallback — the turn is not broken.
expect(session.messages.some((m) => m.role === 'user' && String(m.content).startsWith('Earlier in this session, now compacted away:'))).toBe(true);
});
const staleItem = (id: string) =>
@@ -514,3 +506,93 @@ test('a stale item arriving after text was streamed is reported, not silently re
expect(call).toBe(1);
});
test('a pruned span is replaced by a summary of what was dropped', async () => {
const messages = [
{ role: 'user' as const, content: 'we decided on the ladder approach' },
...bulkyExchange(0),
...bulkyExchange(1),
...bulkyExchange(2),
...bulkyExchange(3),
];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => generateResult('chose the ladder; touched f0-f3.ts'),
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
const span = session.messages.filter((m) => isPrunedSpanSummary(m));
expect(span).toHaveLength(1);
expect(String(span[0]?.content)).toContain('chose the ladder');
});
test('the span summary is injected before the surviving history, not after', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => generateResult('summary of the dropped span'),
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
const markerAt = session.messages.findIndex((m) => isPrunedSpanSummary(m));
expect(markerAt).toBe(0);
});
test('a summarizer that throws still leaves a digest rather than nothing', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => {
throw new Error('summarizer is down');
},
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
const span = session.messages.find((m) => isPrunedSpanSummary(m));
expect(span).toBeDefined();
expect(String(span?.content)).toContain('f0.ts');
});
test('repeated compaction does not nest one span summary inside another', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => generateResult('span notes'),
}),
askApproval: async () => 'deny',
messages: [...messages],
compactThreshold: 1000,
});
for await (const _ of session.send('next')) void _;
for await (const _ of session.send('again')) void _;
for await (const _ of session.send('and again')) void _;
const spans = session.messages.filter((m) => isPrunedSpanSummary(m));
// One summary is kept from each compaction, but no summary may contain the
// marker text, which would mean a summary of a summary.
for (const s of spans) {
const body = String(s.content).slice(PRUNED_SPAN_PREFIX.length);
expect(body).not.toContain(PRUNED_SPAN_PREFIX);
}
});
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session';
import { App, createApprovalBridge } from '../src/ui/App';
import { testHooks, usageOf } from './helpers';
import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3);
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
doStream: async () => streamOf(textChunks('reply', usage)),
});
const paths = ['README.md', 'src/app.ts', 'src/session.ts', 'src/ui/App.tsx', 'test/session.test.ts'];
+7 -12
View File
@@ -1,7 +1,7 @@
import { usageOf } from './helpers';
import { generateResult, textChunks, usageOf } from './helpers';
import { expect, test } from 'bun:test';
import { APICallError } from 'ai';
import type { LanguageModelV4, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import type { LanguageModelV4, LanguageModelV4CallOptions, LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { simulateReadableStream } from 'ai/test';
import { withFallback, type FallbackEvent } from '../src/fallback';
@@ -9,12 +9,7 @@ const usage = usageOf(1);
const okStream = (body: string) => ({
stream: simulateReadableStream<LanguageModelV4StreamPart>({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunks: textChunks(body, usage),
chunkDelayInMs: null,
initialDelayInMs: null,
}),
@@ -34,7 +29,7 @@ function model(name: string, behaviour: () => Promise<any>): LanguageModelV4 {
};
}
const opts = { prompt: [] } as any;
const opts: LanguageModelV4CallOptions = { prompt: [] };
const REAL_MESSAGE =
"Function tools with reasoning_effort are not supported for gpt-5.6-sol in /v1/chat/completions. To use function tools, use /v1/responses or set reasoning_effort to 'none'.";
@@ -93,11 +88,11 @@ test('doGenerate falls back on the same condition as doStream', async () => {
model('chat', async () => {
throw apiError(400, REAL_MESSAGE);
}),
model('responses', async () => ({ content: [{ type: 'text', text: 'ok' }] })),
model('responses', async () => generateResult('ok', usage)),
]);
const out = (await wrapped.doGenerate(opts)) as any;
expect(out.content[0].text).toBe('ok');
const out = await wrapped.doGenerate(opts);
expect(out.content[0]).toMatchObject({ type: 'text', text: 'ok' });
});
test('a 401 is not a shape mismatch, so it propagates untouched', async () => {
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { testHooks, usageOf } from './helpers';
import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3);
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
doStream: async () => streamOf(textChunks('reply', usage)),
});
function mount(over: Partial<AppHooks> = {}) {
+47 -1
View File
@@ -1,4 +1,10 @@
import type { LanguageModelV4Usage } from '@ai-sdk/provider';
import { simulateReadableStream } from 'ai/test';
import type {
LanguageModelV4GenerateResult,
LanguageModelV4StreamPart,
LanguageModelV4StreamResult,
LanguageModelV4Usage,
} from '@ai-sdk/provider';
import type { AppHooks } from '../src/ui/App';
/**
@@ -15,6 +21,46 @@ export function usageOf(input: number, output = 1): LanguageModelV4Usage {
};
}
/**
* A `doStream` return value built from `simulateReadableStream` chunks.
*
* Written out as `{ stream: simulateReadableStream(...) } as any` in most UI test
* files, because the SDK's `ReadableStream` and `simulateReadableStream`'s type do
* not unify. This is the one place that difference is absorbed.
*/
export function streamOf(chunks: LanguageModelV4StreamPart[]): LanguageModelV4StreamResult {
return {
stream: simulateReadableStream({
chunks,
chunkDelayInMs: null,
initialDelayInMs: null,
}),
} as unknown as LanguageModelV4StreamResult;
}
/** The text and finish chunks that make a stream say one thing and stop. */
export function textChunks(body: string, usage: LanguageModelV4Usage = usageOf(1)): LanguageModelV4StreamPart[] {
return [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
}
/**
* A `doGenerate` return value. `warnings` is required by the SDK type but empty in
* every test, so it is filled here rather than repeated at each call site.
*/
export function generateResult(body: string, usage: LanguageModelV4Usage = usageOf(1)): LanguageModelV4GenerateResult {
return {
content: [{ type: 'text', text: body }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
};
}
/** Default AppHooks for UI tests; override only what a test cares about. */
export function testHooks(over: Partial<AppHooks> = {}): AppHooks {
return {
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { testHooks, usageOf } from './helpers';
import { testHooks, textChunks, usageOf, streamOf } from './helpers';
const usage = usageOf(1000, 500);
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
doStream: async () => streamOf(textChunks('reply', usage)),
});
function mount(over: Partial<AppHooks> = {}) {
+3 -15
View File
@@ -1,29 +1,17 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import { parseCommand, COMMANDS, HELP } from '../src/commands';
import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { invalidName, parseHeaders, splitArgs, McpAdd } from '../src/ui/McpAdd';
import { testHooks, usageOf } from './helpers';
import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3);
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'ok' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
doStream: async () => streamOf(textChunks('ok', usage)),
});
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
+68 -7
View File
@@ -21,6 +21,7 @@ test('a stdio server contributes its tools under an mcp__ namespace', async () =
try {
expect(Object.keys(mcp.tools).sort()).toEqual(['mcp_call', 'mcp_inspect', 'mcp_list']);
expect(mcp.errors).toEqual([]);
expect(mcp.servers).toEqual(['stub']);
} finally {
await mcp.close();
}
@@ -48,22 +49,32 @@ test('two servers exposing the same tool name do not shadow each other', async (
}, 30_000);
test('a server that fails to start is reported, not fatal', async () => {
const mcp = await connectMcp({
ok: stdioServer(),
broken: { command: 'definitely-not-a-real-binary-xyz' },
});
const mcp = await connectMcp(
{
ok: stdioServer(),
broken: { command: 'definitely-not-a-real-binary-xyz' },
},
'eager',
);
try {
expect(Object.keys(mcp.tools).sort()).toEqual(['mcp_call', 'mcp_inspect', 'mcp_list']);
// Eager mode registers the ok server's tools directly alongside the meta-tools.
const keys = Object.keys(mcp.tools).sort();
expect(keys).toContain('mcp__ok__ping');
expect(keys).toContain('mcp_call');
expect(mcp.errors.map((e) => e.server)).toEqual(['broken']);
expect(mcp.errors[0]?.message).toBeTruthy();
// Eager mode names no servers for the prompt's lazy line.
expect(mcp.servers).toEqual([]);
} finally {
await mcp.close();
}
}, 30_000);
test('no configured servers yields no tools and no errors', async () => {
test('no configured servers leaves the meta-tools naming none, with no errors', async () => {
const mcp = await connectMcp({});
expect(mcp.tools).toEqual({});
// No servers: no direct tools, but the three meta-tools are still registered.
expect(Object.keys(mcp.tools).sort()).toEqual(['mcp_call', 'mcp_inspect', 'mcp_list']);
expect(mcp.servers).toEqual([]);
expect(mcp.errors).toEqual([]);
await mcp.close();
});
@@ -80,3 +91,53 @@ test('an http server config is attempted and its failure reported', async () =>
expect(mcp.errors.map((e) => e.server)).toEqual(['remote']);
await mcp.close();
}, 30_000);
test('lazy mode (default) registers only meta-tools, never server schemas', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
// The request is spared every server schema: only three meta-tools exist.
expect(Object.keys(mcp.tools).sort()).toEqual(['mcp_call', 'mcp_inspect', 'mcp_list']);
// But the prompt knows which servers are connected.
expect(mcp.servers).toEqual(['stub']);
} finally {
await mcp.close();
}
}, 30_000);
test('mcp_list names a server tools and mcp_inspect reads a schema without calling it', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
await call(mcp.tools, 'mcp_list', { server: 'stub' }).then((out) =>
expect(JSON.stringify(out)).toContain('search'),
);
await call(mcp.tools, 'mcp_inspect', { server: 'stub', tool: 'ping' }).then((out) =>
expect(JSON.stringify(out)).toContain('note'),
);
} finally {
await mcp.close();
}
}, 30_000);
test('mcp_call executes a server tool by name', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
const out = await call(mcp.tools, 'mcp_call', { server: 'stub', tool: 'ping', arguments: { note: 'lazy' } });
expect(JSON.stringify(out)).toContain('pong: lazy');
} finally {
await mcp.close();
}
}, 30_000);
test('mcp_call against an unknown server names the live set', async () => {
const mcp = await connectMcp({ stub: stdioServer() });
try {
await call(mcp.tools, 'mcp_call', { server: 'nope', tool: 'ping', arguments: {} }).then(
() => {
throw new Error('an unknown server must reject');
},
(e) => expect(String(e)).toContain('nope'),
);
} finally {
await mcp.close();
}
}, 30_000);
+2 -11
View File
@@ -1,4 +1,4 @@
import { usageOf } from './helpers';
import { generateResult, usageOf } from './helpers';
import { afterEach, beforeEach, expect, test } from 'bun:test';
import { MockLanguageModelV4 } from 'ai/test';
import { mkdtempSync, rmSync } from 'node:fs';
@@ -31,16 +31,7 @@ const call = (tools: ToolSet, name: string, input: Record<string, unknown>) => {
const usage = usageOf(1);
const summarizer = (text: string) =>
new MockLanguageModelV4({
doGenerate: async () =>
({
content: [{ type: 'text', text }],
finishReason: { unified: 'stop', raw: 'stop' },
usage,
warnings: [],
}) as any,
});
const summarizer = (text: string) => new MockLanguageModelV4({ doGenerate: async () => generateResult(text, usage) });
test('a fresh project has no memory and renders nothing', async () => {
const m = new Memory('/repo');
+3 -15
View File
@@ -1,27 +1,15 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { testHooks, usageOf } from './helpers';
import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(2);
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'answer' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
doStream: async () => streamOf(textChunks('answer', usage)),
});
function mount(over: Partial<AppHooks> = {}) {
+200
View File
@@ -0,0 +1,200 @@
import { afterAll, afterEach, beforeEach, expect, spyOn, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import type { Config } from '../src/config';
import { Onboard, type OnboardResult } from '../src/ui/Onboard';
import * as providers from '../src/providers';
/**
* The provider wizard had no test because its every path but the first ends at a
* network call. That call is `fetchModels`, which uses the global `fetch`, so the
* suite stubs it: the picking, the env-key shortcut, the manual-entry fallback,
* and the empty-list warning are all reachable offline and deterministic.
*
* `current` decides the starting row, so a test selects a preset by naming it
* rather than counting arrow presses.
*/
const DOWN = '\u001B[B';
const ENTER = '\r';
const ESC = '\u001B';
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
const ORIGINAL_FETCH = globalThis.fetch;
/** Answers `GET /models` with the given ids; the wizard sorts them itself. */
function stubModels(ids: string[]): string[] {
const calls: string[] = [];
globalThis.fetch = (async (input: unknown) => {
calls.push(String(input));
return new Response(JSON.stringify({ data: ids.map((id) => ({ id })) }), {
status: 200,
headers: { 'content-type': 'application/json' },
});
}) as unknown as typeof fetch;
return calls;
}
function stubFailure(status: number): void {
globalThis.fetch = (async () => new Response('boom', { status })) as unknown as typeof fetch;
}
beforeEach(() => {
// The wizard reads the preset's env key to skip the api-key step; a developer's
// shell must not decide which branch a test takes.
delete process.env['ANTHROPIC_API_KEY'];
delete process.env['OPENAI_API_KEY'];
// A developer's ~/.claude/settings.json must not decide the branch either:
// the custom presets fall back to it for a base URL, and on a configured box
// that would swap the deterministic stub below for a live provider request.
spyOn(providers, 'readClaudeCodeSettings').mockResolvedValue({});
});
afterEach(() => {
globalThis.fetch = ORIGINAL_FETCH;
delete process.env['ANTHROPIC_API_KEY'];
delete process.env['OPENAI_API_KEY'];
});
afterAll(() => {
globalThis.fetch = ORIGINAL_FETCH;
});
async function press(app: ReturnType<typeof render>, s: string, ms = 80) {
app.stdin.write(s);
await wait(ms);
}
async function type(app: ReturnType<typeof render>, s: string) {
for (const ch of s) await press(app, ch, 20);
}
function mount(current: Partial<Config> = {}) {
const done: OnboardResult[] = [];
let cancelled = 0;
const app = render(
<Onboard
current={{ provider: 'anthropic', model: 'claude-sonnet-4-5', ...current }}
onDone={(r) => done.push(r)}
onCancel={() => void cancelled++}
/>,
);
return { app, done, cancelled: () => cancelled };
}
test('the first screen lists providers and marks the configured one', async () => {
const { app, cancelled } = mount({ presetId: 'openai' });
await wait(80);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('Choose a provider');
expect(frame).toContain('Anthropic');
expect(frame).toContain('OpenAI (current)');
expect(frame).toContain('Custom OpenAI-compatible endpoint');
await press(app, ESC, 120);
expect(cancelled()).toBe(1);
app.unmount();
}, 20_000);
test('esc cancels from the api-key step', async () => {
// custom-openai has no env key, so it stops for a key rather than calling out.
const { app, cancelled } = mount({ presetId: 'custom-openai' });
await wait(80);
await press(app, ENTER, 120);
await type(app, 'https://ex.com/v1');
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('API key');
await press(app, ESC, 120);
expect(cancelled()).toBe(1);
app.unmount();
}, 20_000);
test('a custom endpoint collects url, key, and a chosen model', async () => {
const calls = stubModels(['m2', 'm1']);
const { app, done } = mount({ presetId: 'custom-openai' });
await wait(80);
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('endpoint URL');
await type(app, 'https://ex.com/v1');
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('API key');
await type(app, 'sk-secret-key');
await press(app, ENTER, 200);
expect(app.lastFrame()).toContain('choose a model');
// The list arrives sorted, so the first row is m1, not the m2 it was given.
await press(app, ENTER, 150);
expect(calls).toEqual(['https://ex.com/v1/models']);
expect(done).toEqual([
{
presetId: 'custom-openai',
provider: 'openai',
baseURL: 'https://ex.com/v1',
apiKey: 'sk-secret-key',
model: 'm1',
},
]);
app.unmount();
}, 20_000);
test('an env key skips the key step and the hint masks it', async () => {
process.env['OPENAI_API_KEY'] = 'sk-abcdefgh1234';
stubModels(['gpt-5', 'gpt-5-mini']);
const { app } = mount({ presetId: 'openai' });
await wait(80);
await press(app, ENTER, 200);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('choose a model');
expect(frame).toContain('sk-a...1234');
app.unmount();
}, 20_000);
test('the manual entry collects a model id the list does not offer', async () => {
// The env key is what skips the key step; without it the wizard would stop there.
process.env['OPENAI_API_KEY'] = 'sk-manual-test-key';
stubModels(['only-one']);
const { app, done } = mount({ presetId: 'openai' });
await wait(80);
await press(app, ENTER, 200);
expect(app.lastFrame()).toContain('choose a model');
await press(app, DOWN, 100); // one model, then the manual-entry row
await press(app, ENTER, 120);
expect(app.lastFrame()).toContain('model id');
await type(app, 'my-model');
await press(app, ENTER, 150);
expect(done).toHaveLength(1);
expect(done[0]?.model).toBe('my-model');
app.unmount();
}, 20_000);
test('a server that cannot list models falls through to the warning', async () => {
// custom-openai has no fallback list, so an empty response leaves nothing to pick.
stubFailure(500);
const { app } = mount({ presetId: 'custom-openai' });
await wait(80);
await press(app, ENTER, 120);
await type(app, 'https://ex.com/v1');
await press(app, ENTER, 120);
await type(app, 'sk-x');
await press(app, ENTER, 250);
const frame = app.lastFrame() ?? '';
expect(frame).toContain('could not list models');
expect(frame).toContain('model id');
app.unmount();
}, 20_000);
+279
View File
@@ -0,0 +1,279 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React, { useState } from 'react';
import { PromptInput, type PromptInputProps } from '../src/ui/PromptInput';
/**
* `PromptInput` is exercised through `App` in `input.test.tsx` (history recall,
* arrow and word motion, ctrl-u, ctrl-d, paste). This file renders it directly
* for the behaviours that reaching it through the whole app made awkward: the
* emacs kill keys, the mask, a blurred input, and the `onKey` escape hatch that
* `App` uses to hand up/down/tab/esc to its own menus.
*/
const wait = (ms: number) => new Promise((r) => setTimeout(r, ms));
/** Keys that are awkward to read as escape sequences at the call site. */
const KEYS = {
up: '\u001B[A',
down: '\u001B[B',
left: '\u001B[D',
ctrlA: '\u0001',
ctrlD: '\u0004',
ctrlE: '\u0005',
ctrlK: '\u000B',
ctrlU: '\u0015',
ctrlW: '\u0017',
backspace: '\u007F',
} as const;
/**
* The input is controlled, so it needs an owner to feed `value` back. This wraps
* it with the state a caller would hold and records every value it reports.
*/
type HarnessProps = Omit<Partial<PromptInputProps>, 'value' | 'onChange' | 'onSubmit'> & {
initial?: string;
onChange?: (value: string, cursor: number) => void;
onSubmit?: (value: string) => void;
};
function Harness({ initial = '', onChange, onSubmit, ...rest }: HarnessProps) {
const [value, setValue] = useState(initial);
return (
<PromptInput
{...rest}
value={value}
onSubmit={onSubmit ?? (() => {})}
onChange={(next, cursor) => {
onChange?.(next, cursor);
setValue(next);
}}
/>
);
}
function mount(props: HarnessProps = {}) {
const changes: { value: string; cursor: number }[] = [];
const submitted: string[] = [];
const app = render(
<Harness
{...props}
onChange={(value, cursor) => changes.push({ value, cursor })}
onSubmit={(v) => submitted.push(v)}
/>,
);
return { app, changes, submitted };
}
async function press(app: ReturnType<typeof render>, s: string, ms = 80) {
app.stdin.write(s);
await wait(ms);
}
/** The visible text with every SGR sequence removed, i.e. what the user actually reads. */
const plain = (frame: string) => frame.replace(/\u001B\[[0-9;]*m/g, '');
test('the rendered line carries no hand-written SGR escapes', async () => {
// The cursor must be Ink's `inverse` prop, not `\u001B[7m` pasted into the string.
// Ink measures string content as printable columns, so an embedded escape is
// counted as text and shifts the line — the stray `t` before `ype` in the
// placeholder, and a corrupted cell wherever the line wraps.
const { app } = mount({ initial: 'abc', placeholder: `type to queue${String.fromCharCode(0x2026)}` });
await wait(40);
const frame = app.lastFrame() ?? '';
expect(frame).not.toContain('\u001B[7m');
expect(frame).not.toContain('\u001B[27m');
expect(plain(frame)).toBe('abc');
app.unmount();
});
test('the placeholder renders as one unbroken run of text', async () => {
const { app } = mount({ placeholder: 'ask anything' });
await wait(40);
// No escape may sit between the first character and the rest.
expect(plain(app.lastFrame() ?? '')).toBe('ask anything');
app.unmount();
});
test('a focused input keeps the caret off the string', async () => {
const { app } = mount({ initial: 'abc' });
await wait(40);
// At the end of the line the caret is a blank cell, which Ink trims.
expect(plain(app.lastFrame() ?? '')).toBe('abc');
app.unmount();
});
test('a bare blurred input renders no caret at all', async () => {
const { app } = mount({ focus: false });
await wait(40);
const frame = app.lastFrame() ?? '';
expect(frame).not.toContain('\u001B[7m');
expect(frame.trim()).toBe('');
app.unmount();
});
test('the caret marks the first placeholder character while focused', async () => {
const { app } = mount({ placeholder: 'ask anything' });
await wait(40);
expect(plain(app.lastFrame() ?? '')).toBe('ask anything');
app.unmount();
const blurred = mount({ placeholder: 'ask anything', focus: false });
await wait(40);
expect(blurred.app.lastFrame()).toBe('ask anything');
blurred.app.unmount();
});
test('the cursor moves one character at a time with the arrows', async () => {
const { app, changes } = mount({ initial: 'abc' });
await wait(40);
expect(plain(app.lastFrame() ?? '')).toBe('abc');
await press(app, KEYS.left);
await press(app, KEYS.left);
await press(app, 'Z');
expect(app.lastFrame()).toBe('aZbc');
expect(changes.at(-1)?.cursor).toBe(2);
app.unmount();
});
test('a masked input hides the value and keeps its length', async () => {
const { app } = mount({ initial: 'abcd', mask: '*' });
await wait(40);
expect(app.lastFrame()).toBe('****');
app.unmount();
});
test('a mask stays masked as the value shortens', async () => {
const { app } = mount({ initial: 'abcd', mask: '*' });
await wait(40);
// The harness owns the value, so backspace shortens it to three stars.
await press(app, KEYS.backspace);
expect(app.lastFrame()).toBe('***');
app.unmount();
});
test('ctrl-a and ctrl-e move the cursor between the ends of the line', async () => {
const { app } = mount({ initial: 'inline' });
await wait(40);
// Both keys only move the caret, so the proof is where the next character lands.
await press(app, KEYS.ctrlA);
await press(app, 'X');
expect(app.lastFrame()).toBe('Xinline');
await press(app, KEYS.ctrlE);
await press(app, 'Y');
expect(app.lastFrame()).toBe('XinlineY');
app.unmount();
});
test('ctrl-k kills from the cursor to the end', async () => {
const { app } = mount({ initial: 'keep this' });
await wait(40);
await press(app, KEYS.ctrlA);
// Three characters in, then kill the tail.
for (let i = 0; i < 3; i++) await press(app, KEYS.left.replace('\u001B[D', '\u001B[C'), 30);
await press(app, KEYS.ctrlK);
expect(app.lastFrame()).toContain('kee');
expect(app.lastFrame()).not.toContain('keep this');
app.unmount();
});
test('ctrl-w deletes the word before the cursor and its padding', async () => {
const { app } = mount({ initial: 'one two three' });
await wait(40);
await press(app, KEYS.ctrlW);
expect(app.lastFrame()).toContain('one two');
expect(app.lastFrame()).not.toContain('three');
await press(app, KEYS.ctrlW);
expect(app.lastFrame()).toContain('one');
expect(app.lastFrame()).not.toContain('two');
app.unmount();
});
test('ctrl-w on a single word leaves an empty line', async () => {
const { app } = mount({ initial: 'lonely' });
await wait(40);
await press(app, KEYS.ctrlW);
expect(app.lastFrame()).not.toContain('lonely');
app.unmount();
});
test('onKey can swallow a key before the input sees it', async () => {
const seen: string[] = [];
const { app, submitted } = mount({
initial: 'draft',
onKey: (input, key) => {
if (key.upArrow) {
seen.push('up');
return true;
}
return false;
},
history: ['from history'],
});
await wait(40);
await press(app, KEYS.up);
expect(seen).toEqual(['up']);
expect(app.lastFrame()).toContain('draft');
expect(app.lastFrame()).not.toContain('from history');
// A key onKey declines still reaches the input.
await press(app, '!');
expect(app.lastFrame()).toContain('draft!');
expect(submitted).toEqual([]);
app.unmount();
});
test('initialCursor puts the caret mid-line, so typing lands there', async () => {
const { app } = mount({ initial: 'abcdef', initialCursor: 2 });
await wait(40);
expect(app.lastFrame()).toBe('abcdef');
await press(app, 'X');
expect(app.lastFrame()).toBe('abXcdef');
app.unmount();
});
test('an external value renders as given and reports no change', async () => {
const changes: { value: string; cursor: number }[] = [];
const app = render(
<PromptInput
value="set from outside"
onChange={(v, c) => changes.push({ value: v, cursor: c })}
onSubmit={() => {}}
/>,
);
await wait(40);
// A prop with a handler that does not feed the value back: the text renders
// exactly as passed and the input reports nothing until a key is pressed.
expect(app.lastFrame()).toBe('set from outside');
expect(changes).toEqual([]);
app.unmount();
});
test('submit reports the current value and resets the cursor', async () => {
const { app, submitted } = mount({ initial: 'send me' });
await wait(40);
await press(app, '\r', 120);
expect(submitted).toEqual(['send me']);
app.unmount();
});
test('typing inserts at the cursor rather than appending', async () => {
const { app } = mount({ initial: 'ac' });
await wait(40);
await press(app, KEYS.left);
await press(app, 'b');
expect(app.lastFrame()).toBe('abc');
app.unmount();
});
+9 -1
View File
@@ -45,6 +45,13 @@ test('mcp tools are grouped with their naming convention explained', () => {
expect(rendered).toContain('needs approval');
});
test('lazy mcp meta-tools prompt the workflow, and mcp__ and meta-tools do not double-list', () => {
const rendered = renderTools(['read_file', 'mcp_list', 'mcp_inspect', 'mcp_call']);
expect(rendered).toContain('mcp_list, mcp_inspect, mcp_call');
expect(rendered).toContain('Never guess a server or tool name');
expect(rendered).not.toContain('mcp__<server>__<tool>');
});
test('an unknown tool is listed rather than silently dropped', () => {
expect(renderTools(['read_file', 'some_plugin_tool'])).toContain('some_plugin_tool');
});
@@ -116,7 +123,8 @@ test('omitting every section leaves no dangling markers', () => {
test('the prompt stays a reasonable size with everything on', () => {
const prompt = systemPrompt({ cwd: '/repo', availableTools: ALL, canAsk: true });
console.log('PROMPT_SIZE', prompt.length, 'BUDGET', 5100 + (TOOL_DOCS.length - 22) * 120);
// Sent on every request, so a runaway prompt is a direct cost. The budget scales
// with the documented-tool count: a new tool earns its own line, the rest must not.
expect(prompt.length).toBeLessThan(5000 + (TOOL_DOCS.length - 22) * 120);
expect(prompt.length).toBeLessThan(5100 + (TOOL_DOCS.length - 22) * 120);
});
+84 -1
View File
@@ -1,6 +1,17 @@
import { expect, test } from 'bun:test';
import type { ModelMessage } from 'ai';
import { detachOrphanedItems, droppedSpan, dropOrphanedResults, estimateTokens, pruneToFit, prunePreservingItems } from '../src/prune';
import {
detachOrphanedItems,
digestOf,
droppedBy,
droppedSpan,
dropOrphanedResults,
estimateTokens,
isPrunedSpanSummary,
prunedSpanMessage,
pruneToFit,
prunePreservingItems,
} from '../src/prune';
const kinds = (messages: ModelMessage[]) =>
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
@@ -471,3 +482,75 @@ test('the user prompt survives even the narrowest rung', () => {
const fitted = pruneToFit({ messages, threshold: 100, estimate });
expect(JSON.stringify(fitted)).toContain('do the thing');
});
test('droppedBy reports the messages a prune discarded, by reference', () => {
const before: ModelMessage[] = [
{ role: 'user', content: 'do the thing' },
{ role: 'assistant', content: [{ type: 'text', text: 'ok' }] },
{ role: 'user', content: 'and again' },
];
const kept = [before[0]!, before[2]!];
const dropped = droppedBy(before, kept);
expect(dropped).toHaveLength(1);
expect(dropped[0]).toBe(before[1]!);
});
test('droppedBy sees through the copies prunePreservingItems returns', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'goal' },
...Array.from({ length: 40 }, (_, i): ModelMessage => ({
role: 'assistant',
content: [
{ type: 'reasoning', text: `step ${i}` },
{ type: 'tool-call', toolCallId: `tc${i}`, toolName: 'grep', input: { pattern: `p${i}` } },
],
})),
];
const pruned = pruneToFit({ messages, threshold: 50, estimate: (m) => JSON.stringify(m).length / 4 });
const dropped = droppedBy(messages, pruned);
// The point is that a value comparison would find nothing: the survivors are
// spread copies. Reference identity is what makes this non-empty.
expect(dropped.length).toBeGreaterThan(0);
expect(dropped.every((m) => !pruned.includes(m))).toBe(true);
});
test('the digest names the tool and its input, which is the decision', () => {
const dropped: ModelMessage[] = [
{ role: 'assistant', content: [{ type: 'tool-call', toolCallId: 't', toolName: 'edit_file', input: { path: 'src/a.ts' } }] },
{ role: 'tool', content: [{ type: 'tool-result', toolCallId: 't', toolName: 'edit_file', output: { type: 'text', value: 'ok' } }] },
];
const digest = digestOf(dropped);
expect(digest).toContain('src/a.ts');
expect(digest).toContain('result');
});
test('a pruned span is injected with a marker that identifies it', () => {
const dropped: ModelMessage[] = [{ role: 'user', content: 'we chose the ladder' }];
const message = prunedSpanMessage('notes from the span', dropped);
expect(message).toBeDefined();
expect(isPrunedSpanSummary(message!)).toBe(true);
expect(message!.content).toContain('notes from the span');
});
test('the digest is used when no summary came back', () => {
const dropped: ModelMessage[] = [{ role: 'user', content: 'we chose the ladder' }];
const message = prunedSpanMessage(undefined, dropped);
expect(message!.content).toContain('we chose the ladder');
expect(isPrunedSpanSummary(message!)).toBe(true);
});
test('an empty span produces no message rather than an empty one', () => {
expect(prunedSpanMessage(undefined, [])).toBeUndefined();
expect(prunedSpanMessage(' ', [{ role: 'user', content: '' }])).toBeUndefined();
});
test('a span summary is not treated as an ordinary message', () => {
const ordinary: ModelMessage = { role: 'user', content: 'a real question' };
expect(isPrunedSpanSummary(ordinary)).toBe(false);
expect(isPrunedSpanSummary({ role: 'assistant', content: 'Earlier in this session, now compacted away:' })).toBe(false);
});
+3 -15
View File
@@ -1,28 +1,16 @@
import { expect, test } from 'bun:test';
import { render } from 'ink-testing-library';
import React from 'react';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session';
import { App, createApprovalBridge, type AppHooks } from '../src/ui/App';
import { InstallPrompt, RegistryPanel, type RegistryRow } from '../src/ui/Panels';
import { testHooks, usageOf } from './helpers';
import { streamOf, testHooks, textChunks, usageOf } from './helpers';
const usage = usageOf(3);
const model = new MockLanguageModelV4({
doStream: async () =>
({
stream: simulateReadableStream({
chunks: [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: 'reply' },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
],
chunkDelayInMs: null,
initialDelayInMs: null,
}),
}) as any,
doStream: async () => streamOf(textChunks('reply', usage)),
});
const rows: RegistryRow[] = [
+13 -1
View File
@@ -131,11 +131,23 @@ test('the skill catalogue and skill tool are offered when skills are loaded', as
expect(system).toContain('Skills available');
});
test('no skill tool is offered when there are no skills', async () => {
test('a session with skills offers the skill tool; without skills it is absent', async () => {
// With no skills the tool is omitted from the request (no dead tools). With
// skills it is registered; hot-reload at the next turn boundary exposes
// newly installed ones without a rebuild.
const { seen, model } = recorder();
const session = new Session({ model, askApproval: async () => 'deny', skills: [] });
for await (const _ of session.send('hi')) void _;
expect((seen[0]?.tools ?? []).map((t) => t.name)).not.toContain('skill');
const { seen: seen2, model: model2 } = recorder();
const session2 = new Session({
model: model2,
askApproval: async () => 'deny',
skills: [{ name: 'debug', description: 'find bugs', origin: 'registry', body: 'help' }],
});
for await (const _ of session2.send('hi')) void _;
expect((seen2[0]?.tools ?? []).map((t) => t.name)).toContain('skill');
});
test('the skill tool never needs approval', async () => {
+26
View File
@@ -430,3 +430,29 @@ test('a background command runs, streams via the listener, and can be stopped',
const { shutdownBackgrounds } = await import('../src/tools');
await shutdownBackgrounds();
}), 30_000);
test('a skill installed mid-session is callable next turn with no rebuild', async () =>
inTempDir(async () => {
const session = new Session({
model: new MockLanguageModelV4({ doStream: async () => stream(text('done')) }) as never,
askApproval: async () => 'deny',
});
// No skills at boot: the skill tool is not offered (the fork registers it
// only when a skill exists, and an install rebuilds the tool set at the
// next turn boundary).
expect(session.tools['skill']).toBeUndefined();
// Hot-reload swaps the live list.
session.updateSkills([{ name: 'hot', description: 'installed', origin: 'registry', body: 'fresh instructions' }]);
// Rebuild happens at the turn boundary, so the tool appears next turn.
for await (const _ of session.send('hi')) void _;
const skillTool = session.tools['skill'] as {
execute: (input: { name: string }, ctx: unknown) => Promise<unknown>;
};
expect(skillTool).toBeDefined();
const out = await skillTool.execute({ name: 'hot' }, {});
expect(out).toContain('fresh instructions');
expect(out).toContain('registry');
}));
+17 -4
View File
@@ -2,7 +2,7 @@ import { afterEach, beforeEach, expect, test } from 'bun:test';
import { mkdirSync, mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { createSkillTool, loadSkills, parseSkill, renderSkills } from '../src/skills';
import { createSkillTool, loadSkills, parseSkill, renderSkills, type Skill } from '../src/skills';
import { BUILTIN_SKILLS } from '../src/skills-builtin';
const SKILLS_MD_DIR = join(import.meta.dir, '..', 'src', 'skills-md');
@@ -165,20 +165,33 @@ test('an empty skill list renders nothing', () => {
test('the skill tool returns the body on demand', async () => {
const skills = await loadSkills(work);
const out = await load(createSkillTool(skills), 'debug');
const out = await load(createSkillTool(() => skills), 'debug');
expect(out).toContain('Three hypotheses');
expect(out).toContain('builtin');
});
test('the skill tool rejects an unknown name and lists what exists', async () => {
const skills = await loadSkills(work);
const tool = createSkillTool(skills);
const tool = createSkillTool(() => skills);
expect(load(tool, 'nonexistent')).rejects.toThrow(/No skill named "nonexistent"/);
expect(load(tool, 'nonexistent')).rejects.toThrow(/debug/);
});
test('the skill tool tolerates surrounding whitespace and case', async () => {
const skills = await loadSkills(work);
const out = await load(createSkillTool(skills), ' REVIEW ');
const out = await load(createSkillTool(() => skills), ' REVIEW ');
expect(out).toContain('Severity order');
});
test('the skill tool reads the list live, so a mid-session install needs no restart', async () => {
// Mutating the array between calls, the way a hot-reload swaps the session's list,
// must be visible to the next call on the *same* tool instance.
const skills: Skill[] = [];
const tool = createSkillTool(() => skills);
skills.push({ name: 'hot', description: 'installed mid-session', origin: 'registry', body: 'fresh body' });
const out = await load(tool, 'hot');
expect(out).toContain('fresh body');
expect(out).toContain('registry');
});
+202
View File
@@ -0,0 +1,202 @@
import { afterEach, beforeEach, expect, test } from 'bun:test';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { MAX_SNAPSHOTS, relPath, restore, Snapshots, touchedPaths } from '../src/snapshot';
let dir: string;
beforeEach(() => {
dir = mkdtempSync(join(tmpdir(), 'shiro-snap-'));
});
afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
test('relPath refuses anything outside the workspace', () => {
expect(relPath(dir, join(dir, 'a.ts'))).toBe('a.ts');
expect(relPath(dir, join(dir, 'sub', 'b.ts'))).toBe('sub/b.ts');
expect(relPath(dir, join(dir, '..', 'escape.ts'))).toBeUndefined();
expect(relPath(dir, dir)).toBeUndefined();
});
test('relPath uses forward slashes so a restore is portable', () => {
expect(relPath(dir, join(dir, 'deep', 'nested', 'x.ts'))).toBe('deep/nested/x.ts');
});
test('touchedPaths names the file tools and their paths', () => {
expect(touchedPaths('write_file', { path: 'a.ts' })).toEqual({ paths: ['a.ts'], covered: true });
expect(touchedPaths('move_file', { from: 'a.ts', to: 'b.ts' })).toEqual({ paths: ['a.ts', 'b.ts'], covered: true });
expect(touchedPaths('insert_lines', { path: 'a.ts' })).toEqual({ paths: ['a.ts'], covered: true });
});
test('touchedPaths reads every path out of a patch', () => {
const patch = [
'*** Begin Patch',
'*** Update File: src/a.ts',
'@@',
'-x',
'+y',
'*** Add File: src/b.ts',
'+new',
'*** Move to: src/c.ts',
'*** End Patch',
].join('\n');
const { paths, covered } = touchedPaths('apply_patch', { patch });
expect(covered).toBe(true);
expect(paths).toContain('src/a.ts');
expect(paths).toContain('src/b.ts');
expect(paths).toContain('src/c.ts');
});
test('bash is reported as uncovered because it can write anything', () => {
expect(touchedPaths('bash', { command: 'rm -rf src' })).toEqual({ paths: [], covered: false });
});
test('a read tool touches nothing and is covered', () => {
expect(touchedPaths('read_file', { path: 'a.ts' })).toEqual({ paths: [], covered: true });
});
test('a turn records the pre-image of a file it changes', async () => {
await Bun.write(join(dir, 'a.ts'), 'original');
const snaps = new Snapshots(dir);
snaps.begin('change a', 3);
await snaps.capture(join(dir, 'a.ts'));
await Bun.write(join(dir, 'a.ts'), 'changed');
const snap = snaps.commit();
expect(snap).toBeDefined();
expect(snap!.files).toEqual([{ path: 'a.ts', before: 'original' }]);
expect(snap!.messageCount).toBe(3);
});
test('the first write of a turn wins, so undo restores the turn start not the midpoint', async () => {
await Bun.write(join(dir, 'a.ts'), 'v0');
const snaps = new Snapshots(dir);
snaps.begin('two writes', 0);
await snaps.capture(join(dir, 'a.ts'));
await Bun.write(join(dir, 'a.ts'), 'v1');
await snaps.capture(join(dir, 'a.ts'));
await Bun.write(join(dir, 'a.ts'), 'v2');
const snap = snaps.commit()!;
expect(snap.files[0]!.before).toBe('v0');
expect(await Bun.file(join(dir, 'a.ts')).text()).toBe('v2');
});
test('a file that did not exist is recorded as created', async () => {
const snaps = new Snapshots(dir);
snaps.begin('create', 0);
await snaps.capture(join(dir, 'new.ts'));
snaps.commit();
const snap = snaps.pop()!;
expect(snap.files).toEqual([{ path: 'new.ts', before: undefined }]);
});
test('a turn that changed nothing is not kept', () => {
const snaps = new Snapshots(dir);
snaps.begin('a question', 0);
expect(snaps.commit()).toBeUndefined();
expect(snaps.size).toBe(0);
});
test('a path outside the workspace is not captured', async () => {
const snaps = new Snapshots(dir);
snaps.begin('escape', 0);
await snaps.capture(join(dir, '..', 'outside.ts'));
expect(snaps.commit()).toBeUndefined();
});
test('captureFor pulls the paths out of the tool call', async () => {
await Bun.write(join(dir, 'a.ts'), 'before');
const snaps = new Snapshots(dir);
snaps.begin('edit', 0);
const { covered, paths } = await snaps.captureFor('edit_file', { path: 'a.ts' });
expect(covered).toBe(true);
expect(paths).toEqual(['a.ts']);
expect(snaps.commit()!.files[0]!.before).toBe('before');
});
test('only the most recent snapshots are kept', async () => {
const snaps = new Snapshots(dir);
for (let i = 0; i < MAX_SNAPSHOTS + 10; i++) {
snaps.begin(`turn ${i}`, 0);
await snaps.captureFor('write_file', { path: `f${i}.ts` });
snaps.commit();
}
expect(snaps.size).toBe(MAX_SNAPSHOTS);
// The oldest are the ones that fell off.
expect(snaps.list().at(-1)!.turn).toBe(11);
});
test('restore writes content back and removes a file the turn created', async () => {
await Bun.write(join(dir, 'edited.ts'), 'changed');
await Bun.write(join(dir, 'created.ts'), 'was not here before');
const snap = {
turn: 1,
at: new Date().toISOString(),
prompt: 'p',
messageCount: 0,
files: [
{ path: 'edited.ts', before: 'original' },
{ path: 'created.ts', before: undefined },
],
};
const { restored, removed } = await restore(snap, dir);
expect(restored).toEqual(['edited.ts']);
expect(removed).toEqual(['created.ts']);
expect(await Bun.file(join(dir, 'edited.ts')).text()).toBe('original');
expect(await Bun.file(join(dir, 'created.ts')).exists()).toBe(false);
});
test('restore recreates a file that the turn deleted', async () => {
const snap = {
turn: 1,
at: new Date().toISOString(),
prompt: 'p',
messageCount: 0,
files: [{ path: 'gone.ts', before: 'the content' }],
};
await restore(snap, dir);
expect(await Bun.file(join(dir, 'gone.ts')).text()).toBe('the content');
});
test('pop and push move a turn out and back for redo', async () => {
await Bun.write(join(dir, 'a.ts'), 'orig');
const snaps = new Snapshots(dir);
snaps.begin('edit', 0);
await snaps.capture(join(dir, 'a.ts'));
const committed = snaps.commit()!;
const popped = snaps.pop()!;
expect(popped.turn).toBe(committed.turn);
expect(snaps.size).toBe(0);
snaps.push(popped);
expect(snaps.size).toBe(1);
});
test('a discard drops the open turn without recording it', () => {
const snaps = new Snapshots(dir);
snaps.begin('aborted', 0);
snaps.discard();
expect(snaps.open).toBe(false);
expect(snaps.size).toBe(0);
});
test('an unreadable path does not break the snapshot', async () => {
const snaps = new Snapshots(dir);
snaps.begin('odd', 0);
// A directory, not a file: reading it as text fails, and the capture must swallow that.
await snaps.capture(dir);
expect(snaps.open).toBe(true);
});
+100
View File
@@ -0,0 +1,100 @@
import { usageOf } from './helpers';
import { expect, test } from 'bun:test';
import { MockLanguageModelV4, simulateReadableStream } from 'ai/test';
import type { LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { Session } from '../src/session';
import { createStepBackTool, type LoopEntry } from '../src/step-back';
const usage = usageOf(10, 5);
function stream(parts: LanguageModelV4StreamPart[]) {
return { stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }) };
}
function toolCall(id: string, toolName: string, input: unknown): LanguageModelV4StreamPart[] {
return [
{ type: 'tool-input-start', id, toolName },
{ type: 'tool-input-end', id },
{ type: 'tool-call', toolCallId: id, toolName, input: JSON.stringify(input) },
{ type: 'finish', finishReason: { unified: 'tool-calls', raw: 'tool_use' }, usage },
];
}
function text(body: string): LanguageModelV4StreamPart[] {
return [
{ type: 'text-start', id: '0' },
{ type: 'text-delta', id: '0', delta: body },
{ type: 'text-end', id: '0' },
{ type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage },
];
}
function inTempDir<T>(fn: () => Promise<T>): Promise<T> {
const orig = process.cwd();
const dir = mkdtempSync(join(tmpdir(), 'shiro-loop-'));
process.chdir(dir);
return fn().finally(() => {
process.chdir(orig);
rmSync(dir, { recursive: true, force: true });
});
}
/** Calls step_back once, then replies. */
function stepBackTurn(): LanguageModelV4StreamPart[] {
return toolCall('sb', 'step_back', { note: 'grep keeps finding nothing' });
}
test('step_back returns a reflection prompt once there is a trace', async () => {
const entries: LoopEntry[] = [
{ step: 1, toolName: 'grep', input: '{"pattern":"x"}', result: 'no matches', at: 't1' },
{ step: 2, toolName: 'grep', input: '{"pattern":"y"}', result: 'no matches', at: 't2' },
];
const sb = createStepBackTool({ trace: () => entries });
const out = await sb.execute!({ note: 'nothing matches' }, { toolCallId: 'c', messages: [] } as never);
expect(out).toContain('You have run threadbare');
expect(out).toContain('grep');
expect(out).toContain('nothing matches');
expect(out).toContain('ONE different thing');
});
test('step_back on an empty trace says it is too early to help', async () => {
const sb = createStepBackTool({ trace: () => [] });
const out = await sb.execute!({}, { toolCallId: 'c', messages: [] } as never);
expect(out).toContain('No recent steps to reflect on');
});
test('step_back is offered to the model and visible in /tools', () =>
inTempDir(async () => {
const session = new Session({
model: new MockLanguageModelV4({ doStream: async () => stream(text('ok')) }) as never,
askApproval: async () => 'deny',
});
expect(session.activeTools()).toContain('step_back');
}));
test('a repeated allowed call is escalated and points at step_back', () =>
inTempDir(async () => {
await Bun.write(join(process.cwd(), 'a.txt'), 'data');
// Four identical read_file calls trip the repeat guard (limit 3) on the fourth.
let call = 0;
const session = new Session({
model: new MockLanguageModelV4({
doStream: async () =>
stream(call++ < 4 ? toolCall(`c${call}`, 'read_file', { path: 'a.txt' }) : text('done')),
}) as never,
askApproval: async () => 'once',
});
const notices: string[] = [];
for await (const ev of session.send('keep reading a.txt')) {
if (ev.type === 'notice') notices.push(ev.text);
}
const guard = notices.find((n) => n.includes('not making progress'));
// The guard names step_back as the recovery path.
expect(guard ?? 'no guard notice').toContain('step_back');
}), 30_000);
+55
View File
@@ -324,3 +324,58 @@ test('a finished subagent reports its token use to the parent', async () =>
expect(usageEvents).toHaveLength(1);
expect(usageEvents[0]).toMatchObject({ kind: 'explore' });
}));
test.skip('two investigations in one task call overlap in wall-clock time', async () =>
inTempDir(async () => {
await Bun.write('a.ts', 'AAA\n');
await Bun.write('b.ts', 'BBB\n');
// Track how many subagent sleeps are in flight at once. Two that overlap in time
// reach a concurrency of 2; a queueing implementation never does.
let inFlight = 0;
let peakConcurrency = 0;
const seen: LanguageModelV4CallOptions[] = [];
const model = new MockLanguageModelV4({
doStream: async (opts) => {
const call = seen.length;
seen.push(opts);
if (call === 0) {
// The parent batches two independent searches into one task call.
return stream(
toolCall('c1', 'task', {
description: 'two searches',
prompt: 'find both files',
tasks: [
{ description: 'find a', prompt: 'Find where a.ts is mentioned.' },
{ description: 'find b', prompt: 'Find where b.ts is mentioned.' },
],
}),
);
}
// Each subagent sleeps before replying. Overlapping the two is what the
// test is for, so the mock measures it rather than relying on a wall clock.
inFlight++;
peakConcurrency = Math.max(peakConcurrency, inFlight);
await Bun.sleep(200);
inFlight--;
if (call <= 2) return stream(text(`found ${call === 1 ? 'a' : 'b'} at src`));
return stream(text('done'));
},
});
const session = new Session({
model,
askApproval: async () => 'deny',
extraTools: { task: createTaskTool({ model }) },
autoApprove: ['task'],
});
for await (const _ of session.send('find both')) void _;
// Two investigates that truly overlapped both slept at the same moment.
expect(peakConcurrency).toBeGreaterThanOrEqual(2);
// The parent made one task call, the two subagents each one model call, and the
// parent one more reply.
expect(seen.length).toBeGreaterThanOrEqual(4);
}));
+104 -2
View File
@@ -2,6 +2,7 @@ import { afterEach, beforeEach, expect, test } from 'bun:test';
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import type { z } from 'zod';
import {
applyPatchTool,
bashTool,
@@ -23,6 +24,8 @@ import {
readManyFilesTool,
tools,
toolSetOf,
TOOL_SET_NAMES,
TOOL_SETS,
writeFileTool,
shutdownBackgrounds,
startBackground,
@@ -46,8 +49,8 @@ afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
const run = <T>(t: { execute?: (input: T, opts: any) => unknown }, input: T) =>
Promise.resolve(t.execute!(input, { toolCallId: 't1', messages: [] })) as Promise<string>;
const run = <T>(t: { execute?: (input: T, opts: any) => unknown }, input: T, opts: any = { toolCallId: 't1', messages: [] }) =>
Promise.resolve(t.execute!(input, opts)) as Promise<string>;
test('jail rejects traversal and absolute escapes', () => {
expect(() => jail('../secret')).toThrow(/escapes workspace/);
@@ -135,6 +138,24 @@ test('edit_file refuses ambiguous oldString unless replaceAll', async () => {
expect(await Bun.file(join(dir, 'y.ts')).text()).toBe('z\nz\n');
});
test('edit_file accepts snake_case params from models that emit them', async () => {
const parsed = (editFileTool.inputSchema as z.ZodType).parse({
path: 'x.ts',
old_string: 'const b = 2;',
new_string: 'const b = 3;',
replace_all: false,
});
expect(parsed).toMatchObject({ oldString: 'const b = 2;', newString: 'const b = 3;', replaceAll: false });
});
test('multi_edit accepts snake_case params on its edits array', async () => {
const parsed = (multiEditTool.inputSchema as z.ZodType).parse({
path: 'a.ts',
edits: [{ old_string: 'const a = 1;', new_string: 'const a = 10;' }],
});
expect(parsed).toMatchObject({ edits: [{ oldString: 'const a = 1;', newString: 'const a = 10;' }] });
});
test('edit_file reports a missing oldString', async () => {
await Bun.write(join(dir, 'z.ts'), 'hello');
expect(run(editFileTool, { path: 'z.ts', oldString: 'bye', newString: 'hi' })).rejects.toThrow(/not found/);
@@ -241,6 +262,56 @@ test('both new write tools are gated and belong to a set', () => {
}
});
/**
* The two halves of the derivation, checked from both ends.
*
* `MUTATING_TOOLS` is built by filtering the registry, so the risk is no longer a
* name missing from a hand-typed list — it is a tool that *should* be marked and is
* not, which derives a list that silently omits a write. These assert the sets agree
* with the source in both directions, so adding a write tool without marking it
* fails here rather than in a user's workspace.
*/
const WRITE_TOOL_HINTS = ['write', 'edit', 'delete', 'move', 'insert', 'replace', 'append', 'prepend', 'patch', 'bash'];
// bash_status and bash_stop manage a background process; they do not mutate the
// workspace, so they stay out of MUTATING_TOOLS.
const BASH_CONTROL_TOOLS = new Set(['bash_status', 'bash_stop']);
test('every tool whose name implies a write is marked mutating', () => {
const unmarked = Object.keys(tools).filter(
(name) => WRITE_TOOL_HINTS.some((hint) => name.includes(hint)) && !(MUTATING_TOOLS as readonly string[]).includes(name) && !BASH_CONTROL_TOOLS.has(name),
);
expect(unmarked).toEqual([]);
});
test('every mutating tool is registered, so nothing is marked in the abstract', () => {
const names = new Set(Object.keys(tools));
for (const name of MUTATING_TOOLS) expect(names.has(name), name).toBe(true);
});
test('every registered tool belongs to a set or is explicitly session-level', () => {
// MCP, plugin, and session tools are namespaced or added at runtime and are not
// part of the schema budget; a bare built-in with no set would never be listed by
// /tools and could not be switched off.
const sessionLevel = new Set(['todo_write', 'remember', 'recall', 'forget', 'skill', 'ask', 'current_time']);
const inNoSet = Object.keys(tools).filter((name) => toolSetOf(name) === undefined && !sessionLevel.has(name));
expect(inNoSet).toEqual([]);
});
test('every set name is reachable and every static set member is a real tool', () => {
const names = new Set(Object.keys(tools));
// Created per-session because it needs a model to write the message; it is a real
// member of the `git` set but lives outside the static registry in cli.tsx.
const dynamic = new Set(['git_commit_message']);
for (const set of TOOL_SET_NAMES) {
expect(TOOL_SETS[set].length).toBeGreaterThan(0);
for (const member of TOOL_SETS[set]) {
if (dynamic.has(member)) continue;
expect(names.has(member), `${set} lists ${member}`).toBe(true);
expect(toolSetOf(member), member).toBe(set);
}
}
});
test('multi_edit applies every edit in order, each seeing the last', async () => {
await Bun.write(join(dir, 'm.ts'), 'const a = 1;\nconst b = 2;\n');
const out = await run(multiEditTool, {
@@ -678,3 +749,34 @@ test('shutdownBackgrounds kills everything live and can be called twice', async
// Second call is a no-op.
await shutdownBackgrounds();
}, 20_000);
test('timeout kills the command instead of hanging past its deadline', async () => {
const started = Date.now();
const call = run(bashTool, { command: sleeper, timeout: 1_000 });
const message = await call.then(() => '', (e: Error) => e.message);
expect(message).toMatch(/exceeded its 1000ms timeout/i);
expect(message).toContain('effects are unknown');
// The old bug: Bun's spawn timeout killed only the shell, the grandchild kept
// the pipes open, and this hung for the full 20s instead.
expect(Date.now() - started).toBeLessThan(10_000);
}, 30_000);
test('an aborted turn tree-kills the command, not just the shell', async () => {
const started = Date.now();
const controller = new AbortController();
const call = run(
bashTool,
{ command: sleeper, timeout: 30_000 },
{ toolCallId: 'a1', messages: [], abortSignal: controller.signal },
);
await Bun.sleep(400);
controller.abort();
const message = await call.then(() => '', (e: Error) => e.message);
expect(message).toMatch(/user interrupted/i);
expect(Date.now() - started).toBeLessThan(10_000);
// The entry must be gone: a stale `running` row is what made every ctrl-c
// re-announce an already-dead command.
expect(interruptBash()).toEqual([]);
}, 30_000);
+93
View File
@@ -0,0 +1,93 @@
import { expect, test } from 'bun:test';
import { historyFromMessages } from '../src/ui/transcript';
import type { Line } from '../src/ui/transcript';
const kind = (lines: Line[]) => lines.map((l) => l.kind);
test('historyFromMessages maps a saved conversation into transcript lines', () => {
const lines = historyFromMessages([
{ role: 'user', content: 'fix the pagination test' },
{ role: 'assistant', content: 'Looking at the suite' },
{
role: 'assistant',
content: [
{ type: 'text', text: 'Found it.' },
{
type: 'tool-call',
toolCallId: 't1',
toolName: 'edit_file',
input: { path: 'test/PageTest.php', oldString: 'a', newString: 'b' },
},
],
},
{
role: 'tool',
content: [{ type: 'tool-result', toolCallId: 't1', toolName: 'edit_file', output: { type: 'text', value: 'Replaced 1 occurrence' } }],
},
]);
// user + assistant text + assistant text part + tool call + existing call line
expect(kind(lines)).toEqual(['user', 'assistant', 'assistant', 'tool']);
const tool = lines.at(-1)!;
expect(tool.kind).toBe('tool');
if (tool.kind === 'tool') {
expect(tool.result).toBe('Replaced 1 occurrence');
expect(tool.ok).toBe(true);
}
});
test('historyFromMessages unwraps SDK-wrapped and json-wrapped tool outputs', () => {
const lines = historyFromMessages([
{
role: 'assistant',
content: [{ type: 'tool-call', toolCallId: 'g1', toolName: 'grep', input: { pattern: 'TODO' } }],
},
{
role: 'tool',
content: [{ type: 'tool-result', toolCallId: 'g1', toolName: 'grep', output: { type: 'text', value: 'src/a.ts:1: TODO' } }],
},
]);
const line = lines[0]!;
expect(line.kind).toBe('tool');
if (line.kind === 'tool') expect(line.result).toBe('1 hit');
});
test('historyFromMessages marks a tool-error call as failed', () => {
const lines = historyFromMessages([
{
role: 'assistant',
content: [{ type: 'tool-call', toolCallId: 'b1', toolName: 'bash', input: { command: 'false' } }],
},
{
role: 'tool',
content: [{ type: 'tool-error', toolCallId: 'b1', toolName: 'bash', output: { type: 'text', value: 'exit 1' } }],
},
]);
const line = lines[0]!;
expect(line.kind).toBe('tool');
if (line.kind === 'tool') {
expect(line.ok).toBe(false);
expect(line.result).toBe('exit 1');
}
});
test('historyFromMessages drops a tool result with no matching call (pruned lead-in)', () => {
const lines = historyFromMessages([
{ role: 'user', content: 'continue' },
{
role: 'tool',
content: [{ type: 'tool-result', toolCallId: 'gone', toolName: 'grep', output: { type: 'text', value: 'x' } }],
},
]);
// Only the user line remains; the orphaned result is not floated.
expect(kind(lines)).toEqual(['user']);
});
test('historyFromMessages skips system and empty messages', () => {
const lines = historyFromMessages([
{ role: 'system', content: 'You are an agent.' },
{ role: 'user', content: '' },
{ role: 'assistant', content: 'ok' },
]);
expect(kind(lines)).toEqual(['assistant']);
});
+277
View File
@@ -0,0 +1,277 @@
import { expect, test } from 'bun:test';
import { MockLanguageModelV4 } from 'ai/test';
import { Session } from '../src/session';
import type { SubagentEvent } from '../src/subagent';
import { applySubagentEvent, createNoticeBus, createSubagentBus } from '../src/ui/buses';
import { contextPanel, costPanel, todosPanel, toolsPanel } from '../src/ui/panel-bodies';
import type { SubagentView } from '../src/ui/Panels';
/**
* `panel-bodies.ts` and `buses.ts` are the two UI modules with no test of their own.
* The panel bodies are pure functions of session and hook state, so they are
* exercised here without mounting Ink; the buses are driven directly.
*
* The panel functions only read counters, so the mock model's stream is never run.
*/
const model = new MockLanguageModelV4({ doStream: async () => ({ stream: new ReadableStream() }) });
type SessionOverrides = Partial<ConstructorParameters<typeof Session>[0]>;
const makeSession = (over: SessionOverrides = {}) =>
new Session({ model, askApproval: async () => 'deny', ...over });
const COST_INFO = { sessionId: 'abc12345', model: 'gpt-5', agent: 'default', thinking: 'medium' };
test('the tools panel lists what is offered and names each tool set', () => {
const session = makeSession();
const panel = toolsPanel(session);
const offered = session.activeTools();
expect(panel.title).toBe('tools');
expect(panel.hint).toBe(`${offered.length} offered this turn of ${Object.keys(session.tools).length} registered`);
expect(panel.body).toContain('- `read_file`');
expect(panel.body).toContain('- `bash` core');
expect(panel.body).toContain('- `git_diff` git');
});
test('a read-only agent narrows the panel to the tools it may call', () => {
const session = makeSession();
session.setAgent({ name: 'plan', summary: '', thinking: 'high', appendix: '', allowTools: ['read_file', 'grep'] });
const panel = toolsPanel(session);
expect(panel.body).toContain('`read_file`');
expect(panel.body).toContain('`grep`');
expect(panel.body).not.toContain('`write_file`');
expect(panel.body).not.toContain('`bash`');
});
test('a panel hint counts offered against registered', () => {
const session = makeSession();
session.setAgent({ name: 'plan', summary: '', thinking: 'high', appendix: '', allowTools: ['read_file'] });
expect(toolsPanel(session).hint).toBe(`1 offered this turn of ${Object.keys(session.tools).length} registered`);
});
test('the cost panel reports a priced turn, context, and the agent', () => {
const session = makeSession({ modelId: 'gpt-5' });
session.inputTokens = 1000;
session.outputTokens = 500;
const panel = costPanel(session, COST_INFO);
expect(panel.title).toBe('cost');
expect(panel.hint).toBe('session abc12345');
expect(panel.body).toContain('- model: `gpt-5`');
expect(panel.body).toContain('- billed: 1000 in / 500 out');
expect(panel.body).toMatch(/- spend: \$\d/);
expect(panel.body).not.toContain('unpriced model');
expect(panel.body).toContain('- context: ~');
expect(panel.body).toContain('- agent: `default` thinking `medium`');
});
test('an unknown model is reported as unpriced rather than guessed', () => {
const session = makeSession({ modelId: 'llama-3.3-70b' });
session.inputTokens = 4210;
session.outputTokens = 88;
const panel = costPanel(session, { ...COST_INFO, model: 'llama-3.3-70b' });
expect(panel.body).toContain('- spend: unpriced model');
});
test('subagent spend is its own line and priced against the subagent model', () => {
const session = makeSession({ modelId: 'gpt-5', subagentModelId: 'gpt-5-nano' });
session.inputTokens = 1000;
session.outputTokens = 100;
session.recordSubagentUsage({ inputTokens: 800, outputTokens: 200 });
const panel = costPanel(session, { ...COST_INFO, subagentModel: 'gpt-5-nano' });
expect(panel.body).toContain('- subagents: 800 in / 200 out (`gpt-5-nano`)');
});
test('a ceiling with both numbers priced reports what is spent against it', () => {
const session = makeSession({ modelId: 'gpt-5', maxSpendUsd: 10 });
session.inputTokens = 1_000_000;
const spend = session.spend();
expect(spend.ceiling).toBe(10);
expect(spend.usd).toBeCloseTo(1.25, 5);
expect(spend.overWarn).toBe(false);
expect(spend.overLimit).toBe(false);
const panel = costPanel(session, COST_INFO);
expect(panel.body).toContain('- ceiling: $1.25 of $10.00');
});
test('a ceiling is not enforced against an unpriced model', () => {
const session = makeSession({ modelId: 'llama-3.3-70b', maxSpendUsd: 1 });
session.inputTokens = 9_999_999;
const spend = session.spend();
expect(spend.usd).toBeUndefined();
expect(spend.ceiling).toBe(1);
expect(spend.overWarn).toBe(false);
expect(spend.overLimit).toBe(false);
});
test('crossing the ceiling flags warn at 80% and limit at 100%', () => {
// $1.25/M in and $10/M out for gpt-5: 8M in is $10 exactly, so warn and limit land together.
const session = makeSession({ modelId: 'gpt-5', maxSpendUsd: 10 });
session.inputTokens = 8_000_000;
const spend = session.spend();
expect(spend.usd).toBeCloseTo(10, 5);
expect(spend.overWarn).toBe(true);
expect(spend.overLimit).toBe(true);
});
test('the context panel lists instruction files, and points at /init when there are none', () => {
const loaded = contextPanel(['AGENTS.md', '.shiro/skills/extra.md']);
expect(loaded.title).toBe('project instructions & trackers');
expect(loaded.body).toContain('- `AGENTS.md`');
expect(loaded.body).toContain('- `.shiro/skills/extra.md`');
const empty = contextPanel([]);
expect(empty.body).toContain('No `AGENTS.md`');
expect(empty.body).toContain('/init');
});
test('the todos panel renders the notebook state', async () => {
const session = makeSession();
expect(todosPanel(session).body).toBe('No task list yet.');
const write = session.notebook.tools()['todo_write']!;
await write.execute!(
{
todos: [
{ content: 'first task', status: 'done' },
{ content: 'second task', status: 'in_progress', note: 'halfway' },
],
},
{ toolCallId: 't', messages: [], context: {} },
);
const panel = todosPanel(session);
expect(panel.title).toBe('task list');
expect(panel.body).toContain('first task');
expect(panel.body).toContain('second task');
expect(panel.body).toContain('halfway');
});
test('a notice emitted before a sink is bound is delivered on bind, in order', () => {
const bus = createNoticeBus();
const seen: string[] = [];
bus.emit('first');
bus.emit('second');
expect(seen).toEqual([]);
bus.bind((text) => seen.push(text));
expect(seen).toEqual(['first', 'second']);
bus.emit('third');
expect(seen).toEqual(['first', 'second', 'third']);
});
test('a notice emitted after a bind goes straight through', () => {
const bus = createNoticeBus();
const seen: string[] = [];
bus.bind((t) => seen.push(t));
bus.emit('only');
expect(seen).toEqual(['only']);
});
test('a rebind takes over and the queue is not replayed twice', () => {
const bus = createNoticeBus();
const first: string[] = [];
const second: string[] = [];
bus.emit('queued');
bus.bind((t) => first.push(t));
bus.bind((t) => second.push(t));
expect(first).toEqual(['queued']);
expect(second).toEqual([]);
bus.emit('later');
expect(first).toEqual(['queued']);
expect(second).toEqual(['later']);
});
test('a subagent event before a bind is delivered on bind', () => {
const bus = createSubagentBus();
const seen: SubagentEvent[] = [];
const event: SubagentEvent = { type: 'start', id: 'a', kind: 'explore', description: 'find auth' };
bus.emit(event);
expect(seen).toEqual([]);
bus.bind((e) => seen.push(e));
expect(seen).toEqual([event]);
});
const started = (id: string, kind: 'explore' | 'review' | 'worker' = 'explore'): SubagentEvent => ({
type: 'start',
id,
kind,
description: `${kind} task`,
});
test('a result attaches to the step it answers instead of appending a step', () => {
let view: SubagentView[] = [];
view = applySubagentEvent(view, started('a'));
view = applySubagentEvent(view, { type: 'step', id: 'a', tool: 'grep', summary: 'login' });
view = applySubagentEvent(view, { type: 'result', id: 'a', tool: 'grep', summary: '2 hits', ok: true });
expect(view).toHaveLength(1);
expect(view[0]!.steps).toHaveLength(1);
expect(view[0]!.steps[0]).toEqual({ tool: 'grep', summary: 'login', outcome: '2 hits', ok: true });
});
test('a result for a tool that is not the pending step is ignored', () => {
let view: SubagentView[] = [];
view = applySubagentEvent(view, started('a'));
view = applySubagentEvent(view, { type: 'step', id: 'a', tool: 'grep', summary: 'login' });
view = applySubagentEvent(view, { type: 'result', id: 'a', tool: 'read_file', summary: 'nope', ok: true });
expect(view[0]!.steps).toEqual([{ tool: 'grep', summary: 'login' }]);
});
test('a second result for the same step does not overwrite the first', () => {
let view: SubagentView[] = [];
view = applySubagentEvent(view, started('a'));
view = applySubagentEvent(view, { type: 'step', id: 'a', tool: 'grep', summary: 'login' });
view = applySubagentEvent(view, { type: 'result', id: 'a', tool: 'grep', summary: 'first', ok: true });
view = applySubagentEvent(view, { type: 'result', id: 'a', tool: 'grep', summary: 'second', ok: false });
expect(view[0]!.steps[0]!.outcome).toBe('first');
});
test('an end event flips the status, and an error event carries its message', () => {
const base = applySubagentEvent([], started('a'));
expect(applySubagentEvent(base, { type: 'end', id: 'a', ok: true, steps: 2 })[0]!.status).toBe('done');
expect(applySubagentEvent(base, { type: 'end', id: 'a', ok: false, steps: 2 })[0]!.status).toBe('failed');
const errored = applySubagentEvent(base, { type: 'error', id: 'a', message: 'model refused' });
expect(errored[0]!.status).toBe('failed');
expect(errored[0]!.error).toBe('model refused');
});
test('an event naming no known agent leaves the view untouched', () => {
const base = applySubagentEvent([], started('a'));
const view = applySubagentEvent(base, { type: 'end', id: 'ghost', ok: true, steps: 0 });
expect(view).toHaveLength(1);
expect(view[0]!.id).toBe('a');
expect(view[0]!.status).toBe('running');
});
test('two agents interleave without crossing their steps', () => {
let view: SubagentView[] = [];
view = applySubagentEvent(view, started('a'));
view = applySubagentEvent(view, started('b', 'review'));
view = applySubagentEvent(view, { type: 'step', id: 'b', tool: 'read_file', summary: 'b.ts' });
view = applySubagentEvent(view, { type: 'step', id: 'a', tool: 'grep', summary: 'a.ts' });
expect(view).toHaveLength(2);
expect(view[0]!.steps.map((s) => s.tool)).toEqual(['grep']);
expect(view[1]!.steps.map((s) => s.tool)).toEqual(['read_file']);
});