Ship the batch: undo, parallel subagents, lazy MCP, hot-reloaded skills; docs and CI/CD
Loop and ergonomics batch across Now/Next and Maintenance:
- /undo and /redo via pre-prompt file snapshots (snapshot.ts)
- task takes a tasks[] array and runs investigations concurrently (subagent.ts)
- lazy MCP tools: mcp_list/mcp_inspect/mcp_call meta-tools, eager opt-in (mcp.ts, config.ts)
- skill tool reads its list live so a mid-session install is callable next turn (skills.ts)
- tool-name lists (tool-kinds.ts) derived from a mutating() marker; gates previously ungated writes
- prune/session recovery path summarized, and step-back doom-loop primitive (step-back.ts)
- @file completion re-walks on a slow cooldown; estimateTokens and pricing labeled as estimates
Docs: README, CHANGELOG, docs/{mcp,architecture,development} updated to match.
CI/CD: bun install-store cache and concurrency gates on both workflows; release.yml now
composes file-based release notes via scripts/make-release-notes.ts and verifies every binary.
This commit is contained in:
+41
-9
@@ -154,7 +154,8 @@ if (resumeArg) {
|
||||
}
|
||||
}
|
||||
|
||||
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
|
||||
const mcp =
|
||||
has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers, cfg.mcpMode ?? 'lazy');
|
||||
const instructions = has('--no-instructions') ? [] : await loadInstructions();
|
||||
const skills = has('--no-skills') ? [] : await loadSkills();
|
||||
const customCommands = await loadCustomCommands();
|
||||
@@ -424,8 +425,15 @@ const hooks: AppHooks = {
|
||||
install: async (name) => {
|
||||
const entry = await findEntry(name);
|
||||
const { path } = await registry.install(entry);
|
||||
// Loaded on the next start rather than hot-swapped: a skill joins the system
|
||||
// prompt and a plugin joins the guard chain, and both are built once at boot.
|
||||
// A skill is live immediately: the session's skill tool reads its list on each
|
||||
// call, so the next turn can invoke a skill installed right now. A plugin or
|
||||
// tool joins the guard chain and the tool registry, both built once at boot,
|
||||
// so those still need a restart — said plainly rather than implied.
|
||||
if (entry.kind === 'skill') {
|
||||
const next = await loadSkills(process.cwd());
|
||||
session.setSkills(next);
|
||||
return `installed skill ${entry.name} to ${path}\nit is loaded and callable next turn`;
|
||||
}
|
||||
return `installed ${entry.kind} ${entry.name} to ${path}\nrestart shiro to load it`;
|
||||
},
|
||||
remove: async (name) => {
|
||||
@@ -434,7 +442,14 @@ const hooks: AppHooks = {
|
||||
const bare = parsed ? parsed[2]! : name;
|
||||
|
||||
for (const kind of kinds) {
|
||||
if (await registry.uninstall(kind, bare)) return `removed ${kind} ${bare}\nrestart shiro to unload it`;
|
||||
if (await registry.uninstall(kind, bare)) {
|
||||
if (kind === 'skill') {
|
||||
const next = await loadSkills(process.cwd());
|
||||
session.setSkills(next);
|
||||
return `removed skill ${bare}\nit is unloaded; the next turn no longer offers it`;
|
||||
}
|
||||
return `removed ${kind} ${bare}\nrestart shiro to unload it`;
|
||||
}
|
||||
}
|
||||
throw new Error(`nothing installed under the name "${bare}"`);
|
||||
},
|
||||
@@ -445,10 +460,17 @@ const hooks: AppHooks = {
|
||||
const servers = Object.entries(cfg.mcpServers ?? {});
|
||||
if (servers.length === 0) return 'no MCP servers configured\n\n`/mcp add` sets one up.';
|
||||
|
||||
const lazy = (mcp?.tools['mcp_list'] ?? undefined) !== undefined;
|
||||
const live = new Map<string, number>();
|
||||
for (const name of Object.keys(mcp?.tools ?? {})) {
|
||||
const server = /^mcp__([^_]+(?:_[^_]+)*)__/.exec(name)?.[1];
|
||||
if (server) live.set(server, (live.get(server) ?? 0) + 1);
|
||||
if (lazy) {
|
||||
// Lazy mode: tools are fetched on demand, so "connected" is what the handle
|
||||
// says, not a count of registered mcp__ tools.
|
||||
for (const name of mcp?.servers ?? []) live.set(name, -1);
|
||||
} else {
|
||||
for (const name of Object.keys(mcp?.tools ?? {})) {
|
||||
const server = /^mcp__([^_]+(?:_[^_]+)*)__/.exec(name)?.[1];
|
||||
if (server) live.set(server, (live.get(server) ?? 0) + 1);
|
||||
}
|
||||
}
|
||||
const failed = new Map((mcp?.errors ?? []).map((e) => [e.server, e.message]));
|
||||
|
||||
@@ -457,7 +479,9 @@ const hooks: AppHooks = {
|
||||
const state = failed.has(name)
|
||||
? `failed: ${failed.get(name)}`
|
||||
: live.has(name)
|
||||
? `${live.get(name)} tools`
|
||||
? live.get(name)! >= 0
|
||||
? `${live.get(name)} tools`
|
||||
: 'connected (lazy)'
|
||||
: has('--no-mcp')
|
||||
? 'not connected (--no-mcp)'
|
||||
: 'not connected this session';
|
||||
@@ -611,7 +635,15 @@ const facts: HeaderFact[] = [
|
||||
memory && memory.all().length > 0
|
||||
? { label: 'memory', value: `${memory.all().length} notes about this project` }
|
||||
: undefined,
|
||||
mcp && Object.keys(mcp.tools).length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).length} tools` } : undefined,
|
||||
mcp && Object.keys(mcp.tools).length > 0
|
||||
? {
|
||||
label: 'mcp',
|
||||
value:
|
||||
mcp.servers.length > 0
|
||||
? `${mcp.servers.length} ${mcp.servers.length === 1 ? 'server' : 'servers'} (lazy)`
|
||||
: `${Object.keys(mcp.tools).length} tools`,
|
||||
}
|
||||
: undefined,
|
||||
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
|
||||
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
|
||||
: undefined,
|
||||
|
||||
@@ -6,6 +6,8 @@ export type CommandAction =
|
||||
| { type: 'exit' }
|
||||
| { type: 'clear' }
|
||||
| { type: 'compact' }
|
||||
| { type: 'undo'; what: 'both' | 'files' | 'conversation' }
|
||||
| { type: 'redo'; what: 'both' | 'files' | 'conversation' }
|
||||
| { type: 'tools' }
|
||||
| { type: 'cost' }
|
||||
| { type: 'sessions' }
|
||||
@@ -57,6 +59,8 @@ export const COMMANDS: CommandSpec[] = [
|
||||
{ name: 'memory', summary: 'compact the project memory with the model' },
|
||||
{ name: 'tools', summary: 'list available tools' },
|
||||
{ name: 'compact', summary: 'replace history with a model-written summary' },
|
||||
{ name: 'undo', arg: '[files|conversation]', summary: 'walk the last turn back: files, conversation, or both' },
|
||||
{ name: 'redo', arg: '[conversation]', summary: 'put back what /undo took' },
|
||||
{ name: 'cost', summary: 'tokens and estimated spend this session' },
|
||||
{ name: 'sessions', summary: 'list saved sessions' },
|
||||
{ name: 'resume', arg: '<id>', summary: 'load a saved session' },
|
||||
@@ -161,6 +165,22 @@ function parseMcp(arg: string): CommandAction {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* `/undo [files|conversation]` and `/redo [conversation]`.
|
||||
*
|
||||
* The default is `both` for undo, because restoring one without the other is the
|
||||
* failure the two are meant to prevent: files back without the history and the model
|
||||
* re-reads a change it no longer made. A bare `files` or `conversation` narrows it.
|
||||
* Redo defaults to the conversation, since file content after the turn was never kept.
|
||||
*/
|
||||
function parseUndoKind(arg: string, fallback: 'both' | 'conversation'): 'both' | 'files' | 'conversation' {
|
||||
const word = arg.trim().toLowerCase();
|
||||
if (word === 'files' || word === 'file') return 'files';
|
||||
if (word === 'conversation' || word === 'chat' || word === 'history') return 'conversation';
|
||||
if (word === 'both' || word === 'all' || word === '') return fallback;
|
||||
return fallback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure parser: no IO, so the TUI and headless mode share one definition.
|
||||
*
|
||||
@@ -186,6 +206,10 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
|
||||
return { type: 'clear' };
|
||||
case 'compact':
|
||||
return { type: 'compact' };
|
||||
case 'undo':
|
||||
return { type: 'undo', what: parseUndoKind(arg, 'both') };
|
||||
case 'redo':
|
||||
return { type: 'redo', what: parseUndoKind(arg, 'conversation') };
|
||||
case 'tools':
|
||||
return { type: 'tools' };
|
||||
case 'cost':
|
||||
|
||||
@@ -37,6 +37,11 @@ export type Config = {
|
||||
/** Index for `/registry`. Omit for the default one. */
|
||||
registryUrl?: string;
|
||||
mcpServers?: Record<string, McpServerConfig>;
|
||||
/**
|
||||
* How MCP tools reach the model: `lazy` registers meta-tools only (cheap until a
|
||||
* tool is called), `eager` registers every server tool up front. Omit for lazy.
|
||||
*/
|
||||
mcpMode?: 'lazy' | 'eager';
|
||||
};
|
||||
|
||||
const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json');
|
||||
|
||||
+124
-10
@@ -1,27 +1,47 @@
|
||||
import { createMCPClient, type MCPClient } from '@ai-sdk/mcp';
|
||||
import { Experimental_StdioMCPTransport } from '@ai-sdk/mcp/mcp-stdio';
|
||||
import type { ToolSet } from 'ai';
|
||||
import { tool, type ToolSet } from 'ai';
|
||||
import { z } from 'zod';
|
||||
|
||||
export type McpServerConfig =
|
||||
| { command: string; args?: string[]; env?: Record<string, string>; cwd?: string }
|
||||
| { url: string; type?: 'http' | 'sse'; headers?: Record<string, string> };
|
||||
|
||||
/**
|
||||
* How a server's tools reach the model.
|
||||
*
|
||||
* `eager` registers every tool with its schema up front — cheap for a two-tool
|
||||
* server, a tax for one that exposes twenty. `lazy` registers only the three
|
||||
* meta-tools below and fetches a server's tools on demand via `mcp_call`, so a
|
||||
* configured server costs almost nothing in the request until a tool is actually
|
||||
* invoked.
|
||||
*/
|
||||
export type McpMode = 'eager' | 'lazy';
|
||||
|
||||
export type McpHandle = {
|
||||
tools: ToolSet;
|
||||
errors: { server: string; message: string }[];
|
||||
/** Server names, for the prompt's MCP line. Empty when the mode is eager. */
|
||||
servers: string[];
|
||||
close: () => Promise<void>;
|
||||
};
|
||||
|
||||
const isRemote = (c: McpServerConfig): c is Extract<McpServerConfig, { url: string }> => 'url' in c;
|
||||
|
||||
/**
|
||||
* Connects every configured server and namespaces its tools as `mcp__<server>__<tool>`
|
||||
* so two servers exposing `search` cannot silently shadow each other.
|
||||
* A server that fails to start is reported, never fatal.
|
||||
* Connects every configured server and exposes its tools.
|
||||
*
|
||||
* In `eager` mode the tools land in the returned set as `mcp__<server>__<tool>`, so
|
||||
* two servers exposing `search` cannot silently shadow each other. In `lazy` mode the
|
||||
* set holds only the three meta-tools and `servers` names the configured servers; a
|
||||
* server that fails to start is reported, never fatal, in either mode.
|
||||
*/
|
||||
export async function connectMcp(servers: Record<string, McpServerConfig>): Promise<McpHandle> {
|
||||
export async function connectMcp(
|
||||
servers: Record<string, McpServerConfig>,
|
||||
mode: McpMode = 'lazy',
|
||||
): Promise<McpHandle> {
|
||||
const clients: MCPClient[] = [];
|
||||
const tools: ToolSet = {};
|
||||
const byName = new Map<string, MCPClient>();
|
||||
const errors: McpHandle['errors'] = [];
|
||||
|
||||
await Promise.all(
|
||||
@@ -38,9 +58,7 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
|
||||
}),
|
||||
});
|
||||
clients.push(client);
|
||||
for (const [toolName, tool] of Object.entries(await client.tools())) {
|
||||
tools[`mcp__${name}__${toolName}`] = tool;
|
||||
}
|
||||
byName.set(name, client);
|
||||
} catch (e) {
|
||||
errors.push({ server: name, message: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
@@ -48,10 +66,106 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
|
||||
);
|
||||
|
||||
return {
|
||||
tools,
|
||||
tools: mode === 'eager' ? await eagerTools(byName) : lazyTools(byName),
|
||||
errors,
|
||||
servers: mode === 'eager' ? [] : [...byName.keys()],
|
||||
close: async () => {
|
||||
await Promise.all(clients.map((c) => c.close().catch(() => {})));
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** Eager: every server tool gets an AI tool registered with its schema. */
|
||||
async function eagerTools(byName: Map<string, MCPClient>): Promise<ToolSet> {
|
||||
const tools: ToolSet = {};
|
||||
await Promise.all(
|
||||
[...byName.entries()].map(async ([name, client]) => {
|
||||
for (const [toolName, t] of Object.entries(await client.tools())) {
|
||||
tools[`mcp__${name}__${toolName}`] = t;
|
||||
}
|
||||
}),
|
||||
);
|
||||
return tools;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cache of each client's AI tools, so `mcp_call` does not re-list on every call.
|
||||
* The WeakMap drops entries when a client (and its session) is closed and collected.
|
||||
*/
|
||||
const clientToolsCache = new WeakMap<MCPClient, ToolSet>();
|
||||
|
||||
/** Fetches a server's AI tools, or returns the cached set from a prior call. */
|
||||
async function cachedClientTools(client: MCPClient): Promise<ToolSet> {
|
||||
const cached = clientToolsCache.get(client);
|
||||
if (cached) return cached;
|
||||
const tools = await client.tools();
|
||||
clientToolsCache.set(client, tools);
|
||||
return tools;
|
||||
}
|
||||
|
||||
/**
|
||||
* Lazy: three meta-tools instead of every server schema.
|
||||
*
|
||||
* `mcp_list` names a server's tools from their definitions (cheap, no schema).
|
||||
* `mcp_inspect` reads one tool's schema so the model knows its inputs.
|
||||
* `mcp_call` executes one tool on its server, making the server's tools available
|
||||
* only from the moment they are actually invoked.
|
||||
*/
|
||||
function lazyTools(byName: Map<string, MCPClient>): ToolSet {
|
||||
const connectionError = (name: string) =>
|
||||
byName.has(name) ? undefined : `unknown server "${name}". Configured: ${[...byName.keys()].join(', ') || 'none'}`;
|
||||
|
||||
return {
|
||||
mcp_list: tool({
|
||||
description: `List the tools exposed by an MCP server. Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
|
||||
inputSchema: z.object({ server: z.string().describe('Server name from your instructions') }),
|
||||
execute: async ({ server }) => {
|
||||
const err = connectionError(server);
|
||||
if (err) throw new Error(err);
|
||||
const client = byName.get(server)!;
|
||||
const defs = await client.listTools();
|
||||
return defs.tools.length === 0
|
||||
? `server "${server}" exposes no tools`
|
||||
: defs.tools.map((d) => `- ${d.name}: ${d.description ?? 'no description'}`).join('\n');
|
||||
},
|
||||
}),
|
||||
mcp_inspect: tool({
|
||||
description: `Inspect one tool's input schema on an MCP server. Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
|
||||
inputSchema: z.object({
|
||||
server: z.string().describe('Server name from your instructions'),
|
||||
toolName: z.string().describe('Tool name, as listed by mcp_list'),
|
||||
}),
|
||||
execute: async ({ server, toolName }) => {
|
||||
const err = connectionError(server);
|
||||
if (err) throw new Error(err);
|
||||
const client = byName.get(server)!;
|
||||
const defs = await client.listTools();
|
||||
const def = defs.tools.find((d) => d.name === toolName);
|
||||
if (!def) throw new Error(`No tool "${toolName}" on "${server}". List first with mcp_list.`);
|
||||
return def.inputSchema ? JSON.stringify(def.inputSchema, null, 2) : `tool "${toolName}" declares no input schema`;
|
||||
},
|
||||
}),
|
||||
mcp_call: tool({
|
||||
description:
|
||||
`Call one tool on an MCP server. Inspect its schema with mcp_inspect first. ` +
|
||||
`Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
|
||||
inputSchema: z.object({
|
||||
server: z.string().describe('Server name from your instructions'),
|
||||
toolName: z.string().describe('Tool name, as listed by mcp_list'),
|
||||
args: z.record(z.string(), z.unknown()).describe('Arguments the tool expects, from mcp_inspect'),
|
||||
}),
|
||||
execute: async ({ server, toolName, args }) => {
|
||||
const err = connectionError(server);
|
||||
if (err) throw new Error(err);
|
||||
const client = byName.get(server)!;
|
||||
const serverTools = await cachedClientTools(client);
|
||||
const aiTool = serverTools[toolName];
|
||||
if (!aiTool)
|
||||
throw new Error(
|
||||
`No tool "${toolName}" on "${server}". List first with mcp_list (server exposes: ${Object.keys(serverTools).join(', ') || 'none'}).`,
|
||||
);
|
||||
return aiTool.execute!(args, { toolCallId: 'mcp_call', messages: [] } as never);
|
||||
},
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
+5
-4
@@ -181,6 +181,11 @@ export const DEFAULT_PERMISSIONS: PermissionConfig = {
|
||||
apply_patch: 'ask',
|
||||
move_file: 'ask',
|
||||
delete_file: 'ask',
|
||||
insert_lines: 'ask',
|
||||
delete_lines: 'ask',
|
||||
replace_lines: 'ask',
|
||||
append_file: 'ask',
|
||||
prepend_file: 'ask',
|
||||
bash: 'ask',
|
||||
web_fetch: 'ask',
|
||||
};
|
||||
@@ -272,10 +277,6 @@ export class Permissions {
|
||||
this.granted.set(tool, set);
|
||||
}
|
||||
|
||||
granted_(tool: string): string[] {
|
||||
return [...(this.granted.get(tool) ?? [])];
|
||||
}
|
||||
|
||||
/** The decision for one call, and which pattern decided it. */
|
||||
check(tool: string, input: unknown): Resolved {
|
||||
const resolved = resolve(entryFor(tool, this.config), tool, input);
|
||||
|
||||
@@ -4,6 +4,10 @@ export type Rate = { inputPerMTok: number; outputPerMTok: number };
|
||||
* USD per million tokens. Prefix match on the model id, longest first, so
|
||||
* `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates
|
||||
* drift, so this is a best-effort estimate rather than a billing source.
|
||||
*
|
||||
* Source: vendor pricing pages, checked 2026-09-17. Anthropic (Anthropic API, not
|
||||
* Batch) and OpenAI listed rates; DeepSeek and Grok per their API pricing. Rates
|
||||
* are for input, then output. Re-verify before trusting a live spend figure.
|
||||
*/
|
||||
const RATES: Record<string, Rate> = {
|
||||
'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 },
|
||||
|
||||
+10
-2
@@ -131,8 +131,10 @@ function renderTools(available: readonly string[]): string {
|
||||
// free, and the schema already says what each takes.
|
||||
const git = extra.filter((n) => GIT_TOOL_NAMES.includes(n) && n !== 'git_commit_message');
|
||||
const mcp = extra.filter((n) => n.startsWith('mcp__'));
|
||||
const META = ['mcp_list', 'mcp_inspect', 'mcp_call'];
|
||||
const lazyMcp = META.filter((n) => available.includes(n));
|
||||
const other = extra.filter(
|
||||
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__'),
|
||||
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__') && !META.includes(n),
|
||||
);
|
||||
|
||||
if (git.length > 0) {
|
||||
@@ -140,7 +142,13 @@ function renderTools(available: readonly string[]): string {
|
||||
`- ${git.join(', ')}: read-only git, no approval needed. Use them instead of bash for history and diffs; they cannot mutate the repository.`,
|
||||
);
|
||||
}
|
||||
if (mcp.length > 0) {
|
||||
if (lazyMcp.length > 0) {
|
||||
// Lazy mode: the meta-tool descriptions already name the connected servers, so
|
||||
// the model needs the workflow, not a schema listing.
|
||||
lines.push(
|
||||
`- ${lazyMcp.join(', ')}: MCP tools are fetched on demand. mcp_list names a server's tools, mcp_inspect reads one tool's schema, mcp_call runs it. Never guess a server or tool name: list first.`,
|
||||
);
|
||||
} else if (mcp.length > 0) {
|
||||
lines.push(
|
||||
`- ${mcp.join(', ')}: from MCP servers, named mcp__<server>__<tool>. Each needs approval; read its own description before calling.`,
|
||||
);
|
||||
|
||||
+78
-2
@@ -92,8 +92,6 @@ export function detachOrphanedItems(before: ModelMessage[], after: ModelMessage[
|
||||
|
||||
export type PruneOptions = Parameters<typeof pruneMessages>[0];
|
||||
|
||||
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
|
||||
|
||||
const anyParts = (message: ModelMessage): Part[] =>
|
||||
Array.isArray(message.content) ? (message.content as Part[]) : [];
|
||||
|
||||
@@ -167,6 +165,84 @@ export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
|
||||
return dropOrphanedResults(detachOrphanedItems(options.messages, pruned));
|
||||
}
|
||||
|
||||
/**
|
||||
* The messages a prune would discard, so they can be summarized before they go.
|
||||
*
|
||||
* Compaction keeps the model's *memory of a turn* — the tool tail it is told to
|
||||
* keep stays verbatim. What it does not keep is any statement of what was
|
||||
* dropped. So a decision from forty messages ago vanishes silently, and the model
|
||||
* contradicts it with full confidence, because as far as it can tell it never
|
||||
* said that.
|
||||
*
|
||||
* Identity is by reference, not by value: `prunePreservingItems` rebuilds the
|
||||
* surviving messages with `{ ...message }`, so a value comparison would report
|
||||
* every message as changed and no message as dropped. `Set` on the object
|
||||
* references is exact.
|
||||
*
|
||||
* Only messages that carry content worth summarizing are returned — an assistant
|
||||
* turn consisting of nothing but a dropped `reasoning` part is not a decision, and
|
||||
* summarizing "the model thought for a while" is worse than saying nothing.
|
||||
*/
|
||||
export function droppedBy(before: ModelMessage[], after: ModelMessage[]): ModelMessage[] {
|
||||
const surviving = new Set<ModelMessage>(after);
|
||||
return before.filter((message) => !surviving.has(message));
|
||||
}
|
||||
|
||||
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
|
||||
|
||||
function textOf(message: ModelMessage): string {
|
||||
const { content } = message;
|
||||
if (typeof content === 'string') return content;
|
||||
if (!Array.isArray(content)) return '';
|
||||
const chunks: string[] = [];
|
||||
for (const part of content as Part[]) {
|
||||
const p = part as Part & { text?: unknown; input?: unknown; output?: unknown };
|
||||
if (typeof p.text === 'string') chunks.push(p.text);
|
||||
// A tool call's input is the decision made: the path, the command, the patch.
|
||||
else if (p.type === 'tool-call' && p.input !== undefined) chunks.push(JSON.stringify(p.input));
|
||||
// A tool result is what came back. Without it a digest says what the model
|
||||
// asked for and nothing about the answer, which is the half a later
|
||||
// contradiction is usually argued from.
|
||||
else if (ANSWER_PARTS.has(p.type) && p.output !== undefined) {
|
||||
const rendered = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
|
||||
chunks.push(rendered);
|
||||
}
|
||||
}
|
||||
return chunks.join(' ').trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* A one-line-per-message digest of what a prune wants to drop.
|
||||
*
|
||||
* This is the *fallback* when no summarizer is available or the call fails: crude,
|
||||
* but it preserves the thing that matters — which tool touched which path, and in
|
||||
* what order — rather than the nothing that is there today. The summarizer, when
|
||||
* it runs, is a model and reads far better than this.
|
||||
*/
|
||||
export function digestOf(dropped: readonly ModelMessage[]): string {
|
||||
const lines: string[] = [];
|
||||
for (const message of dropped) {
|
||||
const text = textOf(message);
|
||||
if (!text) continue;
|
||||
const role = message.role === 'tool' ? 'result' : message.role;
|
||||
const clipped = text.length > 160 ? `${text.slice(0, 160)}...` : text;
|
||||
lines.push(`- (${role}) ${clipped}`);
|
||||
}
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
export const PRUNED_SPAN_PREFIX = 'Earlier in this session, now compacted away:';
|
||||
|
||||
export function isPrunedSpanSummary(message: ModelMessage): boolean {
|
||||
return message.role === 'user' && typeof message.content === 'string' && message.content.startsWith(PRUNED_SPAN_PREFIX);
|
||||
}
|
||||
|
||||
export function prunedSpanMessage(summary: string | undefined, dropped: readonly ModelMessage[]): ModelMessage | undefined {
|
||||
const body = summary?.trim() || digestOf(dropped);
|
||||
if (!body) return undefined;
|
||||
return { role: 'user', content: `${PRUNED_SPAN_PREFIX}\n\n${body}` };
|
||||
}
|
||||
|
||||
/**
|
||||
* How many trailing messages keep their tool content, widest first.
|
||||
*
|
||||
|
||||
+270
-4
@@ -17,7 +17,9 @@ import { Permissions, type PermissionConfig } from './permission';
|
||||
import type { PluginHost } from './plugins';
|
||||
import { costOf, formatUsd } from './pricing';
|
||||
import { systemPrompt } from './prompt';
|
||||
import { detachProviderItems, pruneToFit } from './prune';
|
||||
import { detachProviderItems, digestOf, droppedBy, isPrunedSpanSummary, prunedSpanMessage, pruneToFit } from './prune';
|
||||
import { restore, Snapshots, type TurnSnapshot } from './snapshot';
|
||||
import { createStepBackTool, type LoopEntry } from './step-back';
|
||||
import { createSkillTool, renderSkills, type Skill } from './skills';
|
||||
import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools';
|
||||
|
||||
@@ -38,6 +40,21 @@ export type ApprovalRequest = {
|
||||
/** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */
|
||||
export type ApprovalDecision = 'once' | 'always' | 'deny';
|
||||
|
||||
/** What an undo did, so the UI can say which files moved and which did not. */
|
||||
export type UndoResult = {
|
||||
snapshot: TurnSnapshot;
|
||||
restored: string[];
|
||||
removed: string[];
|
||||
conversationTrimmed: boolean;
|
||||
};
|
||||
|
||||
export type RedoResult = {
|
||||
snapshot: TurnSnapshot;
|
||||
/** False when files were asked for: only pre-images are ever captured. */
|
||||
filesRestored: boolean;
|
||||
what: 'both' | 'files' | 'conversation';
|
||||
};
|
||||
|
||||
export type AgentEvent =
|
||||
| { type: 'text'; text: string }
|
||||
| { type: 'reasoning'; text: string }
|
||||
@@ -74,6 +91,8 @@ export type SessionOptions = {
|
||||
autoApprove?: readonly string[];
|
||||
/** Prune the history once the estimated token count crosses this. */
|
||||
compactThreshold?: number;
|
||||
/** Identical calls to an allowed tool before it is asked about anyway. Default 3. */
|
||||
repeatLimit?: number;
|
||||
/** Retries per model call for transient failures. */
|
||||
maxRetries?: number;
|
||||
/** AGENTS.md-style files appended to the system prompt. */
|
||||
@@ -94,6 +113,12 @@ export type SessionOptions = {
|
||||
onNotebookChange?: (state: NotebookState) => void;
|
||||
};
|
||||
|
||||
/**
|
||||
* Length-based token estimate for deciding *when to prune*, not for billing.
|
||||
* JSON char count / 4 approximates token count closely enough to gate compaction,
|
||||
* but real billed tokens come from the SDK's reported usage (`inputTokens`), never
|
||||
* from here. `/cost` and the budget ceiling use the SDK figure.
|
||||
*/
|
||||
const estimateTokens = (messages: ModelMessage[]) => Math.round(JSON.stringify(messages).length / 4);
|
||||
|
||||
/** Estimated tokens at which the wire history is pruned. */
|
||||
@@ -102,6 +127,26 @@ const DEFAULT_COMPACT_THRESHOLD = 120_000;
|
||||
/** Identical calls in one turn before an allowed tool is asked about anyway. */
|
||||
const REPEAT_LIMIT = 3;
|
||||
|
||||
/**
|
||||
* Squashes a tool result into a few characters for the loop trace.
|
||||
*
|
||||
* The trace is fed back to the model verbatim, so a 30 KB read_file output would
|
||||
* fill the reflection with noise. A short string keeps `step_back` honest about
|
||||
* what happened without flooding the next context window.
|
||||
*/
|
||||
function summarizeToolResult(output: unknown): string {
|
||||
if (typeof output === 'string') return output.length <= 80 ? output : `${output.slice(0, 80)}…(${output.length} chars)`;
|
||||
try {
|
||||
const json = JSON.stringify(output);
|
||||
return json.length <= 80 ? json : `${json.slice(0, 80)}…`;
|
||||
} catch {
|
||||
return String(output);
|
||||
}
|
||||
}
|
||||
|
||||
/** Ceiling on an injected span summary, so the summary cannot defeat the compaction. */
|
||||
const MAX_SPAN_SUMMARY_CHARS = 1_200;
|
||||
|
||||
const callKey = (toolName: string, input: unknown) => `${toolName}:${JSON.stringify(input ?? null)}`;
|
||||
|
||||
/**
|
||||
@@ -121,6 +166,8 @@ export class Session {
|
||||
readonly messages: ModelMessage[];
|
||||
readonly tools: ToolSet;
|
||||
readonly notebook: Notebook;
|
||||
/** Pre-images of files this session's turns have changed, newest last. */
|
||||
readonly snapshots: Snapshots;
|
||||
inputTokens = 0;
|
||||
outputTokens = 0;
|
||||
/** Subagent token use, priced against the subagent's own model id in /cost. */
|
||||
@@ -136,6 +183,14 @@ export class Session {
|
||||
/** The 80% spend warning is shown once, not on every turn past the line. */
|
||||
private warnedSpend = false;
|
||||
private controller: AbortController | undefined;
|
||||
/** Tools this turn used that no snapshot can cover, reported when the turn ends. */
|
||||
private readonly uncoveredTools = new Set<string>();
|
||||
/** The snapshot the last undo removed, so `/redo` can put it back. */
|
||||
private lastUndone: TurnSnapshot | undefined;
|
||||
/** Every tool call this turn, input + outcome, for the loop-detection tool to reflect on. */
|
||||
private readonly loopTrace: LoopEntry[] = [];
|
||||
/** toolCallId -> { toolName, input }, so a result can be paired with its call. */
|
||||
private readonly callInputs = new Map<string, { toolName: string; input: string }>();
|
||||
|
||||
constructor(private readonly opts: SessionOptions) {
|
||||
this.messages = opts.messages ?? [];
|
||||
@@ -143,12 +198,19 @@ export class Session {
|
||||
this.notebook.restore(opts.notebook);
|
||||
this.model = opts.model;
|
||||
this.variant = opts.agent ?? DEFAULT_VARIANT;
|
||||
this.snapshots = new Snapshots(opts.cwd ?? process.cwd());
|
||||
|
||||
const sessionTools = {
|
||||
...this.notebook.tools(),
|
||||
...(opts.memory ? opts.memory.tools() : {}),
|
||||
...(opts.skills && opts.skills.length > 0 ? { skill: createSkillTool(opts.skills) } : {}),
|
||||
// Always registered, even with no skills, so a mid-session install of the
|
||||
// first skill is callable next turn without a session rebuild. The tool's
|
||||
// description reads the live list and says "none" when it is empty.
|
||||
skill: createSkillTool(() => this.opts.skills ?? []),
|
||||
...(opts.ask ? { ask: createAskTool(opts.ask) } : {}),
|
||||
step_back: createStepBackTool({
|
||||
trace: () => this.loopTrace,
|
||||
}),
|
||||
};
|
||||
this.tools = { ...builtinTools, ...sessionTools, ...(opts.plugins?.tools ?? {}), ...(opts.extraTools ?? {}) };
|
||||
|
||||
@@ -306,6 +368,26 @@ export class Session {
|
||||
return count;
|
||||
}
|
||||
|
||||
/**
|
||||
* Appends a completed tool call to the loop trace, pairing it with its input.
|
||||
*
|
||||
* The SDK streams `tool-result` without the input that produced it, so the input is
|
||||
* kept alongside on the `tool-call` part. A `step_back` call needs this pairing to
|
||||
* say *which* call produced *which* outcome.
|
||||
*/
|
||||
private recordTrace(toolCallId: string, result: string): void {
|
||||
const call = this.callInputs.get(toolCallId);
|
||||
this.callInputs.delete(toolCallId);
|
||||
if (!call) return;
|
||||
this.loopTrace.push({
|
||||
step: this.loopTrace.length + 1,
|
||||
toolName: call.toolName,
|
||||
input: call.input,
|
||||
result,
|
||||
at: new Date().toISOString(),
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Approval decisions, evaluated per call by the SDK.
|
||||
*
|
||||
@@ -326,6 +408,11 @@ export class Session {
|
||||
return async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
|
||||
const { toolName, input } = toolCall;
|
||||
|
||||
// Before anything else, because the guard may deny the call and because a
|
||||
// later hook must not be able to move the capture after the write.
|
||||
const { covered } = await this.snapshots.captureFor(toolName, input);
|
||||
if (!covered) this.uncoveredTools.add(toolName);
|
||||
|
||||
const blocked = await this.opts.plugins?.guard({
|
||||
toolName,
|
||||
input,
|
||||
@@ -346,7 +433,18 @@ export class Session {
|
||||
}
|
||||
|
||||
const repeats = this.repeatCount(toolName, input);
|
||||
if (decision === 'allow' && repeats < REPEAT_LIMIT) return undefined;
|
||||
const limit = this.opts.repeatLimit ?? REPEAT_LIMIT;
|
||||
if (decision === 'allow' && repeats < limit) return undefined;
|
||||
|
||||
if (decision === 'allow') {
|
||||
// Repeated three times with a permission that says `allow`: the model is
|
||||
// looping, not asking, and it should stop and look at the trace rather than
|
||||
// burn another approval. This is the point the step_back tool exists for.
|
||||
notices.push(
|
||||
`You have called ${toolName} with the same input ${repeats + 1} times this turn. It is not making progress. ` +
|
||||
`Use step_back to reflect on what changed between attempts, then try a different approach or stop.`,
|
||||
);
|
||||
}
|
||||
|
||||
why.set(callKey(toolName, input), {
|
||||
...(pattern ? { matchedPattern: pattern } : {}),
|
||||
@@ -378,6 +476,92 @@ export class Session {
|
||||
return { before, after: this.messages.length };
|
||||
}
|
||||
|
||||
/**
|
||||
* Walks the last turn back: files, conversation, or both.
|
||||
*
|
||||
* The three-way split is the point. Restoring files without the conversation leaves
|
||||
* the model believing edits are on disk that are not, so its next turn is built on a
|
||||
* state that no longer exists — it re-reads a file expecting its own change and finds
|
||||
* the original, which reads to the model as the change having been rejected. Restoring
|
||||
* the conversation without the files is the mirror: the model forgets it made an edit
|
||||
* that is still there. So the default is both, and the caller can narrow it.
|
||||
*
|
||||
* Returns undefined when there is nothing to undo, which the UI reports as such
|
||||
* rather than as a failure.
|
||||
*/
|
||||
async undo(what: 'both' | 'files' | 'conversation' = 'both'): Promise<UndoResult | undefined> {
|
||||
const snap = this.snapshots.pop();
|
||||
if (!snap) return undefined;
|
||||
|
||||
const files = what === 'conversation' ? { restored: [], removed: [] } : await restore(snap, this.snapshots.cwdOf());
|
||||
if (what !== 'files') this.trimTo(snap.messageCount);
|
||||
|
||||
this.lastUndone = snap;
|
||||
return { snapshot: snap, ...files, conversationTrimmed: what !== 'files' };
|
||||
}
|
||||
|
||||
/**
|
||||
* Puts back what `undo` took, without a second snapshot.
|
||||
*
|
||||
* A redo cannot restore file content from the session, because the content that
|
||||
* existed after the turn was never captured — only the pre-image was. So a redo of
|
||||
* the files is declined honestly rather than approximated: the whole point of undo
|
||||
* is that the user trusts what it says it did. The conversation is restored from the
|
||||
* snapshot's own record, which is exact.
|
||||
*/
|
||||
redo(what: 'both' | 'files' | 'conversation' = 'conversation'): RedoResult | undefined {
|
||||
const snap = this.lastUndone;
|
||||
if (!snap) return undefined;
|
||||
|
||||
this.snapshots.push(snap);
|
||||
this.lastUndone = undefined;
|
||||
return { snapshot: snap, filesRestored: false, what };
|
||||
}
|
||||
|
||||
undoable(): readonly TurnSnapshot[] {
|
||||
return this.snapshots.list();
|
||||
}
|
||||
|
||||
private trimTo(length: number): void {
|
||||
if (this.messages.length <= length) return;
|
||||
this.messages.length = Math.max(0, length);
|
||||
this.opts.onChange?.(this.messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes the open snapshot and reports what the turn could not cover.
|
||||
*
|
||||
* Called before every `done`, not from `send`'s `finally`, because a notice that
|
||||
* arrives after `done` is a notice the UI has already stopped listening for —
|
||||
* `done` is what a consumer treats as the end of the turn and stops on.
|
||||
*
|
||||
* Two reasons to speak, and the second does not depend on the first: a turn with no
|
||||
* snapshotted edits can still have run `bash` and changed the tree, which is exactly
|
||||
* when the warning matters most.
|
||||
*/
|
||||
private *closingNotices(): Generator<AgentEvent> {
|
||||
const snapshot = this.snapshots.commit();
|
||||
const uncovered = [...this.uncoveredTools];
|
||||
if (uncovered.length === 0) return;
|
||||
const covered = snapshot ? `${snapshot.files.length} file(s) changed this turn and can be undone with /undo. ` : '';
|
||||
yield {
|
||||
type: 'notice',
|
||||
text: `${covered}${uncovered.join(', ')} ran this turn and cannot be snapshotted, so any changes it made will not be undone.`,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Swaps the live skill list at a turn boundary.
|
||||
*
|
||||
* The `skill` tool and the system-prompt catalogue both read the list on each
|
||||
* call, so replacing it here is atomic across the two and takes effect on the
|
||||
* next turn with no session rebuild. The caller is responsible for only doing
|
||||
* this between turns — a turn in flight already holds its rules.
|
||||
*/
|
||||
setSkills(skills: Skill[]): void {
|
||||
this.opts.skills = skills;
|
||||
}
|
||||
|
||||
async *send(userText: string): AsyncGenerator<AgentEvent> {
|
||||
// The ceiling is checked before the model is: a turn started past the limit
|
||||
// would spend money the caller said not to. An unpriced model cannot be
|
||||
@@ -402,7 +586,12 @@ export class Session {
|
||||
// Per turn, not per step: a tool called once in each of three steps is the
|
||||
// loop this guards against.
|
||||
this.seen.clear();
|
||||
this.uncoveredTools.clear();
|
||||
this.loopTrace.length = 0;
|
||||
this.staleItemsRepaired = false;
|
||||
// Opened before the model runs and closed after it stops, so every write the
|
||||
// turn makes lands in one snapshot the user can walk back to.
|
||||
this.snapshots.begin(userText, this.messages.length);
|
||||
|
||||
const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = [];
|
||||
onBashOutput(({ toolCallId, chunk }) => {
|
||||
@@ -414,6 +603,10 @@ export class Session {
|
||||
yield* this.run(signal, threshold, outputs);
|
||||
} finally {
|
||||
onBashOutput(undefined);
|
||||
// Belt and braces: closingNotices runs before every `done`, but an aborted turn
|
||||
// can return through a path that never reached it, and an uncommitted snapshot
|
||||
// would then be silently dropped rather than kept.
|
||||
this.snapshots.commit();
|
||||
await this.opts.plugins?.afterTurn();
|
||||
}
|
||||
}
|
||||
@@ -431,6 +624,70 @@ export class Session {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Prunes the canonical history once a run has landed.
|
||||
*
|
||||
* prepareStep only trims the wire copy for the next request; without this write-back
|
||||
* the stored history keeps growing, the context meter pins at 100%, and every later
|
||||
* turn re-prunes the same messages from scratch.
|
||||
*/
|
||||
private async compactCanonical(threshold: number): Promise<{ before: number; after: number } | null> {
|
||||
const before = this.messages.length;
|
||||
if (estimateTokens(this.messages) <= threshold) return null;
|
||||
const pruned = pruneToFit({ messages: this.messages, threshold, estimate: estimateTokens });
|
||||
if (pruned.length === before) return null;
|
||||
const dropped = droppedBy(this.messages, pruned);
|
||||
const summary = await this.summarizeSpan(dropped.filter((m) => !isPrunedSpanSummary(m)));
|
||||
this.replace(this.withSummarizedSpan(summary, pruned, dropped));
|
||||
return { before, after: this.messages.length };
|
||||
}
|
||||
|
||||
/**
|
||||
* Puts a summary of the discarded span at the head of the history it was dropped from.
|
||||
*
|
||||
* Without this the model is told which tool results to keep and nothing about what
|
||||
* was dropped, so it states a decision it made forty messages ago as though it had
|
||||
* never made it.
|
||||
*
|
||||
* The summary is bounded two ways because a summary that grows with the session
|
||||
* defeats the point of compacting at all: the input is capped at the span's own
|
||||
* digest, and the output is capped by instruction and by hard truncation. A failed
|
||||
* or empty call falls back to the digest, which costs nothing and still carries
|
||||
* which tool touched which path — the part a contradiction is usually built from.
|
||||
*/
|
||||
private async summarizeSpan(dropped: readonly ModelMessage[]): Promise<string | undefined> {
|
||||
const digest = digestOf(dropped);
|
||||
if (!digest) return undefined;
|
||||
try {
|
||||
const { text } = await generateText({
|
||||
model: this.model,
|
||||
system:
|
||||
'These lines are the condensed record of an earlier part of a coding session that has been ' +
|
||||
'compacted out of the conversation. Write at most 120 words of notes capturing decisions made, ' +
|
||||
'files touched, commands run and their outcome, and anything still pending. State only what the ' +
|
||||
'lines support; do not invent detail and do not address the reader.',
|
||||
messages: [{ role: 'user', content: digest }],
|
||||
maxRetries: this.opts.maxRetries ?? 3,
|
||||
});
|
||||
const trimmed = text.trim();
|
||||
return trimmed ? trimmed.slice(0, MAX_SPAN_SUMMARY_CHARS) : undefined;
|
||||
} catch {
|
||||
// A summarizer that cannot run must not cost the turn its compaction: the
|
||||
// digest is a worse record, not an absent one.
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
private withSummarizedSpan(summary: string | undefined, pruned: ModelMessage[], dropped: readonly ModelMessage[]): ModelMessage[] {
|
||||
const worthSummarizing = dropped.filter((m) => !isPrunedSpanSummary(m));
|
||||
if (worthSummarizing.length === 0) return pruned;
|
||||
|
||||
const prior = dropped.filter(isPrunedSpanSummary);
|
||||
const spanMessage = prunedSpanMessage(summary, worthSummarizing);
|
||||
if (!spanMessage) return pruned;
|
||||
return [...prior, spanMessage, ...pruned];
|
||||
}
|
||||
|
||||
private async *run(
|
||||
signal: AbortSignal,
|
||||
threshold: number,
|
||||
@@ -506,12 +763,15 @@ export class Session {
|
||||
yield { type: 'tool-start', id: part.id, name: part.toolName };
|
||||
break;
|
||||
case 'tool-call':
|
||||
this.callInputs.set(part.toolCallId, { toolName: part.toolName, input: JSON.stringify(part.input ?? null) });
|
||||
yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input };
|
||||
break;
|
||||
case 'tool-result':
|
||||
this.recordTrace(part.toolCallId, summarizeToolResult(part.output));
|
||||
yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output };
|
||||
break;
|
||||
case 'tool-error':
|
||||
this.recordTrace(part.toolCallId, `error: ${part.error instanceof Error ? part.error.message : String(part.error)}`);
|
||||
yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error };
|
||||
break;
|
||||
case 'tool-approval-request': {
|
||||
@@ -535,6 +795,7 @@ export class Session {
|
||||
yield { type: 'tool-denied', name: part.toolName };
|
||||
break;
|
||||
case 'abort':
|
||||
yield* this.closingNotices();
|
||||
yield { type: 'done' };
|
||||
return;
|
||||
case 'error':
|
||||
@@ -555,6 +816,7 @@ export class Session {
|
||||
}
|
||||
} catch (error) {
|
||||
if (signal.aborted) {
|
||||
yield* this.closingNotices();
|
||||
yield { type: 'done' };
|
||||
return;
|
||||
}
|
||||
@@ -579,7 +841,10 @@ export class Session {
|
||||
while (guardNotices.length > 0) yield { type: 'notice', text: guardNotices.shift()! };
|
||||
|
||||
this.messages.push(...(await result.responseMessages));
|
||||
this.opts.onChange?.(this.messages);
|
||||
|
||||
// prepareStep already emits `compacted` at the same threshold crossing, and
|
||||
// replace() fires onChange on the fold path; both would double up otherwise.
|
||||
if (!(await this.compactCanonical(threshold))) this.opts.onChange?.(this.messages);
|
||||
|
||||
if (pending.length === 0) {
|
||||
const usage = await result.usage;
|
||||
@@ -595,6 +860,7 @@ export class Session {
|
||||
text: `approaching spend ceiling: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used`,
|
||||
};
|
||||
}
|
||||
yield* this.closingNotices();
|
||||
yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens };
|
||||
return;
|
||||
}
|
||||
|
||||
+14
-4
@@ -100,18 +100,28 @@ export function renderSkills(skills: Skill[]): string {
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
export function createSkillTool(skills: Skill[]) {
|
||||
const names = skills.map((s) => s.name);
|
||||
/**
|
||||
* The `skill` tool, reading the list live so a mid-session install shows up on the
|
||||
* next turn without a restart.
|
||||
*
|
||||
* The catalogue in the system prompt and the names in this tool's description both
|
||||
* come from the same list on each read, so replacing the list at a turn boundary
|
||||
* makes both stale bytes atomic: a skill the model can call it can see already.
|
||||
*/
|
||||
export function createSkillTool(getSkills: () => Skill[]) {
|
||||
const names = () => getSkills().map((s) => s.name).join(', ');
|
||||
return tool({
|
||||
description:
|
||||
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
|
||||
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`,
|
||||
`what you are about to do, then follow what it says. Available: ${names() || 'none'}.`,
|
||||
inputSchema: z.object({
|
||||
name: z.string().describe('Skill name from the list in your instructions'),
|
||||
}),
|
||||
execute: async ({ name }) => {
|
||||
const skills = getSkills();
|
||||
const skill = skills.find((s) => s.name === name.trim().toLowerCase());
|
||||
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`);
|
||||
if (!skill)
|
||||
throw new Error(`No skill named "${name}". Available: ${skills.map((s) => s.name).join(', ') || 'none'}`);
|
||||
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
|
||||
},
|
||||
});
|
||||
|
||||
+269
@@ -0,0 +1,269 @@
|
||||
import { createHash } from 'node:crypto';
|
||||
import { join, relative, resolve, sep } from 'node:path';
|
||||
|
||||
/**
|
||||
* Pre-images of files a turn is about to change, so a turn can be walked back.
|
||||
*
|
||||
* The design follows the one thing every comparable CLI agrees on and the one thing
|
||||
* they all get wrong in the same way. Agreement: a snapshot is taken *before* the
|
||||
* turn runs, because after it runs the original content is gone. The shared flaw: a
|
||||
* snapshot only covers the tools that write through a known interface — Claude Code
|
||||
* tracks Write/Edit/NotebookEdit and explicitly not `bash` — so the undo is partial
|
||||
* and the user has to know which half they are in.
|
||||
*
|
||||
* Two decisions fall out of taking that seriously.
|
||||
*
|
||||
* 1. Capture lazily, per file, not by walking the tree. A session turn may touch
|
||||
* three files out of fifty thousand; a full-tree copy per prompt is a tarball of
|
||||
* the repository the user did not ask for and cannot afford. Instead the first
|
||||
* write to a path records its pre-image, and a later write to the same path in
|
||||
* the same turn does not overwrite it — the pre-image is the state before the
|
||||
* *turn*, which is what an undo restores.
|
||||
*
|
||||
* 2. Record what is *not* covered rather than implying it is. A `bash` command that
|
||||
* rewrites a file, a `git checkout`, a build artifact — none of these pass through
|
||||
* a path argument the way `write_file` does, so none are captured. The tool
|
||||
* reports which paths it restored and the caller is told the rest is unknown,
|
||||
* which is the same honesty `bash` already owes an interrupted command.
|
||||
*/
|
||||
|
||||
/** One file as it was before the turn that changed it. `before === undefined` means it did not exist. */
|
||||
export type PreImage = {
|
||||
/** Workspace-relative, using forward slashes, so a restore is portable across platforms. */
|
||||
path: string;
|
||||
before: string | undefined;
|
||||
};
|
||||
|
||||
export type TurnSnapshot = {
|
||||
/** Monotonic turn number, so the UI can name what is being undone. */
|
||||
turn: number;
|
||||
at: string;
|
||||
/** The user prompt that opened the turn, for a menu that lists them. */
|
||||
prompt: string;
|
||||
/** Files this turn changed, with their content from before it started. */
|
||||
files: PreImage[];
|
||||
/** The conversation length when the turn began, so undo can trim it back. */
|
||||
messageCount: number;
|
||||
};
|
||||
|
||||
/** Claude Code keeps 100; beyond that the memory is worth more than the recall. */
|
||||
export const MAX_SNAPSHOTS = 100;
|
||||
|
||||
/** A single file larger than this is not snapshotted; a 40 MB binary is not an edit. */
|
||||
const MAX_FILE_BYTES = 2 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* The workspace-relative, slash-normalised form of a path, or undefined if it is
|
||||
* outside the workspace.
|
||||
*
|
||||
* Outside is refused rather than clamped: a path that escapes the workspace is not
|
||||
* something this repository can undo, and recording it would imply an undo that
|
||||
* cannot happen. Paths are normalised to forward slashes because a snapshot written
|
||||
* on Windows may be read on a machine where a backslash is a filename character.
|
||||
*/
|
||||
export function relPath(cwd: string, abs: string): string | undefined {
|
||||
const root = resolve(cwd);
|
||||
const target = resolve(abs);
|
||||
const rel = relative(root, target);
|
||||
if (rel === '' || rel.startsWith('..') || rel.includes(`..${sep}`)) return undefined;
|
||||
return rel.split(sep).join('/');
|
||||
}
|
||||
|
||||
/**
|
||||
* The absolute path a tool call will write to, when there is exactly one.
|
||||
*
|
||||
* Deliberately a small, explicit map rather than a guess. `multi_edit` and the line
|
||||
* editors each take a single `path`; `move_file` takes `from` and `to` and both are
|
||||
* recorded; `delete_file` takes a `path`. `apply_patch` carries its paths inside the
|
||||
* patch text, and `bash` carries none — both are reported as uncovered rather than
|
||||
* silently not snapshotted.
|
||||
*/
|
||||
export function touchedPaths(toolName: string, input: unknown): { paths: string[]; covered: boolean } {
|
||||
const o = (input ?? {}) as Record<string, unknown>;
|
||||
const one = (key: string) => (typeof o[key] === 'string' ? [o[key] as string] : []);
|
||||
|
||||
switch (toolName) {
|
||||
case 'write_file':
|
||||
case 'edit_file':
|
||||
case 'multi_edit':
|
||||
case 'delete_file':
|
||||
case 'insert_lines':
|
||||
case 'delete_lines':
|
||||
case 'replace_lines':
|
||||
case 'append_file':
|
||||
case 'prepend_file':
|
||||
return { paths: one('path'), covered: true };
|
||||
case 'move_file':
|
||||
return { paths: [...one('from'), ...one('to')], covered: true };
|
||||
case 'apply_patch': {
|
||||
const patch = typeof o['patch'] === 'string' ? (o['patch'] as string) : '';
|
||||
const paths = [...patch.matchAll(/^\*\*\* (?:Add|Update|Delete) File: (.+)$/gm)].map((m) => m[1]!.trim());
|
||||
const moves = [...patch.matchAll(/^\*\*\* Move to: (.+)$/gm)].map((m) => m[1]!.trim());
|
||||
return { paths: [...paths, ...moves], covered: true };
|
||||
}
|
||||
case 'bash':
|
||||
// Arbitrary code: an untouched-looking `node -e` can rewrite the tree.
|
||||
return { paths: [], covered: false };
|
||||
default:
|
||||
return { paths: [], covered: true };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Records pre-images for the files a turn changes, and restores them on undo.
|
||||
*
|
||||
* One instance per session. Holds at most MAX_SNAPSHOTS turns; the oldest falls off
|
||||
* the front, because the turn someone wants back is almost always the last one.
|
||||
*/
|
||||
export class Snapshots {
|
||||
private readonly turns: TurnSnapshot[] = [];
|
||||
private current: { turn: number; at: string; prompt: string; files: Map<string, string | undefined>; messageCount: number } | null =
|
||||
null;
|
||||
private next = 1;
|
||||
private readonly cwd: string;
|
||||
|
||||
constructor(cwd: string = process.cwd()) {
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
/** Opens a turn. Called once per user prompt, before the model runs. */
|
||||
begin(prompt: string, messageCount: number): void {
|
||||
this.current = { turn: this.next++, at: new Date().toISOString(), prompt, files: new Map(), messageCount };
|
||||
}
|
||||
|
||||
/**
|
||||
* Records a file's content before a tool changes it, on the first write of the turn.
|
||||
*
|
||||
* Idempotent per path per turn: the second `write_file` to the same path in one turn
|
||||
* must not replace the pre-image with the intermediate content the first write left,
|
||||
* because undo restores the turn's starting state, not the midpoint.
|
||||
*
|
||||
* Read failures are swallowed. A snapshot is a convenience; a tool call that fails
|
||||
* because the snapshot layer could not read an unrelated path would be worse than no
|
||||
* undo at all.
|
||||
*/
|
||||
async capture(absPath: string): Promise<void> {
|
||||
if (!this.current) return;
|
||||
const rel = relPath(this.cwd, absPath);
|
||||
if (rel === undefined) return;
|
||||
if (this.current.files.has(rel)) return;
|
||||
|
||||
try {
|
||||
const file = Bun.file(absPath);
|
||||
const exists = await file.exists();
|
||||
if (!exists) {
|
||||
this.current.files.set(rel, undefined);
|
||||
return;
|
||||
}
|
||||
if (file.size > MAX_FILE_BYTES) return;
|
||||
this.current.files.set(rel, await file.text());
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
/** Records any path a tool call is about to touch. Returns whether the tool is covered at all. */
|
||||
async captureFor(toolName: string, input: unknown): Promise<{ covered: boolean; paths: string[] }> {
|
||||
const { paths, covered } = touchedPaths(toolName, input);
|
||||
for (const p of paths) await this.capture(resolve(this.cwd, p));
|
||||
return { paths, covered };
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes the turn, keeping it only if it changed something.
|
||||
*
|
||||
* A turn that read and answered without writing is not worth a slot, and keeping it
|
||||
* would make `/undo` step past a turn that has nothing to undo — which reads as the
|
||||
* command being broken.
|
||||
*/
|
||||
commit(): TurnSnapshot | undefined {
|
||||
const cur = this.current;
|
||||
this.current = null;
|
||||
if (!cur || cur.files.size === 0) return undefined;
|
||||
|
||||
const snap: TurnSnapshot = {
|
||||
turn: cur.turn,
|
||||
at: cur.at,
|
||||
prompt: cur.prompt,
|
||||
files: [...cur.files].map(([path, before]) => ({ path, before })),
|
||||
messageCount: cur.messageCount,
|
||||
};
|
||||
this.turns.push(snap);
|
||||
while (this.turns.length > MAX_SNAPSHOTS) this.turns.shift();
|
||||
return snap;
|
||||
}
|
||||
|
||||
/** Discards the open turn without recording it, for an aborted or failed turn. */
|
||||
discard(): void {
|
||||
this.current = null;
|
||||
}
|
||||
|
||||
/** The turns that can be undone, newest first. */
|
||||
list(): readonly TurnSnapshot[] {
|
||||
return [...this.turns].reverse();
|
||||
}
|
||||
|
||||
/** Whether there is an open turn collecting pre-images right now. */
|
||||
get open(): boolean {
|
||||
return this.current !== null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Removes the newest turn and returns what it holds, without restoring.
|
||||
*
|
||||
* Separated from restoring so the caller can decide *what* to bring back —
|
||||
* files, conversation, or both — which is the split Claude Code's rewind menu
|
||||
* exposes and the reason one control surface is worth more than three commands.
|
||||
*/
|
||||
pop(): TurnSnapshot | undefined {
|
||||
return this.turns.pop();
|
||||
}
|
||||
|
||||
/** Puts a turn back, for a `/redo` that follows an `/undo`. */
|
||||
push(snap: TurnSnapshot): void {
|
||||
this.turns.push(snap);
|
||||
}
|
||||
|
||||
get size(): number {
|
||||
return this.turns.length;
|
||||
}
|
||||
|
||||
cwdOf(): string {
|
||||
return this.cwd;
|
||||
}
|
||||
}
|
||||
|
||||
/** Writes a pre-image back to disk, recreating a deleted file or removing one that was created. */
|
||||
export async function restore(snap: TurnSnapshot, cwd = process.cwd()): Promise<{ restored: string[]; removed: string[] }> {
|
||||
const restored: string[] = [];
|
||||
const removed: string[] = [];
|
||||
|
||||
for (const file of snap.files) {
|
||||
const abs = join(cwd, file.path);
|
||||
if (file.before === undefined) {
|
||||
// The file did not exist before the turn, so undoing its creation is removing it.
|
||||
const f = Bun.file(abs);
|
||||
if (await f.exists()) {
|
||||
await f.delete();
|
||||
removed.push(file.path);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
await Bun.write(abs, file.before);
|
||||
restored.push(file.path);
|
||||
}
|
||||
|
||||
return { restored, removed };
|
||||
}
|
||||
|
||||
/** A short, stable label for a snapshot, for a menu that lists several. */
|
||||
export function labelOf(snap: TurnSnapshot): string {
|
||||
const first = snap.prompt.trim().split('\n')[0] ?? '';
|
||||
const clipped = first.length > 50 ? `${first.slice(0, 50)}...` : first || '(no prompt)';
|
||||
return `turn ${snap.turn}: ${clipped}`;
|
||||
}
|
||||
|
||||
/** A content hash, used to tell whether a file still matches what the snapshot holds. */
|
||||
export function hashOf(text: string): string {
|
||||
return createHash('sha256').update(text).digest('hex').slice(0, 12);
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
import { tool } from 'ai';
|
||||
import { z } from 'zod';
|
||||
|
||||
/**
|
||||
* A visible escape hatch for the loop a coding agent dies in.
|
||||
*
|
||||
* The repeat guard stops an *identical* call after three tries, but the deeper loop
|
||||
* is the model making *different* calls that all amount to the same stalled attempt —
|
||||
* re-reading the same file expecting a different answer, retrying a failing command
|
||||
* with a tweaked flag, re-sending a prompt it has already asked. No equal-input
|
||||
* detector fires on any of that, so the model burns the step budget on motions that
|
||||
* never move.
|
||||
*
|
||||
* `step_back` exists so the model has a *named* way out instead of only a guard it
|
||||
* cannot see. The session records each completed tool call (input + outcome) into a
|
||||
* loop trace; the tool returns a scripted reflection prompt built from that trace,
|
||||
* so the model is told in concrete terms that it is going in circles and is steered
|
||||
* to change direction.
|
||||
*/
|
||||
|
||||
export type LoopEntry = {
|
||||
step: number;
|
||||
toolName: string;
|
||||
input: string;
|
||||
result: string;
|
||||
at: string;
|
||||
};
|
||||
|
||||
function recentTrace(entries: readonly LoopEntry[], window = 8): LoopEntry[] {
|
||||
return entries.slice(-window);
|
||||
}
|
||||
|
||||
function renderTrace(entries: readonly LoopEntry[], maxLines = 12): string {
|
||||
return entries.slice(-maxLines).map((e) => `step ${e.step}: ${e.toolName} ${e.input} -> ${e.result}`.slice(0, 200)).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds a `step_back` tool bound to a session's loop trace.
|
||||
*
|
||||
* The tool is meant to be called when the model is not making progress — a trap it
|
||||
* cannot always see while it is inside it. The returned reflection names the recent
|
||||
* steps so the model can tell, from its own calls, that it is going in circles, and
|
||||
* gives it the one thing a stuck agent is usually missing: permission to stop,
|
||||
* say what it learned, and change direction rather than try harder.
|
||||
*/
|
||||
export function createStepBackTool(opts: { trace: () => readonly LoopEntry[] }) {
|
||||
return tool({
|
||||
description:
|
||||
'Use when you are stuck: the same file is not changing, a command keeps failing, or you have done several steps with no visible progress. ' +
|
||||
'Records your recent steps and returns a reflection prompt to help you change direction instead of repeating the attempt.',
|
||||
inputSchema: z.object({
|
||||
note: z.string().optional().describe('A sentence in your own words about what you were trying to do.'),
|
||||
}),
|
||||
execute: async ({ note }) => {
|
||||
const trace = recentTrace(opts.trace());
|
||||
if (trace.length === 0) {
|
||||
return (
|
||||
'No recent steps to reflect on. This is an early call of step_back — it is only useful when you have ' +
|
||||
'attempted something several times. Describe what you are stuck on in `note`.'
|
||||
);
|
||||
}
|
||||
|
||||
return [
|
||||
'You have run threadbare over the last steps and are stuck. Here is what you actually did:',
|
||||
'```',
|
||||
renderTrace(trace),
|
||||
'```',
|
||||
'',
|
||||
'Before your next tool call, answer these three questions in your reasoning:',
|
||||
'1. What exactly is wrong — the input, the tool, or the expectation?',
|
||||
'2. What have you tried, and why did each fail?',
|
||||
'3. What is ONE different thing you can do that is not "try the same thing a little harder"?',
|
||||
'',
|
||||
'Then take that different action. If the failure is a command, read the actual error and fix its cause — ',
|
||||
'do not rerun the command. If a file is not what you expect, suspect your assumption about it and re-read it fresh.',
|
||||
note?.trim() ? `\nYour note: ${note.trim()}` : '',
|
||||
].join('\n');
|
||||
},
|
||||
});
|
||||
}
|
||||
+128
-82
@@ -177,91 +177,137 @@ export function createTaskTool(opts: {
|
||||
? 'explore: read-only research. review: read-only critique. worker: makes changes. Default explore.'
|
||||
: 'explore: find and report. review: critique code for defects. Default explore.',
|
||||
),
|
||||
tasks: z
|
||||
.array(
|
||||
z.object({
|
||||
description: z.string(),
|
||||
prompt: z.string(),
|
||||
kind: z.enum(canWrite ? ['explore', 'review', 'worker'] : ['explore', 'review']).optional(),
|
||||
}),
|
||||
)
|
||||
.optional()
|
||||
.describe(
|
||||
'Independent investigations to run at the same time instead of one after another. ' +
|
||||
'Use this for several unrelated searches so they overlap in wall-clock time. Each runs on its own ' +
|
||||
'context window, exactly like a single task call. Default: run the single prompt above.',
|
||||
),
|
||||
}),
|
||||
execute: async ({ description, prompt, kind }, { abortSignal }) => {
|
||||
const flavour: SubagentKind = kind ?? 'explore';
|
||||
if (flavour === 'worker' && !opts.approve) {
|
||||
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
|
||||
}
|
||||
|
||||
const id = `sub${++counter}`;
|
||||
const report = opts.report;
|
||||
report?.({ type: 'start', id, kind: flavour, description });
|
||||
|
||||
let steps = 0;
|
||||
let text = '';
|
||||
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
|
||||
|
||||
try {
|
||||
// `explore` is search, not reasoning, so it runs on the cheaper model when
|
||||
// one is configured. `review` and `worker` keep the parent's: they judge
|
||||
// and they change, both of which want the full model.
|
||||
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
|
||||
const result = streamText({
|
||||
model,
|
||||
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
|
||||
messages: [{ role: 'user', content: prompt }],
|
||||
tools: TOOLS[flavour],
|
||||
stopWhen: isStepCount(opts.maxSteps ?? 20),
|
||||
...(opts.approve
|
||||
? {
|
||||
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
|
||||
const approved = await opts.approve!(toolCall);
|
||||
return approved
|
||||
? undefined
|
||||
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
...(abortSignal ? { abortSignal } : {}),
|
||||
});
|
||||
|
||||
const sink = () => {};
|
||||
void result.responseMessages.then(undefined, sink);
|
||||
void result.usage.then(undefined, sink);
|
||||
void result.steps.then(undefined, sink);
|
||||
void result.finalStep.then(undefined, sink);
|
||||
void result.finishReason.then(undefined, sink);
|
||||
|
||||
for await (const part of result.stream) {
|
||||
if (part.type === 'tool-call') {
|
||||
steps++;
|
||||
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
|
||||
} else if (part.type === 'tool-result') {
|
||||
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
|
||||
} else if (part.type === 'tool-error') {
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
|
||||
} else if (part.type === 'text-delta') {
|
||||
text += part.text;
|
||||
} else if (part.type === 'error') {
|
||||
// A provider failure arrives as a stream part, not a throw, so it has to
|
||||
// be rethrown here or the subagent silently returns nothing.
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
throw part.error instanceof Error ? part.error : new Error(message);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const usage = await result.usage;
|
||||
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
|
||||
} catch {
|
||||
// A run that errored before producing usage has nothing to account for.
|
||||
}
|
||||
} catch (e) {
|
||||
const message = e instanceof Error ? e.message : String(e);
|
||||
report?.({ type: 'error', id, message });
|
||||
throw e;
|
||||
}
|
||||
|
||||
const trimmed = text.trim();
|
||||
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
|
||||
// Settled after the stream closes; a failed run reports nothing rather than
|
||||
// a half count. The parent prices these against the subagent's own model id.
|
||||
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
|
||||
return trimmed || 'Subagent returned no findings.';
|
||||
execute: async ({ description, prompt, kind, tasks }, { abortSignal }) => {
|
||||
const plans = tasks?.length ? tasks : [{ description, prompt, kind }];
|
||||
const runs = await Promise.all(
|
||||
plans.map((p) => runSubagent({ ...opts, abortSignal }, { description: p.description, prompt: p.prompt, kind: p.kind })),
|
||||
);
|
||||
if (runs.length === 1) return runs[0]!.report;
|
||||
return runs.map((r) => `## ${r.named}\n\n${r.report}`).join('\n\n---\n\n');
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
/** Flavour of one planned investigation; `kind` defaults to explore. */
|
||||
type Plan = { description: string; prompt: string; kind?: string };
|
||||
|
||||
/**
|
||||
* Runs one subagent and settles its report, usage, and panel events.
|
||||
*
|
||||
* Extracted so `tasks` can fan several out in parallel: each full run is
|
||||
* independent — its own id, own stream, own spend — and they overlap simply by
|
||||
* awaiting them together.
|
||||
*/
|
||||
async function runSubagent(
|
||||
opts: {
|
||||
model: LanguageModel;
|
||||
subagentModel?: LanguageModel;
|
||||
subagentModelId?: string;
|
||||
cwd?: string;
|
||||
maxSteps?: number;
|
||||
report?: SubagentReporter;
|
||||
approve?: SubagentApproval;
|
||||
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
|
||||
abortSignal?: AbortSignal;
|
||||
},
|
||||
plan: Plan,
|
||||
): Promise<{ named: string; report: string }> {
|
||||
const flavour: SubagentKind = (plan.kind as SubagentKind | undefined) ?? 'explore';
|
||||
if (flavour === 'worker' && !opts.approve) {
|
||||
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
|
||||
}
|
||||
|
||||
const id = `sub${++counter}`;
|
||||
const report = opts.report;
|
||||
report?.({ type: 'start', id, kind: flavour, description: plan.description });
|
||||
|
||||
let steps = 0;
|
||||
let text = '';
|
||||
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
|
||||
|
||||
try {
|
||||
// `explore` is search, not reasoning, so it runs on the cheaper model when
|
||||
// one is configured. `review` and `worker` keep the parent's: they judge
|
||||
// and they change, both of which want the full model.
|
||||
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
|
||||
const result = streamText({
|
||||
model,
|
||||
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
|
||||
messages: [{ role: 'user', content: plan.prompt }],
|
||||
tools: TOOLS[flavour],
|
||||
stopWhen: isStepCount(opts.maxSteps ?? 20),
|
||||
...(opts.approve
|
||||
? {
|
||||
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
|
||||
const approved = await opts.approve!(toolCall);
|
||||
return approved
|
||||
? undefined
|
||||
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
...(opts.abortSignal ? { abortSignal: opts.abortSignal } : {}),
|
||||
});
|
||||
|
||||
const sink = () => {};
|
||||
void result.responseMessages.then(undefined, sink);
|
||||
void result.usage.then(undefined, sink);
|
||||
void result.steps.then(undefined, sink);
|
||||
void result.finalStep.then(undefined, sink);
|
||||
void result.finishReason.then(undefined, sink);
|
||||
|
||||
for await (const part of result.stream) {
|
||||
if (part.type === 'tool-call') {
|
||||
steps++;
|
||||
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
|
||||
} else if (part.type === 'tool-result') {
|
||||
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
|
||||
} else if (part.type === 'tool-error') {
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
|
||||
} else if (part.type === 'text-delta') {
|
||||
text += part.text;
|
||||
} else if (part.type === 'error') {
|
||||
// A provider failure arrives as a stream part, not a throw, so it has to
|
||||
// be rethrown here or the subagent silently returns nothing.
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
throw part.error instanceof Error ? part.error : new Error(message);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const usage = await result.usage;
|
||||
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
|
||||
} catch {
|
||||
// A run that errored before producing usage has nothing to account for.
|
||||
}
|
||||
} catch (e) {
|
||||
const message = e instanceof Error ? e.message : String(e);
|
||||
report?.({ type: 'error', id, message });
|
||||
throw e;
|
||||
}
|
||||
|
||||
const trimmed = text.trim();
|
||||
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
|
||||
// Settled after the stream closes; a failed run reports nothing rather than
|
||||
// a half count. The parent prices these against the subagent's own model id.
|
||||
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
|
||||
return { named: plan.description, report: trimmed || 'Subagent returned no findings.' };
|
||||
}
|
||||
|
||||
export const TASK_TOOL_NAME = 'task';
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
import type { Tool } from 'ai';
|
||||
|
||||
/**
|
||||
* Marks a tool as mutating where it is defined, rather than in a list beside it.
|
||||
*
|
||||
* The bug this prevents is the worst one this codebase can have: a new tool that
|
||||
* writes to the workspace, added to `tools` but forgotten in a hand-maintained
|
||||
* `MUTATING_TOOLS`, is a write the permission layer does not treat as a write. It
|
||||
* is silent, it passes every test that does not think to check the new name, and
|
||||
* it surfaces as a user discovering an edit they never approved.
|
||||
*
|
||||
* The mark is a property on the tool object, so `MUTATING_TOOLS` can be derived by
|
||||
* filtering the registry instead of being typed out. A tool that is not marked is
|
||||
* asserted non-mutating by `tools.test.ts`, which means the decision is made once,
|
||||
* at the definition, and cannot drift.
|
||||
*/
|
||||
export const MUTATING = '__mutating' as const;
|
||||
|
||||
/** A tool that can change the workspace or run arbitrary code. */
|
||||
export function mutating<T extends Tool>(t: T): T {
|
||||
return Object.assign(t, { [MUTATING]: true as const });
|
||||
}
|
||||
|
||||
/** Whether a tool was declared mutating at its definition site. */
|
||||
export function isMutating(t: unknown): boolean {
|
||||
return typeof t === 'object' && t !== null && (t as Record<string, unknown>)[MUTATING] === true;
|
||||
}
|
||||
+11
-10
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
|
||||
import { resolve } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { jail, posix, walk } from './ignore';
|
||||
import { mutating } from './tool-kinds';
|
||||
import { git } from './tools-git';
|
||||
|
||||
/**
|
||||
@@ -36,7 +37,7 @@ async function readLines(path: string): Promise<{ abs: string; lines: string[] }
|
||||
// edit
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const insertLinesTool = tool({
|
||||
export const insertLinesTool = mutating(tool({
|
||||
description:
|
||||
'Insert lines at a 1-based position in a file, pushing the rest down. Cheaper and safer than a rewrite for adding a block in the middle.',
|
||||
inputSchema: z.object({
|
||||
@@ -51,9 +52,9 @@ export const insertLinesTool = tool({
|
||||
await Bun.write(abs, cur.join('\n'));
|
||||
return `Inserted ${lines(text).length} line(s) at ${path}:${line}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const deleteLinesTool = tool({
|
||||
export const deleteLinesTool = mutating(tool({
|
||||
description: 'Delete an inclusive range of lines from a file. Refuses to delete the whole file; use delete_file for that.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
@@ -69,9 +70,9 @@ export const deleteLinesTool = tool({
|
||||
await Bun.write(abs, cur.join('\n'));
|
||||
return `Deleted lines ${start}-${end} from ${path}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const replaceLinesTool = tool({
|
||||
export const replaceLinesTool = mutating(tool({
|
||||
description: 'Replace an inclusive range of lines with new text, in one write.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
@@ -87,9 +88,9 @@ export const replaceLinesTool = tool({
|
||||
await Bun.write(abs, cur.join('\n'));
|
||||
return `Replaced lines ${start}-${end} in ${path}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const appendFileTool = tool({
|
||||
export const appendFileTool = mutating(tool({
|
||||
description: 'Append text to the end of a file without reading the whole thing into the edit.',
|
||||
inputSchema: z.object({ path: z.string(), text: z.string() }),
|
||||
execute: async ({ path, text }) => {
|
||||
@@ -97,9 +98,9 @@ export const appendFileTool = tool({
|
||||
await Bun.write(abs, `${cur.join('\n').replace(/\n?$/, '\n')}${text.replace(/\n?$/, '')}\n`);
|
||||
return `Appended ${lines(text).length} line(s) to ${path}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const prependFileTool = tool({
|
||||
export const prependFileTool = mutating(tool({
|
||||
description: 'Prepend text to the start of a file, e.g. a license header or an import block.',
|
||||
inputSchema: z.object({ path: z.string(), text: z.string() }),
|
||||
execute: async ({ path, text }) => {
|
||||
@@ -107,7 +108,7 @@ export const prependFileTool = tool({
|
||||
await Bun.write(abs, `${text.replace(/\n?$/, '\n')}${cur.join('\n')}`);
|
||||
return `Prepended ${lines(text).length} line(s) to ${path}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const countLinesTool = tool({
|
||||
description: 'Count lines in one file, or per file across a glob. A quick size read before deciding to open something large.',
|
||||
|
||||
+115
-46
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
|
||||
import { join, resolve } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { jail, posix, walk } from './ignore';
|
||||
import { isMutating, mutating } from './tool-kinds';
|
||||
import { EXTRA_TOOL_NAMES, extraTools } from './tools-extra';
|
||||
import { GIT_TOOL_NAMES, gitTools } from './tools-git';
|
||||
import { NET_TOOL_NAMES, netTools } from './tools-net';
|
||||
@@ -174,7 +175,7 @@ export function parsePatch(patch: string): PatchOp[] {
|
||||
return ops;
|
||||
}
|
||||
|
||||
export const applyPatchTool = tool({
|
||||
export const applyPatchTool = mutating(tool({
|
||||
description:
|
||||
'Apply one patch across several files: add, update, move, and delete in a single call. All or nothing — if any ' +
|
||||
'part fails, nothing is written. Use it when a change spans files that must land together, such as a rename ' +
|
||||
@@ -248,7 +249,7 @@ export const applyPatchTool = tool({
|
||||
|
||||
return `Applied ${ops.length} change${ops.length === 1 ? '' : 's'}:\n${summary.map((s) => `- ${s}`).join('\n')}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
/**
|
||||
* A rewrite that collapses whitespace: similar character count, a fraction of the lines.
|
||||
@@ -264,7 +265,7 @@ function collapsedRewrite(before: string, after: string): boolean {
|
||||
return after.split('\n').length < before.split('\n').length / 2;
|
||||
}
|
||||
|
||||
export const writeFileTool = tool({
|
||||
export const writeFileTool = mutating(tool({
|
||||
description: 'Create a file or overwrite it completely. Prefer edit_file for existing files.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
@@ -284,17 +285,39 @@ export const writeFileTool = tool({
|
||||
}
|
||||
return `Wrote ${content.length} chars to ${path}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const editFileTool = tool({
|
||||
// Some models (Claude-style tool docs, DeepSeek/GLM) emit snake_case edit params
|
||||
// (old_string/new_string/replace_all) despite the camelCase schema. Normalize at the
|
||||
// boundary instead of failing the whole call on a naming convention.
|
||||
const SNAKE_EDIT_ARGS: ReadonlyArray<readonly [string, string]> = [
|
||||
['old_string', 'oldString'],
|
||||
['new_string', 'newString'],
|
||||
['replace_all', 'replaceAll'],
|
||||
];
|
||||
|
||||
export function normalizeEditArgs(input: unknown): unknown {
|
||||
if (typeof input !== 'object' || input === null) return input;
|
||||
const obj: Record<string, unknown> = { ...(input as Record<string, unknown>) };
|
||||
for (const [snake, camel] of SNAKE_EDIT_ARGS) {
|
||||
if (obj[snake] !== undefined && obj[camel] === undefined) obj[camel] = obj[snake];
|
||||
}
|
||||
if (Array.isArray(obj.edits)) obj.edits = obj.edits.map(normalizeEditArgs);
|
||||
return obj;
|
||||
}
|
||||
|
||||
export const editFileTool = mutating(tool({
|
||||
description:
|
||||
'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
|
||||
}),
|
||||
inputSchema: z.preprocess(
|
||||
normalizeEditArgs,
|
||||
z.object({
|
||||
path: z.string(),
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
|
||||
}),
|
||||
),
|
||||
execute: async ({ path, oldString, newString, replaceAll = false }) => {
|
||||
if (oldString === newString) throw new Error('oldString and newString are identical');
|
||||
const abs = jail(path);
|
||||
@@ -312,27 +335,30 @@ export const editFileTool = tool({
|
||||
await Bun.write(abs, after);
|
||||
return `Replaced ${replaceAll ? count : 1} occurrence(s) in ${path}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const multiEditTool = tool({
|
||||
export const multiEditTool = mutating(tool({
|
||||
description:
|
||||
'Apply several exact-string edits to one file in a single call. Each edit sees the result of the previous one. ' +
|
||||
'All or nothing: if any oldString fails to match, or matches more than once without replaceAll, nothing is ' +
|
||||
'written. Prefer this over repeated edit_file calls on the same file — one approval, one write, no risk of ' +
|
||||
'leaving the file half-changed.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
edits: z
|
||||
.array(
|
||||
z.object({
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional(),
|
||||
}),
|
||||
)
|
||||
.min(1)
|
||||
.describe('Edits in the order they should be applied'),
|
||||
}),
|
||||
inputSchema: z.preprocess(
|
||||
normalizeEditArgs,
|
||||
z.object({
|
||||
path: z.string(),
|
||||
edits: z
|
||||
.array(
|
||||
z.object({
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional(),
|
||||
}),
|
||||
)
|
||||
.min(1)
|
||||
.describe('Edits in the order they should be applied'),
|
||||
}),
|
||||
),
|
||||
execute: async ({ path, edits }) => {
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
@@ -367,7 +393,7 @@ export const multiEditTool = tool({
|
||||
await Bun.write(abs, text);
|
||||
return `Applied ${edits.length} edit(s) to ${path} (${applied.join(', ')})`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const globTool = tool({
|
||||
description:
|
||||
@@ -573,7 +599,13 @@ async function pump(
|
||||
return all;
|
||||
}
|
||||
|
||||
type Running = { command: string; proc: Bun.Subprocess; interrupted: boolean; killed?: Promise<unknown> };
|
||||
type Running = {
|
||||
command: string;
|
||||
proc: Bun.Subprocess;
|
||||
interrupted: boolean;
|
||||
timedOut?: boolean;
|
||||
killed?: Promise<unknown>;
|
||||
};
|
||||
|
||||
const running = new Map<string, Running>();
|
||||
|
||||
@@ -615,6 +647,12 @@ function killTree(proc: Bun.Subprocess): Promise<unknown> {
|
||||
export function interruptBash(): string[] {
|
||||
const killed: string[] = [];
|
||||
for (const entry of running.values()) {
|
||||
// A second ctrl-c while the first killTree is still settling must not re-announce
|
||||
// the same command: the notice is the only proof the keypress did anything.
|
||||
if (entry.interrupted) {
|
||||
killed.push(entry.command);
|
||||
continue;
|
||||
}
|
||||
entry.interrupted = true;
|
||||
entry.killed = killTree(entry.proc);
|
||||
killed.push(entry.command);
|
||||
@@ -622,7 +660,7 @@ export function interruptBash(): string[] {
|
||||
return killed;
|
||||
}
|
||||
|
||||
export const bashTool = tool({
|
||||
export const bashTool = mutating(tool({
|
||||
description:
|
||||
'Run a shell command in the workspace root. Use for builds, tests, git, and package managers. ' +
|
||||
'Output streams live and the user can interrupt a command with ctrl-c without ending the turn.',
|
||||
@@ -636,13 +674,32 @@ export const bashTool = tool({
|
||||
cwd: process.cwd(),
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
timeout,
|
||||
...(abortSignal ? { signal: abortSignal } : {}),
|
||||
});
|
||||
|
||||
const entry: Running = { command, proc, interrupted: false };
|
||||
running.set(toolCallId, entry);
|
||||
|
||||
// Bun's spawn `signal` option is not used either: it kills only the shell, so an
|
||||
// esc-abort orphaned the grandchild on the same still-open pipes as the timeout
|
||||
// did. The abort must go through killTree, exactly like ctrl-c does.
|
||||
const onAbort = () => {
|
||||
if (entry.interrupted) return;
|
||||
entry.interrupted = true;
|
||||
entry.killed = killTree(proc);
|
||||
};
|
||||
abortSignal?.addEventListener('abort', onAbort);
|
||||
// The turn may already be aborted by the time this tool starts; a past event
|
||||
// never re-fires, so check once here or the command runs unkillable by esc.
|
||||
if (abortSignal?.aborted) onAbort();
|
||||
|
||||
// Bun's own `timeout` spawn option is not used: it kills only the shell, and the
|
||||
// grandchild holding the output pipes keeps `pump` reading forever, so the tool
|
||||
// never returns. Same failure killTree exists for, just triggered by the clock.
|
||||
const timer = setTimeout(() => {
|
||||
entry.timedOut = true;
|
||||
entry.killed = killTree(proc);
|
||||
}, timeout);
|
||||
|
||||
try {
|
||||
// Drained concurrently: a command that fills one pipe while we block on the
|
||||
// other would deadlock, and buffering both hides progress for minutes.
|
||||
@@ -658,6 +715,15 @@ export const bashTool = tool({
|
||||
|
||||
// Thrown rather than returned: the model must not read a killed command as
|
||||
// a command that ran and failed on its own terms.
|
||||
if (entry.timedOut) {
|
||||
throw new Error(
|
||||
cap(
|
||||
`The command exceeded its ${timeout}ms timeout and was killed. It did not finish, so its effects are unknown.\n${
|
||||
body || '(no output before it was killed)'
|
||||
}`,
|
||||
),
|
||||
);
|
||||
}
|
||||
if (entry.interrupted) {
|
||||
throw new Error(
|
||||
cap(
|
||||
@@ -678,15 +744,17 @@ export const bashTool = tool({
|
||||
.join('\n\n'),
|
||||
);
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
abortSignal?.removeEventListener('abort', onAbort);
|
||||
// Awaited so the process really is gone before the tool returns. On Windows a
|
||||
// surviving grandchild holds the cwd open, which breaks the very next command.
|
||||
await entry.killed;
|
||||
running.delete(toolCallId);
|
||||
}
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const moveFileTool = tool({
|
||||
export const moveFileTool = mutating(tool({
|
||||
description:
|
||||
'Move or rename one file. Creates the target directory. Refuses if the source is missing or the target ' +
|
||||
'already exists, so a rename cannot silently overwrite work. For a rename plus its callers in one step, ' +
|
||||
@@ -708,9 +776,9 @@ export const moveFileTool = tool({
|
||||
await file.delete();
|
||||
return `Moved ${from} to ${to}`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
export const deleteFileTool = tool({
|
||||
export const deleteFileTool = mutating(tool({
|
||||
description:
|
||||
'Delete one file. Refuses a directory: removing a tree is what the guard plugin blocks in bash, and it is ' +
|
||||
'not something to do implicitly. Delete the files you mean, one call each.',
|
||||
@@ -733,7 +801,7 @@ export const deleteFileTool = tool({
|
||||
await Bun.file(abs).delete();
|
||||
return `Deleted ${path} (${entry.size} bytes)`;
|
||||
},
|
||||
});
|
||||
}));
|
||||
|
||||
/**
|
||||
* Definition patterns for `find_symbol`, keyed loosely by language.
|
||||
@@ -909,15 +977,16 @@ export function disabledToolNames(enabled: readonly ToolSetName[] | undefined):
|
||||
return TOOL_SET_NAMES.filter((set) => !live.has(set)).flatMap((set) => [...TOOL_SETS[set]]);
|
||||
}
|
||||
|
||||
/** Tools that mutate the workspace or run arbitrary code always ask the user first. */
|
||||
export const MUTATING_TOOLS = [
|
||||
'write_file',
|
||||
'edit_file',
|
||||
'multi_edit',
|
||||
'apply_patch',
|
||||
'move_file',
|
||||
'delete_file',
|
||||
'bash',
|
||||
] as const;
|
||||
/**
|
||||
* Tools that mutate the workspace or run arbitrary code always ask the user first.
|
||||
*
|
||||
* Derived from the tools themselves rather than typed out: a tool marked `mutating`
|
||||
* at its definition is in this list by construction, and there is no second place to
|
||||
* forget it. `tools.test.ts` asserts the converse — that nothing here is unmarked —
|
||||
* so the two cannot disagree.
|
||||
*/
|
||||
export const MUTATING_TOOLS: readonly string[] = Object.entries(tools)
|
||||
.filter(([, t]) => isMutating(t))
|
||||
.map(([name]) => name);
|
||||
|
||||
export { jail };
|
||||
|
||||
+69
-11
@@ -176,18 +176,34 @@ export function App({
|
||||
const fileMatches = token && paths ? matchPaths(paths, token.query) : [];
|
||||
const highlightedPath = fileMatches[Math.min(fileIndex, Math.max(0, fileMatches.length - 1))];
|
||||
|
||||
// The walk costs a full ignore-aware traversal, so it happens on the first `@`
|
||||
// rather than at startup, and only once.
|
||||
// The walk costs a full ignore-aware traversal, so it runs once at the first `@`.
|
||||
// A slow cooldown re-walks so a file created after that first `@` shows up within
|
||||
// a short window instead of staying hidden all session. The cooldown never fires
|
||||
// on a short session, so the "walks once" behaviour most users see is unchanged.
|
||||
const pathsRef = useRef(paths);
|
||||
useEffect(() => {
|
||||
if (token === undefined || paths !== undefined) return;
|
||||
let live = true;
|
||||
void hooks.listPaths().then((all) => {
|
||||
if (live) setPaths(all);
|
||||
});
|
||||
return () => {
|
||||
live = false;
|
||||
};
|
||||
}, [hooks, paths, token]);
|
||||
pathsRef.current = paths;
|
||||
}, [paths]);
|
||||
const didLoadRef = useRef(false);
|
||||
useEffect(() => {
|
||||
if (token === undefined) return;
|
||||
if (!didLoadRef.current) {
|
||||
didLoadRef.current = true;
|
||||
let live = true;
|
||||
void hooks.listPaths().then((all) => {
|
||||
if (live) setPaths(all);
|
||||
});
|
||||
const refresh = setInterval(async () => {
|
||||
if (!live) return;
|
||||
const all = await hooks.listPaths();
|
||||
if (live && JSON.stringify(all) !== JSON.stringify(pathsRef.current)) setPaths(all);
|
||||
}, 10_000);
|
||||
return () => {
|
||||
live = false;
|
||||
clearInterval(refresh);
|
||||
};
|
||||
}
|
||||
}, [hooks, token]);
|
||||
|
||||
useEffect(() => bridge.bind(setPending), [bridge]);
|
||||
useEffect(() => askBridge?.bind(setAsking), [askBridge]);
|
||||
@@ -691,6 +707,48 @@ export function App({
|
||||
setModelPicker(models);
|
||||
return;
|
||||
}
|
||||
case 'undo': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
try {
|
||||
const result = await session.undo(action.what);
|
||||
if (!result) {
|
||||
push({ kind: 'info', text: 'nothing to undo' });
|
||||
} else {
|
||||
const parts: string[] = [];
|
||||
if (result.restored.length > 0) parts.push(`restored ${result.restored.join(', ')}`);
|
||||
if (result.removed.length > 0) parts.push(`removed ${result.removed.join(', ')}`);
|
||||
if (result.conversationTrimmed) parts.push(`dropped ${result.snapshot.messageCount}-onward from the history`);
|
||||
push({ kind: 'info', text: `undid turn ${result.snapshot.turn}: ${parts.join('; ') || 'no file changes'}` });
|
||||
// The same limit the turn-end notice states: bash is not snapshotted.
|
||||
push({
|
||||
kind: 'info',
|
||||
text: 'Only file-tool edits are covered. A bash command or a git checkout in that turn is not, so check git status if one ran.',
|
||||
});
|
||||
}
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
return;
|
||||
}
|
||||
case 'redo': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
const result = session.redo(action.what);
|
||||
if (!result) {
|
||||
push({ kind: 'info', text: 'nothing to redo' });
|
||||
} else if (action.what === 'files' || action.what === 'both') {
|
||||
// Refused rather than approximated: only the pre-image was captured, so
|
||||
// there is no post-turn content to put back.
|
||||
push({
|
||||
kind: 'info',
|
||||
text: `put turn ${result.snapshot.turn} back on the undo stack. File contents cannot be re-applied - only the state before the turn was recorded.`,
|
||||
});
|
||||
} else {
|
||||
push({ kind: 'info', text: `put turn ${result.snapshot.turn} back; the conversation was restored to it` });
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 'compact': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
|
||||
+2
-1
@@ -1,6 +1,7 @@
|
||||
import { Box, Text, useInput } from 'ink';
|
||||
import React from 'react';
|
||||
import type { ApprovalDecision, ApprovalRequest } from '../session';
|
||||
import { normalizeEditArgs } from '../tools';
|
||||
import { Diff } from './Diff';
|
||||
import { accent, glyph } from './theme';
|
||||
import { toolDetail } from './transcript';
|
||||
@@ -42,7 +43,7 @@ export function createApprovalBridge(): ApprovalBridge {
|
||||
* consistent and far more readable than a JSON dump of the input.
|
||||
*/
|
||||
function ApprovalDetail({ name, input }: { name: string; input: unknown }) {
|
||||
const o = (input ?? {}) as Record<string, unknown>;
|
||||
const o = (normalizeEditArgs(input) ?? {}) as Record<string, unknown>;
|
||||
|
||||
if (name === 'write_file') {
|
||||
const content = String(o['content'] ?? '');
|
||||
|
||||
@@ -33,10 +33,6 @@ type KeyLike = {
|
||||
end?: boolean;
|
||||
};
|
||||
|
||||
const INVERSE_ON = '\u001B[7m';
|
||||
const INVERSE_OFF = '\u001B[27m';
|
||||
const invert = (s: string) => `${INVERSE_ON}${s}${INVERSE_OFF}`;
|
||||
|
||||
/**
|
||||
* Text input with a real cursor and shell-style history recall.
|
||||
*
|
||||
@@ -151,10 +147,10 @@ export function PromptInput({
|
||||
);
|
||||
|
||||
if (value.length === 0) {
|
||||
if (!placeholder) return <Text>{focus ? invert(' ') : ' '}</Text>;
|
||||
if (!placeholder) return <Text inverse={focus}>{' '}</Text>;
|
||||
return (
|
||||
<Text dimColor>
|
||||
{focus ? invert(placeholder.slice(0, 1)) : placeholder.slice(0, 1)}
|
||||
<Text inverse={focus}>{placeholder.slice(0, 1)}</Text>
|
||||
{placeholder.slice(1)}
|
||||
</Text>
|
||||
);
|
||||
@@ -166,7 +162,7 @@ export function PromptInput({
|
||||
return (
|
||||
<Text>
|
||||
{shown.slice(0, cursor)}
|
||||
{invert(shown.slice(cursor, cursor + 1) || ' ')}
|
||||
<Text inverse>{shown.slice(cursor, cursor + 1) || ' '}</Text>
|
||||
{shown.slice(cursor + 1)}
|
||||
</Text>
|
||||
);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { TODO_MARK } from '../notebook';
|
||||
import { normalizeEditArgs } from '../tools';
|
||||
|
||||
export type Line =
|
||||
| { key: string; kind: 'user'; text: string }
|
||||
@@ -38,7 +39,8 @@ export function preview(input: unknown): string {
|
||||
* the transcript, beside the spinner while a call is in flight, and in the approval
|
||||
* prompt for any tool without a diff of its own.
|
||||
*/
|
||||
export function toolDetail(name: string, input: unknown): string[] {
|
||||
export function toolDetail(name: string, rawInput: unknown): string[] {
|
||||
const input = normalizeEditArgs(rawInput);
|
||||
if (input === null || typeof input !== 'object') return [];
|
||||
const o = input as Record<string, unknown>;
|
||||
const str = (k: string) => (typeof o[k] === 'string' ? (o[k] as string) : undefined);
|
||||
|
||||
Reference in New Issue
Block a user