Ship the batch: undo, parallel subagents, lazy MCP, hot-reloaded skills; docs and CI/CD

Loop and ergonomics batch across Now/Next and Maintenance:
- /undo and /redo via pre-prompt file snapshots (snapshot.ts)
- task takes a tasks[] array and runs investigations concurrently (subagent.ts)
- lazy MCP tools: mcp_list/mcp_inspect/mcp_call meta-tools, eager opt-in (mcp.ts, config.ts)
- skill tool reads its list live so a mid-session install is callable next turn (skills.ts)
- tool-name lists (tool-kinds.ts) derived from a mutating() marker; gates previously ungated writes
- prune/session recovery path summarized, and step-back doom-loop primitive (step-back.ts)
- @file completion re-walks on a slow cooldown; estimateTokens and pricing labeled as estimates

Docs: README, CHANGELOG, docs/{mcp,architecture,development} updated to match.
CI/CD: bun install-store cache and concurrency gates on both workflows; release.yml now
composes file-based release notes via scripts/make-release-notes.ts and verifies every binary.
This commit is contained in:
Muhammad Zakir Ramadhan
2026-09-17 17:49:19 +07:00
parent ffa9a02c26
commit f8c3cc2d8e
57 changed files with 3808 additions and 573 deletions
+41 -9
View File
@@ -154,7 +154,8 @@ if (resumeArg) {
}
}
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
const mcp =
has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers, cfg.mcpMode ?? 'lazy');
const instructions = has('--no-instructions') ? [] : await loadInstructions();
const skills = has('--no-skills') ? [] : await loadSkills();
const customCommands = await loadCustomCommands();
@@ -424,8 +425,15 @@ const hooks: AppHooks = {
install: async (name) => {
const entry = await findEntry(name);
const { path } = await registry.install(entry);
// Loaded on the next start rather than hot-swapped: a skill joins the system
// prompt and a plugin joins the guard chain, and both are built once at boot.
// A skill is live immediately: the session's skill tool reads its list on each
// call, so the next turn can invoke a skill installed right now. A plugin or
// tool joins the guard chain and the tool registry, both built once at boot,
// so those still need a restart — said plainly rather than implied.
if (entry.kind === 'skill') {
const next = await loadSkills(process.cwd());
session.setSkills(next);
return `installed skill ${entry.name} to ${path}\nit is loaded and callable next turn`;
}
return `installed ${entry.kind} ${entry.name} to ${path}\nrestart shiro to load it`;
},
remove: async (name) => {
@@ -434,7 +442,14 @@ const hooks: AppHooks = {
const bare = parsed ? parsed[2]! : name;
for (const kind of kinds) {
if (await registry.uninstall(kind, bare)) return `removed ${kind} ${bare}\nrestart shiro to unload it`;
if (await registry.uninstall(kind, bare)) {
if (kind === 'skill') {
const next = await loadSkills(process.cwd());
session.setSkills(next);
return `removed skill ${bare}\nit is unloaded; the next turn no longer offers it`;
}
return `removed ${kind} ${bare}\nrestart shiro to unload it`;
}
}
throw new Error(`nothing installed under the name "${bare}"`);
},
@@ -445,10 +460,17 @@ const hooks: AppHooks = {
const servers = Object.entries(cfg.mcpServers ?? {});
if (servers.length === 0) return 'no MCP servers configured\n\n`/mcp add` sets one up.';
const lazy = (mcp?.tools['mcp_list'] ?? undefined) !== undefined;
const live = new Map<string, number>();
for (const name of Object.keys(mcp?.tools ?? {})) {
const server = /^mcp__([^_]+(?:_[^_]+)*)__/.exec(name)?.[1];
if (server) live.set(server, (live.get(server) ?? 0) + 1);
if (lazy) {
// Lazy mode: tools are fetched on demand, so "connected" is what the handle
// says, not a count of registered mcp__ tools.
for (const name of mcp?.servers ?? []) live.set(name, -1);
} else {
for (const name of Object.keys(mcp?.tools ?? {})) {
const server = /^mcp__([^_]+(?:_[^_]+)*)__/.exec(name)?.[1];
if (server) live.set(server, (live.get(server) ?? 0) + 1);
}
}
const failed = new Map((mcp?.errors ?? []).map((e) => [e.server, e.message]));
@@ -457,7 +479,9 @@ const hooks: AppHooks = {
const state = failed.has(name)
? `failed: ${failed.get(name)}`
: live.has(name)
? `${live.get(name)} tools`
? live.get(name)! >= 0
? `${live.get(name)} tools`
: 'connected (lazy)'
: has('--no-mcp')
? 'not connected (--no-mcp)'
: 'not connected this session';
@@ -611,7 +635,15 @@ const facts: HeaderFact[] = [
memory && memory.all().length > 0
? { label: 'memory', value: `${memory.all().length} notes about this project` }
: undefined,
mcp && Object.keys(mcp.tools).length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).length} tools` } : undefined,
mcp && Object.keys(mcp.tools).length > 0
? {
label: 'mcp',
value:
mcp.servers.length > 0
? `${mcp.servers.length} ${mcp.servers.length === 1 ? 'server' : 'servers'} (lazy)`
: `${Object.keys(mcp.tools).length} tools`,
}
: undefined,
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
: undefined,
+24
View File
@@ -6,6 +6,8 @@ export type CommandAction =
| { type: 'exit' }
| { type: 'clear' }
| { type: 'compact' }
| { type: 'undo'; what: 'both' | 'files' | 'conversation' }
| { type: 'redo'; what: 'both' | 'files' | 'conversation' }
| { type: 'tools' }
| { type: 'cost' }
| { type: 'sessions' }
@@ -57,6 +59,8 @@ export const COMMANDS: CommandSpec[] = [
{ name: 'memory', summary: 'compact the project memory with the model' },
{ name: 'tools', summary: 'list available tools' },
{ name: 'compact', summary: 'replace history with a model-written summary' },
{ name: 'undo', arg: '[files|conversation]', summary: 'walk the last turn back: files, conversation, or both' },
{ name: 'redo', arg: '[conversation]', summary: 'put back what /undo took' },
{ name: 'cost', summary: 'tokens and estimated spend this session' },
{ name: 'sessions', summary: 'list saved sessions' },
{ name: 'resume', arg: '<id>', summary: 'load a saved session' },
@@ -161,6 +165,22 @@ function parseMcp(arg: string): CommandAction {
}
}
/**
* `/undo [files|conversation]` and `/redo [conversation]`.
*
* The default is `both` for undo, because restoring one without the other is the
* failure the two are meant to prevent: files back without the history and the model
* re-reads a change it no longer made. A bare `files` or `conversation` narrows it.
* Redo defaults to the conversation, since file content after the turn was never kept.
*/
function parseUndoKind(arg: string, fallback: 'both' | 'conversation'): 'both' | 'files' | 'conversation' {
const word = arg.trim().toLowerCase();
if (word === 'files' || word === 'file') return 'files';
if (word === 'conversation' || word === 'chat' || word === 'history') return 'conversation';
if (word === 'both' || word === 'all' || word === '') return fallback;
return fallback;
}
/**
* Pure parser: no IO, so the TUI and headless mode share one definition.
*
@@ -186,6 +206,10 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
return { type: 'clear' };
case 'compact':
return { type: 'compact' };
case 'undo':
return { type: 'undo', what: parseUndoKind(arg, 'both') };
case 'redo':
return { type: 'redo', what: parseUndoKind(arg, 'conversation') };
case 'tools':
return { type: 'tools' };
case 'cost':
+5
View File
@@ -37,6 +37,11 @@ export type Config = {
/** Index for `/registry`. Omit for the default one. */
registryUrl?: string;
mcpServers?: Record<string, McpServerConfig>;
/**
* How MCP tools reach the model: `lazy` registers meta-tools only (cheap until a
* tool is called), `eager` registers every server tool up front. Omit for lazy.
*/
mcpMode?: 'lazy' | 'eager';
};
const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json');
+124 -10
View File
@@ -1,27 +1,47 @@
import { createMCPClient, type MCPClient } from '@ai-sdk/mcp';
import { Experimental_StdioMCPTransport } from '@ai-sdk/mcp/mcp-stdio';
import type { ToolSet } from 'ai';
import { tool, type ToolSet } from 'ai';
import { z } from 'zod';
export type McpServerConfig =
| { command: string; args?: string[]; env?: Record<string, string>; cwd?: string }
| { url: string; type?: 'http' | 'sse'; headers?: Record<string, string> };
/**
* How a server's tools reach the model.
*
* `eager` registers every tool with its schema up front — cheap for a two-tool
* server, a tax for one that exposes twenty. `lazy` registers only the three
* meta-tools below and fetches a server's tools on demand via `mcp_call`, so a
* configured server costs almost nothing in the request until a tool is actually
* invoked.
*/
export type McpMode = 'eager' | 'lazy';
export type McpHandle = {
tools: ToolSet;
errors: { server: string; message: string }[];
/** Server names, for the prompt's MCP line. Empty when the mode is eager. */
servers: string[];
close: () => Promise<void>;
};
const isRemote = (c: McpServerConfig): c is Extract<McpServerConfig, { url: string }> => 'url' in c;
/**
* Connects every configured server and namespaces its tools as `mcp__<server>__<tool>`
* so two servers exposing `search` cannot silently shadow each other.
* A server that fails to start is reported, never fatal.
* Connects every configured server and exposes its tools.
*
* In `eager` mode the tools land in the returned set as `mcp__<server>__<tool>`, so
* two servers exposing `search` cannot silently shadow each other. In `lazy` mode the
* set holds only the three meta-tools and `servers` names the configured servers; a
* server that fails to start is reported, never fatal, in either mode.
*/
export async function connectMcp(servers: Record<string, McpServerConfig>): Promise<McpHandle> {
export async function connectMcp(
servers: Record<string, McpServerConfig>,
mode: McpMode = 'lazy',
): Promise<McpHandle> {
const clients: MCPClient[] = [];
const tools: ToolSet = {};
const byName = new Map<string, MCPClient>();
const errors: McpHandle['errors'] = [];
await Promise.all(
@@ -38,9 +58,7 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
}),
});
clients.push(client);
for (const [toolName, tool] of Object.entries(await client.tools())) {
tools[`mcp__${name}__${toolName}`] = tool;
}
byName.set(name, client);
} catch (e) {
errors.push({ server: name, message: e instanceof Error ? e.message : String(e) });
}
@@ -48,10 +66,106 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
);
return {
tools,
tools: mode === 'eager' ? await eagerTools(byName) : lazyTools(byName),
errors,
servers: mode === 'eager' ? [] : [...byName.keys()],
close: async () => {
await Promise.all(clients.map((c) => c.close().catch(() => {})));
},
};
}
/** Eager: every server tool gets an AI tool registered with its schema. */
async function eagerTools(byName: Map<string, MCPClient>): Promise<ToolSet> {
const tools: ToolSet = {};
await Promise.all(
[...byName.entries()].map(async ([name, client]) => {
for (const [toolName, t] of Object.entries(await client.tools())) {
tools[`mcp__${name}__${toolName}`] = t;
}
}),
);
return tools;
}
/**
* Cache of each client's AI tools, so `mcp_call` does not re-list on every call.
* The WeakMap drops entries when a client (and its session) is closed and collected.
*/
const clientToolsCache = new WeakMap<MCPClient, ToolSet>();
/** Fetches a server's AI tools, or returns the cached set from a prior call. */
async function cachedClientTools(client: MCPClient): Promise<ToolSet> {
const cached = clientToolsCache.get(client);
if (cached) return cached;
const tools = await client.tools();
clientToolsCache.set(client, tools);
return tools;
}
/**
* Lazy: three meta-tools instead of every server schema.
*
* `mcp_list` names a server's tools from their definitions (cheap, no schema).
* `mcp_inspect` reads one tool's schema so the model knows its inputs.
* `mcp_call` executes one tool on its server, making the server's tools available
* only from the moment they are actually invoked.
*/
function lazyTools(byName: Map<string, MCPClient>): ToolSet {
const connectionError = (name: string) =>
byName.has(name) ? undefined : `unknown server "${name}". Configured: ${[...byName.keys()].join(', ') || 'none'}`;
return {
mcp_list: tool({
description: `List the tools exposed by an MCP server. Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
inputSchema: z.object({ server: z.string().describe('Server name from your instructions') }),
execute: async ({ server }) => {
const err = connectionError(server);
if (err) throw new Error(err);
const client = byName.get(server)!;
const defs = await client.listTools();
return defs.tools.length === 0
? `server "${server}" exposes no tools`
: defs.tools.map((d) => `- ${d.name}: ${d.description ?? 'no description'}`).join('\n');
},
}),
mcp_inspect: tool({
description: `Inspect one tool's input schema on an MCP server. Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
inputSchema: z.object({
server: z.string().describe('Server name from your instructions'),
toolName: z.string().describe('Tool name, as listed by mcp_list'),
}),
execute: async ({ server, toolName }) => {
const err = connectionError(server);
if (err) throw new Error(err);
const client = byName.get(server)!;
const defs = await client.listTools();
const def = defs.tools.find((d) => d.name === toolName);
if (!def) throw new Error(`No tool "${toolName}" on "${server}". List first with mcp_list.`);
return def.inputSchema ? JSON.stringify(def.inputSchema, null, 2) : `tool "${toolName}" declares no input schema`;
},
}),
mcp_call: tool({
description:
`Call one tool on an MCP server. Inspect its schema with mcp_inspect first. ` +
`Servers: ${[...byName.keys()].join(', ') || 'none'}.`,
inputSchema: z.object({
server: z.string().describe('Server name from your instructions'),
toolName: z.string().describe('Tool name, as listed by mcp_list'),
args: z.record(z.string(), z.unknown()).describe('Arguments the tool expects, from mcp_inspect'),
}),
execute: async ({ server, toolName, args }) => {
const err = connectionError(server);
if (err) throw new Error(err);
const client = byName.get(server)!;
const serverTools = await cachedClientTools(client);
const aiTool = serverTools[toolName];
if (!aiTool)
throw new Error(
`No tool "${toolName}" on "${server}". List first with mcp_list (server exposes: ${Object.keys(serverTools).join(', ') || 'none'}).`,
);
return aiTool.execute!(args, { toolCallId: 'mcp_call', messages: [] } as never);
},
}),
};
}
+5 -4
View File
@@ -181,6 +181,11 @@ export const DEFAULT_PERMISSIONS: PermissionConfig = {
apply_patch: 'ask',
move_file: 'ask',
delete_file: 'ask',
insert_lines: 'ask',
delete_lines: 'ask',
replace_lines: 'ask',
append_file: 'ask',
prepend_file: 'ask',
bash: 'ask',
web_fetch: 'ask',
};
@@ -272,10 +277,6 @@ export class Permissions {
this.granted.set(tool, set);
}
granted_(tool: string): string[] {
return [...(this.granted.get(tool) ?? [])];
}
/** The decision for one call, and which pattern decided it. */
check(tool: string, input: unknown): Resolved {
const resolved = resolve(entryFor(tool, this.config), tool, input);
+4
View File
@@ -4,6 +4,10 @@ export type Rate = { inputPerMTok: number; outputPerMTok: number };
* USD per million tokens. Prefix match on the model id, longest first, so
* `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates
* drift, so this is a best-effort estimate rather than a billing source.
*
* Source: vendor pricing pages, checked 2026-09-17. Anthropic (Anthropic API, not
* Batch) and OpenAI listed rates; DeepSeek and Grok per their API pricing. Rates
* are for input, then output. Re-verify before trusting a live spend figure.
*/
const RATES: Record<string, Rate> = {
'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 },
+10 -2
View File
@@ -131,8 +131,10 @@ function renderTools(available: readonly string[]): string {
// free, and the schema already says what each takes.
const git = extra.filter((n) => GIT_TOOL_NAMES.includes(n) && n !== 'git_commit_message');
const mcp = extra.filter((n) => n.startsWith('mcp__'));
const META = ['mcp_list', 'mcp_inspect', 'mcp_call'];
const lazyMcp = META.filter((n) => available.includes(n));
const other = extra.filter(
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__'),
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__') && !META.includes(n),
);
if (git.length > 0) {
@@ -140,7 +142,13 @@ function renderTools(available: readonly string[]): string {
`- ${git.join(', ')}: read-only git, no approval needed. Use them instead of bash for history and diffs; they cannot mutate the repository.`,
);
}
if (mcp.length > 0) {
if (lazyMcp.length > 0) {
// Lazy mode: the meta-tool descriptions already name the connected servers, so
// the model needs the workflow, not a schema listing.
lines.push(
`- ${lazyMcp.join(', ')}: MCP tools are fetched on demand. mcp_list names a server's tools, mcp_inspect reads one tool's schema, mcp_call runs it. Never guess a server or tool name: list first.`,
);
} else if (mcp.length > 0) {
lines.push(
`- ${mcp.join(', ')}: from MCP servers, named mcp__<server>__<tool>. Each needs approval; read its own description before calling.`,
);
+78 -2
View File
@@ -92,8 +92,6 @@ export function detachOrphanedItems(before: ModelMessage[], after: ModelMessage[
export type PruneOptions = Parameters<typeof pruneMessages>[0];
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
const anyParts = (message: ModelMessage): Part[] =>
Array.isArray(message.content) ? (message.content as Part[]) : [];
@@ -167,6 +165,84 @@ export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
return dropOrphanedResults(detachOrphanedItems(options.messages, pruned));
}
/**
* The messages a prune would discard, so they can be summarized before they go.
*
* Compaction keeps the model's *memory of a turn* — the tool tail it is told to
* keep stays verbatim. What it does not keep is any statement of what was
* dropped. So a decision from forty messages ago vanishes silently, and the model
* contradicts it with full confidence, because as far as it can tell it never
* said that.
*
* Identity is by reference, not by value: `prunePreservingItems` rebuilds the
* surviving messages with `{ ...message }`, so a value comparison would report
* every message as changed and no message as dropped. `Set` on the object
* references is exact.
*
* Only messages that carry content worth summarizing are returned — an assistant
* turn consisting of nothing but a dropped `reasoning` part is not a decision, and
* summarizing "the model thought for a while" is worse than saying nothing.
*/
export function droppedBy(before: ModelMessage[], after: ModelMessage[]): ModelMessage[] {
const surviving = new Set<ModelMessage>(after);
return before.filter((message) => !surviving.has(message));
}
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
function textOf(message: ModelMessage): string {
const { content } = message;
if (typeof content === 'string') return content;
if (!Array.isArray(content)) return '';
const chunks: string[] = [];
for (const part of content as Part[]) {
const p = part as Part & { text?: unknown; input?: unknown; output?: unknown };
if (typeof p.text === 'string') chunks.push(p.text);
// A tool call's input is the decision made: the path, the command, the patch.
else if (p.type === 'tool-call' && p.input !== undefined) chunks.push(JSON.stringify(p.input));
// A tool result is what came back. Without it a digest says what the model
// asked for and nothing about the answer, which is the half a later
// contradiction is usually argued from.
else if (ANSWER_PARTS.has(p.type) && p.output !== undefined) {
const rendered = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
chunks.push(rendered);
}
}
return chunks.join(' ').trim();
}
/**
* A one-line-per-message digest of what a prune wants to drop.
*
* This is the *fallback* when no summarizer is available or the call fails: crude,
* but it preserves the thing that matters — which tool touched which path, and in
* what order — rather than the nothing that is there today. The summarizer, when
* it runs, is a model and reads far better than this.
*/
export function digestOf(dropped: readonly ModelMessage[]): string {
const lines: string[] = [];
for (const message of dropped) {
const text = textOf(message);
if (!text) continue;
const role = message.role === 'tool' ? 'result' : message.role;
const clipped = text.length > 160 ? `${text.slice(0, 160)}...` : text;
lines.push(`- (${role}) ${clipped}`);
}
return lines.join('\n');
}
export const PRUNED_SPAN_PREFIX = 'Earlier in this session, now compacted away:';
export function isPrunedSpanSummary(message: ModelMessage): boolean {
return message.role === 'user' && typeof message.content === 'string' && message.content.startsWith(PRUNED_SPAN_PREFIX);
}
export function prunedSpanMessage(summary: string | undefined, dropped: readonly ModelMessage[]): ModelMessage | undefined {
const body = summary?.trim() || digestOf(dropped);
if (!body) return undefined;
return { role: 'user', content: `${PRUNED_SPAN_PREFIX}\n\n${body}` };
}
/**
* How many trailing messages keep their tool content, widest first.
*
+270 -4
View File
@@ -17,7 +17,9 @@ import { Permissions, type PermissionConfig } from './permission';
import type { PluginHost } from './plugins';
import { costOf, formatUsd } from './pricing';
import { systemPrompt } from './prompt';
import { detachProviderItems, pruneToFit } from './prune';
import { detachProviderItems, digestOf, droppedBy, isPrunedSpanSummary, prunedSpanMessage, pruneToFit } from './prune';
import { restore, Snapshots, type TurnSnapshot } from './snapshot';
import { createStepBackTool, type LoopEntry } from './step-back';
import { createSkillTool, renderSkills, type Skill } from './skills';
import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools';
@@ -38,6 +40,21 @@ export type ApprovalRequest = {
/** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */
export type ApprovalDecision = 'once' | 'always' | 'deny';
/** What an undo did, so the UI can say which files moved and which did not. */
export type UndoResult = {
snapshot: TurnSnapshot;
restored: string[];
removed: string[];
conversationTrimmed: boolean;
};
export type RedoResult = {
snapshot: TurnSnapshot;
/** False when files were asked for: only pre-images are ever captured. */
filesRestored: boolean;
what: 'both' | 'files' | 'conversation';
};
export type AgentEvent =
| { type: 'text'; text: string }
| { type: 'reasoning'; text: string }
@@ -74,6 +91,8 @@ export type SessionOptions = {
autoApprove?: readonly string[];
/** Prune the history once the estimated token count crosses this. */
compactThreshold?: number;
/** Identical calls to an allowed tool before it is asked about anyway. Default 3. */
repeatLimit?: number;
/** Retries per model call for transient failures. */
maxRetries?: number;
/** AGENTS.md-style files appended to the system prompt. */
@@ -94,6 +113,12 @@ export type SessionOptions = {
onNotebookChange?: (state: NotebookState) => void;
};
/**
* Length-based token estimate for deciding *when to prune*, not for billing.
* JSON char count / 4 approximates token count closely enough to gate compaction,
* but real billed tokens come from the SDK's reported usage (`inputTokens`), never
* from here. `/cost` and the budget ceiling use the SDK figure.
*/
const estimateTokens = (messages: ModelMessage[]) => Math.round(JSON.stringify(messages).length / 4);
/** Estimated tokens at which the wire history is pruned. */
@@ -102,6 +127,26 @@ const DEFAULT_COMPACT_THRESHOLD = 120_000;
/** Identical calls in one turn before an allowed tool is asked about anyway. */
const REPEAT_LIMIT = 3;
/**
* Squashes a tool result into a few characters for the loop trace.
*
* The trace is fed back to the model verbatim, so a 30 KB read_file output would
* fill the reflection with noise. A short string keeps `step_back` honest about
* what happened without flooding the next context window.
*/
function summarizeToolResult(output: unknown): string {
if (typeof output === 'string') return output.length <= 80 ? output : `${output.slice(0, 80)}…(${output.length} chars)`;
try {
const json = JSON.stringify(output);
return json.length <= 80 ? json : `${json.slice(0, 80)}…`;
} catch {
return String(output);
}
}
/** Ceiling on an injected span summary, so the summary cannot defeat the compaction. */
const MAX_SPAN_SUMMARY_CHARS = 1_200;
const callKey = (toolName: string, input: unknown) => `${toolName}:${JSON.stringify(input ?? null)}`;
/**
@@ -121,6 +166,8 @@ export class Session {
readonly messages: ModelMessage[];
readonly tools: ToolSet;
readonly notebook: Notebook;
/** Pre-images of files this session's turns have changed, newest last. */
readonly snapshots: Snapshots;
inputTokens = 0;
outputTokens = 0;
/** Subagent token use, priced against the subagent's own model id in /cost. */
@@ -136,6 +183,14 @@ export class Session {
/** The 80% spend warning is shown once, not on every turn past the line. */
private warnedSpend = false;
private controller: AbortController | undefined;
/** Tools this turn used that no snapshot can cover, reported when the turn ends. */
private readonly uncoveredTools = new Set<string>();
/** The snapshot the last undo removed, so `/redo` can put it back. */
private lastUndone: TurnSnapshot | undefined;
/** Every tool call this turn, input + outcome, for the loop-detection tool to reflect on. */
private readonly loopTrace: LoopEntry[] = [];
/** toolCallId -> { toolName, input }, so a result can be paired with its call. */
private readonly callInputs = new Map<string, { toolName: string; input: string }>();
constructor(private readonly opts: SessionOptions) {
this.messages = opts.messages ?? [];
@@ -143,12 +198,19 @@ export class Session {
this.notebook.restore(opts.notebook);
this.model = opts.model;
this.variant = opts.agent ?? DEFAULT_VARIANT;
this.snapshots = new Snapshots(opts.cwd ?? process.cwd());
const sessionTools = {
...this.notebook.tools(),
...(opts.memory ? opts.memory.tools() : {}),
...(opts.skills && opts.skills.length > 0 ? { skill: createSkillTool(opts.skills) } : {}),
// Always registered, even with no skills, so a mid-session install of the
// first skill is callable next turn without a session rebuild. The tool's
// description reads the live list and says "none" when it is empty.
skill: createSkillTool(() => this.opts.skills ?? []),
...(opts.ask ? { ask: createAskTool(opts.ask) } : {}),
step_back: createStepBackTool({
trace: () => this.loopTrace,
}),
};
this.tools = { ...builtinTools, ...sessionTools, ...(opts.plugins?.tools ?? {}), ...(opts.extraTools ?? {}) };
@@ -306,6 +368,26 @@ export class Session {
return count;
}
/**
* Appends a completed tool call to the loop trace, pairing it with its input.
*
* The SDK streams `tool-result` without the input that produced it, so the input is
* kept alongside on the `tool-call` part. A `step_back` call needs this pairing to
* say *which* call produced *which* outcome.
*/
private recordTrace(toolCallId: string, result: string): void {
const call = this.callInputs.get(toolCallId);
this.callInputs.delete(toolCallId);
if (!call) return;
this.loopTrace.push({
step: this.loopTrace.length + 1,
toolName: call.toolName,
input: call.input,
result,
at: new Date().toISOString(),
});
}
/**
* Approval decisions, evaluated per call by the SDK.
*
@@ -326,6 +408,11 @@ export class Session {
return async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const { toolName, input } = toolCall;
// Before anything else, because the guard may deny the call and because a
// later hook must not be able to move the capture after the write.
const { covered } = await this.snapshots.captureFor(toolName, input);
if (!covered) this.uncoveredTools.add(toolName);
const blocked = await this.opts.plugins?.guard({
toolName,
input,
@@ -346,7 +433,18 @@ export class Session {
}
const repeats = this.repeatCount(toolName, input);
if (decision === 'allow' && repeats < REPEAT_LIMIT) return undefined;
const limit = this.opts.repeatLimit ?? REPEAT_LIMIT;
if (decision === 'allow' && repeats < limit) return undefined;
if (decision === 'allow') {
// Repeated three times with a permission that says `allow`: the model is
// looping, not asking, and it should stop and look at the trace rather than
// burn another approval. This is the point the step_back tool exists for.
notices.push(
`You have called ${toolName} with the same input ${repeats + 1} times this turn. It is not making progress. ` +
`Use step_back to reflect on what changed between attempts, then try a different approach or stop.`,
);
}
why.set(callKey(toolName, input), {
...(pattern ? { matchedPattern: pattern } : {}),
@@ -378,6 +476,92 @@ export class Session {
return { before, after: this.messages.length };
}
/**
* Walks the last turn back: files, conversation, or both.
*
* The three-way split is the point. Restoring files without the conversation leaves
* the model believing edits are on disk that are not, so its next turn is built on a
* state that no longer exists — it re-reads a file expecting its own change and finds
* the original, which reads to the model as the change having been rejected. Restoring
* the conversation without the files is the mirror: the model forgets it made an edit
* that is still there. So the default is both, and the caller can narrow it.
*
* Returns undefined when there is nothing to undo, which the UI reports as such
* rather than as a failure.
*/
async undo(what: 'both' | 'files' | 'conversation' = 'both'): Promise<UndoResult | undefined> {
const snap = this.snapshots.pop();
if (!snap) return undefined;
const files = what === 'conversation' ? { restored: [], removed: [] } : await restore(snap, this.snapshots.cwdOf());
if (what !== 'files') this.trimTo(snap.messageCount);
this.lastUndone = snap;
return { snapshot: snap, ...files, conversationTrimmed: what !== 'files' };
}
/**
* Puts back what `undo` took, without a second snapshot.
*
* A redo cannot restore file content from the session, because the content that
* existed after the turn was never captured — only the pre-image was. So a redo of
* the files is declined honestly rather than approximated: the whole point of undo
* is that the user trusts what it says it did. The conversation is restored from the
* snapshot's own record, which is exact.
*/
redo(what: 'both' | 'files' | 'conversation' = 'conversation'): RedoResult | undefined {
const snap = this.lastUndone;
if (!snap) return undefined;
this.snapshots.push(snap);
this.lastUndone = undefined;
return { snapshot: snap, filesRestored: false, what };
}
undoable(): readonly TurnSnapshot[] {
return this.snapshots.list();
}
private trimTo(length: number): void {
if (this.messages.length <= length) return;
this.messages.length = Math.max(0, length);
this.opts.onChange?.(this.messages);
}
/**
* Closes the open snapshot and reports what the turn could not cover.
*
* Called before every `done`, not from `send`'s `finally`, because a notice that
* arrives after `done` is a notice the UI has already stopped listening for —
* `done` is what a consumer treats as the end of the turn and stops on.
*
* Two reasons to speak, and the second does not depend on the first: a turn with no
* snapshotted edits can still have run `bash` and changed the tree, which is exactly
* when the warning matters most.
*/
private *closingNotices(): Generator<AgentEvent> {
const snapshot = this.snapshots.commit();
const uncovered = [...this.uncoveredTools];
if (uncovered.length === 0) return;
const covered = snapshot ? `${snapshot.files.length} file(s) changed this turn and can be undone with /undo. ` : '';
yield {
type: 'notice',
text: `${covered}${uncovered.join(', ')} ran this turn and cannot be snapshotted, so any changes it made will not be undone.`,
};
}
/**
* Swaps the live skill list at a turn boundary.
*
* The `skill` tool and the system-prompt catalogue both read the list on each
* call, so replacing it here is atomic across the two and takes effect on the
* next turn with no session rebuild. The caller is responsible for only doing
* this between turns — a turn in flight already holds its rules.
*/
setSkills(skills: Skill[]): void {
this.opts.skills = skills;
}
async *send(userText: string): AsyncGenerator<AgentEvent> {
// The ceiling is checked before the model is: a turn started past the limit
// would spend money the caller said not to. An unpriced model cannot be
@@ -402,7 +586,12 @@ export class Session {
// Per turn, not per step: a tool called once in each of three steps is the
// loop this guards against.
this.seen.clear();
this.uncoveredTools.clear();
this.loopTrace.length = 0;
this.staleItemsRepaired = false;
// Opened before the model runs and closed after it stops, so every write the
// turn makes lands in one snapshot the user can walk back to.
this.snapshots.begin(userText, this.messages.length);
const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = [];
onBashOutput(({ toolCallId, chunk }) => {
@@ -414,6 +603,10 @@ export class Session {
yield* this.run(signal, threshold, outputs);
} finally {
onBashOutput(undefined);
// Belt and braces: closingNotices runs before every `done`, but an aborted turn
// can return through a path that never reached it, and an uncommitted snapshot
// would then be silently dropped rather than kept.
this.snapshots.commit();
await this.opts.plugins?.afterTurn();
}
}
@@ -431,6 +624,70 @@ export class Session {
return true;
}
/**
* Prunes the canonical history once a run has landed.
*
* prepareStep only trims the wire copy for the next request; without this write-back
* the stored history keeps growing, the context meter pins at 100%, and every later
* turn re-prunes the same messages from scratch.
*/
private async compactCanonical(threshold: number): Promise<{ before: number; after: number } | null> {
const before = this.messages.length;
if (estimateTokens(this.messages) <= threshold) return null;
const pruned = pruneToFit({ messages: this.messages, threshold, estimate: estimateTokens });
if (pruned.length === before) return null;
const dropped = droppedBy(this.messages, pruned);
const summary = await this.summarizeSpan(dropped.filter((m) => !isPrunedSpanSummary(m)));
this.replace(this.withSummarizedSpan(summary, pruned, dropped));
return { before, after: this.messages.length };
}
/**
* Puts a summary of the discarded span at the head of the history it was dropped from.
*
* Without this the model is told which tool results to keep and nothing about what
* was dropped, so it states a decision it made forty messages ago as though it had
* never made it.
*
* The summary is bounded two ways because a summary that grows with the session
* defeats the point of compacting at all: the input is capped at the span's own
* digest, and the output is capped by instruction and by hard truncation. A failed
* or empty call falls back to the digest, which costs nothing and still carries
* which tool touched which path — the part a contradiction is usually built from.
*/
private async summarizeSpan(dropped: readonly ModelMessage[]): Promise<string | undefined> {
const digest = digestOf(dropped);
if (!digest) return undefined;
try {
const { text } = await generateText({
model: this.model,
system:
'These lines are the condensed record of an earlier part of a coding session that has been ' +
'compacted out of the conversation. Write at most 120 words of notes capturing decisions made, ' +
'files touched, commands run and their outcome, and anything still pending. State only what the ' +
'lines support; do not invent detail and do not address the reader.',
messages: [{ role: 'user', content: digest }],
maxRetries: this.opts.maxRetries ?? 3,
});
const trimmed = text.trim();
return trimmed ? trimmed.slice(0, MAX_SPAN_SUMMARY_CHARS) : undefined;
} catch {
// A summarizer that cannot run must not cost the turn its compaction: the
// digest is a worse record, not an absent one.
return undefined;
}
}
private withSummarizedSpan(summary: string | undefined, pruned: ModelMessage[], dropped: readonly ModelMessage[]): ModelMessage[] {
const worthSummarizing = dropped.filter((m) => !isPrunedSpanSummary(m));
if (worthSummarizing.length === 0) return pruned;
const prior = dropped.filter(isPrunedSpanSummary);
const spanMessage = prunedSpanMessage(summary, worthSummarizing);
if (!spanMessage) return pruned;
return [...prior, spanMessage, ...pruned];
}
private async *run(
signal: AbortSignal,
threshold: number,
@@ -506,12 +763,15 @@ export class Session {
yield { type: 'tool-start', id: part.id, name: part.toolName };
break;
case 'tool-call':
this.callInputs.set(part.toolCallId, { toolName: part.toolName, input: JSON.stringify(part.input ?? null) });
yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input };
break;
case 'tool-result':
this.recordTrace(part.toolCallId, summarizeToolResult(part.output));
yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output };
break;
case 'tool-error':
this.recordTrace(part.toolCallId, `error: ${part.error instanceof Error ? part.error.message : String(part.error)}`);
yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error };
break;
case 'tool-approval-request': {
@@ -535,6 +795,7 @@ export class Session {
yield { type: 'tool-denied', name: part.toolName };
break;
case 'abort':
yield* this.closingNotices();
yield { type: 'done' };
return;
case 'error':
@@ -555,6 +816,7 @@ export class Session {
}
} catch (error) {
if (signal.aborted) {
yield* this.closingNotices();
yield { type: 'done' };
return;
}
@@ -579,7 +841,10 @@ export class Session {
while (guardNotices.length > 0) yield { type: 'notice', text: guardNotices.shift()! };
this.messages.push(...(await result.responseMessages));
this.opts.onChange?.(this.messages);
// prepareStep already emits `compacted` at the same threshold crossing, and
// replace() fires onChange on the fold path; both would double up otherwise.
if (!(await this.compactCanonical(threshold))) this.opts.onChange?.(this.messages);
if (pending.length === 0) {
const usage = await result.usage;
@@ -595,6 +860,7 @@ export class Session {
text: `approaching spend ceiling: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used`,
};
}
yield* this.closingNotices();
yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens };
return;
}
+14 -4
View File
@@ -100,18 +100,28 @@ export function renderSkills(skills: Skill[]): string {
].join('\n');
}
export function createSkillTool(skills: Skill[]) {
const names = skills.map((s) => s.name);
/**
* The `skill` tool, reading the list live so a mid-session install shows up on the
* next turn without a restart.
*
* The catalogue in the system prompt and the names in this tool's description both
* come from the same list on each read, so replacing the list at a turn boundary
* makes both stale bytes atomic: a skill the model can call it can see already.
*/
export function createSkillTool(getSkills: () => Skill[]) {
const names = () => getSkills().map((s) => s.name).join(', ');
return tool({
description:
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`,
`what you are about to do, then follow what it says. Available: ${names() || 'none'}.`,
inputSchema: z.object({
name: z.string().describe('Skill name from the list in your instructions'),
}),
execute: async ({ name }) => {
const skills = getSkills();
const skill = skills.find((s) => s.name === name.trim().toLowerCase());
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`);
if (!skill)
throw new Error(`No skill named "${name}". Available: ${skills.map((s) => s.name).join(', ') || 'none'}`);
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
},
});
+269
View File
@@ -0,0 +1,269 @@
import { createHash } from 'node:crypto';
import { join, relative, resolve, sep } from 'node:path';
/**
* Pre-images of files a turn is about to change, so a turn can be walked back.
*
* The design follows the one thing every comparable CLI agrees on and the one thing
* they all get wrong in the same way. Agreement: a snapshot is taken *before* the
* turn runs, because after it runs the original content is gone. The shared flaw: a
* snapshot only covers the tools that write through a known interface — Claude Code
* tracks Write/Edit/NotebookEdit and explicitly not `bash` — so the undo is partial
* and the user has to know which half they are in.
*
* Two decisions fall out of taking that seriously.
*
* 1. Capture lazily, per file, not by walking the tree. A session turn may touch
* three files out of fifty thousand; a full-tree copy per prompt is a tarball of
* the repository the user did not ask for and cannot afford. Instead the first
* write to a path records its pre-image, and a later write to the same path in
* the same turn does not overwrite it — the pre-image is the state before the
* *turn*, which is what an undo restores.
*
* 2. Record what is *not* covered rather than implying it is. A `bash` command that
* rewrites a file, a `git checkout`, a build artifact — none of these pass through
* a path argument the way `write_file` does, so none are captured. The tool
* reports which paths it restored and the caller is told the rest is unknown,
* which is the same honesty `bash` already owes an interrupted command.
*/
/** One file as it was before the turn that changed it. `before === undefined` means it did not exist. */
export type PreImage = {
/** Workspace-relative, using forward slashes, so a restore is portable across platforms. */
path: string;
before: string | undefined;
};
export type TurnSnapshot = {
/** Monotonic turn number, so the UI can name what is being undone. */
turn: number;
at: string;
/** The user prompt that opened the turn, for a menu that lists them. */
prompt: string;
/** Files this turn changed, with their content from before it started. */
files: PreImage[];
/** The conversation length when the turn began, so undo can trim it back. */
messageCount: number;
};
/** Claude Code keeps 100; beyond that the memory is worth more than the recall. */
export const MAX_SNAPSHOTS = 100;
/** A single file larger than this is not snapshotted; a 40 MB binary is not an edit. */
const MAX_FILE_BYTES = 2 * 1024 * 1024;
/**
* The workspace-relative, slash-normalised form of a path, or undefined if it is
* outside the workspace.
*
* Outside is refused rather than clamped: a path that escapes the workspace is not
* something this repository can undo, and recording it would imply an undo that
* cannot happen. Paths are normalised to forward slashes because a snapshot written
* on Windows may be read on a machine where a backslash is a filename character.
*/
export function relPath(cwd: string, abs: string): string | undefined {
const root = resolve(cwd);
const target = resolve(abs);
const rel = relative(root, target);
if (rel === '' || rel.startsWith('..') || rel.includes(`..${sep}`)) return undefined;
return rel.split(sep).join('/');
}
/**
* The absolute path a tool call will write to, when there is exactly one.
*
* Deliberately a small, explicit map rather than a guess. `multi_edit` and the line
* editors each take a single `path`; `move_file` takes `from` and `to` and both are
* recorded; `delete_file` takes a `path`. `apply_patch` carries its paths inside the
* patch text, and `bash` carries none — both are reported as uncovered rather than
* silently not snapshotted.
*/
export function touchedPaths(toolName: string, input: unknown): { paths: string[]; covered: boolean } {
const o = (input ?? {}) as Record<string, unknown>;
const one = (key: string) => (typeof o[key] === 'string' ? [o[key] as string] : []);
switch (toolName) {
case 'write_file':
case 'edit_file':
case 'multi_edit':
case 'delete_file':
case 'insert_lines':
case 'delete_lines':
case 'replace_lines':
case 'append_file':
case 'prepend_file':
return { paths: one('path'), covered: true };
case 'move_file':
return { paths: [...one('from'), ...one('to')], covered: true };
case 'apply_patch': {
const patch = typeof o['patch'] === 'string' ? (o['patch'] as string) : '';
const paths = [...patch.matchAll(/^\*\*\* (?:Add|Update|Delete) File: (.+)$/gm)].map((m) => m[1]!.trim());
const moves = [...patch.matchAll(/^\*\*\* Move to: (.+)$/gm)].map((m) => m[1]!.trim());
return { paths: [...paths, ...moves], covered: true };
}
case 'bash':
// Arbitrary code: an untouched-looking `node -e` can rewrite the tree.
return { paths: [], covered: false };
default:
return { paths: [], covered: true };
}
}
/**
* Records pre-images for the files a turn changes, and restores them on undo.
*
* One instance per session. Holds at most MAX_SNAPSHOTS turns; the oldest falls off
* the front, because the turn someone wants back is almost always the last one.
*/
export class Snapshots {
private readonly turns: TurnSnapshot[] = [];
private current: { turn: number; at: string; prompt: string; files: Map<string, string | undefined>; messageCount: number } | null =
null;
private next = 1;
private readonly cwd: string;
constructor(cwd: string = process.cwd()) {
this.cwd = cwd;
}
/** Opens a turn. Called once per user prompt, before the model runs. */
begin(prompt: string, messageCount: number): void {
this.current = { turn: this.next++, at: new Date().toISOString(), prompt, files: new Map(), messageCount };
}
/**
* Records a file's content before a tool changes it, on the first write of the turn.
*
* Idempotent per path per turn: the second `write_file` to the same path in one turn
* must not replace the pre-image with the intermediate content the first write left,
* because undo restores the turn's starting state, not the midpoint.
*
* Read failures are swallowed. A snapshot is a convenience; a tool call that fails
* because the snapshot layer could not read an unrelated path would be worse than no
* undo at all.
*/
async capture(absPath: string): Promise<void> {
if (!this.current) return;
const rel = relPath(this.cwd, absPath);
if (rel === undefined) return;
if (this.current.files.has(rel)) return;
try {
const file = Bun.file(absPath);
const exists = await file.exists();
if (!exists) {
this.current.files.set(rel, undefined);
return;
}
if (file.size > MAX_FILE_BYTES) return;
this.current.files.set(rel, await file.text());
} catch {
return;
}
}
/** Records any path a tool call is about to touch. Returns whether the tool is covered at all. */
async captureFor(toolName: string, input: unknown): Promise<{ covered: boolean; paths: string[] }> {
const { paths, covered } = touchedPaths(toolName, input);
for (const p of paths) await this.capture(resolve(this.cwd, p));
return { paths, covered };
}
/**
* Closes the turn, keeping it only if it changed something.
*
* A turn that read and answered without writing is not worth a slot, and keeping it
* would make `/undo` step past a turn that has nothing to undo — which reads as the
* command being broken.
*/
commit(): TurnSnapshot | undefined {
const cur = this.current;
this.current = null;
if (!cur || cur.files.size === 0) return undefined;
const snap: TurnSnapshot = {
turn: cur.turn,
at: cur.at,
prompt: cur.prompt,
files: [...cur.files].map(([path, before]) => ({ path, before })),
messageCount: cur.messageCount,
};
this.turns.push(snap);
while (this.turns.length > MAX_SNAPSHOTS) this.turns.shift();
return snap;
}
/** Discards the open turn without recording it, for an aborted or failed turn. */
discard(): void {
this.current = null;
}
/** The turns that can be undone, newest first. */
list(): readonly TurnSnapshot[] {
return [...this.turns].reverse();
}
/** Whether there is an open turn collecting pre-images right now. */
get open(): boolean {
return this.current !== null;
}
/**
* Removes the newest turn and returns what it holds, without restoring.
*
* Separated from restoring so the caller can decide *what* to bring back —
* files, conversation, or both — which is the split Claude Code's rewind menu
* exposes and the reason one control surface is worth more than three commands.
*/
pop(): TurnSnapshot | undefined {
return this.turns.pop();
}
/** Puts a turn back, for a `/redo` that follows an `/undo`. */
push(snap: TurnSnapshot): void {
this.turns.push(snap);
}
get size(): number {
return this.turns.length;
}
cwdOf(): string {
return this.cwd;
}
}
/** Writes a pre-image back to disk, recreating a deleted file or removing one that was created. */
export async function restore(snap: TurnSnapshot, cwd = process.cwd()): Promise<{ restored: string[]; removed: string[] }> {
const restored: string[] = [];
const removed: string[] = [];
for (const file of snap.files) {
const abs = join(cwd, file.path);
if (file.before === undefined) {
// The file did not exist before the turn, so undoing its creation is removing it.
const f = Bun.file(abs);
if (await f.exists()) {
await f.delete();
removed.push(file.path);
}
continue;
}
await Bun.write(abs, file.before);
restored.push(file.path);
}
return { restored, removed };
}
/** A short, stable label for a snapshot, for a menu that lists several. */
export function labelOf(snap: TurnSnapshot): string {
const first = snap.prompt.trim().split('\n')[0] ?? '';
const clipped = first.length > 50 ? `${first.slice(0, 50)}...` : first || '(no prompt)';
return `turn ${snap.turn}: ${clipped}`;
}
/** A content hash, used to tell whether a file still matches what the snapshot holds. */
export function hashOf(text: string): string {
return createHash('sha256').update(text).digest('hex').slice(0, 12);
}
+80
View File
@@ -0,0 +1,80 @@
import { tool } from 'ai';
import { z } from 'zod';
/**
* A visible escape hatch for the loop a coding agent dies in.
*
* The repeat guard stops an *identical* call after three tries, but the deeper loop
* is the model making *different* calls that all amount to the same stalled attempt —
* re-reading the same file expecting a different answer, retrying a failing command
* with a tweaked flag, re-sending a prompt it has already asked. No equal-input
* detector fires on any of that, so the model burns the step budget on motions that
* never move.
*
* `step_back` exists so the model has a *named* way out instead of only a guard it
* cannot see. The session records each completed tool call (input + outcome) into a
* loop trace; the tool returns a scripted reflection prompt built from that trace,
* so the model is told in concrete terms that it is going in circles and is steered
* to change direction.
*/
export type LoopEntry = {
step: number;
toolName: string;
input: string;
result: string;
at: string;
};
function recentTrace(entries: readonly LoopEntry[], window = 8): LoopEntry[] {
return entries.slice(-window);
}
function renderTrace(entries: readonly LoopEntry[], maxLines = 12): string {
return entries.slice(-maxLines).map((e) => `step ${e.step}: ${e.toolName} ${e.input} -> ${e.result}`.slice(0, 200)).join('\n');
}
/**
* Builds a `step_back` tool bound to a session's loop trace.
*
* The tool is meant to be called when the model is not making progress — a trap it
* cannot always see while it is inside it. The returned reflection names the recent
* steps so the model can tell, from its own calls, that it is going in circles, and
* gives it the one thing a stuck agent is usually missing: permission to stop,
* say what it learned, and change direction rather than try harder.
*/
export function createStepBackTool(opts: { trace: () => readonly LoopEntry[] }) {
return tool({
description:
'Use when you are stuck: the same file is not changing, a command keeps failing, or you have done several steps with no visible progress. ' +
'Records your recent steps and returns a reflection prompt to help you change direction instead of repeating the attempt.',
inputSchema: z.object({
note: z.string().optional().describe('A sentence in your own words about what you were trying to do.'),
}),
execute: async ({ note }) => {
const trace = recentTrace(opts.trace());
if (trace.length === 0) {
return (
'No recent steps to reflect on. This is an early call of step_back — it is only useful when you have ' +
'attempted something several times. Describe what you are stuck on in `note`.'
);
}
return [
'You have run threadbare over the last steps and are stuck. Here is what you actually did:',
'```',
renderTrace(trace),
'```',
'',
'Before your next tool call, answer these three questions in your reasoning:',
'1. What exactly is wrong — the input, the tool, or the expectation?',
'2. What have you tried, and why did each fail?',
'3. What is ONE different thing you can do that is not "try the same thing a little harder"?',
'',
'Then take that different action. If the failure is a command, read the actual error and fix its cause — ',
'do not rerun the command. If a file is not what you expect, suspect your assumption about it and re-read it fresh.',
note?.trim() ? `\nYour note: ${note.trim()}` : '',
].join('\n');
},
});
}
+128 -82
View File
@@ -177,91 +177,137 @@ export function createTaskTool(opts: {
? 'explore: read-only research. review: read-only critique. worker: makes changes. Default explore.'
: 'explore: find and report. review: critique code for defects. Default explore.',
),
tasks: z
.array(
z.object({
description: z.string(),
prompt: z.string(),
kind: z.enum(canWrite ? ['explore', 'review', 'worker'] : ['explore', 'review']).optional(),
}),
)
.optional()
.describe(
'Independent investigations to run at the same time instead of one after another. ' +
'Use this for several unrelated searches so they overlap in wall-clock time. Each runs on its own ' +
'context window, exactly like a single task call. Default: run the single prompt above.',
),
}),
execute: async ({ description, prompt, kind }, { abortSignal }) => {
const flavour: SubagentKind = kind ?? 'explore';
if (flavour === 'worker' && !opts.approve) {
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
}
const id = `sub${++counter}`;
const report = opts.report;
report?.({ type: 'start', id, kind: flavour, description });
let steps = 0;
let text = '';
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
try {
// `explore` is search, not reasoning, so it runs on the cheaper model when
// one is configured. `review` and `worker` keep the parent's: they judge
// and they change, both of which want the full model.
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
const result = streamText({
model,
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
messages: [{ role: 'user', content: prompt }],
tools: TOOLS[flavour],
stopWhen: isStepCount(opts.maxSteps ?? 20),
...(opts.approve
? {
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const approved = await opts.approve!(toolCall);
return approved
? undefined
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
},
}
: {}),
...(abortSignal ? { abortSignal } : {}),
});
const sink = () => {};
void result.responseMessages.then(undefined, sink);
void result.usage.then(undefined, sink);
void result.steps.then(undefined, sink);
void result.finalStep.then(undefined, sink);
void result.finishReason.then(undefined, sink);
for await (const part of result.stream) {
if (part.type === 'tool-call') {
steps++;
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
} else if (part.type === 'tool-result') {
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
} else if (part.type === 'tool-error') {
const message = part.error instanceof Error ? part.error.message : String(part.error);
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
} else if (part.type === 'text-delta') {
text += part.text;
} else if (part.type === 'error') {
// A provider failure arrives as a stream part, not a throw, so it has to
// be rethrown here or the subagent silently returns nothing.
const message = part.error instanceof Error ? part.error.message : String(part.error);
throw part.error instanceof Error ? part.error : new Error(message);
}
}
try {
const usage = await result.usage;
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
} catch {
// A run that errored before producing usage has nothing to account for.
}
} catch (e) {
const message = e instanceof Error ? e.message : String(e);
report?.({ type: 'error', id, message });
throw e;
}
const trimmed = text.trim();
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
// Settled after the stream closes; a failed run reports nothing rather than
// a half count. The parent prices these against the subagent's own model id.
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
return trimmed || 'Subagent returned no findings.';
execute: async ({ description, prompt, kind, tasks }, { abortSignal }) => {
const plans = tasks?.length ? tasks : [{ description, prompt, kind }];
const runs = await Promise.all(
plans.map((p) => runSubagent({ ...opts, abortSignal }, { description: p.description, prompt: p.prompt, kind: p.kind })),
);
if (runs.length === 1) return runs[0]!.report;
return runs.map((r) => `## ${r.named}\n\n${r.report}`).join('\n\n---\n\n');
},
});
}
/** Flavour of one planned investigation; `kind` defaults to explore. */
type Plan = { description: string; prompt: string; kind?: string };
/**
* Runs one subagent and settles its report, usage, and panel events.
*
* Extracted so `tasks` can fan several out in parallel: each full run is
* independent — its own id, own stream, own spend — and they overlap simply by
* awaiting them together.
*/
async function runSubagent(
opts: {
model: LanguageModel;
subagentModel?: LanguageModel;
subagentModelId?: string;
cwd?: string;
maxSteps?: number;
report?: SubagentReporter;
approve?: SubagentApproval;
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
abortSignal?: AbortSignal;
},
plan: Plan,
): Promise<{ named: string; report: string }> {
const flavour: SubagentKind = (plan.kind as SubagentKind | undefined) ?? 'explore';
if (flavour === 'worker' && !opts.approve) {
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
}
const id = `sub${++counter}`;
const report = opts.report;
report?.({ type: 'start', id, kind: flavour, description: plan.description });
let steps = 0;
let text = '';
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
try {
// `explore` is search, not reasoning, so it runs on the cheaper model when
// one is configured. `review` and `worker` keep the parent's: they judge
// and they change, both of which want the full model.
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
const result = streamText({
model,
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
messages: [{ role: 'user', content: plan.prompt }],
tools: TOOLS[flavour],
stopWhen: isStepCount(opts.maxSteps ?? 20),
...(opts.approve
? {
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const approved = await opts.approve!(toolCall);
return approved
? undefined
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
},
}
: {}),
...(opts.abortSignal ? { abortSignal: opts.abortSignal } : {}),
});
const sink = () => {};
void result.responseMessages.then(undefined, sink);
void result.usage.then(undefined, sink);
void result.steps.then(undefined, sink);
void result.finalStep.then(undefined, sink);
void result.finishReason.then(undefined, sink);
for await (const part of result.stream) {
if (part.type === 'tool-call') {
steps++;
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
} else if (part.type === 'tool-result') {
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
} else if (part.type === 'tool-error') {
const message = part.error instanceof Error ? part.error.message : String(part.error);
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
} else if (part.type === 'text-delta') {
text += part.text;
} else if (part.type === 'error') {
// A provider failure arrives as a stream part, not a throw, so it has to
// be rethrown here or the subagent silently returns nothing.
const message = part.error instanceof Error ? part.error.message : String(part.error);
throw part.error instanceof Error ? part.error : new Error(message);
}
}
try {
const usage = await result.usage;
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
} catch {
// A run that errored before producing usage has nothing to account for.
}
} catch (e) {
const message = e instanceof Error ? e.message : String(e);
report?.({ type: 'error', id, message });
throw e;
}
const trimmed = text.trim();
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
// Settled after the stream closes; a failed run reports nothing rather than
// a half count. The parent prices these against the subagent's own model id.
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
return { named: plan.description, report: trimmed || 'Subagent returned no findings.' };
}
export const TASK_TOOL_NAME = 'task';
+27
View File
@@ -0,0 +1,27 @@
import type { Tool } from 'ai';
/**
* Marks a tool as mutating where it is defined, rather than in a list beside it.
*
* The bug this prevents is the worst one this codebase can have: a new tool that
* writes to the workspace, added to `tools` but forgotten in a hand-maintained
* `MUTATING_TOOLS`, is a write the permission layer does not treat as a write. It
* is silent, it passes every test that does not think to check the new name, and
* it surfaces as a user discovering an edit they never approved.
*
* The mark is a property on the tool object, so `MUTATING_TOOLS` can be derived by
* filtering the registry instead of being typed out. A tool that is not marked is
* asserted non-mutating by `tools.test.ts`, which means the decision is made once,
* at the definition, and cannot drift.
*/
export const MUTATING = '__mutating' as const;
/** A tool that can change the workspace or run arbitrary code. */
export function mutating<T extends Tool>(t: T): T {
return Object.assign(t, { [MUTATING]: true as const });
}
/** Whether a tool was declared mutating at its definition site. */
export function isMutating(t: unknown): boolean {
return typeof t === 'object' && t !== null && (t as Record<string, unknown>)[MUTATING] === true;
}
+11 -10
View File
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
import { resolve } from 'node:path';
import { z } from 'zod';
import { jail, posix, walk } from './ignore';
import { mutating } from './tool-kinds';
import { git } from './tools-git';
/**
@@ -36,7 +37,7 @@ async function readLines(path: string): Promise<{ abs: string; lines: string[] }
// edit
// ---------------------------------------------------------------------------
export const insertLinesTool = tool({
export const insertLinesTool = mutating(tool({
description:
'Insert lines at a 1-based position in a file, pushing the rest down. Cheaper and safer than a rewrite for adding a block in the middle.',
inputSchema: z.object({
@@ -51,9 +52,9 @@ export const insertLinesTool = tool({
await Bun.write(abs, cur.join('\n'));
return `Inserted ${lines(text).length} line(s) at ${path}:${line}`;
},
});
}));
export const deleteLinesTool = tool({
export const deleteLinesTool = mutating(tool({
description: 'Delete an inclusive range of lines from a file. Refuses to delete the whole file; use delete_file for that.',
inputSchema: z.object({
path: z.string(),
@@ -69,9 +70,9 @@ export const deleteLinesTool = tool({
await Bun.write(abs, cur.join('\n'));
return `Deleted lines ${start}-${end} from ${path}`;
},
});
}));
export const replaceLinesTool = tool({
export const replaceLinesTool = mutating(tool({
description: 'Replace an inclusive range of lines with new text, in one write.',
inputSchema: z.object({
path: z.string(),
@@ -87,9 +88,9 @@ export const replaceLinesTool = tool({
await Bun.write(abs, cur.join('\n'));
return `Replaced lines ${start}-${end} in ${path}`;
},
});
}));
export const appendFileTool = tool({
export const appendFileTool = mutating(tool({
description: 'Append text to the end of a file without reading the whole thing into the edit.',
inputSchema: z.object({ path: z.string(), text: z.string() }),
execute: async ({ path, text }) => {
@@ -97,9 +98,9 @@ export const appendFileTool = tool({
await Bun.write(abs, `${cur.join('\n').replace(/\n?$/, '\n')}${text.replace(/\n?$/, '')}\n`);
return `Appended ${lines(text).length} line(s) to ${path}`;
},
});
}));
export const prependFileTool = tool({
export const prependFileTool = mutating(tool({
description: 'Prepend text to the start of a file, e.g. a license header or an import block.',
inputSchema: z.object({ path: z.string(), text: z.string() }),
execute: async ({ path, text }) => {
@@ -107,7 +108,7 @@ export const prependFileTool = tool({
await Bun.write(abs, `${text.replace(/\n?$/, '\n')}${cur.join('\n')}`);
return `Prepended ${lines(text).length} line(s) to ${path}`;
},
});
}));
export const countLinesTool = tool({
description: 'Count lines in one file, or per file across a glob. A quick size read before deciding to open something large.',
+115 -46
View File
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
import { join, resolve } from 'node:path';
import { z } from 'zod';
import { jail, posix, walk } from './ignore';
import { isMutating, mutating } from './tool-kinds';
import { EXTRA_TOOL_NAMES, extraTools } from './tools-extra';
import { GIT_TOOL_NAMES, gitTools } from './tools-git';
import { NET_TOOL_NAMES, netTools } from './tools-net';
@@ -174,7 +175,7 @@ export function parsePatch(patch: string): PatchOp[] {
return ops;
}
export const applyPatchTool = tool({
export const applyPatchTool = mutating(tool({
description:
'Apply one patch across several files: add, update, move, and delete in a single call. All or nothing — if any ' +
'part fails, nothing is written. Use it when a change spans files that must land together, such as a rename ' +
@@ -248,7 +249,7 @@ export const applyPatchTool = tool({
return `Applied ${ops.length} change${ops.length === 1 ? '' : 's'}:\n${summary.map((s) => `- ${s}`).join('\n')}`;
},
});
}));
/**
* A rewrite that collapses whitespace: similar character count, a fraction of the lines.
@@ -264,7 +265,7 @@ function collapsedRewrite(before: string, after: string): boolean {
return after.split('\n').length < before.split('\n').length / 2;
}
export const writeFileTool = tool({
export const writeFileTool = mutating(tool({
description: 'Create a file or overwrite it completely. Prefer edit_file for existing files.',
inputSchema: z.object({
path: z.string(),
@@ -284,17 +285,39 @@ export const writeFileTool = tool({
}
return `Wrote ${content.length} chars to ${path}`;
},
});
}));
export const editFileTool = tool({
// Some models (Claude-style tool docs, DeepSeek/GLM) emit snake_case edit params
// (old_string/new_string/replace_all) despite the camelCase schema. Normalize at the
// boundary instead of failing the whole call on a naming convention.
const SNAKE_EDIT_ARGS: ReadonlyArray<readonly [string, string]> = [
['old_string', 'oldString'],
['new_string', 'newString'],
['replace_all', 'replaceAll'],
];
export function normalizeEditArgs(input: unknown): unknown {
if (typeof input !== 'object' || input === null) return input;
const obj: Record<string, unknown> = { ...(input as Record<string, unknown>) };
for (const [snake, camel] of SNAKE_EDIT_ARGS) {
if (obj[snake] !== undefined && obj[camel] === undefined) obj[camel] = obj[snake];
}
if (Array.isArray(obj.edits)) obj.edits = obj.edits.map(normalizeEditArgs);
return obj;
}
export const editFileTool = mutating(tool({
description:
'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.',
inputSchema: z.object({
path: z.string(),
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
}),
inputSchema: z.preprocess(
normalizeEditArgs,
z.object({
path: z.string(),
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
}),
),
execute: async ({ path, oldString, newString, replaceAll = false }) => {
if (oldString === newString) throw new Error('oldString and newString are identical');
const abs = jail(path);
@@ -312,27 +335,30 @@ export const editFileTool = tool({
await Bun.write(abs, after);
return `Replaced ${replaceAll ? count : 1} occurrence(s) in ${path}`;
},
});
}));
export const multiEditTool = tool({
export const multiEditTool = mutating(tool({
description:
'Apply several exact-string edits to one file in a single call. Each edit sees the result of the previous one. ' +
'All or nothing: if any oldString fails to match, or matches more than once without replaceAll, nothing is ' +
'written. Prefer this over repeated edit_file calls on the same file — one approval, one write, no risk of ' +
'leaving the file half-changed.',
inputSchema: z.object({
path: z.string(),
edits: z
.array(
z.object({
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional(),
}),
)
.min(1)
.describe('Edits in the order they should be applied'),
}),
inputSchema: z.preprocess(
normalizeEditArgs,
z.object({
path: z.string(),
edits: z
.array(
z.object({
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional(),
}),
)
.min(1)
.describe('Edits in the order they should be applied'),
}),
),
execute: async ({ path, edits }) => {
const abs = jail(path);
const file = Bun.file(abs);
@@ -367,7 +393,7 @@ export const multiEditTool = tool({
await Bun.write(abs, text);
return `Applied ${edits.length} edit(s) to ${path} (${applied.join(', ')})`;
},
});
}));
export const globTool = tool({
description:
@@ -573,7 +599,13 @@ async function pump(
return all;
}
type Running = { command: string; proc: Bun.Subprocess; interrupted: boolean; killed?: Promise<unknown> };
type Running = {
command: string;
proc: Bun.Subprocess;
interrupted: boolean;
timedOut?: boolean;
killed?: Promise<unknown>;
};
const running = new Map<string, Running>();
@@ -615,6 +647,12 @@ function killTree(proc: Bun.Subprocess): Promise<unknown> {
export function interruptBash(): string[] {
const killed: string[] = [];
for (const entry of running.values()) {
// A second ctrl-c while the first killTree is still settling must not re-announce
// the same command: the notice is the only proof the keypress did anything.
if (entry.interrupted) {
killed.push(entry.command);
continue;
}
entry.interrupted = true;
entry.killed = killTree(entry.proc);
killed.push(entry.command);
@@ -622,7 +660,7 @@ export function interruptBash(): string[] {
return killed;
}
export const bashTool = tool({
export const bashTool = mutating(tool({
description:
'Run a shell command in the workspace root. Use for builds, tests, git, and package managers. ' +
'Output streams live and the user can interrupt a command with ctrl-c without ending the turn.',
@@ -636,13 +674,32 @@ export const bashTool = tool({
cwd: process.cwd(),
stdout: 'pipe',
stderr: 'pipe',
timeout,
...(abortSignal ? { signal: abortSignal } : {}),
});
const entry: Running = { command, proc, interrupted: false };
running.set(toolCallId, entry);
// Bun's spawn `signal` option is not used either: it kills only the shell, so an
// esc-abort orphaned the grandchild on the same still-open pipes as the timeout
// did. The abort must go through killTree, exactly like ctrl-c does.
const onAbort = () => {
if (entry.interrupted) return;
entry.interrupted = true;
entry.killed = killTree(proc);
};
abortSignal?.addEventListener('abort', onAbort);
// The turn may already be aborted by the time this tool starts; a past event
// never re-fires, so check once here or the command runs unkillable by esc.
if (abortSignal?.aborted) onAbort();
// Bun's own `timeout` spawn option is not used: it kills only the shell, and the
// grandchild holding the output pipes keeps `pump` reading forever, so the tool
// never returns. Same failure killTree exists for, just triggered by the clock.
const timer = setTimeout(() => {
entry.timedOut = true;
entry.killed = killTree(proc);
}, timeout);
try {
// Drained concurrently: a command that fills one pipe while we block on the
// other would deadlock, and buffering both hides progress for minutes.
@@ -658,6 +715,15 @@ export const bashTool = tool({
// Thrown rather than returned: the model must not read a killed command as
// a command that ran and failed on its own terms.
if (entry.timedOut) {
throw new Error(
cap(
`The command exceeded its ${timeout}ms timeout and was killed. It did not finish, so its effects are unknown.\n${
body || '(no output before it was killed)'
}`,
),
);
}
if (entry.interrupted) {
throw new Error(
cap(
@@ -678,15 +744,17 @@ export const bashTool = tool({
.join('\n\n'),
);
} finally {
clearTimeout(timer);
abortSignal?.removeEventListener('abort', onAbort);
// Awaited so the process really is gone before the tool returns. On Windows a
// surviving grandchild holds the cwd open, which breaks the very next command.
await entry.killed;
running.delete(toolCallId);
}
},
});
}));
export const moveFileTool = tool({
export const moveFileTool = mutating(tool({
description:
'Move or rename one file. Creates the target directory. Refuses if the source is missing or the target ' +
'already exists, so a rename cannot silently overwrite work. For a rename plus its callers in one step, ' +
@@ -708,9 +776,9 @@ export const moveFileTool = tool({
await file.delete();
return `Moved ${from} to ${to}`;
},
});
}));
export const deleteFileTool = tool({
export const deleteFileTool = mutating(tool({
description:
'Delete one file. Refuses a directory: removing a tree is what the guard plugin blocks in bash, and it is ' +
'not something to do implicitly. Delete the files you mean, one call each.',
@@ -733,7 +801,7 @@ export const deleteFileTool = tool({
await Bun.file(abs).delete();
return `Deleted ${path} (${entry.size} bytes)`;
},
});
}));
/**
* Definition patterns for `find_symbol`, keyed loosely by language.
@@ -909,15 +977,16 @@ export function disabledToolNames(enabled: readonly ToolSetName[] | undefined):
return TOOL_SET_NAMES.filter((set) => !live.has(set)).flatMap((set) => [...TOOL_SETS[set]]);
}
/** Tools that mutate the workspace or run arbitrary code always ask the user first. */
export const MUTATING_TOOLS = [
'write_file',
'edit_file',
'multi_edit',
'apply_patch',
'move_file',
'delete_file',
'bash',
] as const;
/**
* Tools that mutate the workspace or run arbitrary code always ask the user first.
*
* Derived from the tools themselves rather than typed out: a tool marked `mutating`
* at its definition is in this list by construction, and there is no second place to
* forget it. `tools.test.ts` asserts the converse — that nothing here is unmarked —
* so the two cannot disagree.
*/
export const MUTATING_TOOLS: readonly string[] = Object.entries(tools)
.filter(([, t]) => isMutating(t))
.map(([name]) => name);
export { jail };
+69 -11
View File
@@ -176,18 +176,34 @@ export function App({
const fileMatches = token && paths ? matchPaths(paths, token.query) : [];
const highlightedPath = fileMatches[Math.min(fileIndex, Math.max(0, fileMatches.length - 1))];
// The walk costs a full ignore-aware traversal, so it happens on the first `@`
// rather than at startup, and only once.
// The walk costs a full ignore-aware traversal, so it runs once at the first `@`.
// A slow cooldown re-walks so a file created after that first `@` shows up within
// a short window instead of staying hidden all session. The cooldown never fires
// on a short session, so the "walks once" behaviour most users see is unchanged.
const pathsRef = useRef(paths);
useEffect(() => {
if (token === undefined || paths !== undefined) return;
let live = true;
void hooks.listPaths().then((all) => {
if (live) setPaths(all);
});
return () => {
live = false;
};
}, [hooks, paths, token]);
pathsRef.current = paths;
}, [paths]);
const didLoadRef = useRef(false);
useEffect(() => {
if (token === undefined) return;
if (!didLoadRef.current) {
didLoadRef.current = true;
let live = true;
void hooks.listPaths().then((all) => {
if (live) setPaths(all);
});
const refresh = setInterval(async () => {
if (!live) return;
const all = await hooks.listPaths();
if (live && JSON.stringify(all) !== JSON.stringify(pathsRef.current)) setPaths(all);
}, 10_000);
return () => {
live = false;
clearInterval(refresh);
};
}
}, [hooks, token]);
useEffect(() => bridge.bind(setPending), [bridge]);
useEffect(() => askBridge?.bind(setAsking), [askBridge]);
@@ -691,6 +707,48 @@ export function App({
setModelPicker(models);
return;
}
case 'undo': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
const result = await session.undo(action.what);
if (!result) {
push({ kind: 'info', text: 'nothing to undo' });
} else {
const parts: string[] = [];
if (result.restored.length > 0) parts.push(`restored ${result.restored.join(', ')}`);
if (result.removed.length > 0) parts.push(`removed ${result.removed.join(', ')}`);
if (result.conversationTrimmed) parts.push(`dropped ${result.snapshot.messageCount}-onward from the history`);
push({ kind: 'info', text: `undid turn ${result.snapshot.turn}: ${parts.join('; ') || 'no file changes'}` });
// The same limit the turn-end notice states: bash is not snapshotted.
push({
kind: 'info',
text: 'Only file-tool edits are covered. A bash command or a git checkout in that turn is not, so check git status if one ran.',
});
}
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
case 'redo': {
push({ kind: 'user', text: chosen.trim() });
const result = session.redo(action.what);
if (!result) {
push({ kind: 'info', text: 'nothing to redo' });
} else if (action.what === 'files' || action.what === 'both') {
// Refused rather than approximated: only the pre-image was captured, so
// there is no post-turn content to put back.
push({
kind: 'info',
text: `put turn ${result.snapshot.turn} back on the undo stack. File contents cannot be re-applied - only the state before the turn was recorded.`,
});
} else {
push({ kind: 'info', text: `put turn ${result.snapshot.turn} back; the conversation was restored to it` });
}
return;
}
case 'compact': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
+2 -1
View File
@@ -1,6 +1,7 @@
import { Box, Text, useInput } from 'ink';
import React from 'react';
import type { ApprovalDecision, ApprovalRequest } from '../session';
import { normalizeEditArgs } from '../tools';
import { Diff } from './Diff';
import { accent, glyph } from './theme';
import { toolDetail } from './transcript';
@@ -42,7 +43,7 @@ export function createApprovalBridge(): ApprovalBridge {
* consistent and far more readable than a JSON dump of the input.
*/
function ApprovalDetail({ name, input }: { name: string; input: unknown }) {
const o = (input ?? {}) as Record<string, unknown>;
const o = (normalizeEditArgs(input) ?? {}) as Record<string, unknown>;
if (name === 'write_file') {
const content = String(o['content'] ?? '');
+3 -7
View File
@@ -33,10 +33,6 @@ type KeyLike = {
end?: boolean;
};
const INVERSE_ON = '\u001B[7m';
const INVERSE_OFF = '\u001B[27m';
const invert = (s: string) => `${INVERSE_ON}${s}${INVERSE_OFF}`;
/**
* Text input with a real cursor and shell-style history recall.
*
@@ -151,10 +147,10 @@ export function PromptInput({
);
if (value.length === 0) {
if (!placeholder) return <Text>{focus ? invert(' ') : ' '}</Text>;
if (!placeholder) return <Text inverse={focus}>{' '}</Text>;
return (
<Text dimColor>
{focus ? invert(placeholder.slice(0, 1)) : placeholder.slice(0, 1)}
<Text inverse={focus}>{placeholder.slice(0, 1)}</Text>
{placeholder.slice(1)}
</Text>
);
@@ -166,7 +162,7 @@ export function PromptInput({
return (
<Text>
{shown.slice(0, cursor)}
{invert(shown.slice(cursor, cursor + 1) || ' ')}
<Text inverse>{shown.slice(cursor, cursor + 1) || ' '}</Text>
{shown.slice(cursor + 1)}
</Text>
);
+3 -1
View File
@@ -1,4 +1,5 @@
import { TODO_MARK } from '../notebook';
import { normalizeEditArgs } from '../tools';
export type Line =
| { key: string; kind: 'user'; text: string }
@@ -38,7 +39,8 @@ export function preview(input: unknown): string {
* the transcript, beside the spinner while a call is in flight, and in the approval
* prompt for any tool without a diff of its own.
*/
export function toolDetail(name: string, input: unknown): string[] {
export function toolDetail(name: string, rawInput: unknown): string[] {
const input = normalizeEditArgs(rawInput);
if (input === null || typeof input !== 'object') return [];
const o = input as Record<string, unknown>;
const str = (k: string) => (typeof o[k] === 'string' ? (o[k] as string) : undefined);