Merge remote-tracking branch 'refs/remotes/upstream/main'
ci / check (macos-latest) (push) Canceled after 0s
ci / check (ubuntu-latest) (push) Canceled after 0s
ci / check (windows-latest) (push) Canceled after 0s

# Conflicts:
#	ROADMAP.md
#	TODO.md
#	src/cli.tsx
#	src/config.ts
#	src/mcp.ts
#	src/permission.ts
#	src/prompt.ts
#	src/session.ts
#	src/snapshot.ts
#	src/subagent.ts
#	src/tools-extra.ts
#	src/tools.ts
#	src/ui/App.tsx
#	test/mcp.test.ts
#	test/prune.test.ts
#	test/session.test.ts
#	test/tools.test.ts
This commit is contained in:
asepharyana
2026-09-21 20:43:35 +07:00
59 changed files with 3481 additions and 463 deletions
+3 -2
View File
@@ -168,7 +168,8 @@ if (resumeArg) {
}
}
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
const mcp =
has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers, cfg.mcpMode ?? 'lazy');
const instructions = has('--no-instructions') ? [] : await loadInstructions();
const skills = has('--no-skills') ? [] : await loadSkills();
const customCommands = await loadCustomCommands();
@@ -765,7 +766,7 @@ const facts: HeaderFact[] = [
memory && memory.all().length > 0
? { label: 'memory', value: `${memory.all().length} notes about this project` }
: undefined,
mcp && Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length} tools${(mcp.tools as Record<string, unknown>)['mcp_list'] ? ' (mcp_list/mcp_inspect/mcp_call)' : ''}` } : undefined,
mcp && Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length} tools${(mcp.tools as Record<string, unknown>)['mcp_list'] ? ' (mcp_list/mcp_inspect/mcp_call)' : ''}` } : undefined,
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
: undefined,
+24 -8
View File
@@ -6,6 +6,8 @@ export type CommandAction =
| { type: 'exit' }
| { type: 'clear' }
| { type: 'compact' }
| { type: 'undo'; what: 'both' | 'files' | 'conversation' }
| { type: 'redo'; what: 'both' | 'files' | 'conversation' }
| { type: 'tools' }
| { type: 'cost' }
| { type: 'sessions' }
@@ -26,8 +28,6 @@ export type CommandAction =
| { type: 'info'; text: string }
| { type: 'model'; model: string }
| { type: 'resume'; id: string }
| { type: 'undo' }
| { type: 'redo' }
| { type: 'changes' }
| { type: 'diff'; action: 'raw' | 'review' }
| { type: 'bash'; action: 'list' | 'stop' | 'stop-all'; arg?: string }
@@ -66,12 +66,12 @@ export const COMMANDS: CommandSpec[] = [
{ name: 'memory', summary: 'compact the project memory with the model' },
{ name: 'tools', summary: 'list available tools' },
{ name: 'compact', summary: 'replace history with a model-written summary' },
{ name: 'undo', arg: '[files|conversation]', summary: 'walk the last turn back: files, conversation, or both' },
{ name: 'redo', arg: '[conversation]', summary: 'put back what /undo took' },
{ name: 'cost', summary: 'tokens and estimated spend this session' },
{ name: 'sessions', summary: 'list saved sessions' },
{ name: 'resume', arg: '<id>', summary: 'load a saved session' },
{ name: 'save', summary: 'write the session to disk now' },
{ name: 'undo', summary: 'undo the last turn — restores files and conversation (bash effects are not snapshotted)' },
{ name: 'redo', summary: 'redo the last undone turn' },
{ name: 'changes', summary: 'show what the last turn changed on disk' },
{ name: 'diff', arg: '[review]', summary: 'diff the last turn; /diff review shows per-hunk file:line blocks' },
{ name: 'bash', arg: '[list|stop <id>|stop all]', summary: 'list or stop background commands started with bash background: true' },
@@ -179,6 +179,22 @@ function parseMcp(arg: string): CommandAction {
}
}
/**
* `/undo [files|conversation]` and `/redo [conversation]`.
*
* The default is `both` for undo, because restoring one without the other is the
* failure the two are meant to prevent: files back without the history and the model
* re-reads a change it no longer made. A bare `files` or `conversation` narrows it.
* Redo defaults to the conversation, since file content after the turn was never kept.
*/
function parseUndoKind(arg: string, fallback: 'both' | 'conversation'): 'both' | 'files' | 'conversation' {
const word = arg.trim().toLowerCase();
if (word === 'files' || word === 'file') return 'files';
if (word === 'conversation' || word === 'chat' || word === 'history') return 'conversation';
if (word === 'both' || word === 'all' || word === '') return fallback;
return fallback;
}
/**
* Pure parser: no IO, so the TUI and headless mode share one definition.
*
@@ -204,6 +220,10 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
return { type: 'clear' };
case 'compact':
return { type: 'compact' };
case 'undo':
return { type: 'undo', what: parseUndoKind(arg, 'both') };
case 'redo':
return { type: 'redo', what: parseUndoKind(arg, 'conversation') };
case 'tools':
return { type: 'tools' };
case 'cost':
@@ -243,10 +263,6 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
return arg ? { type: 'model', model: arg } : { type: 'models' };
case 'resume':
return arg ? { type: 'resume', id: arg } : { type: 'info', text: 'usage: /resume <session-id>' };
case 'undo':
return { type: 'undo' };
case 'redo':
return { type: 'redo' };
case 'changes':
return { type: 'changes' };
case 'diff': {
+6 -1
View File
@@ -53,7 +53,7 @@ export type Config = {
/** Install unsigned registry entries. Default false — signed entries are required. */
registryAllowUnsigned?: boolean;
mcpServers?: Record<string, McpServerConfig>;
/** A check command (e.g. `tsc --watch`) run in the UI only, never in model context. */
/** A check command (e.g. `tsc --watch`) run in the UI only, never in model context. */
diagnostics?: string;
/**
* When a turn ends normally but the task list still has work, keep going with
@@ -61,6 +61,11 @@ export type Config = {
* `true` on, `false` off, or `{ "maxTurns": n }` to bound it. Default on.
*/
continueWhileTodos?: boolean | { maxTurns?: number };
/**
* How MCP tools reach the model: `lazy` registers meta-tools only (cheap until a
* tool is called), `eager` registers every server tool up front. Omit for lazy.
*/
mcpMode?: 'lazy' | 'eager';
};
const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json');
+26 -9
View File
@@ -7,6 +7,17 @@ export type McpServerConfig =
| ({ command: string; args?: string[]; env?: Record<string, string>; cwd?: string } & { expose?: 'direct' | 'meta' })
| ({ url: string; type?: 'http' | 'sse'; headers?: Record<string, string> } & { expose?: 'direct' | 'meta' });
/**
* How a server's tools reach the model.
*
* `eager` registers every tool with its schema up front — cheap for a two-tool
* server, a tax for one that exposes twenty. `lazy` registers only the three
* meta-tools below and fetches a server's tools on demand via `mcp_call`, so a
* configured server costs almost nothing in the request until a tool is actually
* invoked.
*/
export type McpMode = 'eager' | 'lazy';
export type McpHandle = {
/** Live clients keyed by server name — only for servers that connected. */
clients: Map<string, MCPClient>;
@@ -15,6 +26,8 @@ export type McpHandle = {
/** Tools to merge into the session: direct mcp__* + 3 meta tools when any server exists. */
tools: ToolSet;
errors: { server: string; message: string }[];
/** Server names, for the prompt's MCP line. Empty when the mode is eager. */
servers: string[];
close: () => Promise<void>;
};
@@ -143,7 +156,10 @@ export function createMcpMetaTools(handle: McpHandle): ToolSet {
* through the 3 meta-tools so their schemas cost nothing until used.
* A server that fails to start is reported, never fatal.
*/
export async function connectMcp(servers: Record<string, McpServerConfig>): Promise<McpHandle> {
export async function connectMcp(
servers: Record<string, McpServerConfig>,
mode: McpMode = 'lazy',
): Promise<McpHandle> {
const clients = new Map<string, MCPClient>();
const tools: ToolSet = {};
const errors: McpHandle['errors'] = [];
@@ -161,8 +177,8 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
...(cfg.cwd ? { cwd: cfg.cwd } : {}),
}),
});
clients.set(name, client);
if (isDirect(cfg)) {
clients.set(name, client);
if (mode === 'eager' || isDirect(cfg)) {
for (const [toolName, t] of Object.entries(await client.tools())) {
tools[`mcp__${name}__${toolName}`] = t;
}
@@ -173,11 +189,12 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
}),
);
const handle: McpHandle = {
const handle: McpHandle = {
clients,
configs: servers,
tools,
errors,
servers: mode === 'eager' ? [] : [...clients.keys()],
close: async () => {
await Promise.all([...clients.values()].map((c) => c.close().catch(() => {})));
},
@@ -185,11 +202,11 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
toolCache.set(handle, new Map());
const hasAnyServer = Object.keys(servers).length > 0;
const hasMetaServer = Object.entries(servers).some(([, cfg]) => !isDirect(cfg));
if (hasMetaServer) {
const meta = createMcpMetaTools(handle);
Object.assign(tools, meta);
}
// The three meta-tools are always registered: a session with servers whose
// connection failed can still call mcp_list and be told why, and a direct
// server's own tools land alongside them.
const meta = createMcpMetaTools(handle);
Object.assign(tools, meta);
if (hasAnyServer) {
Object.defineProperty(tools, '__mcpServerNames', { value: Object.keys(servers), enumerable: false, writable: true, configurable: true });
}
-4
View File
@@ -325,10 +325,6 @@ export class Permissions {
this.granted.set(tool, set);
}
granted_(tool: string): string[] {
return [...(this.granted.get(tool) ?? [])];
}
/** The decision for one call, and which pattern decided it. */
check(tool: string, input: unknown): Resolved {
const resolved = resolve(entryFor(tool, this.config), tool, input);
+4
View File
@@ -13,6 +13,10 @@ export type Rate = { inputPerMTok: number; outputPerMTok: number };
* USD per million tokens. Prefix match on the model id, longest first, so
* `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates
* drift, so this is a best-effort estimate rather than a billing source.
*
* Source: vendor pricing pages, checked 2026-09-17. Anthropic (Anthropic API, not
* Batch) and OpenAI listed rates; DeepSeek and Grok per their API pricing. Rates
* are for input, then output. Re-verify before trusting a live spend figure.
*/
const RATES: Record<string, Rate> = {
'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 },
+10 -2
View File
@@ -148,8 +148,10 @@ function renderTools(available: readonly string[]): string {
// free, and the schema already says what each takes.
const git = extra.filter((n) => GIT_TOOL_NAMES.includes(n) && n !== 'git_commit_message');
const mcpDirect = extra.filter((n) => n.startsWith('mcp__'));
const META = ['mcp_list', 'mcp_inspect', 'mcp_call'];
const lazyMcp = META.filter((n) => available.includes(n));
const other = extra.filter(
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__'),
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__') && !META.includes(n),
);
if (git.length > 0) {
@@ -157,7 +159,13 @@ function renderTools(available: readonly string[]): string {
`- ${git.join(', ')}: read-only git, no approval needed. Use them instead of bash for history and diffs; they cannot mutate the repository.`,
);
}
if (mcpDirect.length > 0) {
if (lazyMcp.length > 0) {
// Lazy mode: the meta-tool descriptions already name the connected servers, so
// the model needs the workflow, not a schema listing.
lines.push(
`- ${lazyMcp.join(', ')}: MCP tools are fetched on demand. mcp_list names a server's tools, mcp_inspect reads one tool's schema, mcp_call runs it. Never guess a server or tool name: list first.`,
);
} else if (mcpDirect.length > 0) {
lines.push(
`- ${mcpDirect.join(', ')}: from MCP servers exposed direct (mcp__<server>__<tool>). Each needs approval.`,
);
+78 -2
View File
@@ -92,8 +92,6 @@ export function detachOrphanedItems(before: ModelMessage[], after: ModelMessage[
export type PruneOptions = Parameters<typeof pruneMessages>[0];
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
const anyParts = (message: ModelMessage): Part[] =>
Array.isArray(message.content) ? (message.content as Part[]) : [];
@@ -167,6 +165,84 @@ export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
return dropOrphanedResults(detachOrphanedItems(options.messages, pruned));
}
/**
* The messages a prune would discard, so they can be summarized before they go.
*
* Compaction keeps the model's *memory of a turn* — the tool tail it is told to
* keep stays verbatim. What it does not keep is any statement of what was
* dropped. So a decision from forty messages ago vanishes silently, and the model
* contradicts it with full confidence, because as far as it can tell it never
* said that.
*
* Identity is by reference, not by value: `prunePreservingItems` rebuilds the
* surviving messages with `{ ...message }`, so a value comparison would report
* every message as changed and no message as dropped. `Set` on the object
* references is exact.
*
* Only messages that carry content worth summarizing are returned — an assistant
* turn consisting of nothing but a dropped `reasoning` part is not a decision, and
* summarizing "the model thought for a while" is worse than saying nothing.
*/
export function droppedBy(before: ModelMessage[], after: ModelMessage[]): ModelMessage[] {
const surviving = new Set<ModelMessage>(after);
return before.filter((message) => !surviving.has(message));
}
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
function textOf(message: ModelMessage): string {
const { content } = message;
if (typeof content === 'string') return content;
if (!Array.isArray(content)) return '';
const chunks: string[] = [];
for (const part of content as Part[]) {
const p = part as Part & { text?: unknown; input?: unknown; output?: unknown };
if (typeof p.text === 'string') chunks.push(p.text);
// A tool call's input is the decision made: the path, the command, the patch.
else if (p.type === 'tool-call' && p.input !== undefined) chunks.push(JSON.stringify(p.input));
// A tool result is what came back. Without it a digest says what the model
// asked for and nothing about the answer, which is the half a later
// contradiction is usually argued from.
else if (ANSWER_PARTS.has(p.type) && p.output !== undefined) {
const rendered = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
chunks.push(rendered);
}
}
return chunks.join(' ').trim();
}
/**
* A one-line-per-message digest of what a prune wants to drop.
*
* This is the *fallback* when no summarizer is available or the call fails: crude,
* but it preserves the thing that matters — which tool touched which path, and in
* what order — rather than the nothing that is there today. The summarizer, when
* it runs, is a model and reads far better than this.
*/
export function digestOf(dropped: readonly ModelMessage[]): string {
const lines: string[] = [];
for (const message of dropped) {
const text = textOf(message);
if (!text) continue;
const role = message.role === 'tool' ? 'result' : message.role;
const clipped = text.length > 160 ? `${text.slice(0, 160)}...` : text;
lines.push(`- (${role}) ${clipped}`);
}
return lines.join('\n');
}
export const PRUNED_SPAN_PREFIX = 'Earlier in this session, now compacted away:';
export function isPrunedSpanSummary(message: ModelMessage): boolean {
return message.role === 'user' && typeof message.content === 'string' && message.content.startsWith(PRUNED_SPAN_PREFIX);
}
export function prunedSpanMessage(summary: string | undefined, dropped: readonly ModelMessage[]): ModelMessage | undefined {
const body = summary?.trim() || digestOf(dropped);
if (!body) return undefined;
return { role: 'user', content: `${PRUNED_SPAN_PREFIX}\n\n${body}` };
}
/**
* How many trailing messages keep their tool content, widest first.
*
+98 -6
View File
@@ -18,12 +18,13 @@ import { Permissions, type PermissionConfig } from './permission';
import type { PluginHost } from './plugins';
import { costOf, formatUsd } from './pricing';
import { systemPrompt } from './prompt';
import { detachProviderItems, droppedSpan, estimateTokens as pruneEstimateTokens, pruneToFit } from './prune';
import { detachProviderItems, droppedSpan, estimateTokens as pruneEstimateTokens, prunedSpanMessage, pruneToFit } from './prune';
import { walk } from './ignore';
import { createStepBackTool, type LoopEntry } from './step-back';
import { createSkillTool, renderSkills, type Skill } from './skills';
import { suggestSkillsFromTranscript, writeAutoSkill } from './skill-learner';
import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools';
import { onBeforeWrite, SnapshotStack, type FileState } from './snapshot';
import { onBeforeWrite, type FileState, restore, Snapshots, SnapshotStack, type TurnSnapshot } from './snapshot';
import { dirname, join, resolve } from 'node:path';
import { existsSync, readFileSync, statSync, readdirSync } from 'node:fs';
@@ -48,6 +49,21 @@ export type ChangeSummary = {
deleted: string[];
};
/** What `/undo` removed, so the UI can name the files. */
export type UndoResult = {
snapshot: TurnSnapshot;
restored: string[];
removed: string[];
conversationTrimmed: boolean;
};
/** What `/redo` put back. Files are never re-restored (no post-image), only the conversation is. */
export type RedoResult = {
snapshot: TurnSnapshot;
filesRestored: false;
what: 'both' | 'files' | 'conversation';
};
/** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */
export type ApprovalDecision = 'once' | 'always' | 'deny';
@@ -102,6 +118,8 @@ export type SessionOptions = {
plugins?: PluginHost;
/** Where an `ask` tool call goes. Omit in headless runs. */
ask?: AskFn;
/** How many identical calls this turn trip the repeat guard (default REPEAT_LIMIT). */
repeatLimit?: number;
messages?: ModelMessage[];
onChange?: (messages: ModelMessage[]) => void;
/** Live stdout/stderr from bash, for a UI that wants progress. */
@@ -153,6 +171,23 @@ const REPEAT_LIMIT = 3;
const callKey = (toolName: string, input: unknown) => `${toolName}:${JSON.stringify(input ?? null)}`;
/**
* Squashes a tool result into a few characters for the loop trace.
*
* The trace is fed to the model verbatim by `step_back`, so a 30 KB read_file
* output must not flood the next context window: keep the first 80 chars and
* say how long the original was.
*/
function summarizeToolResult(output: unknown): string {
if (typeof output === 'string') return output.length <= 80 ? output : `${output.slice(0, 80)}…(${output.length} chars)`;
try {
const json = JSON.stringify(output);
return json.length <= 80 ? json : `${json.slice(0, 80)}…`;
} catch {
return String(output);
}
}
/**
* The provider rejected an `item_reference` because it no longer holds that item:
* 404 "Item with id 'msg_...' not found". Retrying the same history repeats it, so
@@ -241,11 +276,16 @@ export class Session {
private pendingHost: PluginHost | undefined;
/** Calls seen this turn, for the repeat guard. Cleared per turn, not per step. */
private readonly seen = new Map<string, number>();
/** Every tool call this turn, input + outcome, for the loop-diagnosis tool. */
private readonly loopTrace: LoopEntry[] = [];
/** Tool-call inputs by call id, for the trace: `step_back` reads what was asked. */
private readonly callInputs = new Map<string, { toolName: string; input: string }>();
/** One stale-item repair per turn, so a repeating 404 cannot loop the run. */
private staleItemsRepaired = false;
/** The 80% spend warning is shown once, not on every turn past the line. */
private warnedSpend = false;
private controller: AbortController | undefined;
/** The undo stack: approved snapshotted tool calls, popped by `/undo`. */
private readonly snapshots = new SnapshotStack();
private turnBeforeLen = 0;
private turnBeforeFiles = new Map<string, FileState>();
@@ -311,6 +351,10 @@ export class Session {
...this.notebook.tools(),
...(this.opts.memory ? this.opts.memory.tools() : {}),
...skillTool,
// The loop escape hatch: step_back reflects on the session's loop trace
// and steers the model off a stalled attempt. Always registered, like the
// repo tools, so a circling model can reach it without a permission grant.
step_back: createStepBackTool({ trace: () => this.loopTrace }),
...(this.opts.ask ? { ask: createAskTool(this.opts.ask) } : {}),
};
const tools: ToolSet = { ...builtinTools, ...sessionTools, ...(this.pluginHost?.tools ?? {}), ...(this.opts.extraTools ?? {}) };
@@ -365,6 +409,15 @@ export class Session {
this.pluginHost = host;
this.rebuild();
}
/**
* Swaps the session's skill set (hot-reload path from the CLI). The next
* `buildSessionTools` picks the new list up via `currentSkills`.
*/
setSkills(skills: Skill[]): void {
this.currentSkills = skills;
}
private drainPendingHotReload(): void {
let changed = false;
if (this.pendingSkills !== undefined) { this.currentSkills = this.pendingSkills; this.pendingSkills = undefined; changed = true; }
@@ -912,6 +965,23 @@ export class Session {
return count;
}
/**
* Records one tool outcome for `step_back`: what was called and what came back.
*
* The trace is capped per turn so a chatty loop cannot grow it without bound.
*/
private recordTrace(callId: string, result: string): void {
const entry = this.callInputs.get(callId);
this.loopTrace.push({
step: this.loopTrace.length,
toolName: entry?.toolName ?? 'unknown',
input: entry?.input ?? '',
result,
at: new Date().toISOString(),
});
if (this.loopTrace.length > 50) this.loopTrace.shift();
}
/**
* Approval decisions, evaluated per call by the SDK.
*
@@ -952,7 +1022,18 @@ export class Session {
}
const repeats = this.repeatCount(toolName, input);
if (decision === 'allow' && repeats < REPEAT_LIMIT) return undefined;
const limit = this.opts.repeatLimit ?? REPEAT_LIMIT;
if (decision === 'allow' && repeats < limit) return undefined;
if (decision === 'allow') {
// Repeated with a permission that says `allow`: the model is looping, not
// asking, and it should stop and look at the trace rather than burn another
// approval. This is the point the step_back tool exists for.
notices.push(
`You have called ${toolName} with the same input ${repeats + 1} times this turn. It is not making progress. ` +
`Use step_back to reflect on what changed between attempts, then try a different approach or stop.`,
);
}
why.set(callKey(toolName, input), {
...(pattern ? { matchedPattern: pattern } : {}),
@@ -1052,6 +1133,8 @@ export class Session {
// Per turn, not per step: a tool called once in each of three steps is the
// loop this guards against.
this.seen.clear();
this.loopTrace.length = 0;
this.callInputs.clear();
this.staleItemsRepaired = false;
const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = [];
@@ -1081,7 +1164,7 @@ export class Session {
}
// tail for redo: the messages added by this turn
const tail = this.messages.slice(this.turnBeforeLen).map((m) => ({ ...m, content: typeof m.content === 'string' ? m.content : JSON.parse(JSON.stringify(m.content)) } as import('ai').ModelMessage));
const snap: import('./snapshot').TurnSnapshot & { _tail?: import('ai').ModelMessage[] } = {
const snap: { beforeLen: number; afterLen: number; beforeFiles: Map<string, FileState>; afterFiles: Map<string, FileState>; _tail?: import('ai').ModelMessage[] } = {
beforeLen: this.turnBeforeLen,
afterLen: this.messages.length,
beforeFiles: new Map(this.turnBeforeFiles),
@@ -1266,12 +1349,18 @@ export class Session {
break;
case 'tool-call':
if (part.toolName === 'todo_write') this.todoWrittenThisTurn = true;
this.callInputs.set(part.toolCallId, {
toolName: part.toolName,
input: callKey(part.toolName, part.input),
});
yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input };
break;
case 'tool-result':
this.recordTrace(part.toolCallId, summarizeToolResult(part.output));
yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output };
break;
case 'tool-error':
this.recordTrace(part.toolCallId, `error: ${part.error instanceof Error ? part.error.message : String(part.error)}`);
yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error };
break;
case 'tool-approval-request': {
@@ -1348,10 +1437,13 @@ export class Session {
this.opts.onChange?.(this.messages);
// Lossless compaction: summarize what the wire pruned so future turns keep it.
// The note leads the surviving history: it stands in for the dropped span,
// so the model sees it first, then the history it actually kept.
if (compactionSpan && compactionSpan.length > 0) {
const retained = await summarizeDiscarded(compactionSpan, this.model);
if (retained) {
this.messages.push({ role: 'user', content: `Note (retained from compacted history):\n${retained}` });
const note = prunedSpanMessage(retained, compactionSpan);
if (note) {
this.messages.unshift(note);
this.opts.onChange?.(this.messages);
}
compactionSpan = undefined;
+14 -4
View File
@@ -100,18 +100,28 @@ export function renderSkills(skills: Skill[]): string {
].join('\n');
}
export function createSkillTool(skills: Skill[]) {
const names = skills.map((s) => s.name);
/**
* The `skill` tool, reading the list live so a mid-session install shows up on the
* next turn without a restart.
*
* The catalogue in the system prompt and the names in this tool's description both
* come from the same list on each read, so replacing the list at a turn boundary
* makes both stale bytes atomic: a skill the model can call it can see already.
*/
export function createSkillTool(getSkills: () => Skill[]) {
const names = () => getSkills().map((s) => s.name).join(', ');
return tool({
description:
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`,
`what you are about to do, then follow what it says. Available: ${names() || 'none'}.`,
inputSchema: z.object({
name: z.string().describe('Skill name from the list in your instructions'),
}),
execute: async ({ name }) => {
const skills = getSkills();
const skill = skills.find((s) => s.name === name.trim().toLowerCase());
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`);
if (!skill)
throw new Error(`No skill named "${name}". Available: ${skills.map((s) => s.name).join(', ') || 'none'}`);
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
},
});
+262 -12
View File
@@ -1,3 +1,6 @@
import { createHash } from 'node:crypto';
import { join, relative, resolve, sep } from 'node:path';
/**
* Per-turn file snapshots for /undo and /redo.
*
@@ -10,16 +13,263 @@
* history stack (cap 100) and owns undo/redo.
*/
export type FileState = { existed: boolean; content: string | null };
/** One file as it was before the turn that changed it. `before === undefined` means it did not exist. */
export type PreImage = {
/** Workspace-relative, using forward slashes, so a restore is portable across platforms. */
path: string;
before: string | undefined;
};
export type TurnSnapshot = {
/** Messages length before the turn's user message was pushed. */
/** Monotonic turn number, so the UI can name what is being undone. */
turn: number;
at: string;
/** The user prompt that opened the turn, for a menu that lists them. */
prompt: string;
/** Files this turn changed, with their content from before it started. */
files: PreImage[];
/** The conversation length when the turn began, so undo can trim it back. */
messageCount: number;
};
/** Claude Code keeps 100; beyond that the memory is worth more than the recall. */
export const MAX_SNAPSHOTS = 100;
/** A single file larger than this is not snapshotted; a 40 MB binary is not an edit. */
const MAX_FILE_BYTES = 2 * 1024 * 1024;
/**
* The workspace-relative, slash-normalised form of a path, or undefined if it is
* outside the workspace.
*
* Outside is refused rather than clamped: a path that escapes the workspace is not
* something this repository can undo, and recording it would imply an undo that
* cannot happen. Paths are normalised to forward slashes because a snapshot written
* on Windows may be read on a machine where a backslash is a filename character.
*/
export function relPath(cwd: string, abs: string): string | undefined {
const root = resolve(cwd);
const target = resolve(abs);
const rel = relative(root, target);
if (rel === '' || rel.startsWith('..') || rel.includes(`..${sep}`)) return undefined;
return rel.split(sep).join('/');
}
/**
* The absolute path a tool call will write to, when there is exactly one.
*
* Deliberately a small, explicit map rather than a guess. `multi_edit` and the line
* editors each take a single `path`; `move_file` takes `from` and `to` and both are
* recorded; `delete_file` takes a `path`. `apply_patch` carries its paths inside the
* patch text, and `bash` carries none — both are reported as uncovered rather than
* silently not snapshotted.
*/
export function touchedPaths(toolName: string, input: unknown): { paths: string[]; covered: boolean } {
const o = (input ?? {}) as Record<string, unknown>;
const one = (key: string) => (typeof o[key] === 'string' ? [o[key] as string] : []);
switch (toolName) {
case 'write_file':
case 'edit_file':
case 'multi_edit':
case 'delete_file':
case 'insert_lines':
case 'delete_lines':
case 'replace_lines':
case 'append_file':
case 'prepend_file':
return { paths: one('path'), covered: true };
case 'move_file':
return { paths: [...one('from'), ...one('to')], covered: true };
case 'apply_patch': {
const patch = typeof o['patch'] === 'string' ? (o['patch'] as string) : '';
const paths = [...patch.matchAll(/^\*\*\* (?:Add|Update|Delete) File: (.+)$/gm)].map((m) => m[1]!.trim());
const moves = [...patch.matchAll(/^\*\*\* Move to: (.+)$/gm)].map((m) => m[1]!.trim());
return { paths: [...paths, ...moves], covered: true };
}
case 'bash':
// Arbitrary code: an untouched-looking `node -e` can rewrite the tree.
return { paths: [], covered: false };
default:
return { paths: [], covered: true };
}
}
/**
* Records pre-images for the files a turn changes, and restores them on undo.
*
* One instance per session. Holds at most MAX_SNAPSHOTS turns; the oldest falls off
* the front, because the turn someone wants back is almost always the last one.
*/
export class Snapshots {
private readonly turns: TurnSnapshot[] = [];
private current: { turn: number; at: string; prompt: string; files: Map<string, string | undefined>; messageCount: number } | null =
null;
private next = 1;
private readonly cwd: string;
constructor(cwd: string = process.cwd()) {
this.cwd = cwd;
}
/** Opens a turn. Called once per user prompt, before the model runs. */
begin(prompt: string, messageCount: number): void {
this.current = { turn: this.next++, at: new Date().toISOString(), prompt, files: new Map(), messageCount };
}
/**
* Records a file's content before a tool changes it, on the first write of the turn.
*
* Idempotent per path per turn: the second `write_file` to the same path in one turn
* must not replace the pre-image with the intermediate content the first write left,
* because undo restores the turn's starting state, not the midpoint.
*
* Read failures are swallowed. A snapshot is a convenience; a tool call that fails
* because the snapshot layer could not read an unrelated path would be worse than no
* undo at all.
*/
async capture(absPath: string): Promise<void> {
if (!this.current) return;
const rel = relPath(this.cwd, absPath);
if (rel === undefined) return;
if (this.current.files.has(rel)) return;
try {
const file = Bun.file(absPath);
const exists = await file.exists();
if (!exists) {
this.current.files.set(rel, undefined);
return;
}
if (file.size > MAX_FILE_BYTES) return;
this.current.files.set(rel, await file.text());
} catch {
return;
}
}
/** Records any path a tool call is about to touch. Returns whether the tool is covered at all. */
async captureFor(toolName: string, input: unknown): Promise<{ covered: boolean; paths: string[] }> {
const { paths, covered } = touchedPaths(toolName, input);
for (const p of paths) await this.capture(resolve(this.cwd, p));
return { paths, covered };
}
/**
* Closes the turn, keeping it only if it changed something.
*
* A turn that read and answered without writing is not worth a slot, and keeping it
* would make `/undo` step past a turn that has nothing to undo — which reads as the
* command being broken.
*/
commit(): TurnSnapshot | undefined {
const cur = this.current;
this.current = null;
if (!cur || cur.files.size === 0) return undefined;
const snap: TurnSnapshot = {
turn: cur.turn,
at: cur.at,
prompt: cur.prompt,
files: [...cur.files].map(([path, before]) => ({ path, before })),
messageCount: cur.messageCount,
};
this.turns.push(snap);
while (this.turns.length > MAX_SNAPSHOTS) this.turns.shift();
return snap;
}
/** Discards the open turn without recording it, for an aborted or failed turn. */
discard(): void {
this.current = null;
}
/** The turns that can be undone, newest first. */
list(): readonly TurnSnapshot[] {
return [...this.turns].reverse();
}
/** Whether there is an open turn collecting pre-images right now. */
get open(): boolean {
return this.current !== null;
}
/**
* Removes the newest turn and returns what it holds, without restoring.
*
* Separated from restoring so the caller can decide *what* to bring back —
* files, conversation, or both — which is the split Claude Code's rewind menu
* exposes and the reason one control surface is worth more than three commands.
*/
pop(): TurnSnapshot | undefined {
return this.turns.pop();
}
/** Puts a turn back, for a `/redo` that follows an `/undo`. */
push(snap: TurnSnapshot): void {
this.turns.push(snap);
}
get size(): number {
return this.turns.length;
}
cwdOf(): string {
return this.cwd;
}
clear(): void {
this.turns.length = 0;
this.current = null;
}
}
/** Writes a pre-image back to disk, recreating a deleted file or removing one that was created. */
export async function restore(snap: TurnSnapshot, cwd = process.cwd()): Promise<{ restored: string[]; removed: string[] }> {
const restored: string[] = [];
const removed: string[] = [];
for (const file of snap.files) {
const abs = join(cwd, file.path);
if (file.before === undefined) {
// The file did not exist before the turn, so undoing its creation is removing it.
const f = Bun.file(abs);
if (await f.exists()) {
await f.delete();
removed.push(file.path);
}
continue;
}
await Bun.write(abs, file.before);
restored.push(file.path);
}
return { restored, removed };
}
/** A short, stable label for a snapshot, for a menu that lists several. */
export function labelOf(snap: TurnSnapshot): string {
const first = snap.prompt.trim().split('\n')[0] ?? '';
const clipped = first.length > 50 ? `${first.slice(0, 50)}...` : first || '(no prompt)';
return `turn ${snap.turn}: ${clipped}`;
}
/** A content hash, used to tell whether a file still matches what the snapshot holds. */
export function hashOf(text: string): string {
return createHash('sha256').update(text).digest('hex').slice(0, 12);
}
/* -------------------------------------------------------------------------- */
/* Fork's SnapshotStack — kept for the fork's undo/redo tests and API surface. */
/* -------------------------------------------------------------------------- */
export type FileState = { existed: boolean; content: string | null };
/** The fork's undo stack entry shape: before/after message lengths + file maps. */
type StackEntry = {
beforeLen: number;
/** Messages length after the turn completed (including tool results). */
afterLen: number;
/** File state before the turn, keyed by absolute path. Only files the turn touched. */
beforeFiles: Map<string, FileState>;
/** File state after the turn, for redo. */
afterFiles: Map<string, FileState>;
};
@@ -37,10 +287,10 @@ export async function recordBeforeWrite(abs: string): Promise<void> {
}
export class SnapshotStack {
private readonly history: TurnSnapshot[] = [];
private readonly future: TurnSnapshot[] = [];
private readonly history: StackEntry[] = [];
private readonly future: StackEntry[] = [];
push(entry: TurnSnapshot): void {
push(entry: StackEntry): void {
this.history.push(entry);
if (this.history.length > MAX_HISTORY) this.history.shift();
this.future.length = 0;
@@ -50,17 +300,17 @@ export class SnapshotStack {
canRedo(): boolean { return this.future.length > 0; }
/** The most recent snapshot without consuming it — lets a caller diff the last turn. */
peek(): TurnSnapshot | undefined {
peek(): StackEntry | undefined {
return this.history.at(-1);
}
popForUndo(): TurnSnapshot | undefined {
popForUndo(): StackEntry | undefined {
const e = this.history.pop();
if (e) this.future.push(e);
return e;
}
popForRedo(): TurnSnapshot | undefined {
popForRedo(): StackEntry | undefined {
const e = this.future.pop();
if (e) this.history.push(e);
return e;
@@ -74,4 +324,4 @@ export class SnapshotStack {
depth(): { undo: number; redo: number } {
return { undo: this.history.length, redo: this.future.length };
}
}
}
+80
View File
@@ -0,0 +1,80 @@
import { tool } from 'ai';
import { z } from 'zod';
/**
* A visible escape hatch for the loop a coding agent dies in.
*
* The repeat guard stops an *identical* call after three tries, but the deeper loop
* is the model making *different* calls that all amount to the same stalled attempt —
* re-reading the same file expecting a different answer, retrying a failing command
* with a tweaked flag, re-sending a prompt it has already asked. No equal-input
* detector fires on any of that, so the model burns the step budget on motions that
* never move.
*
* `step_back` exists so the model has a *named* way out instead of only a guard it
* cannot see. The session records each completed tool call (input + outcome) into a
* loop trace; the tool returns a scripted reflection prompt built from that trace,
* so the model is told in concrete terms that it is going in circles and is steered
* to change direction.
*/
export type LoopEntry = {
step: number;
toolName: string;
input: string;
result: string;
at: string;
};
function recentTrace(entries: readonly LoopEntry[], window = 8): LoopEntry[] {
return entries.slice(-window);
}
function renderTrace(entries: readonly LoopEntry[], maxLines = 12): string {
return entries.slice(-maxLines).map((e) => `step ${e.step}: ${e.toolName} ${e.input} -> ${e.result}`.slice(0, 200)).join('\n');
}
/**
* Builds a `step_back` tool bound to a session's loop trace.
*
* The tool is meant to be called when the model is not making progress — a trap it
* cannot always see while it is inside it. The returned reflection names the recent
* steps so the model can tell, from its own calls, that it is going in circles, and
* gives it the one thing a stuck agent is usually missing: permission to stop,
* say what it learned, and change direction rather than try harder.
*/
export function createStepBackTool(opts: { trace: () => readonly LoopEntry[] }) {
return tool({
description:
'Use when you are stuck: the same file is not changing, a command keeps failing, or you have done several steps with no visible progress. ' +
'Records your recent steps and returns a reflection prompt to help you change direction instead of repeating the attempt.',
inputSchema: z.object({
note: z.string().optional().describe('A sentence in your own words about what you were trying to do.'),
}),
execute: async ({ note }) => {
const trace = recentTrace(opts.trace());
if (trace.length === 0) {
return (
'No recent steps to reflect on. This is an early call of step_back — it is only useful when you have ' +
'attempted something several times. Describe what you are stuck on in `note`.'
);
}
return [
'You have run threadbare over the last steps and are stuck. Here is what you actually did:',
'```',
renderTrace(trace),
'```',
'',
'Before your next tool call, answer these three questions in your reasoning:',
'1. What exactly is wrong — the input, the tool, or the expectation?',
'2. What have you tried, and why did each fail?',
'3. What is ONE different thing you can do that is not "try the same thing a little harder"?',
'',
'Then take that different action. If the failure is a command, read the actual error and fix its cause — ',
'do not rerun the command. If a file is not what you expect, suspect your assumption about it and re-read it fresh.',
note?.trim() ? `\nYour note: ${note.trim()}` : '',
].join('\n');
},
});
}
+108 -1
View File
@@ -290,7 +290,7 @@ export function createTaskTool(opts: {
'the user exactly as yours are. Use it for a self-contained task whose intermediate steps you do not ' +
'need to see; keep work you must supervise step by step in your own turn.'
: '') +
'\nDo not delegate something you can answer with a single grep. For independent pieces of work, pass `tasks` to run them in parallel instead of calling task several times sequentially.',
'\nDo not delegate something you can answer with a single grep. For independent pieces of work, pass `tasks` to run them in parallel instead of calling task several times sequentially.',
inputSchema: z.union([singleSchema, batchSchema]),
execute: async (input, { abortSignal }) => {
const asBatch = input as { tasks?: TaskSpec[]; description?: string; prompt?: string; kind?: SubagentKind };
@@ -337,4 +337,111 @@ export function createTaskTool(opts: {
});
}
/** Flavour of one planned investigation; `kind` defaults to explore. */
type Plan = { description: string; prompt: string; kind?: string };
/**
* Runs one subagent and settles its report, usage, and panel events.
*
* Extracted so `tasks` can fan several out in parallel: each full run is
* independent — its own id, own stream, own spend — and they overlap simply by
* awaiting them together.
*/
async function runSubagent(
opts: {
model: LanguageModel;
subagentModel?: LanguageModel;
subagentModelId?: string;
cwd?: string;
maxSteps?: number;
report?: SubagentReporter;
approve?: SubagentApproval;
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
abortSignal?: AbortSignal;
},
plan: Plan,
): Promise<{ named: string; report: string }> {
const flavour: SubagentKind = (plan.kind as SubagentKind | undefined) ?? 'explore';
if (flavour === 'worker' && !opts.approve) {
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
}
const id = `sub${++counter}`;
const report = opts.report;
report?.({ type: 'start', id, kind: flavour, description: plan.description });
let steps = 0;
let text = '';
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
try {
// `explore` is search, not reasoning, so it runs on the cheaper model when
// one is configured. `review` and `worker` keep the parent's: they judge
// and they change, both of which want the full model.
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
const result = streamText({
model,
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
messages: [{ role: 'user', content: plan.prompt }],
tools: TOOLS[flavour],
stopWhen: isStepCount(opts.maxSteps ?? 20),
...(opts.approve
? {
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
const approved = await opts.approve!(toolCall);
return approved
? undefined
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
},
}
: {}),
...(opts.abortSignal ? { abortSignal: opts.abortSignal } : {}),
});
const sink = () => {};
void result.responseMessages.then(undefined, sink);
void result.usage.then(undefined, sink);
void result.steps.then(undefined, sink);
void result.finalStep.then(undefined, sink);
void result.finishReason.then(undefined, sink);
for await (const part of result.stream) {
if (part.type === 'tool-call') {
steps++;
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
} else if (part.type === 'tool-result') {
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
} else if (part.type === 'tool-error') {
const message = part.error instanceof Error ? part.error.message : String(part.error);
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
} else if (part.type === 'text-delta') {
text += part.text;
} else if (part.type === 'error') {
// A provider failure arrives as a stream part, not a throw, so it has to
// be rethrown here or the subagent silently returns nothing.
const message = part.error instanceof Error ? part.error.message : String(part.error);
throw part.error instanceof Error ? part.error : new Error(message);
}
}
try {
const usage = await result.usage;
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
} catch {
// A run that errored before producing usage has nothing to account for.
}
} catch (e) {
const message = e instanceof Error ? e.message : String(e);
report?.({ type: 'error', id, message });
throw e;
}
const trimmed = text.trim();
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
// Settled after the stream closes; a failed run reports nothing rather than
// a half count. The parent prices these against the subagent's own model id.
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
return { named: plan.description, report: trimmed || 'Subagent returned no findings.' };
}
export const TASK_TOOL_NAME = 'task';
+27
View File
@@ -0,0 +1,27 @@
import type { Tool } from 'ai';
/**
* Marks a tool as mutating where it is defined, rather than in a list beside it.
*
* The bug this prevents is the worst one this codebase can have: a new tool that
* writes to the workspace, added to `tools` but forgotten in a hand-maintained
* `MUTATING_TOOLS`, is a write the permission layer does not treat as a write. It
* is silent, it passes every test that does not think to check the new name, and
* it surfaces as a user discovering an edit they never approved.
*
* The mark is a property on the tool object, so `MUTATING_TOOLS` can be derived by
* filtering the registry instead of being typed out. A tool that is not marked is
* asserted non-mutating by `tools.test.ts`, which means the decision is made once,
* at the definition, and cannot drift.
*/
export const MUTATING = '__mutating' as const;
/** A tool that can change the workspace or run arbitrary code. */
export function mutating<T extends Tool>(t: T): T {
return Object.assign(t, { [MUTATING]: true as const });
}
/** Whether a tool was declared mutating at its definition site. */
export function isMutating(t: unknown): boolean {
return typeof t === 'object' && t !== null && (t as Record<string, unknown>)[MUTATING] === true;
}
+88 -22
View File
@@ -291,16 +291,37 @@ export const writeFileTool = withMeta({ set: 'core', mutating: true }, tool({
return `Wrote ${content.length} chars to ${path}`;
},
}));
// Some models (Claude-style tool docs, DeepSeek/GLM) emit snake_case edit params
// (old_string/new_string/replace_all) despite the camelCase schema. Normalize at the
// boundary instead of failing the whole call on a naming convention.
const SNAKE_EDIT_ARGS: ReadonlyArray<readonly [string, string]> = [
['old_string', 'oldString'],
['new_string', 'newString'],
['replace_all', 'replaceAll'],
];
export function normalizeEditArgs(input: unknown): unknown {
if (typeof input !== 'object' || input === null) return input;
const obj: Record<string, unknown> = { ...(input as Record<string, unknown>) };
for (const [snake, camel] of SNAKE_EDIT_ARGS) {
if (obj[snake] !== undefined && obj[camel] === undefined) obj[camel] = obj[snake];
}
if (Array.isArray(obj.edits)) obj.edits = obj.edits.map(normalizeEditArgs);
return obj;
}
export const editFileTool = withMeta({ set: 'core', mutating: true }, tool({
description:
'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.',
inputSchema: z.object({
path: z.string(),
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
}),
inputSchema: z.preprocess(
normalizeEditArgs,
z.object({
path: z.string(),
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
}),
),
execute: async ({ path, oldString, newString, replaceAll = false }) => {
if (oldString === newString) throw new Error('oldString and newString are identical');
const abs = jail(path);
@@ -327,19 +348,22 @@ export const multiEditTool = withMeta({ set: 'edit-plus', mutating: true }, tool
'All or nothing: if any oldString fails to match, or matches more than once without replaceAll, nothing is ' +
'written. Prefer this over repeated edit_file calls on the same file — one approval, one write, no risk of ' +
'leaving the file half-changed.',
inputSchema: z.object({
path: z.string(),
edits: z
.array(
z.object({
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional(),
}),
)
.min(1)
.describe('Edits in the order they should be applied'),
}),
inputSchema: z.preprocess(
normalizeEditArgs,
z.object({
path: z.string(),
edits: z
.array(
z.object({
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
newString: z.string().describe('Replacement text'),
replaceAll: z.boolean().optional(),
}),
)
.min(1)
.describe('Edits in the order they should be applied'),
}),
),
execute: async ({ path, edits }) => {
const abs = jail(path);
await recordBeforeWrite(abs);
@@ -581,7 +605,13 @@ async function pump(
return all;
}
type Running = { command: string; proc: Bun.Subprocess; interrupted: boolean; killed?: Promise<unknown> };
type Running = {
command: string;
proc: Bun.Subprocess;
interrupted: boolean;
timedOut?: boolean;
killed?: Promise<unknown>;
};
const running = new Map<string, Running>();
@@ -823,6 +853,12 @@ function killTree(proc: Bun.Subprocess): Promise<unknown> {
export function interruptBash(): string[] {
const killed: string[] = [];
for (const entry of running.values()) {
// A second ctrl-c while the first killTree is still settling must not re-announce
// the same command: the notice is the only proof the keypress did anything.
if (entry.interrupted) {
killed.push(entry.command);
continue;
}
entry.interrupted = true;
entry.killed = killTree(entry.proc);
killed.push(entry.command);
@@ -854,13 +890,32 @@ export const bashTool = withMeta({ set: 'core', mutating: true }, tool({
cwd: process.cwd(),
stdout: 'pipe',
stderr: 'pipe',
timeout,
...(abortSignal ? { signal: abortSignal } : {}),
});
const entry: Running = { command, proc, interrupted: false };
running.set(toolCallId, entry);
// Bun's spawn `signal` option is not used either: it kills only the shell, so an
// esc-abort orphaned the grandchild on the same still-open pipes as the timeout
// did. The abort must go through killTree, exactly like ctrl-c does.
const onAbort = () => {
if (entry.interrupted) return;
entry.interrupted = true;
entry.killed = killTree(proc);
};
abortSignal?.addEventListener('abort', onAbort);
// The turn may already be aborted by the time this tool starts; a past event
// never re-fires, so check once here or the command runs unkillable by esc.
if (abortSignal?.aborted) onAbort();
// Bun's own `timeout` spawn option is not used: it kills only the shell, and the
// grandchild holding the output pipes keeps `pump` reading forever, so the tool
// never returns. Same failure killTree exists for, just triggered by the clock.
const timer = setTimeout(() => {
entry.timedOut = true;
entry.killed = killTree(proc);
}, timeout);
try {
// Drained concurrently: a command that fills one pipe while we block on the
// other would deadlock, and buffering both hides progress for minutes.
@@ -876,6 +931,15 @@ export const bashTool = withMeta({ set: 'core', mutating: true }, tool({
// Thrown rather than returned: the model must not read a killed command as
// a command that ran and failed on its own terms.
if (entry.timedOut) {
throw new Error(
cap(
`The command exceeded its ${timeout}ms timeout and was killed. It did not finish, so its effects are unknown.\n${
body || '(no output before it was killed)'
}`,
),
);
}
if (entry.interrupted) {
throw new Error(
cap(
@@ -896,6 +960,8 @@ export const bashTool = withMeta({ set: 'core', mutating: true }, tool({
.join('\n\n'),
);
} finally {
clearTimeout(timer);
abortSignal?.removeEventListener('abort', onAbort);
// Awaited so the process really is gone before the tool returns. On Windows a
// surviving grandchild holds the cwd open, which breaks the very next command.
await entry.killed;
+61 -35
View File
@@ -38,7 +38,7 @@ import { CommandMenu, InstallConfirm, Picker } from './Pickers';
import { contextPanel, costPanel, todosPanel, toolsPanel, changesPanel, diffPanel, diffReviewPanel, workflowPanel } from './panel-bodies';
import { PromptInput } from './PromptInput';
import { accent, glyph } from './theme';
import { nextKey, resultSummary, toolDetail, withResult, type Line, type NewLine } from './transcript';
import { historyFromMessages, nextKey, resultSummary, toolDetail, withResult, type Line, type NewLine } from './transcript';
export { createApprovalBridge, createNoticeBus, createSubagentBus, applySubagentEvent };
export type { ApprovalBridge, NoticeBus, SubagentBus };
@@ -144,7 +144,11 @@ export function App({
stdout.off('resize', onResize);
};
}, [stdout]);
const [history, setHistory] = useState<Line[]>([]);
// Seeded from the resumed history: a session loaded with -r/-c should show its
// saved conversation rather than a blank transcript. Only read once, at mount.
const [history, setHistory] = useState<Line[]>(() =>
session.messages.length === 0 ? [] : historyFromMessages(session.messages as { role?: string; content?: unknown }[]),
);
const [draft, setDraft] = useState('');
const [live, setLive] = useState('');
const [busy, setBusy] = useState(false);
@@ -196,17 +200,33 @@ export function App({
// The walk costs a full ignore-aware traversal, so it happens on the first `@`
// rather than at startup, and re-runs when files change (listPaths is cached
// in hook, but App keeps seq so a stale `paths` is dropped).
// in hook, but App keeps seq so a stale `paths` is dropped). A slow cooldown
// also re-walks so a file created after that first `@` shows up within a short
// window instead of staying hidden all session.
const pathsRef = useRef(paths);
useEffect(() => {
if (token === undefined || paths !== undefined) return;
let live = true;
void hooks.listPaths().then((all) => {
if (live) setPaths(all);
});
return () => {
live = false;
};
}, [hooks, paths, token]);
pathsRef.current = paths;
}, [paths]);
const didLoadRef = useRef(false);
useEffect(() => {
if (token === undefined) return;
if (!didLoadRef.current) {
didLoadRef.current = true;
let live = true;
void hooks.listPaths().then((all) => {
if (live) setPaths(all);
});
const refresh = setInterval(async () => {
if (!live) return;
const all = await hooks.listPaths();
if (live && JSON.stringify(all) !== JSON.stringify(pathsRef.current)) setPaths(all);
}, 10_000);
return () => {
live = false;
clearInterval(refresh);
};
}
}, [hooks, token]);
// A file mutated this turn: drop the cached walk so next `@` re-walks.
const seq = hooks.fileChangeSeq();
@@ -712,34 +732,18 @@ export function App({
push({ kind: 'user', text: chosen.trim() });
try {
const msg = await hooks.resumeSession(action.id);
setHistory([]);
// Reflect the freshly loaded history: session.messages now holds the
// restored wire messages, and the transcript must show them again.
setHistory(
session.messages.length === 0
? []
: historyFromMessages(session.messages as { role?: string; content?: unknown }[]),
);
push({ kind: 'info', text: msg });
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
return;
case 'undo': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
push({ kind: 'info', text: await session.undo() });
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
case 'redo': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
push({ kind: 'info', text: await session.redo() });
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
case 'changes': {
push({ kind: 'user', text: chosen.trim() });
setPanel(changesPanel(session));
@@ -843,6 +847,28 @@ export function App({
setModelPicker(models);
return;
}
case 'undo': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
push({ kind: 'info', text: await session.undo() });
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
case 'redo': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
push({ kind: 'info', text: await session.redo() });
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
case 'compact': {
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
+2 -1
View File
@@ -1,6 +1,7 @@
import { Box, Text, useInput } from 'ink';
import React from 'react';
import type { ApprovalDecision, ApprovalRequest } from '../session';
import { normalizeEditArgs } from '../tools';
import { Diff } from './Diff';
import { accent, glyph } from './theme';
import { toolDetail } from './transcript';
@@ -42,7 +43,7 @@ export function createApprovalBridge(): ApprovalBridge {
* consistent and far more readable than a JSON dump of the input.
*/
function ApprovalDetail({ name, input }: { name: string; input: unknown }) {
const o = (input ?? {}) as Record<string, unknown>;
const o = (normalizeEditArgs(input) ?? {}) as Record<string, unknown>;
if (name === 'write_file') {
const content = String(o['content'] ?? '');
+3 -7
View File
@@ -33,10 +33,6 @@ type KeyLike = {
end?: boolean;
};
const INVERSE_ON = '\u001B[7m';
const INVERSE_OFF = '\u001B[27m';
const invert = (s: string) => `${INVERSE_ON}${s}${INVERSE_OFF}`;
/**
* Text input with a real cursor and shell-style history recall.
*
@@ -151,10 +147,10 @@ export function PromptInput({
);
if (value.length === 0) {
if (!placeholder) return <Text>{focus ? invert(' ') : ' '}</Text>;
if (!placeholder) return <Text inverse={focus}>{' '}</Text>;
return (
<Text dimColor>
{focus ? invert(placeholder.slice(0, 1)) : placeholder.slice(0, 1)}
<Text inverse={focus}>{placeholder.slice(0, 1)}</Text>
{placeholder.slice(1)}
</Text>
);
@@ -166,7 +162,7 @@ export function PromptInput({
return (
<Text>
{shown.slice(0, cursor)}
{invert(shown.slice(cursor, cursor + 1) || ' ')}
<Text inverse>{shown.slice(cursor, cursor + 1) || ' '}</Text>
{shown.slice(cursor + 1)}
</Text>
);
+99 -1
View File
@@ -1,4 +1,5 @@
import { TODO_MARK } from '../notebook';
import { normalizeEditArgs } from '../tools';
export type Line =
| { key: string; kind: 'user'; text: string }
@@ -38,7 +39,8 @@ export function preview(input: unknown): string {
* the transcript, beside the spinner while a call is in flight, and in the approval
* prompt for any tool without a diff of its own.
*/
export function toolDetail(name: string, input: unknown): string[] {
export function toolDetail(name: string, rawInput: unknown): string[] {
const input = normalizeEditArgs(rawInput);
if (input === null || typeof input !== 'object') return [];
const o = input as Record<string, unknown>;
const str = (k: string) => (typeof o[k] === 'string' ? (o[k] as string) : undefined);
@@ -195,6 +197,102 @@ export function withResult(lines: Line[], name: string, result: string, ok: bool
return lines;
}
/**
* The tool output that was stored on a `tool-result`. The SDK json-wraps a
* string return as `{ type: 'text', value }`, so a restored message needs the
* same unwrap the live stream already produced at save time.
*/
function toolResultText(output: unknown): string {
if (typeof output === 'string') return output;
if (
output !== null &&
typeof output === 'object' &&
'value' in output &&
typeof (output as { value: unknown }).value === 'string'
) {
return (output as { value: string }).value;
}
try {
return JSON.stringify(output);
} catch {
return String(output);
}
}
type StoredPart = {
type?: string;
toolName?: string;
input?: unknown;
output?: unknown;
text?: unknown;
};
/**
* The transcript lines a saved `ModelMessage[]` becomes, so a resumed session
* renders its history instead of starting blank.
*
* Mirrors how the live loop paints: user strings as user lines, assistant text
* as an assistant line, assistant `tool-call` parts as tool lines, and each
* `role: 'tool'` result attached to the newest unanswered call of that name just
* like `withResult` does. A result with no matching call (a pruned lead-in) is
* dropped rather than left floating.
*/
export function historyFromMessages(messages: readonly { role?: string; content?: unknown }[]): Line[] {
const lines: Line[] = [];
const textOf = (content: unknown): string =>
typeof content === 'string'
? content
: Array.isArray(content)
? (content as StoredPart[]).filter((p) => p.type === 'text' && typeof p.text === 'string').map((p) => p.text as string).join('')
: '';
const toolPartsOf = (content: unknown): StoredPart[] =>
Array.isArray(content) ? (content as StoredPart[]).filter((p) => p.type === 'tool-call') : [];
for (const m of messages) {
switch (m.role) {
case 'user': {
const text = textOf(m.content).trim();
if (text) lines.push({ key: nextKey(), kind: 'user', text });
break;
}
case 'assistant': {
const text = textOf(m.content).trim();
if (text) lines.push({ key: nextKey(), kind: 'assistant', text });
for (const p of toolPartsOf(m.content)) {
const name = p.toolName ?? '';
if (!name) continue;
lines.push({ key: nextKey(), kind: 'tool', name, detail: toolDetail(name, p.input), ok: true });
}
break;
}
case 'tool': {
const parts = Array.isArray(m.content) ? (m.content as StoredPart[]) : [];
for (const p of parts) {
if (p.type !== 'tool-result' && p.type !== 'tool-error') continue;
const name = p.toolName ?? '';
const result =
p.type === 'tool-error'
? toolResultText(p.output) || 'tool failed'
: resultSummary(name, toolResultText(p.output));
for (let i = lines.length - 1; i >= 0; i--) {
const line = lines[i]!;
if (line.kind !== 'tool' || line.name !== name || line.result !== undefined) continue;
lines[i] = { ...line, result, ok: p.type !== 'tool-error' };
break;
}
}
break;
}
default:
break;
}
}
return lines;
}
/** A task list as markdown, for the `/todos` panel. */
export const todoLines = (todos: readonly { status: keyof typeof TODO_MARK; content: string; note?: string }[]) =>
todos.length > 0
+1 -1
View File
@@ -5,7 +5,7 @@
* fails inside the shipped binary. A constant is compiled in and always correct.
* `scripts/release.ts` checks it against the release tag so the two cannot drift.
*/
export const VERSION = '1.0.0';
export const VERSION = '1.0.1';
/** What `--version` prints: enough to identify a build from a bug report. */
export function versionLine(): string {