Merge remote-tracking branch 'refs/remotes/upstream/main'
# Conflicts: # ROADMAP.md # TODO.md # src/cli.tsx # src/config.ts # src/mcp.ts # src/permission.ts # src/prompt.ts # src/session.ts # src/snapshot.ts # src/subagent.ts # src/tools-extra.ts # src/tools.ts # src/ui/App.tsx # test/mcp.test.ts # test/prune.test.ts # test/session.test.ts # test/tools.test.ts
This commit is contained in:
+3
-2
@@ -168,7 +168,8 @@ if (resumeArg) {
|
||||
}
|
||||
}
|
||||
|
||||
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
|
||||
const mcp =
|
||||
has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers, cfg.mcpMode ?? 'lazy');
|
||||
const instructions = has('--no-instructions') ? [] : await loadInstructions();
|
||||
const skills = has('--no-skills') ? [] : await loadSkills();
|
||||
const customCommands = await loadCustomCommands();
|
||||
@@ -765,7 +766,7 @@ const facts: HeaderFact[] = [
|
||||
memory && memory.all().length > 0
|
||||
? { label: 'memory', value: `${memory.all().length} notes about this project` }
|
||||
: undefined,
|
||||
mcp && Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length} tools${(mcp.tools as Record<string, unknown>)['mcp_list'] ? ' (mcp_list/mcp_inspect/mcp_call)' : ''}` } : undefined,
|
||||
mcp && Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).filter((k) => k !== '__mcpServerNames').length} tools${(mcp.tools as Record<string, unknown>)['mcp_list'] ? ' (mcp_list/mcp_inspect/mcp_call)' : ''}` } : undefined,
|
||||
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
|
||||
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
|
||||
: undefined,
|
||||
|
||||
+24
-8
@@ -6,6 +6,8 @@ export type CommandAction =
|
||||
| { type: 'exit' }
|
||||
| { type: 'clear' }
|
||||
| { type: 'compact' }
|
||||
| { type: 'undo'; what: 'both' | 'files' | 'conversation' }
|
||||
| { type: 'redo'; what: 'both' | 'files' | 'conversation' }
|
||||
| { type: 'tools' }
|
||||
| { type: 'cost' }
|
||||
| { type: 'sessions' }
|
||||
@@ -26,8 +28,6 @@ export type CommandAction =
|
||||
| { type: 'info'; text: string }
|
||||
| { type: 'model'; model: string }
|
||||
| { type: 'resume'; id: string }
|
||||
| { type: 'undo' }
|
||||
| { type: 'redo' }
|
||||
| { type: 'changes' }
|
||||
| { type: 'diff'; action: 'raw' | 'review' }
|
||||
| { type: 'bash'; action: 'list' | 'stop' | 'stop-all'; arg?: string }
|
||||
@@ -66,12 +66,12 @@ export const COMMANDS: CommandSpec[] = [
|
||||
{ name: 'memory', summary: 'compact the project memory with the model' },
|
||||
{ name: 'tools', summary: 'list available tools' },
|
||||
{ name: 'compact', summary: 'replace history with a model-written summary' },
|
||||
{ name: 'undo', arg: '[files|conversation]', summary: 'walk the last turn back: files, conversation, or both' },
|
||||
{ name: 'redo', arg: '[conversation]', summary: 'put back what /undo took' },
|
||||
{ name: 'cost', summary: 'tokens and estimated spend this session' },
|
||||
{ name: 'sessions', summary: 'list saved sessions' },
|
||||
{ name: 'resume', arg: '<id>', summary: 'load a saved session' },
|
||||
{ name: 'save', summary: 'write the session to disk now' },
|
||||
{ name: 'undo', summary: 'undo the last turn — restores files and conversation (bash effects are not snapshotted)' },
|
||||
{ name: 'redo', summary: 'redo the last undone turn' },
|
||||
{ name: 'changes', summary: 'show what the last turn changed on disk' },
|
||||
{ name: 'diff', arg: '[review]', summary: 'diff the last turn; /diff review shows per-hunk file:line blocks' },
|
||||
{ name: 'bash', arg: '[list|stop <id>|stop all]', summary: 'list or stop background commands started with bash background: true' },
|
||||
@@ -179,6 +179,22 @@ function parseMcp(arg: string): CommandAction {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* `/undo [files|conversation]` and `/redo [conversation]`.
|
||||
*
|
||||
* The default is `both` for undo, because restoring one without the other is the
|
||||
* failure the two are meant to prevent: files back without the history and the model
|
||||
* re-reads a change it no longer made. A bare `files` or `conversation` narrows it.
|
||||
* Redo defaults to the conversation, since file content after the turn was never kept.
|
||||
*/
|
||||
function parseUndoKind(arg: string, fallback: 'both' | 'conversation'): 'both' | 'files' | 'conversation' {
|
||||
const word = arg.trim().toLowerCase();
|
||||
if (word === 'files' || word === 'file') return 'files';
|
||||
if (word === 'conversation' || word === 'chat' || word === 'history') return 'conversation';
|
||||
if (word === 'both' || word === 'all' || word === '') return fallback;
|
||||
return fallback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure parser: no IO, so the TUI and headless mode share one definition.
|
||||
*
|
||||
@@ -204,6 +220,10 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
|
||||
return { type: 'clear' };
|
||||
case 'compact':
|
||||
return { type: 'compact' };
|
||||
case 'undo':
|
||||
return { type: 'undo', what: parseUndoKind(arg, 'both') };
|
||||
case 'redo':
|
||||
return { type: 'redo', what: parseUndoKind(arg, 'conversation') };
|
||||
case 'tools':
|
||||
return { type: 'tools' };
|
||||
case 'cost':
|
||||
@@ -243,10 +263,6 @@ export function parseCommand(raw: string, custom: readonly CustomCommand[] = [])
|
||||
return arg ? { type: 'model', model: arg } : { type: 'models' };
|
||||
case 'resume':
|
||||
return arg ? { type: 'resume', id: arg } : { type: 'info', text: 'usage: /resume <session-id>' };
|
||||
case 'undo':
|
||||
return { type: 'undo' };
|
||||
case 'redo':
|
||||
return { type: 'redo' };
|
||||
case 'changes':
|
||||
return { type: 'changes' };
|
||||
case 'diff': {
|
||||
|
||||
+6
-1
@@ -53,7 +53,7 @@ export type Config = {
|
||||
/** Install unsigned registry entries. Default false — signed entries are required. */
|
||||
registryAllowUnsigned?: boolean;
|
||||
mcpServers?: Record<string, McpServerConfig>;
|
||||
/** A check command (e.g. `tsc --watch`) run in the UI only, never in model context. */
|
||||
/** A check command (e.g. `tsc --watch`) run in the UI only, never in model context. */
|
||||
diagnostics?: string;
|
||||
/**
|
||||
* When a turn ends normally but the task list still has work, keep going with
|
||||
@@ -61,6 +61,11 @@ export type Config = {
|
||||
* `true` on, `false` off, or `{ "maxTurns": n }` to bound it. Default on.
|
||||
*/
|
||||
continueWhileTodos?: boolean | { maxTurns?: number };
|
||||
/**
|
||||
* How MCP tools reach the model: `lazy` registers meta-tools only (cheap until a
|
||||
* tool is called), `eager` registers every server tool up front. Omit for lazy.
|
||||
*/
|
||||
mcpMode?: 'lazy' | 'eager';
|
||||
};
|
||||
|
||||
const configPath = () => join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko', 'config.json');
|
||||
|
||||
+26
-9
@@ -7,6 +7,17 @@ export type McpServerConfig =
|
||||
| ({ command: string; args?: string[]; env?: Record<string, string>; cwd?: string } & { expose?: 'direct' | 'meta' })
|
||||
| ({ url: string; type?: 'http' | 'sse'; headers?: Record<string, string> } & { expose?: 'direct' | 'meta' });
|
||||
|
||||
/**
|
||||
* How a server's tools reach the model.
|
||||
*
|
||||
* `eager` registers every tool with its schema up front — cheap for a two-tool
|
||||
* server, a tax for one that exposes twenty. `lazy` registers only the three
|
||||
* meta-tools below and fetches a server's tools on demand via `mcp_call`, so a
|
||||
* configured server costs almost nothing in the request until a tool is actually
|
||||
* invoked.
|
||||
*/
|
||||
export type McpMode = 'eager' | 'lazy';
|
||||
|
||||
export type McpHandle = {
|
||||
/** Live clients keyed by server name — only for servers that connected. */
|
||||
clients: Map<string, MCPClient>;
|
||||
@@ -15,6 +26,8 @@ export type McpHandle = {
|
||||
/** Tools to merge into the session: direct mcp__* + 3 meta tools when any server exists. */
|
||||
tools: ToolSet;
|
||||
errors: { server: string; message: string }[];
|
||||
/** Server names, for the prompt's MCP line. Empty when the mode is eager. */
|
||||
servers: string[];
|
||||
close: () => Promise<void>;
|
||||
};
|
||||
|
||||
@@ -143,7 +156,10 @@ export function createMcpMetaTools(handle: McpHandle): ToolSet {
|
||||
* through the 3 meta-tools so their schemas cost nothing until used.
|
||||
* A server that fails to start is reported, never fatal.
|
||||
*/
|
||||
export async function connectMcp(servers: Record<string, McpServerConfig>): Promise<McpHandle> {
|
||||
export async function connectMcp(
|
||||
servers: Record<string, McpServerConfig>,
|
||||
mode: McpMode = 'lazy',
|
||||
): Promise<McpHandle> {
|
||||
const clients = new Map<string, MCPClient>();
|
||||
const tools: ToolSet = {};
|
||||
const errors: McpHandle['errors'] = [];
|
||||
@@ -161,8 +177,8 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
|
||||
...(cfg.cwd ? { cwd: cfg.cwd } : {}),
|
||||
}),
|
||||
});
|
||||
clients.set(name, client);
|
||||
if (isDirect(cfg)) {
|
||||
clients.set(name, client);
|
||||
if (mode === 'eager' || isDirect(cfg)) {
|
||||
for (const [toolName, t] of Object.entries(await client.tools())) {
|
||||
tools[`mcp__${name}__${toolName}`] = t;
|
||||
}
|
||||
@@ -173,11 +189,12 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
|
||||
}),
|
||||
);
|
||||
|
||||
const handle: McpHandle = {
|
||||
const handle: McpHandle = {
|
||||
clients,
|
||||
configs: servers,
|
||||
tools,
|
||||
errors,
|
||||
servers: mode === 'eager' ? [] : [...clients.keys()],
|
||||
close: async () => {
|
||||
await Promise.all([...clients.values()].map((c) => c.close().catch(() => {})));
|
||||
},
|
||||
@@ -185,11 +202,11 @@ export async function connectMcp(servers: Record<string, McpServerConfig>): Prom
|
||||
toolCache.set(handle, new Map());
|
||||
|
||||
const hasAnyServer = Object.keys(servers).length > 0;
|
||||
const hasMetaServer = Object.entries(servers).some(([, cfg]) => !isDirect(cfg));
|
||||
if (hasMetaServer) {
|
||||
const meta = createMcpMetaTools(handle);
|
||||
Object.assign(tools, meta);
|
||||
}
|
||||
// The three meta-tools are always registered: a session with servers whose
|
||||
// connection failed can still call mcp_list and be told why, and a direct
|
||||
// server's own tools land alongside them.
|
||||
const meta = createMcpMetaTools(handle);
|
||||
Object.assign(tools, meta);
|
||||
if (hasAnyServer) {
|
||||
Object.defineProperty(tools, '__mcpServerNames', { value: Object.keys(servers), enumerable: false, writable: true, configurable: true });
|
||||
}
|
||||
|
||||
@@ -325,10 +325,6 @@ export class Permissions {
|
||||
this.granted.set(tool, set);
|
||||
}
|
||||
|
||||
granted_(tool: string): string[] {
|
||||
return [...(this.granted.get(tool) ?? [])];
|
||||
}
|
||||
|
||||
/** The decision for one call, and which pattern decided it. */
|
||||
check(tool: string, input: unknown): Resolved {
|
||||
const resolved = resolve(entryFor(tool, this.config), tool, input);
|
||||
|
||||
@@ -13,6 +13,10 @@ export type Rate = { inputPerMTok: number; outputPerMTok: number };
|
||||
* USD per million tokens. Prefix match on the model id, longest first, so
|
||||
* `claude-sonnet-4-5-20250929` resolves via `claude-sonnet-4-5`. Published rates
|
||||
* drift, so this is a best-effort estimate rather than a billing source.
|
||||
*
|
||||
* Source: vendor pricing pages, checked 2026-09-17. Anthropic (Anthropic API, not
|
||||
* Batch) and OpenAI listed rates; DeepSeek and Grok per their API pricing. Rates
|
||||
* are for input, then output. Re-verify before trusting a live spend figure.
|
||||
*/
|
||||
const RATES: Record<string, Rate> = {
|
||||
'claude-opus-4': { inputPerMTok: 15, outputPerMTok: 75 },
|
||||
|
||||
+10
-2
@@ -148,8 +148,10 @@ function renderTools(available: readonly string[]): string {
|
||||
// free, and the schema already says what each takes.
|
||||
const git = extra.filter((n) => GIT_TOOL_NAMES.includes(n) && n !== 'git_commit_message');
|
||||
const mcpDirect = extra.filter((n) => n.startsWith('mcp__'));
|
||||
const META = ['mcp_list', 'mcp_inspect', 'mcp_call'];
|
||||
const lazyMcp = META.filter((n) => available.includes(n));
|
||||
const other = extra.filter(
|
||||
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__'),
|
||||
(n) => (!GIT_TOOL_NAMES.includes(n) || n === 'git_commit_message') && !n.startsWith('mcp__') && !META.includes(n),
|
||||
);
|
||||
|
||||
if (git.length > 0) {
|
||||
@@ -157,7 +159,13 @@ function renderTools(available: readonly string[]): string {
|
||||
`- ${git.join(', ')}: read-only git, no approval needed. Use them instead of bash for history and diffs; they cannot mutate the repository.`,
|
||||
);
|
||||
}
|
||||
if (mcpDirect.length > 0) {
|
||||
if (lazyMcp.length > 0) {
|
||||
// Lazy mode: the meta-tool descriptions already name the connected servers, so
|
||||
// the model needs the workflow, not a schema listing.
|
||||
lines.push(
|
||||
`- ${lazyMcp.join(', ')}: MCP tools are fetched on demand. mcp_list names a server's tools, mcp_inspect reads one tool's schema, mcp_call runs it. Never guess a server or tool name: list first.`,
|
||||
);
|
||||
} else if (mcpDirect.length > 0) {
|
||||
lines.push(
|
||||
`- ${mcpDirect.join(', ')}: from MCP servers exposed direct (mcp__<server>__<tool>). Each needs approval.`,
|
||||
);
|
||||
|
||||
+78
-2
@@ -92,8 +92,6 @@ export function detachOrphanedItems(before: ModelMessage[], after: ModelMessage[
|
||||
|
||||
export type PruneOptions = Parameters<typeof pruneMessages>[0];
|
||||
|
||||
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
|
||||
|
||||
const anyParts = (message: ModelMessage): Part[] =>
|
||||
Array.isArray(message.content) ? (message.content as Part[]) : [];
|
||||
|
||||
@@ -167,6 +165,84 @@ export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
|
||||
return dropOrphanedResults(detachOrphanedItems(options.messages, pruned));
|
||||
}
|
||||
|
||||
/**
|
||||
* The messages a prune would discard, so they can be summarized before they go.
|
||||
*
|
||||
* Compaction keeps the model's *memory of a turn* — the tool tail it is told to
|
||||
* keep stays verbatim. What it does not keep is any statement of what was
|
||||
* dropped. So a decision from forty messages ago vanishes silently, and the model
|
||||
* contradicts it with full confidence, because as far as it can tell it never
|
||||
* said that.
|
||||
*
|
||||
* Identity is by reference, not by value: `prunePreservingItems` rebuilds the
|
||||
* surviving messages with `{ ...message }`, so a value comparison would report
|
||||
* every message as changed and no message as dropped. `Set` on the object
|
||||
* references is exact.
|
||||
*
|
||||
* Only messages that carry content worth summarizing are returned — an assistant
|
||||
* turn consisting of nothing but a dropped `reasoning` part is not a decision, and
|
||||
* summarizing "the model thought for a while" is worse than saying nothing.
|
||||
*/
|
||||
export function droppedBy(before: ModelMessage[], after: ModelMessage[]): ModelMessage[] {
|
||||
const surviving = new Set<ModelMessage>(after);
|
||||
return before.filter((message) => !surviving.has(message));
|
||||
}
|
||||
|
||||
const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
|
||||
|
||||
function textOf(message: ModelMessage): string {
|
||||
const { content } = message;
|
||||
if (typeof content === 'string') return content;
|
||||
if (!Array.isArray(content)) return '';
|
||||
const chunks: string[] = [];
|
||||
for (const part of content as Part[]) {
|
||||
const p = part as Part & { text?: unknown; input?: unknown; output?: unknown };
|
||||
if (typeof p.text === 'string') chunks.push(p.text);
|
||||
// A tool call's input is the decision made: the path, the command, the patch.
|
||||
else if (p.type === 'tool-call' && p.input !== undefined) chunks.push(JSON.stringify(p.input));
|
||||
// A tool result is what came back. Without it a digest says what the model
|
||||
// asked for and nothing about the answer, which is the half a later
|
||||
// contradiction is usually argued from.
|
||||
else if (ANSWER_PARTS.has(p.type) && p.output !== undefined) {
|
||||
const rendered = typeof p.output === 'string' ? p.output : JSON.stringify(p.output);
|
||||
chunks.push(rendered);
|
||||
}
|
||||
}
|
||||
return chunks.join(' ').trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* A one-line-per-message digest of what a prune wants to drop.
|
||||
*
|
||||
* This is the *fallback* when no summarizer is available or the call fails: crude,
|
||||
* but it preserves the thing that matters — which tool touched which path, and in
|
||||
* what order — rather than the nothing that is there today. The summarizer, when
|
||||
* it runs, is a model and reads far better than this.
|
||||
*/
|
||||
export function digestOf(dropped: readonly ModelMessage[]): string {
|
||||
const lines: string[] = [];
|
||||
for (const message of dropped) {
|
||||
const text = textOf(message);
|
||||
if (!text) continue;
|
||||
const role = message.role === 'tool' ? 'result' : message.role;
|
||||
const clipped = text.length > 160 ? `${text.slice(0, 160)}...` : text;
|
||||
lines.push(`- (${role}) ${clipped}`);
|
||||
}
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
export const PRUNED_SPAN_PREFIX = 'Earlier in this session, now compacted away:';
|
||||
|
||||
export function isPrunedSpanSummary(message: ModelMessage): boolean {
|
||||
return message.role === 'user' && typeof message.content === 'string' && message.content.startsWith(PRUNED_SPAN_PREFIX);
|
||||
}
|
||||
|
||||
export function prunedSpanMessage(summary: string | undefined, dropped: readonly ModelMessage[]): ModelMessage | undefined {
|
||||
const body = summary?.trim() || digestOf(dropped);
|
||||
if (!body) return undefined;
|
||||
return { role: 'user', content: `${PRUNED_SPAN_PREFIX}\n\n${body}` };
|
||||
}
|
||||
|
||||
/**
|
||||
* How many trailing messages keep their tool content, widest first.
|
||||
*
|
||||
|
||||
+98
-6
@@ -18,12 +18,13 @@ import { Permissions, type PermissionConfig } from './permission';
|
||||
import type { PluginHost } from './plugins';
|
||||
import { costOf, formatUsd } from './pricing';
|
||||
import { systemPrompt } from './prompt';
|
||||
import { detachProviderItems, droppedSpan, estimateTokens as pruneEstimateTokens, pruneToFit } from './prune';
|
||||
import { detachProviderItems, droppedSpan, estimateTokens as pruneEstimateTokens, prunedSpanMessage, pruneToFit } from './prune';
|
||||
import { walk } from './ignore';
|
||||
import { createStepBackTool, type LoopEntry } from './step-back';
|
||||
import { createSkillTool, renderSkills, type Skill } from './skills';
|
||||
import { suggestSkillsFromTranscript, writeAutoSkill } from './skill-learner';
|
||||
import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools';
|
||||
import { onBeforeWrite, SnapshotStack, type FileState } from './snapshot';
|
||||
import { onBeforeWrite, type FileState, restore, Snapshots, SnapshotStack, type TurnSnapshot } from './snapshot';
|
||||
import { dirname, join, resolve } from 'node:path';
|
||||
import { existsSync, readFileSync, statSync, readdirSync } from 'node:fs';
|
||||
|
||||
@@ -48,6 +49,21 @@ export type ChangeSummary = {
|
||||
deleted: string[];
|
||||
};
|
||||
|
||||
/** What `/undo` removed, so the UI can name the files. */
|
||||
export type UndoResult = {
|
||||
snapshot: TurnSnapshot;
|
||||
restored: string[];
|
||||
removed: string[];
|
||||
conversationTrimmed: boolean;
|
||||
};
|
||||
|
||||
/** What `/redo` put back. Files are never re-restored (no post-image), only the conversation is. */
|
||||
export type RedoResult = {
|
||||
snapshot: TurnSnapshot;
|
||||
filesRestored: false;
|
||||
what: 'both' | 'files' | 'conversation';
|
||||
};
|
||||
|
||||
/** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */
|
||||
export type ApprovalDecision = 'once' | 'always' | 'deny';
|
||||
|
||||
@@ -102,6 +118,8 @@ export type SessionOptions = {
|
||||
plugins?: PluginHost;
|
||||
/** Where an `ask` tool call goes. Omit in headless runs. */
|
||||
ask?: AskFn;
|
||||
/** How many identical calls this turn trip the repeat guard (default REPEAT_LIMIT). */
|
||||
repeatLimit?: number;
|
||||
messages?: ModelMessage[];
|
||||
onChange?: (messages: ModelMessage[]) => void;
|
||||
/** Live stdout/stderr from bash, for a UI that wants progress. */
|
||||
@@ -153,6 +171,23 @@ const REPEAT_LIMIT = 3;
|
||||
|
||||
const callKey = (toolName: string, input: unknown) => `${toolName}:${JSON.stringify(input ?? null)}`;
|
||||
|
||||
/**
|
||||
* Squashes a tool result into a few characters for the loop trace.
|
||||
*
|
||||
* The trace is fed to the model verbatim by `step_back`, so a 30 KB read_file
|
||||
* output must not flood the next context window: keep the first 80 chars and
|
||||
* say how long the original was.
|
||||
*/
|
||||
function summarizeToolResult(output: unknown): string {
|
||||
if (typeof output === 'string') return output.length <= 80 ? output : `${output.slice(0, 80)}…(${output.length} chars)`;
|
||||
try {
|
||||
const json = JSON.stringify(output);
|
||||
return json.length <= 80 ? json : `${json.slice(0, 80)}…`;
|
||||
} catch {
|
||||
return String(output);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The provider rejected an `item_reference` because it no longer holds that item:
|
||||
* 404 "Item with id 'msg_...' not found". Retrying the same history repeats it, so
|
||||
@@ -241,11 +276,16 @@ export class Session {
|
||||
private pendingHost: PluginHost | undefined;
|
||||
/** Calls seen this turn, for the repeat guard. Cleared per turn, not per step. */
|
||||
private readonly seen = new Map<string, number>();
|
||||
/** Every tool call this turn, input + outcome, for the loop-diagnosis tool. */
|
||||
private readonly loopTrace: LoopEntry[] = [];
|
||||
/** Tool-call inputs by call id, for the trace: `step_back` reads what was asked. */
|
||||
private readonly callInputs = new Map<string, { toolName: string; input: string }>();
|
||||
/** One stale-item repair per turn, so a repeating 404 cannot loop the run. */
|
||||
private staleItemsRepaired = false;
|
||||
/** The 80% spend warning is shown once, not on every turn past the line. */
|
||||
private warnedSpend = false;
|
||||
private controller: AbortController | undefined;
|
||||
/** The undo stack: approved snapshotted tool calls, popped by `/undo`. */
|
||||
private readonly snapshots = new SnapshotStack();
|
||||
private turnBeforeLen = 0;
|
||||
private turnBeforeFiles = new Map<string, FileState>();
|
||||
@@ -311,6 +351,10 @@ export class Session {
|
||||
...this.notebook.tools(),
|
||||
...(this.opts.memory ? this.opts.memory.tools() : {}),
|
||||
...skillTool,
|
||||
// The loop escape hatch: step_back reflects on the session's loop trace
|
||||
// and steers the model off a stalled attempt. Always registered, like the
|
||||
// repo tools, so a circling model can reach it without a permission grant.
|
||||
step_back: createStepBackTool({ trace: () => this.loopTrace }),
|
||||
...(this.opts.ask ? { ask: createAskTool(this.opts.ask) } : {}),
|
||||
};
|
||||
const tools: ToolSet = { ...builtinTools, ...sessionTools, ...(this.pluginHost?.tools ?? {}), ...(this.opts.extraTools ?? {}) };
|
||||
@@ -365,6 +409,15 @@ export class Session {
|
||||
this.pluginHost = host;
|
||||
this.rebuild();
|
||||
}
|
||||
|
||||
/**
|
||||
* Swaps the session's skill set (hot-reload path from the CLI). The next
|
||||
* `buildSessionTools` picks the new list up via `currentSkills`.
|
||||
*/
|
||||
setSkills(skills: Skill[]): void {
|
||||
this.currentSkills = skills;
|
||||
}
|
||||
|
||||
private drainPendingHotReload(): void {
|
||||
let changed = false;
|
||||
if (this.pendingSkills !== undefined) { this.currentSkills = this.pendingSkills; this.pendingSkills = undefined; changed = true; }
|
||||
@@ -912,6 +965,23 @@ export class Session {
|
||||
return count;
|
||||
}
|
||||
|
||||
/**
|
||||
* Records one tool outcome for `step_back`: what was called and what came back.
|
||||
*
|
||||
* The trace is capped per turn so a chatty loop cannot grow it without bound.
|
||||
*/
|
||||
private recordTrace(callId: string, result: string): void {
|
||||
const entry = this.callInputs.get(callId);
|
||||
this.loopTrace.push({
|
||||
step: this.loopTrace.length,
|
||||
toolName: entry?.toolName ?? 'unknown',
|
||||
input: entry?.input ?? '',
|
||||
result,
|
||||
at: new Date().toISOString(),
|
||||
});
|
||||
if (this.loopTrace.length > 50) this.loopTrace.shift();
|
||||
}
|
||||
|
||||
/**
|
||||
* Approval decisions, evaluated per call by the SDK.
|
||||
*
|
||||
@@ -952,7 +1022,18 @@ export class Session {
|
||||
}
|
||||
|
||||
const repeats = this.repeatCount(toolName, input);
|
||||
if (decision === 'allow' && repeats < REPEAT_LIMIT) return undefined;
|
||||
const limit = this.opts.repeatLimit ?? REPEAT_LIMIT;
|
||||
if (decision === 'allow' && repeats < limit) return undefined;
|
||||
|
||||
if (decision === 'allow') {
|
||||
// Repeated with a permission that says `allow`: the model is looping, not
|
||||
// asking, and it should stop and look at the trace rather than burn another
|
||||
// approval. This is the point the step_back tool exists for.
|
||||
notices.push(
|
||||
`You have called ${toolName} with the same input ${repeats + 1} times this turn. It is not making progress. ` +
|
||||
`Use step_back to reflect on what changed between attempts, then try a different approach or stop.`,
|
||||
);
|
||||
}
|
||||
|
||||
why.set(callKey(toolName, input), {
|
||||
...(pattern ? { matchedPattern: pattern } : {}),
|
||||
@@ -1052,6 +1133,8 @@ export class Session {
|
||||
// Per turn, not per step: a tool called once in each of three steps is the
|
||||
// loop this guards against.
|
||||
this.seen.clear();
|
||||
this.loopTrace.length = 0;
|
||||
this.callInputs.clear();
|
||||
this.staleItemsRepaired = false;
|
||||
|
||||
const outputs: Extract<AgentEvent, { type: 'tool-output' }>[] = [];
|
||||
@@ -1081,7 +1164,7 @@ export class Session {
|
||||
}
|
||||
// tail for redo: the messages added by this turn
|
||||
const tail = this.messages.slice(this.turnBeforeLen).map((m) => ({ ...m, content: typeof m.content === 'string' ? m.content : JSON.parse(JSON.stringify(m.content)) } as import('ai').ModelMessage));
|
||||
const snap: import('./snapshot').TurnSnapshot & { _tail?: import('ai').ModelMessage[] } = {
|
||||
const snap: { beforeLen: number; afterLen: number; beforeFiles: Map<string, FileState>; afterFiles: Map<string, FileState>; _tail?: import('ai').ModelMessage[] } = {
|
||||
beforeLen: this.turnBeforeLen,
|
||||
afterLen: this.messages.length,
|
||||
beforeFiles: new Map(this.turnBeforeFiles),
|
||||
@@ -1266,12 +1349,18 @@ export class Session {
|
||||
break;
|
||||
case 'tool-call':
|
||||
if (part.toolName === 'todo_write') this.todoWrittenThisTurn = true;
|
||||
this.callInputs.set(part.toolCallId, {
|
||||
toolName: part.toolName,
|
||||
input: callKey(part.toolName, part.input),
|
||||
});
|
||||
yield { type: 'tool-call', id: part.toolCallId, name: part.toolName, input: part.input };
|
||||
break;
|
||||
case 'tool-result':
|
||||
this.recordTrace(part.toolCallId, summarizeToolResult(part.output));
|
||||
yield { type: 'tool-result', id: part.toolCallId, name: part.toolName, output: part.output };
|
||||
break;
|
||||
case 'tool-error':
|
||||
this.recordTrace(part.toolCallId, `error: ${part.error instanceof Error ? part.error.message : String(part.error)}`);
|
||||
yield { type: 'tool-error', id: part.toolCallId, name: part.toolName, error: part.error };
|
||||
break;
|
||||
case 'tool-approval-request': {
|
||||
@@ -1348,10 +1437,13 @@ export class Session {
|
||||
this.opts.onChange?.(this.messages);
|
||||
|
||||
// Lossless compaction: summarize what the wire pruned so future turns keep it.
|
||||
// The note leads the surviving history: it stands in for the dropped span,
|
||||
// so the model sees it first, then the history it actually kept.
|
||||
if (compactionSpan && compactionSpan.length > 0) {
|
||||
const retained = await summarizeDiscarded(compactionSpan, this.model);
|
||||
if (retained) {
|
||||
this.messages.push({ role: 'user', content: `Note (retained from compacted history):\n${retained}` });
|
||||
const note = prunedSpanMessage(retained, compactionSpan);
|
||||
if (note) {
|
||||
this.messages.unshift(note);
|
||||
this.opts.onChange?.(this.messages);
|
||||
}
|
||||
compactionSpan = undefined;
|
||||
|
||||
+14
-4
@@ -100,18 +100,28 @@ export function renderSkills(skills: Skill[]): string {
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
export function createSkillTool(skills: Skill[]) {
|
||||
const names = skills.map((s) => s.name);
|
||||
/**
|
||||
* The `skill` tool, reading the list live so a mid-session install shows up on the
|
||||
* next turn without a restart.
|
||||
*
|
||||
* The catalogue in the system prompt and the names in this tool's description both
|
||||
* come from the same list on each read, so replacing the list at a turn boundary
|
||||
* makes both stale bytes atomic: a skill the model can call it can see already.
|
||||
*/
|
||||
export function createSkillTool(getSkills: () => Skill[]) {
|
||||
const names = () => getSkills().map((s) => s.name).join(', ');
|
||||
return tool({
|
||||
description:
|
||||
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
|
||||
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`,
|
||||
`what you are about to do, then follow what it says. Available: ${names() || 'none'}.`,
|
||||
inputSchema: z.object({
|
||||
name: z.string().describe('Skill name from the list in your instructions'),
|
||||
}),
|
||||
execute: async ({ name }) => {
|
||||
const skills = getSkills();
|
||||
const skill = skills.find((s) => s.name === name.trim().toLowerCase());
|
||||
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`);
|
||||
if (!skill)
|
||||
throw new Error(`No skill named "${name}". Available: ${skills.map((s) => s.name).join(', ') || 'none'}`);
|
||||
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
|
||||
},
|
||||
});
|
||||
|
||||
+262
-12
@@ -1,3 +1,6 @@
|
||||
import { createHash } from 'node:crypto';
|
||||
import { join, relative, resolve, sep } from 'node:path';
|
||||
|
||||
/**
|
||||
* Per-turn file snapshots for /undo and /redo.
|
||||
*
|
||||
@@ -10,16 +13,263 @@
|
||||
* history stack (cap 100) and owns undo/redo.
|
||||
*/
|
||||
|
||||
export type FileState = { existed: boolean; content: string | null };
|
||||
/** One file as it was before the turn that changed it. `before === undefined` means it did not exist. */
|
||||
export type PreImage = {
|
||||
/** Workspace-relative, using forward slashes, so a restore is portable across platforms. */
|
||||
path: string;
|
||||
before: string | undefined;
|
||||
};
|
||||
|
||||
export type TurnSnapshot = {
|
||||
/** Messages length before the turn's user message was pushed. */
|
||||
/** Monotonic turn number, so the UI can name what is being undone. */
|
||||
turn: number;
|
||||
at: string;
|
||||
/** The user prompt that opened the turn, for a menu that lists them. */
|
||||
prompt: string;
|
||||
/** Files this turn changed, with their content from before it started. */
|
||||
files: PreImage[];
|
||||
/** The conversation length when the turn began, so undo can trim it back. */
|
||||
messageCount: number;
|
||||
};
|
||||
|
||||
/** Claude Code keeps 100; beyond that the memory is worth more than the recall. */
|
||||
export const MAX_SNAPSHOTS = 100;
|
||||
|
||||
/** A single file larger than this is not snapshotted; a 40 MB binary is not an edit. */
|
||||
const MAX_FILE_BYTES = 2 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* The workspace-relative, slash-normalised form of a path, or undefined if it is
|
||||
* outside the workspace.
|
||||
*
|
||||
* Outside is refused rather than clamped: a path that escapes the workspace is not
|
||||
* something this repository can undo, and recording it would imply an undo that
|
||||
* cannot happen. Paths are normalised to forward slashes because a snapshot written
|
||||
* on Windows may be read on a machine where a backslash is a filename character.
|
||||
*/
|
||||
export function relPath(cwd: string, abs: string): string | undefined {
|
||||
const root = resolve(cwd);
|
||||
const target = resolve(abs);
|
||||
const rel = relative(root, target);
|
||||
if (rel === '' || rel.startsWith('..') || rel.includes(`..${sep}`)) return undefined;
|
||||
return rel.split(sep).join('/');
|
||||
}
|
||||
|
||||
/**
|
||||
* The absolute path a tool call will write to, when there is exactly one.
|
||||
*
|
||||
* Deliberately a small, explicit map rather than a guess. `multi_edit` and the line
|
||||
* editors each take a single `path`; `move_file` takes `from` and `to` and both are
|
||||
* recorded; `delete_file` takes a `path`. `apply_patch` carries its paths inside the
|
||||
* patch text, and `bash` carries none — both are reported as uncovered rather than
|
||||
* silently not snapshotted.
|
||||
*/
|
||||
export function touchedPaths(toolName: string, input: unknown): { paths: string[]; covered: boolean } {
|
||||
const o = (input ?? {}) as Record<string, unknown>;
|
||||
const one = (key: string) => (typeof o[key] === 'string' ? [o[key] as string] : []);
|
||||
|
||||
switch (toolName) {
|
||||
case 'write_file':
|
||||
case 'edit_file':
|
||||
case 'multi_edit':
|
||||
case 'delete_file':
|
||||
case 'insert_lines':
|
||||
case 'delete_lines':
|
||||
case 'replace_lines':
|
||||
case 'append_file':
|
||||
case 'prepend_file':
|
||||
return { paths: one('path'), covered: true };
|
||||
case 'move_file':
|
||||
return { paths: [...one('from'), ...one('to')], covered: true };
|
||||
case 'apply_patch': {
|
||||
const patch = typeof o['patch'] === 'string' ? (o['patch'] as string) : '';
|
||||
const paths = [...patch.matchAll(/^\*\*\* (?:Add|Update|Delete) File: (.+)$/gm)].map((m) => m[1]!.trim());
|
||||
const moves = [...patch.matchAll(/^\*\*\* Move to: (.+)$/gm)].map((m) => m[1]!.trim());
|
||||
return { paths: [...paths, ...moves], covered: true };
|
||||
}
|
||||
case 'bash':
|
||||
// Arbitrary code: an untouched-looking `node -e` can rewrite the tree.
|
||||
return { paths: [], covered: false };
|
||||
default:
|
||||
return { paths: [], covered: true };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Records pre-images for the files a turn changes, and restores them on undo.
|
||||
*
|
||||
* One instance per session. Holds at most MAX_SNAPSHOTS turns; the oldest falls off
|
||||
* the front, because the turn someone wants back is almost always the last one.
|
||||
*/
|
||||
export class Snapshots {
|
||||
private readonly turns: TurnSnapshot[] = [];
|
||||
private current: { turn: number; at: string; prompt: string; files: Map<string, string | undefined>; messageCount: number } | null =
|
||||
null;
|
||||
private next = 1;
|
||||
private readonly cwd: string;
|
||||
|
||||
constructor(cwd: string = process.cwd()) {
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
/** Opens a turn. Called once per user prompt, before the model runs. */
|
||||
begin(prompt: string, messageCount: number): void {
|
||||
this.current = { turn: this.next++, at: new Date().toISOString(), prompt, files: new Map(), messageCount };
|
||||
}
|
||||
|
||||
/**
|
||||
* Records a file's content before a tool changes it, on the first write of the turn.
|
||||
*
|
||||
* Idempotent per path per turn: the second `write_file` to the same path in one turn
|
||||
* must not replace the pre-image with the intermediate content the first write left,
|
||||
* because undo restores the turn's starting state, not the midpoint.
|
||||
*
|
||||
* Read failures are swallowed. A snapshot is a convenience; a tool call that fails
|
||||
* because the snapshot layer could not read an unrelated path would be worse than no
|
||||
* undo at all.
|
||||
*/
|
||||
async capture(absPath: string): Promise<void> {
|
||||
if (!this.current) return;
|
||||
const rel = relPath(this.cwd, absPath);
|
||||
if (rel === undefined) return;
|
||||
if (this.current.files.has(rel)) return;
|
||||
|
||||
try {
|
||||
const file = Bun.file(absPath);
|
||||
const exists = await file.exists();
|
||||
if (!exists) {
|
||||
this.current.files.set(rel, undefined);
|
||||
return;
|
||||
}
|
||||
if (file.size > MAX_FILE_BYTES) return;
|
||||
this.current.files.set(rel, await file.text());
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
/** Records any path a tool call is about to touch. Returns whether the tool is covered at all. */
|
||||
async captureFor(toolName: string, input: unknown): Promise<{ covered: boolean; paths: string[] }> {
|
||||
const { paths, covered } = touchedPaths(toolName, input);
|
||||
for (const p of paths) await this.capture(resolve(this.cwd, p));
|
||||
return { paths, covered };
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes the turn, keeping it only if it changed something.
|
||||
*
|
||||
* A turn that read and answered without writing is not worth a slot, and keeping it
|
||||
* would make `/undo` step past a turn that has nothing to undo — which reads as the
|
||||
* command being broken.
|
||||
*/
|
||||
commit(): TurnSnapshot | undefined {
|
||||
const cur = this.current;
|
||||
this.current = null;
|
||||
if (!cur || cur.files.size === 0) return undefined;
|
||||
|
||||
const snap: TurnSnapshot = {
|
||||
turn: cur.turn,
|
||||
at: cur.at,
|
||||
prompt: cur.prompt,
|
||||
files: [...cur.files].map(([path, before]) => ({ path, before })),
|
||||
messageCount: cur.messageCount,
|
||||
};
|
||||
this.turns.push(snap);
|
||||
while (this.turns.length > MAX_SNAPSHOTS) this.turns.shift();
|
||||
return snap;
|
||||
}
|
||||
|
||||
/** Discards the open turn without recording it, for an aborted or failed turn. */
|
||||
discard(): void {
|
||||
this.current = null;
|
||||
}
|
||||
|
||||
/** The turns that can be undone, newest first. */
|
||||
list(): readonly TurnSnapshot[] {
|
||||
return [...this.turns].reverse();
|
||||
}
|
||||
|
||||
/** Whether there is an open turn collecting pre-images right now. */
|
||||
get open(): boolean {
|
||||
return this.current !== null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Removes the newest turn and returns what it holds, without restoring.
|
||||
*
|
||||
* Separated from restoring so the caller can decide *what* to bring back —
|
||||
* files, conversation, or both — which is the split Claude Code's rewind menu
|
||||
* exposes and the reason one control surface is worth more than three commands.
|
||||
*/
|
||||
pop(): TurnSnapshot | undefined {
|
||||
return this.turns.pop();
|
||||
}
|
||||
|
||||
/** Puts a turn back, for a `/redo` that follows an `/undo`. */
|
||||
push(snap: TurnSnapshot): void {
|
||||
this.turns.push(snap);
|
||||
}
|
||||
|
||||
get size(): number {
|
||||
return this.turns.length;
|
||||
}
|
||||
|
||||
cwdOf(): string {
|
||||
return this.cwd;
|
||||
}
|
||||
|
||||
clear(): void {
|
||||
this.turns.length = 0;
|
||||
this.current = null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Writes a pre-image back to disk, recreating a deleted file or removing one that was created. */
|
||||
export async function restore(snap: TurnSnapshot, cwd = process.cwd()): Promise<{ restored: string[]; removed: string[] }> {
|
||||
const restored: string[] = [];
|
||||
const removed: string[] = [];
|
||||
|
||||
for (const file of snap.files) {
|
||||
const abs = join(cwd, file.path);
|
||||
if (file.before === undefined) {
|
||||
// The file did not exist before the turn, so undoing its creation is removing it.
|
||||
const f = Bun.file(abs);
|
||||
if (await f.exists()) {
|
||||
await f.delete();
|
||||
removed.push(file.path);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
await Bun.write(abs, file.before);
|
||||
restored.push(file.path);
|
||||
}
|
||||
|
||||
return { restored, removed };
|
||||
}
|
||||
|
||||
/** A short, stable label for a snapshot, for a menu that lists several. */
|
||||
export function labelOf(snap: TurnSnapshot): string {
|
||||
const first = snap.prompt.trim().split('\n')[0] ?? '';
|
||||
const clipped = first.length > 50 ? `${first.slice(0, 50)}...` : first || '(no prompt)';
|
||||
return `turn ${snap.turn}: ${clipped}`;
|
||||
}
|
||||
|
||||
/** A content hash, used to tell whether a file still matches what the snapshot holds. */
|
||||
export function hashOf(text: string): string {
|
||||
return createHash('sha256').update(text).digest('hex').slice(0, 12);
|
||||
}
|
||||
|
||||
/* -------------------------------------------------------------------------- */
|
||||
/* Fork's SnapshotStack — kept for the fork's undo/redo tests and API surface. */
|
||||
/* -------------------------------------------------------------------------- */
|
||||
|
||||
export type FileState = { existed: boolean; content: string | null };
|
||||
|
||||
/** The fork's undo stack entry shape: before/after message lengths + file maps. */
|
||||
type StackEntry = {
|
||||
beforeLen: number;
|
||||
/** Messages length after the turn completed (including tool results). */
|
||||
afterLen: number;
|
||||
/** File state before the turn, keyed by absolute path. Only files the turn touched. */
|
||||
beforeFiles: Map<string, FileState>;
|
||||
/** File state after the turn, for redo. */
|
||||
afterFiles: Map<string, FileState>;
|
||||
};
|
||||
|
||||
@@ -37,10 +287,10 @@ export async function recordBeforeWrite(abs: string): Promise<void> {
|
||||
}
|
||||
|
||||
export class SnapshotStack {
|
||||
private readonly history: TurnSnapshot[] = [];
|
||||
private readonly future: TurnSnapshot[] = [];
|
||||
private readonly history: StackEntry[] = [];
|
||||
private readonly future: StackEntry[] = [];
|
||||
|
||||
push(entry: TurnSnapshot): void {
|
||||
push(entry: StackEntry): void {
|
||||
this.history.push(entry);
|
||||
if (this.history.length > MAX_HISTORY) this.history.shift();
|
||||
this.future.length = 0;
|
||||
@@ -50,17 +300,17 @@ export class SnapshotStack {
|
||||
canRedo(): boolean { return this.future.length > 0; }
|
||||
|
||||
/** The most recent snapshot without consuming it — lets a caller diff the last turn. */
|
||||
peek(): TurnSnapshot | undefined {
|
||||
peek(): StackEntry | undefined {
|
||||
return this.history.at(-1);
|
||||
}
|
||||
|
||||
popForUndo(): TurnSnapshot | undefined {
|
||||
popForUndo(): StackEntry | undefined {
|
||||
const e = this.history.pop();
|
||||
if (e) this.future.push(e);
|
||||
return e;
|
||||
}
|
||||
|
||||
popForRedo(): TurnSnapshot | undefined {
|
||||
popForRedo(): StackEntry | undefined {
|
||||
const e = this.future.pop();
|
||||
if (e) this.history.push(e);
|
||||
return e;
|
||||
@@ -74,4 +324,4 @@ export class SnapshotStack {
|
||||
depth(): { undo: number; redo: number } {
|
||||
return { undo: this.history.length, redo: this.future.length };
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
import { tool } from 'ai';
|
||||
import { z } from 'zod';
|
||||
|
||||
/**
|
||||
* A visible escape hatch for the loop a coding agent dies in.
|
||||
*
|
||||
* The repeat guard stops an *identical* call after three tries, but the deeper loop
|
||||
* is the model making *different* calls that all amount to the same stalled attempt —
|
||||
* re-reading the same file expecting a different answer, retrying a failing command
|
||||
* with a tweaked flag, re-sending a prompt it has already asked. No equal-input
|
||||
* detector fires on any of that, so the model burns the step budget on motions that
|
||||
* never move.
|
||||
*
|
||||
* `step_back` exists so the model has a *named* way out instead of only a guard it
|
||||
* cannot see. The session records each completed tool call (input + outcome) into a
|
||||
* loop trace; the tool returns a scripted reflection prompt built from that trace,
|
||||
* so the model is told in concrete terms that it is going in circles and is steered
|
||||
* to change direction.
|
||||
*/
|
||||
|
||||
export type LoopEntry = {
|
||||
step: number;
|
||||
toolName: string;
|
||||
input: string;
|
||||
result: string;
|
||||
at: string;
|
||||
};
|
||||
|
||||
function recentTrace(entries: readonly LoopEntry[], window = 8): LoopEntry[] {
|
||||
return entries.slice(-window);
|
||||
}
|
||||
|
||||
function renderTrace(entries: readonly LoopEntry[], maxLines = 12): string {
|
||||
return entries.slice(-maxLines).map((e) => `step ${e.step}: ${e.toolName} ${e.input} -> ${e.result}`.slice(0, 200)).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds a `step_back` tool bound to a session's loop trace.
|
||||
*
|
||||
* The tool is meant to be called when the model is not making progress — a trap it
|
||||
* cannot always see while it is inside it. The returned reflection names the recent
|
||||
* steps so the model can tell, from its own calls, that it is going in circles, and
|
||||
* gives it the one thing a stuck agent is usually missing: permission to stop,
|
||||
* say what it learned, and change direction rather than try harder.
|
||||
*/
|
||||
export function createStepBackTool(opts: { trace: () => readonly LoopEntry[] }) {
|
||||
return tool({
|
||||
description:
|
||||
'Use when you are stuck: the same file is not changing, a command keeps failing, or you have done several steps with no visible progress. ' +
|
||||
'Records your recent steps and returns a reflection prompt to help you change direction instead of repeating the attempt.',
|
||||
inputSchema: z.object({
|
||||
note: z.string().optional().describe('A sentence in your own words about what you were trying to do.'),
|
||||
}),
|
||||
execute: async ({ note }) => {
|
||||
const trace = recentTrace(opts.trace());
|
||||
if (trace.length === 0) {
|
||||
return (
|
||||
'No recent steps to reflect on. This is an early call of step_back — it is only useful when you have ' +
|
||||
'attempted something several times. Describe what you are stuck on in `note`.'
|
||||
);
|
||||
}
|
||||
|
||||
return [
|
||||
'You have run threadbare over the last steps and are stuck. Here is what you actually did:',
|
||||
'```',
|
||||
renderTrace(trace),
|
||||
'```',
|
||||
'',
|
||||
'Before your next tool call, answer these three questions in your reasoning:',
|
||||
'1. What exactly is wrong — the input, the tool, or the expectation?',
|
||||
'2. What have you tried, and why did each fail?',
|
||||
'3. What is ONE different thing you can do that is not "try the same thing a little harder"?',
|
||||
'',
|
||||
'Then take that different action. If the failure is a command, read the actual error and fix its cause — ',
|
||||
'do not rerun the command. If a file is not what you expect, suspect your assumption about it and re-read it fresh.',
|
||||
note?.trim() ? `\nYour note: ${note.trim()}` : '',
|
||||
].join('\n');
|
||||
},
|
||||
});
|
||||
}
|
||||
+108
-1
@@ -290,7 +290,7 @@ export function createTaskTool(opts: {
|
||||
'the user exactly as yours are. Use it for a self-contained task whose intermediate steps you do not ' +
|
||||
'need to see; keep work you must supervise step by step in your own turn.'
|
||||
: '') +
|
||||
'\nDo not delegate something you can answer with a single grep. For independent pieces of work, pass `tasks` to run them in parallel instead of calling task several times sequentially.',
|
||||
'\nDo not delegate something you can answer with a single grep. For independent pieces of work, pass `tasks` to run them in parallel instead of calling task several times sequentially.',
|
||||
inputSchema: z.union([singleSchema, batchSchema]),
|
||||
execute: async (input, { abortSignal }) => {
|
||||
const asBatch = input as { tasks?: TaskSpec[]; description?: string; prompt?: string; kind?: SubagentKind };
|
||||
@@ -337,4 +337,111 @@ export function createTaskTool(opts: {
|
||||
});
|
||||
}
|
||||
|
||||
/** Flavour of one planned investigation; `kind` defaults to explore. */
|
||||
type Plan = { description: string; prompt: string; kind?: string };
|
||||
|
||||
/**
|
||||
* Runs one subagent and settles its report, usage, and panel events.
|
||||
*
|
||||
* Extracted so `tasks` can fan several out in parallel: each full run is
|
||||
* independent — its own id, own stream, own spend — and they overlap simply by
|
||||
* awaiting them together.
|
||||
*/
|
||||
async function runSubagent(
|
||||
opts: {
|
||||
model: LanguageModel;
|
||||
subagentModel?: LanguageModel;
|
||||
subagentModelId?: string;
|
||||
cwd?: string;
|
||||
maxSteps?: number;
|
||||
report?: SubagentReporter;
|
||||
approve?: SubagentApproval;
|
||||
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
|
||||
abortSignal?: AbortSignal;
|
||||
},
|
||||
plan: Plan,
|
||||
): Promise<{ named: string; report: string }> {
|
||||
const flavour: SubagentKind = (plan.kind as SubagentKind | undefined) ?? 'explore';
|
||||
if (flavour === 'worker' && !opts.approve) {
|
||||
throw new Error('The worker kind needs an approval channel, which this session has not provided.');
|
||||
}
|
||||
|
||||
const id = `sub${++counter}`;
|
||||
const report = opts.report;
|
||||
report?.({ type: 'start', id, kind: flavour, description: plan.description });
|
||||
|
||||
let steps = 0;
|
||||
let text = '';
|
||||
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
|
||||
|
||||
try {
|
||||
// `explore` is search, not reasoning, so it runs on the cheaper model when
|
||||
// one is configured. `review` and `worker` keep the parent's: they judge
|
||||
// and they change, both of which want the full model.
|
||||
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
|
||||
const result = streamText({
|
||||
model,
|
||||
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
|
||||
messages: [{ role: 'user', content: plan.prompt }],
|
||||
tools: TOOLS[flavour],
|
||||
stopWhen: isStepCount(opts.maxSteps ?? 20),
|
||||
...(opts.approve
|
||||
? {
|
||||
toolApproval: async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => {
|
||||
const approved = await opts.approve!(toolCall);
|
||||
return approved
|
||||
? undefined
|
||||
: { type: 'denied' as const, reason: 'The user denied this call. Stop and report it.' };
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
...(opts.abortSignal ? { abortSignal: opts.abortSignal } : {}),
|
||||
});
|
||||
|
||||
const sink = () => {};
|
||||
void result.responseMessages.then(undefined, sink);
|
||||
void result.usage.then(undefined, sink);
|
||||
void result.steps.then(undefined, sink);
|
||||
void result.finalStep.then(undefined, sink);
|
||||
void result.finishReason.then(undefined, sink);
|
||||
|
||||
for await (const part of result.stream) {
|
||||
if (part.type === 'tool-call') {
|
||||
steps++;
|
||||
report?.({ type: 'step', id, tool: part.toolName, summary: summarize(part.input) });
|
||||
} else if (part.type === 'tool-result') {
|
||||
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(part.output), ok: true });
|
||||
} else if (part.type === 'tool-error') {
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
report?.({ type: 'result', id, tool: part.toolName, summary: outcome(message), ok: false });
|
||||
} else if (part.type === 'text-delta') {
|
||||
text += part.text;
|
||||
} else if (part.type === 'error') {
|
||||
// A provider failure arrives as a stream part, not a throw, so it has to
|
||||
// be rethrown here or the subagent silently returns nothing.
|
||||
const message = part.error instanceof Error ? part.error.message : String(part.error);
|
||||
throw part.error instanceof Error ? part.error : new Error(message);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const usage = await result.usage;
|
||||
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
|
||||
} catch {
|
||||
// A run that errored before producing usage has nothing to account for.
|
||||
}
|
||||
} catch (e) {
|
||||
const message = e instanceof Error ? e.message : String(e);
|
||||
report?.({ type: 'error', id, message });
|
||||
throw e;
|
||||
}
|
||||
|
||||
const trimmed = text.trim();
|
||||
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
|
||||
// Settled after the stream closes; a failed run reports nothing rather than
|
||||
// a half count. The parent prices these against the subagent's own model id.
|
||||
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
|
||||
return { named: plan.description, report: trimmed || 'Subagent returned no findings.' };
|
||||
}
|
||||
|
||||
export const TASK_TOOL_NAME = 'task';
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
import type { Tool } from 'ai';
|
||||
|
||||
/**
|
||||
* Marks a tool as mutating where it is defined, rather than in a list beside it.
|
||||
*
|
||||
* The bug this prevents is the worst one this codebase can have: a new tool that
|
||||
* writes to the workspace, added to `tools` but forgotten in a hand-maintained
|
||||
* `MUTATING_TOOLS`, is a write the permission layer does not treat as a write. It
|
||||
* is silent, it passes every test that does not think to check the new name, and
|
||||
* it surfaces as a user discovering an edit they never approved.
|
||||
*
|
||||
* The mark is a property on the tool object, so `MUTATING_TOOLS` can be derived by
|
||||
* filtering the registry instead of being typed out. A tool that is not marked is
|
||||
* asserted non-mutating by `tools.test.ts`, which means the decision is made once,
|
||||
* at the definition, and cannot drift.
|
||||
*/
|
||||
export const MUTATING = '__mutating' as const;
|
||||
|
||||
/** A tool that can change the workspace or run arbitrary code. */
|
||||
export function mutating<T extends Tool>(t: T): T {
|
||||
return Object.assign(t, { [MUTATING]: true as const });
|
||||
}
|
||||
|
||||
/** Whether a tool was declared mutating at its definition site. */
|
||||
export function isMutating(t: unknown): boolean {
|
||||
return typeof t === 'object' && t !== null && (t as Record<string, unknown>)[MUTATING] === true;
|
||||
}
|
||||
+88
-22
@@ -291,16 +291,37 @@ export const writeFileTool = withMeta({ set: 'core', mutating: true }, tool({
|
||||
return `Wrote ${content.length} chars to ${path}`;
|
||||
},
|
||||
}));
|
||||
// Some models (Claude-style tool docs, DeepSeek/GLM) emit snake_case edit params
|
||||
// (old_string/new_string/replace_all) despite the camelCase schema. Normalize at the
|
||||
// boundary instead of failing the whole call on a naming convention.
|
||||
const SNAKE_EDIT_ARGS: ReadonlyArray<readonly [string, string]> = [
|
||||
['old_string', 'oldString'],
|
||||
['new_string', 'newString'],
|
||||
['replace_all', 'replaceAll'],
|
||||
];
|
||||
|
||||
export function normalizeEditArgs(input: unknown): unknown {
|
||||
if (typeof input !== 'object' || input === null) return input;
|
||||
const obj: Record<string, unknown> = { ...(input as Record<string, unknown>) };
|
||||
for (const [snake, camel] of SNAKE_EDIT_ARGS) {
|
||||
if (obj[snake] !== undefined && obj[camel] === undefined) obj[camel] = obj[snake];
|
||||
}
|
||||
if (Array.isArray(obj.edits)) obj.edits = obj.edits.map(normalizeEditArgs);
|
||||
return obj;
|
||||
}
|
||||
|
||||
export const editFileTool = withMeta({ set: 'core', mutating: true }, tool({
|
||||
description:
|
||||
'Replace an exact string in a file. oldString must appear exactly once unless replaceAll is true. Include surrounding context to make oldString unique.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
|
||||
}),
|
||||
inputSchema: z.preprocess(
|
||||
normalizeEditArgs,
|
||||
z.object({
|
||||
path: z.string(),
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional().describe('Replace every occurrence instead of requiring exactly one'),
|
||||
}),
|
||||
),
|
||||
execute: async ({ path, oldString, newString, replaceAll = false }) => {
|
||||
if (oldString === newString) throw new Error('oldString and newString are identical');
|
||||
const abs = jail(path);
|
||||
@@ -327,19 +348,22 @@ export const multiEditTool = withMeta({ set: 'edit-plus', mutating: true }, tool
|
||||
'All or nothing: if any oldString fails to match, or matches more than once without replaceAll, nothing is ' +
|
||||
'written. Prefer this over repeated edit_file calls on the same file — one approval, one write, no risk of ' +
|
||||
'leaving the file half-changed.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
edits: z
|
||||
.array(
|
||||
z.object({
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional(),
|
||||
}),
|
||||
)
|
||||
.min(1)
|
||||
.describe('Edits in the order they should be applied'),
|
||||
}),
|
||||
inputSchema: z.preprocess(
|
||||
normalizeEditArgs,
|
||||
z.object({
|
||||
path: z.string(),
|
||||
edits: z
|
||||
.array(
|
||||
z.object({
|
||||
oldString: z.string().describe('Exact text to find, including whitespace and indentation'),
|
||||
newString: z.string().describe('Replacement text'),
|
||||
replaceAll: z.boolean().optional(),
|
||||
}),
|
||||
)
|
||||
.min(1)
|
||||
.describe('Edits in the order they should be applied'),
|
||||
}),
|
||||
),
|
||||
execute: async ({ path, edits }) => {
|
||||
const abs = jail(path);
|
||||
await recordBeforeWrite(abs);
|
||||
@@ -581,7 +605,13 @@ async function pump(
|
||||
return all;
|
||||
}
|
||||
|
||||
type Running = { command: string; proc: Bun.Subprocess; interrupted: boolean; killed?: Promise<unknown> };
|
||||
type Running = {
|
||||
command: string;
|
||||
proc: Bun.Subprocess;
|
||||
interrupted: boolean;
|
||||
timedOut?: boolean;
|
||||
killed?: Promise<unknown>;
|
||||
};
|
||||
|
||||
const running = new Map<string, Running>();
|
||||
|
||||
@@ -823,6 +853,12 @@ function killTree(proc: Bun.Subprocess): Promise<unknown> {
|
||||
export function interruptBash(): string[] {
|
||||
const killed: string[] = [];
|
||||
for (const entry of running.values()) {
|
||||
// A second ctrl-c while the first killTree is still settling must not re-announce
|
||||
// the same command: the notice is the only proof the keypress did anything.
|
||||
if (entry.interrupted) {
|
||||
killed.push(entry.command);
|
||||
continue;
|
||||
}
|
||||
entry.interrupted = true;
|
||||
entry.killed = killTree(entry.proc);
|
||||
killed.push(entry.command);
|
||||
@@ -854,13 +890,32 @@ export const bashTool = withMeta({ set: 'core', mutating: true }, tool({
|
||||
cwd: process.cwd(),
|
||||
stdout: 'pipe',
|
||||
stderr: 'pipe',
|
||||
timeout,
|
||||
...(abortSignal ? { signal: abortSignal } : {}),
|
||||
});
|
||||
|
||||
const entry: Running = { command, proc, interrupted: false };
|
||||
running.set(toolCallId, entry);
|
||||
|
||||
// Bun's spawn `signal` option is not used either: it kills only the shell, so an
|
||||
// esc-abort orphaned the grandchild on the same still-open pipes as the timeout
|
||||
// did. The abort must go through killTree, exactly like ctrl-c does.
|
||||
const onAbort = () => {
|
||||
if (entry.interrupted) return;
|
||||
entry.interrupted = true;
|
||||
entry.killed = killTree(proc);
|
||||
};
|
||||
abortSignal?.addEventListener('abort', onAbort);
|
||||
// The turn may already be aborted by the time this tool starts; a past event
|
||||
// never re-fires, so check once here or the command runs unkillable by esc.
|
||||
if (abortSignal?.aborted) onAbort();
|
||||
|
||||
// Bun's own `timeout` spawn option is not used: it kills only the shell, and the
|
||||
// grandchild holding the output pipes keeps `pump` reading forever, so the tool
|
||||
// never returns. Same failure killTree exists for, just triggered by the clock.
|
||||
const timer = setTimeout(() => {
|
||||
entry.timedOut = true;
|
||||
entry.killed = killTree(proc);
|
||||
}, timeout);
|
||||
|
||||
try {
|
||||
// Drained concurrently: a command that fills one pipe while we block on the
|
||||
// other would deadlock, and buffering both hides progress for minutes.
|
||||
@@ -876,6 +931,15 @@ export const bashTool = withMeta({ set: 'core', mutating: true }, tool({
|
||||
|
||||
// Thrown rather than returned: the model must not read a killed command as
|
||||
// a command that ran and failed on its own terms.
|
||||
if (entry.timedOut) {
|
||||
throw new Error(
|
||||
cap(
|
||||
`The command exceeded its ${timeout}ms timeout and was killed. It did not finish, so its effects are unknown.\n${
|
||||
body || '(no output before it was killed)'
|
||||
}`,
|
||||
),
|
||||
);
|
||||
}
|
||||
if (entry.interrupted) {
|
||||
throw new Error(
|
||||
cap(
|
||||
@@ -896,6 +960,8 @@ export const bashTool = withMeta({ set: 'core', mutating: true }, tool({
|
||||
.join('\n\n'),
|
||||
);
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
abortSignal?.removeEventListener('abort', onAbort);
|
||||
// Awaited so the process really is gone before the tool returns. On Windows a
|
||||
// surviving grandchild holds the cwd open, which breaks the very next command.
|
||||
await entry.killed;
|
||||
|
||||
+61
-35
@@ -38,7 +38,7 @@ import { CommandMenu, InstallConfirm, Picker } from './Pickers';
|
||||
import { contextPanel, costPanel, todosPanel, toolsPanel, changesPanel, diffPanel, diffReviewPanel, workflowPanel } from './panel-bodies';
|
||||
import { PromptInput } from './PromptInput';
|
||||
import { accent, glyph } from './theme';
|
||||
import { nextKey, resultSummary, toolDetail, withResult, type Line, type NewLine } from './transcript';
|
||||
import { historyFromMessages, nextKey, resultSummary, toolDetail, withResult, type Line, type NewLine } from './transcript';
|
||||
|
||||
export { createApprovalBridge, createNoticeBus, createSubagentBus, applySubagentEvent };
|
||||
export type { ApprovalBridge, NoticeBus, SubagentBus };
|
||||
@@ -144,7 +144,11 @@ export function App({
|
||||
stdout.off('resize', onResize);
|
||||
};
|
||||
}, [stdout]);
|
||||
const [history, setHistory] = useState<Line[]>([]);
|
||||
// Seeded from the resumed history: a session loaded with -r/-c should show its
|
||||
// saved conversation rather than a blank transcript. Only read once, at mount.
|
||||
const [history, setHistory] = useState<Line[]>(() =>
|
||||
session.messages.length === 0 ? [] : historyFromMessages(session.messages as { role?: string; content?: unknown }[]),
|
||||
);
|
||||
const [draft, setDraft] = useState('');
|
||||
const [live, setLive] = useState('');
|
||||
const [busy, setBusy] = useState(false);
|
||||
@@ -196,17 +200,33 @@ export function App({
|
||||
|
||||
// The walk costs a full ignore-aware traversal, so it happens on the first `@`
|
||||
// rather than at startup, and re-runs when files change (listPaths is cached
|
||||
// in hook, but App keeps seq so a stale `paths` is dropped).
|
||||
// in hook, but App keeps seq so a stale `paths` is dropped). A slow cooldown
|
||||
// also re-walks so a file created after that first `@` shows up within a short
|
||||
// window instead of staying hidden all session.
|
||||
const pathsRef = useRef(paths);
|
||||
useEffect(() => {
|
||||
if (token === undefined || paths !== undefined) return;
|
||||
let live = true;
|
||||
void hooks.listPaths().then((all) => {
|
||||
if (live) setPaths(all);
|
||||
});
|
||||
return () => {
|
||||
live = false;
|
||||
};
|
||||
}, [hooks, paths, token]);
|
||||
pathsRef.current = paths;
|
||||
}, [paths]);
|
||||
const didLoadRef = useRef(false);
|
||||
useEffect(() => {
|
||||
if (token === undefined) return;
|
||||
if (!didLoadRef.current) {
|
||||
didLoadRef.current = true;
|
||||
let live = true;
|
||||
void hooks.listPaths().then((all) => {
|
||||
if (live) setPaths(all);
|
||||
});
|
||||
const refresh = setInterval(async () => {
|
||||
if (!live) return;
|
||||
const all = await hooks.listPaths();
|
||||
if (live && JSON.stringify(all) !== JSON.stringify(pathsRef.current)) setPaths(all);
|
||||
}, 10_000);
|
||||
return () => {
|
||||
live = false;
|
||||
clearInterval(refresh);
|
||||
};
|
||||
}
|
||||
}, [hooks, token]);
|
||||
|
||||
// A file mutated this turn: drop the cached walk so next `@` re-walks.
|
||||
const seq = hooks.fileChangeSeq();
|
||||
@@ -712,34 +732,18 @@ export function App({
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
try {
|
||||
const msg = await hooks.resumeSession(action.id);
|
||||
setHistory([]);
|
||||
// Reflect the freshly loaded history: session.messages now holds the
|
||||
// restored wire messages, and the transcript must show them again.
|
||||
setHistory(
|
||||
session.messages.length === 0
|
||||
? []
|
||||
: historyFromMessages(session.messages as { role?: string; content?: unknown }[]),
|
||||
);
|
||||
push({ kind: 'info', text: msg });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
return;
|
||||
case 'undo': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
try {
|
||||
push({ kind: 'info', text: await session.undo() });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
return;
|
||||
}
|
||||
case 'redo': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
try {
|
||||
push({ kind: 'info', text: await session.redo() });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
return;
|
||||
}
|
||||
case 'changes': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setPanel(changesPanel(session));
|
||||
@@ -843,6 +847,28 @@ export function App({
|
||||
setModelPicker(models);
|
||||
return;
|
||||
}
|
||||
case 'undo': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
try {
|
||||
push({ kind: 'info', text: await session.undo() });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
return;
|
||||
}
|
||||
case 'redo': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
try {
|
||||
push({ kind: 'info', text: await session.redo() });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
return;
|
||||
}
|
||||
case 'compact': {
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
|
||||
+2
-1
@@ -1,6 +1,7 @@
|
||||
import { Box, Text, useInput } from 'ink';
|
||||
import React from 'react';
|
||||
import type { ApprovalDecision, ApprovalRequest } from '../session';
|
||||
import { normalizeEditArgs } from '../tools';
|
||||
import { Diff } from './Diff';
|
||||
import { accent, glyph } from './theme';
|
||||
import { toolDetail } from './transcript';
|
||||
@@ -42,7 +43,7 @@ export function createApprovalBridge(): ApprovalBridge {
|
||||
* consistent and far more readable than a JSON dump of the input.
|
||||
*/
|
||||
function ApprovalDetail({ name, input }: { name: string; input: unknown }) {
|
||||
const o = (input ?? {}) as Record<string, unknown>;
|
||||
const o = (normalizeEditArgs(input) ?? {}) as Record<string, unknown>;
|
||||
|
||||
if (name === 'write_file') {
|
||||
const content = String(o['content'] ?? '');
|
||||
|
||||
@@ -33,10 +33,6 @@ type KeyLike = {
|
||||
end?: boolean;
|
||||
};
|
||||
|
||||
const INVERSE_ON = '\u001B[7m';
|
||||
const INVERSE_OFF = '\u001B[27m';
|
||||
const invert = (s: string) => `${INVERSE_ON}${s}${INVERSE_OFF}`;
|
||||
|
||||
/**
|
||||
* Text input with a real cursor and shell-style history recall.
|
||||
*
|
||||
@@ -151,10 +147,10 @@ export function PromptInput({
|
||||
);
|
||||
|
||||
if (value.length === 0) {
|
||||
if (!placeholder) return <Text>{focus ? invert(' ') : ' '}</Text>;
|
||||
if (!placeholder) return <Text inverse={focus}>{' '}</Text>;
|
||||
return (
|
||||
<Text dimColor>
|
||||
{focus ? invert(placeholder.slice(0, 1)) : placeholder.slice(0, 1)}
|
||||
<Text inverse={focus}>{placeholder.slice(0, 1)}</Text>
|
||||
{placeholder.slice(1)}
|
||||
</Text>
|
||||
);
|
||||
@@ -166,7 +162,7 @@ export function PromptInput({
|
||||
return (
|
||||
<Text>
|
||||
{shown.slice(0, cursor)}
|
||||
{invert(shown.slice(cursor, cursor + 1) || ' ')}
|
||||
<Text inverse>{shown.slice(cursor, cursor + 1) || ' '}</Text>
|
||||
{shown.slice(cursor + 1)}
|
||||
</Text>
|
||||
);
|
||||
|
||||
+99
-1
@@ -1,4 +1,5 @@
|
||||
import { TODO_MARK } from '../notebook';
|
||||
import { normalizeEditArgs } from '../tools';
|
||||
|
||||
export type Line =
|
||||
| { key: string; kind: 'user'; text: string }
|
||||
@@ -38,7 +39,8 @@ export function preview(input: unknown): string {
|
||||
* the transcript, beside the spinner while a call is in flight, and in the approval
|
||||
* prompt for any tool without a diff of its own.
|
||||
*/
|
||||
export function toolDetail(name: string, input: unknown): string[] {
|
||||
export function toolDetail(name: string, rawInput: unknown): string[] {
|
||||
const input = normalizeEditArgs(rawInput);
|
||||
if (input === null || typeof input !== 'object') return [];
|
||||
const o = input as Record<string, unknown>;
|
||||
const str = (k: string) => (typeof o[k] === 'string' ? (o[k] as string) : undefined);
|
||||
@@ -195,6 +197,102 @@ export function withResult(lines: Line[], name: string, result: string, ok: bool
|
||||
return lines;
|
||||
}
|
||||
|
||||
/**
|
||||
* The tool output that was stored on a `tool-result`. The SDK json-wraps a
|
||||
* string return as `{ type: 'text', value }`, so a restored message needs the
|
||||
* same unwrap the live stream already produced at save time.
|
||||
*/
|
||||
function toolResultText(output: unknown): string {
|
||||
if (typeof output === 'string') return output;
|
||||
if (
|
||||
output !== null &&
|
||||
typeof output === 'object' &&
|
||||
'value' in output &&
|
||||
typeof (output as { value: unknown }).value === 'string'
|
||||
) {
|
||||
return (output as { value: string }).value;
|
||||
}
|
||||
try {
|
||||
return JSON.stringify(output);
|
||||
} catch {
|
||||
return String(output);
|
||||
}
|
||||
}
|
||||
|
||||
type StoredPart = {
|
||||
type?: string;
|
||||
toolName?: string;
|
||||
input?: unknown;
|
||||
output?: unknown;
|
||||
text?: unknown;
|
||||
};
|
||||
|
||||
/**
|
||||
* The transcript lines a saved `ModelMessage[]` becomes, so a resumed session
|
||||
* renders its history instead of starting blank.
|
||||
*
|
||||
* Mirrors how the live loop paints: user strings as user lines, assistant text
|
||||
* as an assistant line, assistant `tool-call` parts as tool lines, and each
|
||||
* `role: 'tool'` result attached to the newest unanswered call of that name just
|
||||
* like `withResult` does. A result with no matching call (a pruned lead-in) is
|
||||
* dropped rather than left floating.
|
||||
*/
|
||||
export function historyFromMessages(messages: readonly { role?: string; content?: unknown }[]): Line[] {
|
||||
const lines: Line[] = [];
|
||||
|
||||
const textOf = (content: unknown): string =>
|
||||
typeof content === 'string'
|
||||
? content
|
||||
: Array.isArray(content)
|
||||
? (content as StoredPart[]).filter((p) => p.type === 'text' && typeof p.text === 'string').map((p) => p.text as string).join('')
|
||||
: '';
|
||||
|
||||
const toolPartsOf = (content: unknown): StoredPart[] =>
|
||||
Array.isArray(content) ? (content as StoredPart[]).filter((p) => p.type === 'tool-call') : [];
|
||||
|
||||
for (const m of messages) {
|
||||
switch (m.role) {
|
||||
case 'user': {
|
||||
const text = textOf(m.content).trim();
|
||||
if (text) lines.push({ key: nextKey(), kind: 'user', text });
|
||||
break;
|
||||
}
|
||||
case 'assistant': {
|
||||
const text = textOf(m.content).trim();
|
||||
if (text) lines.push({ key: nextKey(), kind: 'assistant', text });
|
||||
for (const p of toolPartsOf(m.content)) {
|
||||
const name = p.toolName ?? '';
|
||||
if (!name) continue;
|
||||
lines.push({ key: nextKey(), kind: 'tool', name, detail: toolDetail(name, p.input), ok: true });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 'tool': {
|
||||
const parts = Array.isArray(m.content) ? (m.content as StoredPart[]) : [];
|
||||
for (const p of parts) {
|
||||
if (p.type !== 'tool-result' && p.type !== 'tool-error') continue;
|
||||
const name = p.toolName ?? '';
|
||||
const result =
|
||||
p.type === 'tool-error'
|
||||
? toolResultText(p.output) || 'tool failed'
|
||||
: resultSummary(name, toolResultText(p.output));
|
||||
for (let i = lines.length - 1; i >= 0; i--) {
|
||||
const line = lines[i]!;
|
||||
if (line.kind !== 'tool' || line.name !== name || line.result !== undefined) continue;
|
||||
lines[i] = { ...line, result, ok: p.type !== 'tool-error' };
|
||||
break;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return lines;
|
||||
}
|
||||
|
||||
/** A task list as markdown, for the `/todos` panel. */
|
||||
export const todoLines = (todos: readonly { status: keyof typeof TODO_MARK; content: string; note?: string }[]) =>
|
||||
todos.length > 0
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@
|
||||
* fails inside the shipped binary. A constant is compiled in and always correct.
|
||||
* `scripts/release.ts` checks it against the release tag so the two cannot drift.
|
||||
*/
|
||||
export const VERSION = '1.0.0';
|
||||
export const VERSION = '1.0.1';
|
||||
|
||||
/** What `--version` prints: enough to identify a build from a bug report. */
|
||||
export function versionLine(): string {
|
||||
|
||||
Reference in New Issue
Block a user