release 1.0.0: cost control, 41 tools, 29 skills, custom commands, auto-load

This commit is contained in:
Muhammad Zakir Ramadhan
2026-09-07 19:59:58 +07:00
parent a22d8e13b1
commit ffa9a02c26
97 changed files with 4791 additions and 788 deletions
+186
View File
@@ -0,0 +1,186 @@
import { tool, type ToolSet } from 'ai';
import { homedir } from 'node:os';
import { join } from 'node:path';
import { z } from 'zod';
import { jail } from './ignore';
import { manifestToPlugin, parseManifest, type PluginManifest } from './registry';
import type { Plugin } from './plugins';
/**
* Auto-registration and auto-loading of external skills, tools, and plugins.
*
* Everything here is *data*, never code — the same rule the registry enforces.
* An external tool is a bounded manifest (a shell template through the guard, an
* HTTP fetch, or a file read), an external plugin a refusal manifest, an external
* skill a markdown body. Loading arbitrary code from disk would let an entry read
* every file the agent can read and lie about what it blocks, so it is not offered.
*
* Directories, later shadowing earlier by name:
* ~/.shiro-neko/{tools,plugins,skills} (user)
* .shiro/{tools,plugins,skills} (project)
* Skills already load through skills.ts; this module adds tools and plugins and
* the one place cli turns them all on.
*/
const home = () => process.env['SHIRO_HOME'] ?? homedir();
export type LoadError = { name: string; message: string };
function dirs(kind: 'tools' | 'plugins' | 'skills', cwd: string): string[] {
return [join(home(), '.shiro-neko', kind), join(cwd, '.shiro', kind)];
}
async function scan(dir: string, ext: string): Promise<string[]> {
const files: string[] = [];
try {
for await (const f of new Bun.Glob(`*.${ext}`).scan({ cwd: dir, onlyFiles: true })) files.push(f);
} catch {
return [];
}
return files.sort();
}
const MAX_PATTERN = 200;
const nameSchema = z.string().min(1).max(40).regex(/^[a-z0-9][a-z0-9-_]*$/i);
// ---------------------------------------------------------------------------
// External tools, as bounded manifests.
// ---------------------------------------------------------------------------
/**
* Three kinds of tool, each with a ceiling on what it can do. None runs arbitrary
* code: `shell` interpolates a fixed template and runs it through the guard and
* the platform shell, `http` fetches a fixed URL, `read` returns a fixed file's
* contents (jailed to the workspace). The input is a single optional `arg` string
* substituted into a `{arg}` placeholder, so a manifest cannot take structure it
* was not declared for.
*/
const toolManifestSchema = z.object({
name: nameSchema,
description: z.string().min(1).max(300),
kind: z.enum(['shell', 'http', 'read']),
/** The template with an optional `{arg}` placeholder. */
command: z.string().max(500).optional(),
url: z.string().max(500).optional(),
path: z.string().max(300).optional(),
/** Set false to require approval before running. Default true (auto-approved). */
autoApprove: z.boolean().optional(),
});
export type ToolManifest = z.infer<typeof toolManifestSchema>;
export function parseToolManifest(source: string): ToolManifest {
let raw: unknown;
try {
raw = JSON.parse(source);
} catch {
throw new Error('the tool manifest is not valid JSON');
}
const parsed = toolManifestSchema.safeParse(raw);
if (!parsed.success) {
throw new Error(`the tool manifest is malformed: ${parsed.error.issues[0]?.message ?? 'unknown reason'}`);
}
const m = parsed.data;
if (m.kind === 'shell' && !m.command) throw new Error(`shell tool "${m.name}" needs a command template`);
if (m.kind === 'http' && !m.url) throw new Error(`http tool "${m.name}" needs a url`);
if (m.kind === 'read' && !m.path) throw new Error(`read tool "${m.name}" needs a path`);
return m;
}
const MAX_TOOL_OUTPUT = 30_000;
const cap = (s: string) => (s.length <= MAX_TOOL_OUTPUT ? s : `${s.slice(0, MAX_TOOL_OUTPUT)}\n... [truncated]`);
/** The guard an external shell tool runs through, supplied by cli so it shares the real chain. */
export type ShellGuard = (command: string) => Promise<string | undefined>;
/**
* A manifest as a live tool. The guard is applied to every `shell` invocation, so
* an external tool cannot smuggle a destructive command past the user any more
* than a built-in bash call can.
*/
export function manifestToTool(manifest: ToolManifest, guard: ShellGuard) {
const inputSchema = z.object({ arg: z.string().optional().describe('optional argument substituted into {arg}') });
const substitute = (template: string, arg: string) => template.replaceAll('{arg}', arg);
return tool({
description: `${manifest.description} (external ${manifest.kind} tool)`,
inputSchema,
execute: async ({ arg = '' }) => {
if (manifest.kind === 'read') {
const abs = jail(substitute(manifest.path!, arg));
const file = Bun.file(abs);
if (!(await file.exists())) throw new Error(`no such file: ${manifest.path}`);
return cap(await file.text());
}
if (manifest.kind === 'http') {
const url = substitute(manifest.url!, arg);
if (!/^https:\/\//i.test(url)) throw new Error(`http tools may only fetch https URLs, got: ${url}`);
const res = await fetch(url, { redirect: 'follow', signal: AbortSignal.timeout(20_000) });
if (!res.ok) throw new Error(`${url} returned ${res.status}`);
return cap(await res.text());
}
const command = substitute(manifest.command!, arg);
const blocked = await guard(command);
if (blocked) throw new Error(`refused: ${blocked}`);
const shell = process.platform === 'win32' ? ['cmd', '/c', command] : ['bash', '-lc', command];
const proc = Bun.spawn(shell, { stdout: 'pipe', stderr: 'pipe' });
const [out, err, code] = await Promise.all([
new Response(proc.stdout).text(),
new Response(proc.stderr).text(),
proc.exited,
]);
if (code !== 0) throw new Error(`exited ${code}: ${err.trim().slice(0, 300)}`);
return cap(out.trim() || '(no output)');
},
});
}
export type ExternalTools = { tools: ToolSet; autoApprove: string[]; errors: LoadError[] };
/** Loads every external tool manifest, project shadowing user by name. Bad files are reported and skipped. */
export async function loadExternalTools(cwd: string, guard: ShellGuard): Promise<ExternalTools> {
const tools: ToolSet = {};
const autoApprove: string[] = [];
const errors: LoadError[] = [];
for (const dir of dirs('tools', cwd)) {
for (const file of await scan(dir, 'json')) {
const fallback = file.replace(/\.json$/i, '');
try {
const manifest = parseToolManifest(await Bun.file(join(dir, file)).text());
tools[manifest.name] = manifestToTool(manifest, guard);
if (manifest.autoApprove !== false) autoApprove.push(manifest.name);
} catch (e) {
errors.push({ name: fallback, message: e instanceof Error ? e.message : String(e) });
}
}
}
return { tools, autoApprove, errors };
}
// ---------------------------------------------------------------------------
// External plugins, as refusal manifests (same shape the registry installs).
// ---------------------------------------------------------------------------
export type ExternalPlugins = { plugins: Plugin[]; errors: LoadError[] };
/** Loads refusal-manifest plugins from disk, merging with any already installed via the registry. */
export async function loadExternalPlugins(cwd: string): Promise<ExternalPlugins> {
const byName = new Map<string, Plugin>();
const errors: LoadError[] = [];
for (const dir of dirs('plugins', cwd)) {
for (const file of await scan(dir, 'json')) {
const fallback = file.replace(/\.json$/i, '');
try {
const manifest: PluginManifest = parseManifest(await Bun.file(join(dir, file)).text());
byName.set(manifest.name, manifestToPlugin(manifest));
} catch (e) {
errors.push({ name: fallback, message: e instanceof Error ? e.message : String(e) });
}
}
}
return { plugins: [...byName.values()], errors };
}
+88 -29
View File
@@ -3,6 +3,7 @@ import { render } from 'ink';
import React from 'react';
import type { LanguageModel, ModelMessage } from 'ai';
import { resolveAgent, VARIANTS, isThinkingLevel, type AgentVariant } from './agents';
import { loadExternalPlugins, loadExternalTools } from './autoload';
import { configPath, loadConfig, missingKeyMessage, resolveModel, writeConfigFile, type Config } from './config';
import type { FallbackEvent } from './fallback';
import { farewell } from './farewell';
@@ -18,12 +19,14 @@ import { createHost } from './plugins';
import { fetchModels, presetById } from './providers';
import * as registry from './registry';
import { Session } from './session';
import { loadCustomCommands } from './custom-commands';
import { loadSkills } from './skills';
import * as store from './store';
import { createTaskTool, type SubagentApproval } from './subagent';
import { VERSION, versionLine } from './version';
import { createAskBridge } from './ui/Ask';
import { App, createApprovalBridge, createNoticeBus, createSubagentBus, type AppHooks } from './ui/App';
import { Header, type HeaderFact } from './ui/Header';
import type { RegistryRow as AppRegistryRow } from './ui/Panels';
// SDK warnings go straight to stderr, which tears up the Ink render.
@@ -154,6 +157,7 @@ if (resumeArg) {
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
const instructions = has('--no-instructions') ? [] : await loadInstructions();
const skills = has('--no-skills') ? [] : await loadSkills();
const customCommands = await loadCustomCommands();
const promptHistory = await store.loadHistory();
const installedPlugins = has('--no-plugins') ? { plugins: [], errors: [] } : await registry.loadInstalledPlugins();
@@ -202,9 +206,29 @@ const enabledPlugins = has('--no-plugins') ? [] : (cfg.plugins ?? DEFAULT_ENABLE
const pluginErrors = enabledPlugins
.filter((name) => !BUILTIN_PLUGINS.some((p) => p.name === name))
.map((name) => ({ plugin: name, message: 'no such plugin' }));
// External skills, tools, and plugins auto-load from ~/.shiro-neko/<kind> and
// .shiro/<kind>. All are data, never code; a bad file is reported, not fatal.
const externalPlugins = has('--no-plugins') ? { plugins: [], errors: [] } : await loadExternalPlugins(process.cwd());
const plugins = createHost(
[...BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)), ...installedPlugins.plugins],
[...pluginErrors, ...installedPlugins.errors],
[
...BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)),
...installedPlugins.plugins,
...externalPlugins.plugins,
],
[
...pluginErrors,
...installedPlugins.errors,
...externalPlugins.errors.map((e) => ({ plugin: e.name, message: e.message })),
],
);
// External shell tools run through the same guard chain as a built-in bash call,
// so an installed tool cannot do what the agent itself may not. Late-bound because
// the host above is what runs the chain.
const externalTools = await loadExternalTools(process.cwd(), async (command) =>
plugins.guard({ toolName: 'bash', input: { command }, cwd: process.cwd() }),
);
const memory = has('--no-memory') ? undefined : new Memory(process.cwd(), languageModel);
@@ -261,8 +285,23 @@ const subagentGate: SubagentApproval = (req) => {
return approveSubagent(req);
};
// A subagent doing search rather than reasoning can run on a cheaper model.
// It resolves against the same provider and key, so a configured `subagentModel`
// never needs a second credential.
const subagentModel =
cfg.subagentModel && cfg.subagentModel !== cfg.model && cfg.apiKey
? resolveModel({ ...cfg, model: cfg.subagentModel }, reportFallback)
: (languageModel ?? unconfiguredModel);
// Late-bound like `approveSubagent`: the task tool is built into `extraTools`
// before the Session that owns the spend ledger exists, so the usage callback is
// wired after construction.
let recordSubagent: (usage: { inputTokens: number; outputTokens: number }) => void = () => {};
const session = new Session({
model: languageModel ?? unconfiguredModel,
modelId: cfg.model,
...(cfg.subagentModel ? { subagentModelId: cfg.subagentModel } : {}),
askApproval: bridge.ask,
yolo,
instructions,
@@ -276,8 +315,10 @@ const session = new Session({
...(memory ? { memory } : {}),
...(record.notebook ? { notebook: record.notebook } : {}),
...(cfg.maxRetries !== undefined ? { maxRetries: cfg.maxRetries } : {}),
...(cfg.maxSpendUsd !== undefined ? { maxSpendUsd: cfg.maxSpendUsd } : {}),
extraTools: {
...(mcp?.tools ?? {}),
...externalTools.tools,
git_commit_message: createCommitMessageTool({
model: languageModel ?? unconfiguredModel,
...(headless ? {} : { cwd: process.cwd() }),
@@ -287,6 +328,9 @@ const session = new Session({
: {
task: createTaskTool({
model: languageModel ?? unconfiguredModel,
subagentModel,
subagentModelId: cfg.subagentModel,
onUsage: (u) => recordSubagent(u),
...(headless ? {} : { report: subagents.emit }),
// A worker's writes go through the parent's rules and the parent's
// prompt. Headless has nobody to answer, so `worker` is withheld there
@@ -295,7 +339,7 @@ const session = new Session({
}),
}),
},
autoApprove: ['task', 'git_commit_message'],
autoApprove: ['task', 'git_commit_message', ...externalTools.autoApprove],
messages: [...record.messages],
onChange: (messages) => {
// Debounced so a long tool loop does not hit the disk on every step.
@@ -305,6 +349,7 @@ const session = new Session({
});
approveSubagent = session.approveForSubagent();
recordSubagent = (u) => session.recordSubagentUsage(u);
async function shutdown(code: number): Promise<never> {
clearTimeout(saveTimer);
@@ -344,6 +389,7 @@ const hooks: AppHooks = {
for await (const rel of walk({ limit: 5000 })) found.push(rel);
return found;
},
customCommands: () => customCommands,
registry: {
list: async () => {
const entries = await registry.fetchIndex(cfg.registryUrl);
@@ -548,35 +594,46 @@ const hooks: AppHooks = {
},
};
const header = [
needsProvider
? `shiro-neko ${VERSION} no provider configured`
: `shiro-neko ${VERSION} ${cfg.provider}/${record.model} session ${record.id.slice(0, 8)}`,
`agent: ${agentVariant.name} thinking: ${agentVariant.thinking}`,
`cwd: ${process.cwd()}`,
restored ? `resumed ${record.messages.length} messages` : undefined,
// The welcome dashboard's environment facts, in scan order. Anything that should
// stop the user — a failed plugin, `--yolo`, a missing key — is given a tone so it
// lifts out of the quiet metadata rather than blending into it.
const facts: HeaderFact[] = [
{ label: 'agent', value: `${agentVariant.name} thinking ${agentVariant.thinking}` },
restored ? { label: 'resumed', value: `${record.messages.length} messages` } : undefined,
instructions.length > 0
? `instructions: ${instructions.map((i) => i.path.split(/[\\/]/).at(-1)).join(', ')}`
: 'no AGENTS.md found - /init writes one',
skills.length > 0 ? `skills: ${skills.map((s) => s.name).join(', ')}` : undefined,
plugins.plugins.length > 0 ? `plugins: ${plugins.plugins.map((p) => p.name).join(', ')}` : undefined,
...plugins.errors.map((e) => `plugin ${e.plugin}: ${e.message}`),
memory && memory.all().length > 0 ? `memory: ${memory.all().length} notes about this project` : undefined,
mcp && Object.keys(mcp.tools).length > 0 ? `mcp: ${Object.keys(mcp.tools).length} tools` : undefined,
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
? `mcp: ${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`
? { label: 'instructions', value: instructions.map((i) => i.path.split(/[\\/]/).at(-1)!).join(', ') }
: { label: 'instructions', value: 'none - /init writes an AGENTS.md', tone: 'info' },
skills.length > 0 ? { label: 'skills', value: skills.map((s) => s.name).join(', ') } : undefined,
plugins.plugins.length > 0
? { label: 'plugins', value: plugins.plugins.map((p) => p.name).join(', ') }
: undefined,
...(mcp?.errors ?? []).map((e) => `mcp ${e.server} failed: ${e.message}`),
...plugins.errors.map((e) => ({ label: 'plugin error', value: `${e.plugin}: ${e.message}`, tone: 'err' as const })),
memory && memory.all().length > 0
? { label: 'memory', value: `${memory.all().length} notes about this project` }
: undefined,
mcp && Object.keys(mcp.tools).length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).length} tools` } : undefined,
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
: undefined,
...(mcp?.errors ?? []).map((e) => ({ label: 'mcp error', value: `${e.server}: ${e.message}`, tone: 'err' as const })),
yolo
? 'approvals: OFF (--yolo), but deny rules and the guard still apply'
? { label: 'approvals', value: 'OFF (--yolo) - deny rules and the guard still apply', tone: 'warn' as const }
: cfg.permission
? `approvals: rules for ${Object.keys(cfg.permission).join(', ')}, defaults elsewhere`
: 'approvals: ask for write_file, edit_file, multi_edit, apply_patch, move_file, delete_file, bash, web_fetch, mcp__*',
cfg.toolSets ? `tool sets: core, ${cfg.toolSets.join(', ')}` : undefined,
'/help for commands',
]
.filter(Boolean)
.join('\n');
? { label: 'approvals', value: `rules for ${Object.keys(cfg.permission).join(', ')}, defaults elsewhere` }
: { label: 'approvals', value: 'ask for writes, bash, web_fetch, mcp' },
cfg.toolSets ? { label: 'tool sets', value: `core, ${cfg.toolSets.join(', ')}` } : undefined,
].filter((f): f is HeaderFact => f !== undefined);
const headerNode = (
<Header
version={VERSION}
{...(needsProvider ? {} : { provider: cfg.provider, model: record.model })}
sessionId={record.id.slice(0, 8)}
cwd={process.cwd()}
title={restored ? record.title : undefined}
facts={facts}
/>
);
// ctrl-c has to reach the App: with a command running it kills that command and
// keeps the turn. Ink's own handler would exit the process before we saw the key.
@@ -584,7 +641,9 @@ const app = render(
<App
session={session}
bridge={bridge}
header={header}
header=""
headerNode={headerNode}
version={VERSION}
hooks={hooks}
notices={notices}
askBridge={askBridge}
+18 -6
View File
@@ -1,3 +1,5 @@
import type { CustomCommand } from './custom-commands';
export type CommandAction =
| { type: 'none' }
| { type: 'prompt'; text: string }
@@ -24,6 +26,8 @@ export type CommandAction =
| { type: 'info'; text: string }
| { type: 'model'; model: string }
| { type: 'resume'; id: string }
/** A custom command from a markdown file, expanded against its arguments. */
| { type: 'custom'; command: CustomCommand; args: string[] }
| { type: 'unknown'; name: string };
export type CommandSpec = {
@@ -81,11 +85,12 @@ export const HELP = [
* An exact name sorts first so pressing enter on `/model` cannot run `/models`.
* Aliases stay hidden to keep the list short.
*/
export function matchCommands(input: string): CommandSpec[] {
export function matchCommands(input: string, custom: readonly CustomCommand[] = []): CommandSpec[] {
if (!input.startsWith('/')) return [];
const typed = input.slice(1).toLowerCase();
if (typed.includes(' ')) return [];
const hits = COMMANDS.filter((c) => c.name.startsWith(typed));
const customSpecs: CommandSpec[] = custom.map((c) => ({ name: c.name, summary: c.description }));
const hits = [...COMMANDS, ...customSpecs].filter((c) => c.name.startsWith(typed));
const exact = hits.findIndex((c) => c.name === typed);
return exact > 0 ? [hits[exact]!, ...hits.filter((_, i) => i !== exact)] : hits;
}
@@ -156,8 +161,13 @@ function parseMcp(arg: string): CommandAction {
}
}
/** Pure parser: no IO, so the TUI and headless mode share one definition. */
export function parseCommand(raw: string): CommandAction {
/**
* Pure parser: no IO, so the TUI and headless mode share one definition.
*
* Custom commands are consulted only after every built-in name misses, so a
* markdown file can add a command but never shadow one that ships with the binary.
*/
export function parseCommand(raw: string, custom: readonly CustomCommand[] = []): CommandAction {
const input = raw.trim();
if (!input) return { type: 'none' };
if (!input.startsWith('/')) return { type: 'prompt', text: input };
@@ -215,7 +225,9 @@ export function parseCommand(raw: string): CommandAction {
return arg ? { type: 'model', model: arg } : { type: 'models' };
case 'resume':
return arg ? { type: 'resume', id: arg } : { type: 'info', text: 'usage: /resume <session-id>' };
default:
return { type: 'unknown', name };
default: {
const cmd = custom.find((c) => c.name === name);
return cmd ? { type: 'custom', command: cmd, args: arg ? arg.split(/\s+/) : [] } : { type: 'unknown', name };
}
}
}
+6
View File
@@ -20,6 +20,10 @@ export type Config = {
presetId?: string;
/** Retries per model call for transient failures. SDK default is 2. */
maxRetries?: number;
/** USD ceiling for a session's spend: warn at 80%, refuse the next turn at 100%. */
maxSpendUsd?: number;
/** Model id for subagents; omit to share the parent's. */
subagentModel?: string;
/** Default agent variant name. */
agent?: string;
/** Default thinking level. */
@@ -90,6 +94,8 @@ export async function loadConfig(): Promise<Config> {
apiKey: process.env['SHIRO_API_KEY'] ?? file.apiKey ?? process.env[ENV_KEY[provider]],
...(file.presetId ? { presetId: file.presetId } : {}),
...(file.maxRetries !== undefined ? { maxRetries: file.maxRetries } : {}),
...(typeof file.maxSpendUsd === 'number' && file.maxSpendUsd > 0 ? { maxSpendUsd: file.maxSpendUsd } : {}),
...(file.subagentModel ? { subagentModel: file.subagentModel } : {}),
...(file.agent ? { agent: file.agent } : {}),
...(file.thinking ? { thinking: file.thinking } : {}),
...(Array.isArray(file.plugins) ? { plugins: file.plugins } : {}),
+121
View File
@@ -0,0 +1,121 @@
import { homedir } from 'node:os';
import { join } from 'node:path';
import { guardPlugin } from './plugins-builtin';
/**
* Custom slash commands read from markdown files.
*
* `.shiro/commands/<name>.md` in the project and `~/.shiro-neko/commands/<name>.md`
* for the user. The filename is the command; the body becomes the prompt. A project
* command shadows a user command of the same name, so a repo can specialise a
* personal default.
*/
export type CustomCommand = {
name: string;
/** One-line summary for the `/` menu, from frontmatter or the first body line. */
description: string;
/** Agent to run it under, when frontmatter sets one. */
agent?: string;
/** The prompt template, before substitution. */
body: string;
origin: 'project' | 'user';
path: string;
};
const MAX_BODY = 20_000;
/** Reads frontmatter `description` and `agent`; everything after the `---` fence is the prompt. */
function parse(name: string, source: string, origin: CustomCommand['origin'], path: string): CustomCommand | undefined {
const match = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/.exec(source.trimStart());
const meta: Record<string, string> = {};
let body = source;
if (match) {
for (const line of match[1]!.split(/\r?\n/)) {
const kv = /^([A-Za-z_-]+)\s*:\s*(.*)$/.exec(line.trim());
if (kv) meta[kv[1]!.toLowerCase()] = kv[2]!.replace(/^["']|["']$/g, '').trim();
}
body = match[2]!;
}
const trimmed = body.trim().slice(0, MAX_BODY);
if (!trimmed) return undefined;
const description = meta['description'] ?? trimmed.split('\n').find((l) => l.trim().length > 0)?.trim().slice(0, 60) ?? name;
return {
name,
description,
...(meta['agent'] ? { agent: meta['agent'] } : {}),
body: trimmed,
origin,
path,
};
}
function commandDirs(cwd: string): { dir: string; origin: CustomCommand['origin'] }[] {
const home = join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko');
return [
{ dir: join(home, 'commands'), origin: 'user' },
{ dir: join(cwd, '.shiro', 'commands'), origin: 'project' },
];
}
/** Loads every custom command, project shadowing user by name. A file that fails to parse is skipped. */
export async function loadCustomCommands(cwd = process.cwd()): Promise<CustomCommand[]> {
const byName = new Map<string, CustomCommand>();
for (const { dir, origin } of commandDirs(cwd)) {
let files: string[] = [];
try {
for await (const f of new Bun.Glob('*.md').scan({ cwd: dir, onlyFiles: true })) files.push(f);
} catch {
continue;
}
for (const file of files.sort()) {
const name = file.replace(/\.md$/i, '');
if (!/^[a-z0-9][a-z0-9-_]*$/i.test(name)) continue;
const path = join(dir, file);
try {
const cmd = parse(name, await Bun.file(path).text(), origin, path);
if (cmd) byName.set(cmd.name, cmd);
} catch {
continue;
}
}
}
return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name));
}
/** Runs a `` !`cmd` `` substitution through the guard before executing it. */
async function runSubstitution(command: string): Promise<string> {
const blocked = await guardPlugin.beforeToolCall!({ toolName: 'bash', input: { command }, cwd: process.cwd() });
if (blocked) throw new Error(`shell substitution refused: ${blocked}`);
// The same shell bash uses, so a substitution and a bash call agree on syntax.
const shell = process.platform === 'win32' ? ['cmd', '/c', command] : ['bash', '-lc', command];
const proc = Bun.spawn(shell, { stdout: 'pipe', stderr: 'pipe' });
const [out, err, code] = await Promise.all([
new Response(proc.stdout).text(),
new Response(proc.stderr).text(),
proc.exited,
]);
if (code !== 0) throw new Error(`shell substitution \`!${command}\` exited ${code}: ${err.trim().slice(0, 200)}`);
return out.trim();
}
/**
* Expands a command's body against the arguments it was typed with.
*
* `$ARGUMENTS` is the whole argument string, `$1`, `$2`, … the positionals, and
* `` !`cmd` `` runs a shell command and inlines its output — each such command
* passed through the guard first, so a custom command cannot smuggle a destructive
* call past the user the way a plain bash call cannot.
*/
export async function expandCommand(cmd: CustomCommand, args: string[]): Promise<string> {
let out = cmd.body;
out = out.replaceAll('$ARGUMENTS', args.join(' '));
out = out.replace(/\$(\d+)/g, (_, i) => args[Number(i) - 1] ?? '');
const substitutions = [...out.matchAll(/!`([^`]+)`/g)];
for (const m of substitutions) {
const value = await runSubstitution(m[1]!);
out = out.replace(m[0], value);
}
return out.trim();
}
+133 -3
View File
@@ -218,6 +218,122 @@ export const formatPlugin: Plugin = {
},
};
/**
* Bash command patterns a guard refuses, shared by several small plugins.
*
* Each plugin owns one concern so it can be toggled alone; they are data (a name,
* a pattern list, an appendix), never code beyond the matcher they all share.
*/
const bashRefusal = (patterns: { re: RegExp; why: string }[]) => {
return ({ toolName, input }: Parameters<NonNullable<Plugin['beforeToolCall']>>[0]) => {
if (toolName !== 'bash') return undefined;
const command = String((input as { command?: unknown } | null)?.command ?? '');
if (!command) return undefined;
for (const { re, why } of patterns) {
if (re.test(command)) return `refusing "${command.slice(0, 120)}" (${why}). Run it yourself if it is really needed.`;
}
return undefined;
};
};
export const noForcePushPlugin: Plugin = {
name: 'no-force-push',
description: 'refuses any push that rewrites remote history',
appendix: 'The no-force-push plugin refuses force pushes. Ask the user to run one by hand if it is truly intended.',
beforeToolCall: bashRefusal([
{ re: /\bgit\s+push\b[^|]*(--force\b|--force-with-lease\b|\s-f\b)/, why: 'rewrites remote history' },
{ re: /\bgit\s+push\b[^|]*\s+\+/, why: 'a force push via refspec' },
]),
};
export const noMainCommitPlugin: Plugin = {
name: 'no-main-commit',
description: 'refuses to commit directly to main or master',
appendix: 'The no-main-commit plugin refuses to commit to main/master. Create a branch and commit there instead.',
beforeToolCall: bashRefusal([
{ re: /\bgit\s+(commit|merge)\b[^|]*\b(main|master)\b/, why: 'touches the default branch directly' },
{ re: /\bgit\s+checkout\s+(main|master)\b[^|]*&&[^|]*\bcommit\b/, why: 'commits on the default branch' },
]),
};
export const noRootPlugin: Plugin = {
name: 'no-root',
description: 'refuses commands run with sudo or as an elevated shell',
appendix: 'The no-root plugin refuses sudo and elevation. Nothing the agent does should need it; ask the user to run it themselves.',
beforeToolCall: bashRefusal([
{ re: /(^|\s)sudo\b/, why: 'elevated privileges' },
{ re: /\brunas\b|\bStart-Process\b[^|]*-Verb\s+RunAs/i, why: 'an elevated process' },
]),
};
export const noNetPipePlugin: Plugin = {
name: 'no-net-pipe',
description: 'refuses to execute anything downloaded straight into a shell',
appendix: 'The no-net-pipe plugin refuses piping a download into an interpreter. Download, review the file, then run it.',
beforeToolCall: bashRefusal([
{ re: /\b(curl|wget)\b[^|]*\|\s*(ba|z|k)?sh\b|\b(curl|wget)\b[^|]*\|\s*(node|python|ruby|perl|bun)\b/i, why: 'executes a download unseen' },
{ re: /\biex\b|\bInvoke-Expression\b[^|]*\b(iwr|Invoke-WebRequest|curl)\b/i, why: 'executes a download unseen' },
]),
};
export const noGitConfigPlugin: Plugin = {
name: 'no-git-config',
description: 'refuses to change git configuration or global state',
appendix: 'The no-git-config plugin refuses to edit git config. Tell the user the exact config change to make themselves.',
beforeToolCall: bashRefusal([
{ re: /\bgit\s+config\b[^|]*(--global|--system)/, why: 'changes global git configuration' },
{ re: /\bgit\s+config\b[^|]*(user\.(name|email)|core\.(sshCommand|editor|pager))\s+\S/, why: 'changes how git identifies or runs' },
]),
};
export const noEnvWritePlugin: Plugin = {
name: 'no-env-write',
description: 'refuses to print or export secrets into the shell environment',
appendix: 'The no-env-write plugin refuses to export or echo credentials into the environment. The user sets their own secrets.',
beforeToolCall: bashRefusal([
{ re: /\b(export|setx?)\s+[A-Z_]*(KEY|TOKEN|SECRET|PASSWORD|PASSWD)\s*=/i, why: 'writes a credential into the environment' },
{ re: /\becho\b[^|]*\b(api[_-]?key|secret|token|password)\b[^|]*>>?\s*\S/i, why: 'writes a credential to a file' },
]),
};
export const conventionalCommitPlugin: Plugin = {
name: 'conventional-commit',
description: 'nudges commit messages toward the conventional format',
appendix:
'The conventional-commit plugin is advisory: write commit subjects as type(scope): summary, e.g. ' +
'`fix(auth): reject expired tokens`. Types: feat, fix, docs, style, refactor, perf, test, build, ci, chore, revert.',
};
export const testsFirstPlugin: Plugin = {
name: 'tests-first',
description: 'reminds the agent to pin behaviour with a failing test before fixing',
appendix:
'The tests-first plugin is advisory: for a bug, write or find the test that reproduces it before changing code. ' +
'Watch it fail, then fix, then watch it pass. A fix without a failing-then-passing test is unverified.',
};
export const smallDiffsPlugin: Plugin = {
name: 'small-diffs',
description: 'reminds the agent to keep a change focused on one thing',
appendix:
'The small-diffs plugin is advisory: one change does one thing. Do not tidy, rename, or reformat outside the ' +
'task. A diff that is hard to review is usually two diffs wearing one coat — split it.',
};
export const confirmDeletePlugin: Plugin = {
name: 'confirm-delete',
description: 'refuses delete calls that name broad or ambiguous paths',
appendix: 'The confirm-delete plugin refuses deletes that name a directory or a wildcard. Delete one explicit file at a time.',
beforeToolCall: ({ toolName, input }) => {
if (toolName !== 'delete_file') return undefined;
const path = String((input as { path?: unknown } | null)?.path ?? '');
if (/[*?[\]]/.test(path) || path.endsWith('/') || path === '.' || path === '') {
return `refusing to delete "${path}" (ambiguous or broad). Delete one explicit file.`;
}
return undefined;
},
};
export const BUILTIN_PLUGINS: Plugin[] = [
guardPlugin,
secretsPlugin,
@@ -225,15 +341,29 @@ export const BUILTIN_PLUGINS: Plugin[] = [
bellPlugin,
timePlugin,
formatPlugin,
noForcePushPlugin,
noMainCommitPlugin,
noRootPlugin,
noNetPipePlugin,
noGitConfigPlugin,
noEnvWritePlugin,
conventionalCommitPlugin,
testsFirstPlugin,
smallDiffsPlugin,
confirmDeletePlugin,
];
/**
* Enabled unless the config turns them off.
*
* `guard`, `secrets`, and `protect` are refusals, so they are on: a user who has to
* opt into a safety check does not have it. `bell` and `format` both act on their
* own — one makes noise, the other writes files — so they are opt-in.
* opt into a safety check does not have it. The four narrow safety refusals
* (`no-force-push`, `no-net-pipe`, `no-root`, `no-env-write`) are on for the same
* reason — each blocks a single irreversible class of mistake. `bell` and `format`
* act on their own, and the advisory/opinionated plugins (`no-main-commit`,
* `conventional-commit`, `tests-first`, `small-diffs`, `confirm-delete`,
* `no-git-config`) encode a workflow preference, so all of those are opt-in.
*/
export const DEFAULT_ENABLED = ['guard', 'secrets', 'protect', 'time'];
export const DEFAULT_ENABLED = ['guard', 'secrets', 'protect', 'time', 'no-force-push', 'no-net-pipe', 'no-root', 'no-env-write'];
export { DESTRUCTIVE, SECRET_PATHS, PROTECTED_PATHS };
+49 -4
View File
@@ -43,6 +43,28 @@ const TOOL_DOCS: ToolDoc[] = [
name: 'grep',
line: 'search contents. Prefer it over reading many files; scope with include to keep results small.',
},
{ name: 'find_symbol', line: 'jump to where a function, class, or type is defined. Use it before grep when you want a declaration, not every use.' },
{ name: 'json_query', line: 'read one value from a JSON file by dotted path, e.g. scripts.build, instead of reading it whole.' },
{ name: 'insert_lines', line: 'insert a block at a line number, pushing the rest down. Cheaper than a rewrite for adding to the middle of a file.' },
{ name: 'delete_lines', line: 'delete a line range. Refuses the whole file; use delete_file for that.' },
{ name: 'replace_lines', line: 'replace a line range with new text in one write.' },
{ name: 'append_file', line: 'add to the end of a file without a full rewrite.' },
{ name: 'prepend_file', line: 'add to the top of a file, e.g. a header or an import block.' },
{ name: 'count_lines', line: 'line counts for one file or a glob. A size read before opening something large.' },
{ name: 'tree', line: 'indented directory tree, ignore-aware. Scan a broad shape faster than list_dir.' },
{ name: 'file_info', line: 'size, line count, modified time, text or binary, for one file.' },
{ name: 'find_files', line: 'find files whose name contains a substring, e.g. "auth". Not a glob.' },
{ name: 'recent_files', line: 'files modified most recently. Find what a tool just touched.' },
{ name: 'changed_files', line: 'the working-tree delta git reports, at a glance.' },
{ name: 'git_log_file', line: 'commits that touched one file, newest first.' },
{ name: 'git_diff_commits', line: 'diff between two refs, optionally one path.' },
{ name: 'git_show_file', line: 'a file\'s contents at a ref, e.g. auth.ts at HEAD~3.' },
{ name: 'git_current_branch', line: 'current branch with upstream and ahead/behind.' },
{ name: 'git_changed_in_ref', line: 'files changed between a ref and the working tree, names only.' },
{ name: 'outline', line: 'top-level declarations of a source file. Read it before opening a large file.' },
{ name: 'read_symbol', line: 'the full body of one definition by name.' },
{ name: 'env_info', line: 'platform, shell, and which runtimes are installed, before writing a command.' },
{ name: 'count_tokens', line: 'estimate the token cost of a file or string before sending it.' },
{
name: 'edit_file',
line: 'oldString must match byte-for-byte including indentation, and be unique. Include surrounding lines to disambiguate. Prefer several small edits over one large rewrite.',
@@ -143,6 +165,7 @@ export function systemPrompt(parts: PromptParts): string {
const toolNames = availableTools ?? TOOL_DOCS.map((d) => d.name);
const canRun = toolNames.includes('bash');
const canDelegate = toolNames.includes('task');
const approvalTools = toolNames.filter((name) =>
['write_file', 'edit_file', 'multi_edit', 'apply_patch', 'move_file', 'delete_file', 'bash', 'web_fetch'].includes(
name,
@@ -150,19 +173,35 @@ export function systemPrompt(parts: PromptParts): string {
);
const workflow = [
'- Read before you write. Ground every claim about the code in something you actually opened.',
'- Make the smallest change that solves the task. A bugfix diff contains only the bug.',
'- Read before you write. Ground every claim about the code in something you actually opened. Never describe code you have not read.',
'- Make the smallest change that solves the task. A bugfix diff contains only the bug; a feature diff contains only the feature.',
'- Match the existing style, libraries, and conventions. Sample a neighbouring file before inventing a pattern.',
approvalTools.length > 0
? `- ${approvalTools.join(', ')} need the user to approve each call. If one is denied, stop and ask what to do instead of working around it.`
: '- You have no tools that change anything this turn. Investigate and report; do not describe edits as if you had made them.',
canRun
? "- After changing code, verify it: run the project's build or tests. \"Should work\" is not verification."
? "- After changing code, verify it: run the project's build or tests. \"Should work\" is not verification; output you saw is."
: '- You cannot run commands this turn, so say what should be run to verify rather than claiming it passes.',
'- When something fails twice, stop and re-read the error literally. Check that the code you think is running is the code that is running.',
].join('\n');
// The failure loop is its own block so a stuck model has a procedure, not a vague
// instruction to "try harder". Written as discrete steps because a model in a loop
// needs an exit, not encouragement.
const recovery = [
'- Fail once: read the error literally and fix the thing it names, not the thing you expected.',
'- Fail twice on the same attempt: stop. Confirm the code running is the code you think — right file, fresh build, no stale cache or shadowed import.',
'- Fail three times: change strategy, not parameters. Reproduce smaller, print the value at the failure point, or ask. Do not re-run the same call hoping for a different result.',
].join('\n');
const delegation = canDelegate
? `- Delegate with task for a search across many files or a self-contained change you need not watch. Its prompt must stand alone — it sees none of this conversation. Keep work you must supervise in your own turn.`
: '';
const workflow2 = [
canAsk
? '- Ask rather than guess when two readings of the request lead to different work. Decide small things yourself and say what you assumed.'
: '- No one can answer a question this run. Decide yourself and state the assumption plainly.',
'- Long sessions compact as context fills. Record what stays true with remember; restate the goal on a long task.',
].join('\n');
return `You are Shiro Neko, a coding agent working in the user's terminal.
@@ -178,6 +217,12 @@ ${renderTools(toolNames)}
How to work
${workflow}
When something fails
${recovery}
${delegation ? `\nDelegating\n${delegation}\n` : ''}
Working with the user
${workflow2}
How to reply
- Lead with the outcome. The user wants to know what happened, not what you are about to do.
- No preamble, no restating the task, no summary of your own summary.
+67
View File
@@ -15,6 +15,7 @@ import type { Memory } from './memory';
import { Notebook, type NotebookState } from './notebook';
import { Permissions, type PermissionConfig } from './permission';
import type { PluginHost } from './plugins';
import { costOf, formatUsd } from './pricing';
import { systemPrompt } from './prompt';
import { detachProviderItems, pruneToFit } from './prune';
import { createSkillTool, renderSkills, type Skill } from './skills';
@@ -53,10 +54,16 @@ export type AgentEvent =
export type SessionOptions = {
model: LanguageModel;
/** Model id, for pricing the session's spend against the ceiling. */
modelId?: string;
/** Subagent model id, when it differs; its spend prices against this. */
subagentModelId?: string;
askApproval: (req: ApprovalRequest) => Promise<ApprovalDecision>;
yolo?: boolean;
cwd?: string;
maxSteps?: number;
/** USD ceiling: warn at 80%, refuse the next turn at 100%. */
maxSpendUsd?: number;
/** MCP and subagent tools merged on top of the built-ins. */
extraTools?: ToolSet;
/** Tool sets offered this session; omit for all of them. `core` is always on. */
@@ -116,6 +123,9 @@ export class Session {
readonly notebook: Notebook;
inputTokens = 0;
outputTokens = 0;
/** Subagent token use, priced against the subagent's own model id in /cost. */
subagentInputTokens = 0;
subagentOutputTokens = 0;
private model: LanguageModel;
private variant: AgentVariant;
private readonly permissions: Permissions;
@@ -123,6 +133,8 @@ export class Session {
private readonly seen = new Map<string, number>();
/** One stale-item repair per turn, so a repeating 404 cannot loop the run. */
private staleItemsRepaired = false;
/** The 80% spend warning is shown once, not on every turn past the line. */
private warnedSpend = false;
private controller: AbortController | undefined;
constructor(private readonly opts: SessionOptions) {
@@ -213,10 +225,19 @@ export class Session {
this.messages.length = 0;
this.inputTokens = 0;
this.outputTokens = 0;
this.subagentInputTokens = 0;
this.subagentOutputTokens = 0;
this.warnedSpend = false;
this.notebook.clear();
this.opts.onChange?.(this.messages);
}
/** A subagent's finished run, folded into the session's spend and the /cost split. */
recordSubagentUsage(usage: { inputTokens: number; outputTokens: number }): void {
this.subagentInputTokens += usage.inputTokens;
this.subagentOutputTokens += usage.outputTokens;
}
replace(messages: ModelMessage[]): void {
this.messages.length = 0;
this.messages.push(...messages);
@@ -236,6 +257,27 @@ export class Session {
return this.opts.compactThreshold ?? DEFAULT_COMPACT_THRESHOLD;
}
/**
* The session's spend so far and the configured ceiling, for the UI's status
* and the refuse-the-next-turn check. Unpriced models report no spend: a
* ceiling cannot be enforced against a model we cannot price.
*/
spend(): { usd?: number; ceiling?: number; overWarn: boolean; overLimit: boolean } {
const ceiling = this.opts.maxSpendUsd;
const parent = costOf(this.opts.modelId ?? '', this.inputTokens, this.outputTokens);
const sub =
this.subagentInputTokens + this.subagentOutputTokens > 0
? costOf(this.opts.subagentModelId ?? this.opts.modelId ?? '', this.subagentInputTokens, this.subagentOutputTokens)
: 0;
// Spend is only knowable when every part is priced; an unpriced piece means
// the total is a lower bound, so the ceiling is not enforced against it.
const usd = parent === undefined || sub === undefined ? undefined : parent + sub;
if (ceiling === undefined || usd === undefined) {
return { ...(usd !== undefined ? { usd } : {}), ...(ceiling !== undefined ? { ceiling } : {}), overWarn: false, overLimit: false };
}
return { usd, ceiling, overWarn: usd >= ceiling * 0.8, overLimit: usd >= ceiling };
}
private systemFor(): string {
return systemPrompt({
cwd: this.opts.cwd ?? process.cwd(),
@@ -337,6 +379,21 @@ export class Session {
}
async *send(userText: string): AsyncGenerator<AgentEvent> {
// The ceiling is checked before the model is: a turn started past the limit
// would spend money the caller said not to. An unpriced model cannot be
// measured, so it is never refused here — the ceiling simply cannot see it.
const spend = this.spend();
if (spend.overLimit) {
yield {
type: 'error',
error: new Error(
`spend ceiling reached: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used. Raise maxSpendUsd or start a new session.`,
),
};
yield { type: 'done' };
return;
}
this.messages.push({ role: 'user', content: userText });
this.opts.onChange?.(this.messages);
this.controller = new AbortController();
@@ -528,6 +585,16 @@ export class Session {
const usage = await result.usage;
this.inputTokens += usage.inputTokens ?? 0;
this.outputTokens += usage.outputTokens ?? 0;
// Warn as the ceiling comes into view, once, so a long session is not
// surprised by a refusal it never saw coming.
const spend = this.spend();
if (spend.overWarn && !this.warnedSpend) {
this.warnedSpend = true;
yield {
type: 'notice',
text: `approaching spend ceiling: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used`,
};
}
yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens };
return;
}
+64 -414
View File
@@ -1,420 +1,70 @@
/**
* Skills bundled with the binary.
*
* These are string constants rather than files on disk because `bun build --compile`
* only embeds modules reachable through imports; a directory of .md files would be
* missing from the shipped binary.
* Each skill is a Markdown file in `src/skills-md/`, loaded here as a raw-text import.
* The `.md` file is the single source of truth — frontmatter and body in proper
* Markdown — so skills are edited and reviewed as Markdown, not as escaped strings
* inside TypeScript. Bun inlines every text import into the compiled binary, so the
* folder ships with `bun build --compile` exactly as the old string constants did.
*/
import accessibility from './skills-md/accessibility.md' with { type: 'text' };
import apiDesign from './skills-md/api-design.md' with { type: 'text' };
import ciCd from './skills-md/ci-cd.md' with { type: 'text' };
import commit from './skills-md/commit.md' with { type: 'text' };
import data from './skills-md/data.md' with { type: 'text' };
import db from './skills-md/db.md' with { type: 'text' };
import debug from './skills-md/debug.md' with { type: 'text' };
import deps from './skills-md/deps.md' with { type: 'text' };
import docker from './skills-md/docker.md' with { type: 'text' };
import docs from './skills-md/docs.md' with { type: 'text' };
import frontend from './skills-md/frontend.md' with { type: 'text' };
import gitWorkflow from './skills-md/git-workflow.md' with { type: 'text' };
import i18n from './skills-md/i18n.md' with { type: 'text' };
import incident from './skills-md/incident.md' with { type: 'text' };
import logging from './skills-md/logging.md' with { type: 'text' };
import migrate from './skills-md/migrate.md' with { type: 'text' };
import onboarding from './skills-md/onboarding.md' with { type: 'text' };
import optimizeSql from './skills-md/optimize-sql.md' with { type: 'text' };
import perf from './skills-md/perf.md' with { type: 'text' };
import perfFrontend from './skills-md/perf-frontend.md' with { type: 'text' };
import plan from './skills-md/plan.md' with { type: 'text' };
import readme from './skills-md/readme.md' with { type: 'text' };
import refactor from './skills-md/refactor.md' with { type: 'text' };
import release from './skills-md/release.md' with { type: 'text' };
import review from './skills-md/review.md' with { type: 'text' };
import security from './skills-md/security.md' with { type: 'text' };
import test from './skills-md/test.md' with { type: 'text' };
import uxCopy from './skills-md/ux-copy.md' with { type: 'text' };
import verify from './skills-md/verify.md' with { type: 'text' };
export const BUILTIN_SKILLS: { name: string; source: string }[] = [
{
name: 'debug',
source: `---
name: debug
description: Track down a bug whose cause is not obvious. Use when a test fails for unclear reasons, behaviour differs between environments, or an earlier fix did not hold.
---
# Debugging
Do not guess. A guess that happens to work leaves the real cause in place.
## Reproduce first
Find the smallest command that shows the failure and record it with \`remember\`. If you
cannot reproduce it, say so and ask what the user did differently — do not proceed on a
hypothesis you cannot test.
## Three hypotheses, then evidence
Write down at least three causes that would produce this exact symptom. Rank them by how
cheap they are to disprove, then disprove them in that order. State which one you are
testing before you test it.
Evidence means observed output: a log line, a failing assertion, a value printed at the
point of failure. "It should be X" is not evidence.
## Bisect when the space is large
- Recent regression: check what changed last.
- Unclear layer: assert the value at each boundary until one is wrong.
- Intermittent: run it in a loop and capture the failing case, do not reason about it abstractly.
## Fix the cause
Once you know the cause, fix that and nothing else. Do not tidy surrounding code in the
same change — a bugfix diff should contain only the bug.
Write a test that fails before the fix and passes after. If you cannot express the bug as
a test, say why.
## After two failed attempts
Stop. Re-read the error text literally, character by character. Check your assumption
about which code is actually running: the wrong file, a stale build, a shadowed import,
or a cached dependency accounts for most "impossible" bugs.
`,
},
{
name: 'review',
source: `---
name: review
description: Review a diff or a file for defects. Use when asked to review, critique, or check code before it ships.
---
# Code review
Severity order. Do not lead with style.
1. **Incorrect behaviour** — wrong result, wrong edge case, wrong state after failure.
2. **Missing validation at trust boundaries** — user input, network responses, file contents,
anything crossing a process line. Internal calls need no defensive checks.
3. **Security** — injection, path traversal, secrets in logs or errors, missing authz.
4. **Resource handling** — unclosed handles, unbounded growth, unawaited promises.
5. **Clarity** — only when it will cause a future defect.
## For each finding
State file and line, what breaks, and the change. Show the fix as code when it is short.
Skip anything a formatter would fix. Skip preference. If a choice is defensible, leave it.
## Say when it is fine
A review that invents problems to look thorough is worse than a short one. If the change
is correct, say so and stop.
## Verify, do not assume
Read the surrounding code before calling something a bug. A "missing" null check often
exists one level up. Run the tests if that is what settles it.
`,
},
{
name: 'refactor',
source: `---
name: refactor
description: Restructure code without changing behaviour. Use when asked to refactor, clean up, extract, or reorganise.
---
# Refactoring
Behaviour must not change. That is the whole constraint.
## Establish the safety net first
Run the existing tests and record that they pass. If the code has no tests, write one that
pins current behaviour — including the ugly parts — before touching anything. Refactoring
untested code is rewriting it.
## Then move in small steps
One transformation at a time, tests green between each. Rename, then extract, then move —
not all three in one edit. A large refactor that fails leaves you unable to tell which step
broke it.
## What not to do
- Do not fix bugs while refactoring. Note them, finish, fix separately.
- Do not add abstraction for a single caller. Duplication beats a premature interface.
- Do not widen the scope. The request was this code, not its neighbours.
- Do not change public API unless asked; if it must change, say so first.
## Done means
Tests pass, behaviour is identical, and the diff is smaller than the reader feared.
`,
},
{
name: 'test',
source: `---
name: test
description: Write or repair tests. Use when adding coverage, fixing a flaky test, or asked how something should be tested.
---
# Testing
A test earns its place by failing when the code is wrong.
## Match the project
Read two existing test files first. Use their runner, their assertion style, their file
layout, their naming. A test that looks foreign is a test nobody maintains.
## Test behaviour, not implementation
Assert on what a caller observes. A test that reaches into private state breaks on every
refactor and catches nothing.
Cover: the normal case, the boundaries, and the failure. Failure cases catch more real
defects than happy paths.
## Never do this
- Do not assert what the code currently returns without knowing it is correct — that pins
the bug.
- Do not weaken an assertion to make a test pass. If it fails, either the code or the
expectation is wrong; find out which.
- Do not delete a failing test. It is telling you something.
## Flaky tests
A test that passes alone and fails in a suite is a shared-state problem: a global, a
temp directory, a port, an unawaited promise, or ordering. Find which, do not add a retry.
## Verify
Run the test and watch it fail before the fix, pass after. A test you never saw fail is
not known to work.
`,
},
{
name: 'verify',
source: `---
name: verify
description: Confirm a change actually works by using it, not by reading it. Use before reporting a task complete, or when asked whether something works.
---
# Verification
A green test suite says the tests pass. It does not say the feature works.
## Run the artifact, not the source
Build it and use it the way a user would:
- **CLI** — build the binary and run it. Happy path, bad input, \`--help\`. Read the output.
- **HTTP service** — start it and \`curl\` the endpoint. Check the status and the body.
- **Library** — write a throwaway script that imports and calls the new code end to end.
- **Script or job** — run it against real input and inspect what it produced.
Delete the throwaway afterwards.
## What counts as evidence
Command output you actually saw. Paste the relevant lines, not a summary of them.
These are not evidence:
- "The tests pass" for a change tests do not cover.
- "The types check" for anything about runtime behaviour.
- "It should work now" for anything at all.
## Check the failure path too
Feed it the input you expect to be rejected and confirm it is rejected, with a message
that says why. A feature that works only on correct input is half-built.
## Report what you did not verify
Say plainly what you could not run and why: a missing credential, a service you cannot
start, a platform you are not on. An honest gap is useful; a claim that hides one is not.
## When verification fails
The defect is yours to fix in this turn. Do not report the task complete with a note that
it did not work.
`,
},
{
name: 'commit',
source: `---
name: commit
description: Stage and commit work. Use when asked to commit, or to split existing changes into commits.
---
# Committing
Never commit unless the user asked. If it is unclear whether they did, ask.
## Look before you stage
\`git_status\` and \`git_diff\` first. You are looking for two things:
1. Changes that are not yours. Another agent or the user may share this worktree, and
\`git add .\` takes their half-finished work with yours.
2. Files that should never be committed: \`.env\`, credentials, keys, large build output,
anything a \`.gitignore\` rule was supposed to catch and did not. Flag these to the user
rather than committing them.
Stage the specific paths you changed. \`git add .\` is how unrelated work ends up in a
commit that then has to be reverted whole.
## One commit, one reason
If the diff does two unrelated things, make two commits. A commit that both fixes a bug and
renames a module cannot be reverted, cherry-picked, or bisected usefully.
## The message
Match the repository's existing style — read \`git_log\` before writing one. Failing that:
- A subject line under 70 characters, imperative, saying what changed.
- A body explaining *why*, when the reason is not obvious from the diff. Wrap at 72.
- No "as requested", no restating the diff line by line, no emoji unless the repo uses them.
## Do not
- Do not \`--amend\` a commit that has been pushed. Write a new one.
- Do not \`--no-verify\`. If a hook rejects the commit, the hook found something.
- Do not \`git push\` unless asked, and never force-push without being asked explicitly.
- Do not commit and then immediately fix it up with a second commit. Get it right, or say
what is wrong.
## After committing
Report the short hash and the subject. If a hook rewrote files, say so and confirm the
final state is what was intended.
`,
},
{
name: 'security',
source: `---
name: security
description: Review code for security defects, or write code that handles untrusted input. Use when touching authentication, user input, file paths, shell commands, SQL, or anything reachable from the network.
---
# Security
Find the trust boundary first. Everything crossing it is hostile until parsed.
## The boundaries in most codebases
- Request bodies, query strings, headers, cookies.
- File contents and filenames, including paths a user supplied.
- Environment variables in a multi-tenant deployment.
- Anything a model or a third-party API returned.
Inside a boundary, values are already validated and re-checking them is noise. At the
boundary, nothing is optional.
## What to look for, in order
1. **Injection.** String-built SQL, shell commands assembled from input, \`eval\`, template
rendering with user data as the template rather than the data. The fix is parameters and
argument arrays, never escaping.
2. **Missing authorisation.** An endpoint that checks *who* you are but not *what* you may
touch. Look for an id taken from the request and used without an ownership check.
3. **Path traversal.** \`../\` in anything joined onto a filesystem root. Resolve, then verify
the result is still inside the root — a prefix check on the raw input misses
\`a/../../secret\`.
4. **Secrets in the wrong place.** Keys in source, in logs, in error messages, in a commit.
A secret that reached a log is a secret to rotate.
5. **Server-side request forgery.** A URL from input, fetched. Block private and loopback
addresses by *resolved* address, and re-check every redirect hop.
6. **Weak crypto and hand-rolled auth.** Homemade token formats, \`Math.random\` for anything
security-bearing, comparisons on secrets that are not constant time.
## What not to do
Do not report a finding you cannot trace to a concrete input. "This could be unsafe" without
a path from an attacker-controlled value to the sink is noise that buries the real one.
Do not fix a symptom at one caller when the sink is shared. Grep every caller and fix the
seam once.
## Reporting
File, line, the path from input to sink, and the fix. Say plainly when a thing that looks
dangerous is actually fine, and why — a reviewer's confidence is worth as much as a finding.
`,
},
{
name: 'perf',
source: `---
name: perf
description: Make something faster, or find out why it is slow. Use when a command, request, test suite, or build takes longer than it should.
---
# Performance
Measure first. A change made without a number before it is a guess with extra steps.
## Get a number
Time the actual operation, not a proxy for it. \`time\`, the framework's own timing output,
or a loop around the slow call with a timestamp either side. Record the baseline with
\`remember\` so the comparison survives compaction.
If you cannot measure it, say so and stop. Optimising an unmeasured path is how a codebase
accumulates complexity that buys nothing.
## Find where the time goes
- **Wall-clock dominated by one call?** Look there and nowhere else.
- **Spread evenly?** Suspect the loop around it: an O(n²) walk, a query per row, a file read
per iteration.
- **Idle time?** It is waiting: a sequential chain of independent awaits, an unpooled
connection, a lock.
The usual culprits, in the order they actually appear: N+1 queries, work repeated inside a
loop that could be hoisted, a missing index, sequential awaits that could run together,
reading a whole file to use one line, and re-parsing something that could be parsed once.
## Change one thing
One change, then re-measure. Two changes together and you do not know which one paid — and
one of them may have cost.
## Stop when it is fast enough
State the target before you start: "the test suite under a minute", "the endpoint under
200ms". Past the target, further work is complexity with no user on the other end of it.
## Report
Baseline, change, new number, and what you did not do. A 40% win with one line changed is a
better report than a 45% win that restructured a module.
`,
},
{
name: 'migrate',
source: `---
name: migrate
description: Upgrade a dependency, framework, or language version across a codebase. Use when a major version bump, a deprecation, or a breaking API change has to be applied.
---
# Migration
The failure mode is a half-applied migration: it compiles, most tests pass, and one code
path still uses the old API.
## Read the changelog before the code
Find what actually broke. A major version usually has a migration guide; read it and list
the changes that apply to this codebase specifically. Below 1.0, treat a minor bump as
breaking — semver promises nothing there.
## Find every call site before changing one
Grep for the old API across the whole repository, including tests, scripts, config, CI
workflows, Dockerfiles, and documentation. A version literal pinned in a workflow while the
manifest says something else is a split-brain deploy.
Write the list down with \`todo_write\`. The list is the migration; the edits are mechanical.
## Change in one shape
Apply the same transformation everywhere rather than improving each site as you pass
through it. A migration mixed with refactoring cannot be reviewed, and cannot be reverted
if the upgrade turns out to be wrong.
\`apply_patch\` is the tool for this: one atomic patch across the files that must land
together.
## Verify at the boundary that broke
Type checks catch signature changes and miss behaviour changes. Run the tests, then actually
use the thing that was upgraded: start the server, run the CLI, execute the query. A green
suite over an untested upgrade path proves the suite did not cover it.
## Never hand-merge a lockfile
On a conflict, take either side whole and regenerate with the package manager. The resolver
owns that file.
## Report
The version before and after, every file class touched, what you verified by running, and
anything the changelog said applies that you deliberately did not do.
`,
},
{ name: 'accessibility', source: accessibility },
{ name: 'api-design', source: apiDesign },
{ name: 'ci-cd', source: ciCd },
{ name: 'commit', source: commit },
{ name: 'data', source: data },
{ name: 'db', source: db },
{ name: 'debug', source: debug },
{ name: 'deps', source: deps },
{ name: 'docker', source: docker },
{ name: 'docs', source: docs },
{ name: 'frontend', source: frontend },
{ name: 'git-workflow', source: gitWorkflow },
{ name: 'i18n', source: i18n },
{ name: 'incident', source: incident },
{ name: 'logging', source: logging },
{ name: 'migrate', source: migrate },
{ name: 'onboarding', source: onboarding },
{ name: 'optimize-sql', source: optimizeSql },
{ name: 'perf', source: perf },
{ name: 'perf-frontend', source: perfFrontend },
{ name: 'plan', source: plan },
{ name: 'readme', source: readme },
{ name: 'refactor', source: refactor },
{ name: 'release', source: release },
{ name: 'review', source: review },
{ name: 'security', source: security },
{ name: 'test', source: test },
{ name: 'ux-copy', source: uxCopy },
{ name: 'verify', source: verify },
];
+42
View File
@@ -0,0 +1,42 @@
---
name: accessibility
description: Make a UI accessible. Use when adding a feature that must work with a keyboard or screen reader, fixing contrast or focus issues, or reviewing for WCAG.
---
# Accessibility
Accessibility is usability for everyone, including people using a keyboard, a screen reader, a
magnifier, or a noisy display. Build it in, not on.
## Semantic HTML does the heavy lifting
A `<button>`, `<a>`, `<input>`, `<nav>`, `<main>` carries behaviour and meaning for free
that a `<div>` with a click handler does not. Reach for the native element first; add ARIA only
when no native element fits. The first rule of ARIA is do not use ARIA if a native element exists.
## Keyboard is the baseline
- Every interactive element is reachable and operable with Tab and Enter/Space alone.
- A visible focus indicator on everything — never `outline: none` without a replacement.
- Logical tab order following the visual order, and focus managed into and out of modals,
menus, and dialogs (trapped while open, returned to the trigger on close).
## Screen readers hear structure
- Headings in order (`h1` once, then down a level at a time) so the page has a navigable outline.
- Every `<input>` has a `<label>`; every icon-only button has an accessible name; every image
has alt text that conveys its point (or empty alt when it is purely decorative).
- Dynamic changes announce themselves: a toast, an error, a loaded region uses a live region so
it is heard, not just seen.
## Contrast and meaning
Text meets 4.5:1 against its background (3:1 for large text). Colour is never the only carrier
of meaning — pair it with an icon, a label, or a pattern. A red-only "error" is invisible to a
colour-blind user.
## Test it the way it is used
Tab through the whole flow. Turn on a screen reader and listen. Zoom to 200% and 400%. Run an
automated checker for the mechanical half — then do the manual half it cannot cover, because
most accessibility failures are not machine-detectable.
+45
View File
@@ -0,0 +1,45 @@
---
name: api-design
description: Design or revise an HTTP or library API. Use when adding an endpoint, shaping request/response bodies, naming resources, or reviewing an API for consistency.
---
# API design
An API is a contract. Every choice is a promise you cannot take back without a major version.
## Resource before action
Name things, not verbs. `POST /users` to create, not `POST /createUser`. The URL is the
noun; the method is the verb. When you reach for a verb in the path, that is a sign the
resource is missing — `POST /users/:id/deactivations` reads better than `/deactivateUser`
when the operation has state of its own.
## Shape the body for the reader
- Field names are `snake_case` or `camelCase`, picked once for the whole API. A body that
mixes both is a body nobody documented.
- Return the object, not a wrapper, unless the wrapper carries something: `{ "user": {...} }`
only when there is also pagination, a cursor, or an error envelope.
- Errors have a stable shape: a machine-readable `code`, a human `message`, and the field
that failed. A client should never have to parse the message.
## Status codes mean something
- `201` for a created resource, with the resource in the body.
- `204` for success with nothing to return.
- `400` for a body that failed validation, `401` unauthenticated, `403` authenticated but
not allowed, `404` not found or not allowed to know, `409` a conflict with current state,
`422` well-formed but semantically wrong.
- Never `200` with an error in the body. A client checking only the status will treat it as
success.
## Idempotency and safety
GET, PUT, DELETE must be safe to retry: same request, same state. POST is not. If a client can
double-submit, provide an idempotency key or a natural unique constraint, and say which.
## Version when you must, not before
Add fields freely; removing or renaming is a break. If you are not yet committed, say so with
a `beta` or `v0` marker rather than locking a shape you have not used. Document the contract
you guarantee, not the implementation that happens to produce it.
+39
View File
@@ -0,0 +1,39 @@
---
name: ci-cd
description: Write or repair CI/CD pipelines and workflow files. Use when a build fails in CI but not locally, when adding a workflow, or when caching, matrix, or deploy steps need design.
---
# CI/CD
CI is a second machine that does not have your setup. "Works on my machine" means the pipeline
is missing something your machine has.
## Reproduce the environment, not the symptom
When CI fails and local passes, the difference is the environment: the toolchain version, an
uncommitted file, a cache, an env var, the OS. Diff those before touching the code. Read the
failing log literally — the first error, not the last, which is usually a downstream echo.
## Pin everything that can move
- Toolchain versions (`node`, `bun`, `python`), action versions, base images. `latest`
is a build that breaks on a day you did nothing.
- Lockfiles go in the repo and the install respects them (`--frozen-lockfile`, `ci`). An
install that re-resolves in CI is a different build from the one you tested.
## Cache the expensive, deterministic part
Dependencies are the cache; build output usually is not. Key the cache on the lockfile hash so
a changed dependency invalidates it. A cache that is too broad serves stale artifacts; too
narrow saves nothing.
## Fail fast, in the right order
Cheap checks first: lint and typecheck before the test matrix, tests before the deploy. A
pipeline that deploys before it verifies publishes the bug it was built to catch.
## Secrets and deploys
Secrets live in the CI secret store, never in the file, and are masked in logs. A deploy step
is gated: on a tag, on a protected branch, on a manual approval — never on every push. Assume
every log line is public and write the pipeline accordingly.
+54
View File
@@ -0,0 +1,54 @@
---
name: commit
description: Stage and commit work. Use when asked to commit, or to split existing changes into commits.
---
# Committing
Never commit unless the user asked. If it is unclear whether they did, ask. A commit is a
durable statement about shared history, not a save-point.
## Look before you stage
`git_status` and `git_diff` first — read the whole diff you are about to commit. You are
looking for three things:
1. **Changes that are not yours.** Another agent or the user may share this worktree, and
`git add .` takes their half-finished work with yours.
2. **Files that should never be committed:** `.env`, credentials, keys, large build
output, anything a `.gitignore` rule was supposed to catch and did not. Flag these to
the user rather than committing them — a committed secret is a secret to rotate.
3. **Your own accidents:** debug prints, commented-out code, a stray `TODO`, a file you
opened and saved by mistake. Revert them before staging, not in a follow-up commit.
Stage the specific paths you changed. `git add .` is how unrelated work ends up in a
commit that then has to be reverted whole.
## One commit, one reason
If the diff does two unrelated things, make two commits. A commit that both fixes a bug
and renames a module cannot be reverted, cherry-picked, or bisected usefully. Each commit
should pass the tests on its own — a series of broken commits defeats `git bisect`.
## The message
Match the repository's existing style — read `git_log` before writing one. Failing that:
- A subject line under 70 characters, imperative mood, saying what changed: "Fix off-by-one
in pagination", not "fixed a bug" or "changes".
- A body explaining *why* when the reason is not obvious from the diff. Wrap at 72.
- No "as requested", no restating the diff line by line, no emoji unless the repo uses them,
no sign-off noise the repo does not already use.
## Do not
- Do not `--amend` a commit that has been pushed. Write a new one.
- Do not `--no-verify`. If a hook rejects the commit, the hook found something — read it.
- Do not `git push` unless asked, and never force-push without being asked explicitly.
- Do not commit and then immediately fix it up with a second commit. Get it right, or say
what is wrong.
## After committing
Report the short hash and the subject. If a hook rewrote files, say so, and confirm the
final state — `git_status` again — is what was intended.
+40
View File
@@ -0,0 +1,40 @@
---
name: data
description: Process, validate, or transform data. Use when parsing files, cleaning datasets, designing a data pipeline, or debugging a transform that produces wrong output.
---
# Data
Bad data fails silently and far away from where it entered. Validate at the boundary, keep the
raw, and make every transform checkable.
## Validate at the boundary
Parse and validate when data enters the system, not when it is used. A schema check at the edge
turns a corrupt record into a clear rejection; skipping it turns the same record into a wrong
answer three layers later. Reject loudly, with the record and the reason — never coerce and
carry on.
## Keep the raw
Store the untransformed input alongside the derived. When a transform turns out to be wrong,
the raw lets you recompute; without it, the information is gone. Derived data is rebuildable;
source data is not.
## Transformations are pure and tested
A transform takes input and returns output with no hidden state, so it can be tested on a
fixture and re-run safely. Test the edge cases that actually occur in data: the empty field,
the wrong type, the unexpected null, the duplicate, the encoding that is not UTF-8.
## Duplicates, nulls, and ranges are the usual corruption
Check for: unexpected duplicates on a key, nulls where a value is required, values outside a
sane range (a negative age, a date in the future), and referential breaks (an id pointing at
nothing). These four catch most real-world data problems before they reach a report.
## Idempotent pipelines
A step that can be re-run without duplicating or corrupting its output is a step you can retry
after a failure. Key on a stable id and upsert rather than blind-insert. A pipeline you cannot
safely re-run is a pipeline you will one day have to fix by hand at 2am.
+38
View File
@@ -0,0 +1,38 @@
---
name: db
description: Design schemas, write migrations, or fix query and data problems. Use when adding a table, writing a migration, debugging a slow query, or choosing keys and indexes.
---
# Databases
The schema is the hardest thing to change in the whole system. Design it for the queries, not
the object model.
## Keys and constraints are the real schema
- Every table has a primary key; prefer a surrogate `id` unless a natural key is truly stable.
- Foreign keys and `NOT NULL` are not optional decoration — they are the constraints that stop
bad data at the door instead of in application code six months later.
- Unique constraints belong on the thing that must be unique (email, slug), enforced by the
database, not by a check-then-insert that races.
## Migrations are one-way and additive where possible
- Never edit a migration that has run anywhere. Add a new one.
- Destructive changes (drop column, rename, change type) are two migrations: add the new shape,
deploy code that writes both, then remove the old in a later release. A single migration that
renames a column breaks every old copy of the app still running.
- Test a migration against real data volume. `ALTER` on ten rows is instant; on ten million it
locks the table.
## Indexes follow the queries
Index the columns you filter and join on, in the order the query uses them. A composite index
`(a, b)` serves `WHERE a` and `WHERE a, b` but not `WHERE b` alone. Read the query plan
(`EXPLAIN`) before adding one — a guess is an index that costs writes and serves nothing.
## The N+1 is the default bug
A query per row in a loop is the most common database performance defect. Fetch the set with a
join or a batched `WHERE id IN (...)`. If a page does one query per item, that is the fix
before any caching.
+63
View File
@@ -0,0 +1,63 @@
---
name: debug
description: Track down a bug whose cause is not obvious. Use when a test fails for unclear reasons, behaviour differs between environments, or an earlier fix did not hold.
---
# Debugging
Do not guess. A guess that happens to work leaves the real cause in place, and it will
fire again — usually in production, usually at a worse time.
## Reproduce first
Find the smallest command that shows the failure and record it with `remember`. If you
cannot reproduce it, say so and ask what the user did differently — do not proceed on a
hypothesis you cannot test.
Shrink the reproduction until it is minimal: one input, one call, one assertion. Every
moving part you leave in is a place the bug can hide. A reproduction that takes thirty
steps will not get run often enough to confirm the fix.
## Three hypotheses, then evidence
Write down at least three causes that would produce this exact symptom — not "the code is
wrong" but specific mechanisms: "the offset is off by one when the page is empty", "the
cache is read before the write lands". Rank them by how cheap they are to disprove, then
disprove them in that order. State which one you are testing before you test it.
Evidence means observed output: a log line, a failing assertion, a value printed at the
point of failure. "It should be X" is not evidence. When the evidence contradicts your
favoured hypothesis, the hypothesis is wrong — do not explain the evidence away.
## Localise before you fix
Assert the value at each boundary until one is wrong. The bug lives between the last
boundary where the value is right and the first where it is wrong. Fixing before you have
that bracket means editing the wrong place and learning nothing.
## Bisect when the space is large
- Recent regression: `git bisect` or read what changed last. The bug arrived in a commit;
find which one.
- Unclear layer: assert the value at each boundary until one is wrong.
- Intermittent: run it in a loop and capture the failing case with full logging. Do not
reason about a race abstractly — make it happen on demand, then it is no longer
intermittent.
## Fix the cause, not the symptom
Once you know the cause, fix that and nothing else. Do not tidy surrounding code in the
same change — a bugfix diff should contain only the bug, so it can be reverted whole if it
is wrong.
Write a test that fails before the fix and passes after. Watch it fail first; a test you
never saw fail proves nothing. If you cannot express the bug as a test, say why — and say
what you ran instead to confirm the fix.
## After two failed attempts
Stop. Re-read the error text literally, character by character — most "impossible" bugs
are a misread message. Then check your assumption about which code is actually running:
the wrong file, a stale build, a shadowed import, a cached dependency, or an env var that
differs from your shell. Verify by printing something at the point you *think* executes;
if it does not print, that is your answer.
+39
View File
@@ -0,0 +1,39 @@
---
name: deps
description: Manage dependencies: choosing, adding, updating, or removing them. Use when evaluating a library, resolving a version conflict, pruning unused deps, or hardening the supply chain.
---
# Dependencies
Every dependency is code you did not write but now maintain. Add deliberately, prune regularly.
## Choose on maintenance, not features
Before adding: is it actively maintained (recent commits, responsive issues), widely used, and
small enough to be worth it? A dependency that saves a day and is abandoned in a year costs a
week. For something small and stable, a dozen lines in your own codebase often beats a package.
## Pin and lock
Exact versions in the manifest for anything that matters, a lockfile committed, and installs
that respect it. A `^` range means your build tomorrow differs from your build today. The
lockfile is the build's memory; do not delete it to "fix" a conflict — resolve the conflict.
## Update on a schedule, read the changelog
Routine small updates beat a yearly painful one. For a major bump: read the changelog and the
migration guide, find every call site of the changed API, and apply one shape of change (see
the migrate skill). Update one thing at a time so a regression has an obvious cause.
## Know your transitive tree
A direct dependency drags in dozens of transitive ones. Audit the tree for: known
vulnerabilities (`audit`/SCA tooling), abandoned packages deep in it, and duplicate copies of
the same library at different versions bloating the bundle. Remove what you no longer use — an
unused dependency is attack surface and install time for nothing.
## Supply chain is a trust decision
A package runs its install scripts with your permissions. Prefer packages with provenance and a
reproducible build, be wary of sudden ownership transfers, and pin so a hijacked publish does
not reach you automatically. The lockfile is also your audit trail of exactly what shipped.
+41
View File
@@ -0,0 +1,41 @@
---
name: docker
description: Write or fix Dockerfiles and container setups. Use when an image is too large, a build is slow, a container will not start, or layering and caching need design.
---
# Docker
An image is a build artifact. Small, reproducible, and boring is the goal.
## Layer cache is the whole speed game
Order instructions from least to most frequently changed: base image, then dependency
manifests, then `install`, then source copy, then build. Copying `.` before installing
dependencies means every code change re-runs the install — the single most common Dockerfile
mistake.
## Small images, on purpose
- Use multi-stage builds: build in a full toolchain stage, copy only the artifact into a slim
runtime stage. The compiler does not ship to production.
- Pick a slim or distroless base unless you need the tooling. Alpine is small but musl breaks
some binaries; know why you chose it.
- One `RUN` with `&&` for related steps, cleaning up in the same layer — a separate `RUN rm`
does not shrink the image, the data is still in the earlier layer.
## The container is not a VM
- One process per container, as PID 1, so signals work. Use an init if the app spawns children.
- Do not run as root. Add a user and `USER` it.
- Read-only filesystem where possible; write to a mounted volume for anything that must persist.
Nothing in the image is writable state.
## .dockerignore is as important as the Dockerfile
Exclude `.git`, `node_modules`, build output, and any secret file. A context that sends the
whole repo is slow, and a secret copied into an image layer is a secret to rotate.
## Healthcheck and logs
The process logs to stdout/stderr, never to a file inside the container — the runtime collects
it. Add a `HEALTHCHECK` that proves the service answers, not just that the process exists.
+39
View File
@@ -0,0 +1,39 @@
---
name: docs
description: Write or update documentation, READMEs, and guides. Use when asked to document a feature, write usage docs, or bring docs back in line with the code.
---
# Documentation
Docs lie by omission. Write only what you have verified in the code.
## Ground every claim in the source
Before documenting a behaviour, read it. A flag, a default, an error message — open the
code and quote what it actually does, not what the name suggests. The most damaging doc
line is the confident one that was true two versions ago. If the code and the existing
docs disagree, the code is right; say so and fix the doc.
## Answer the reader's actual question
A reader opens a doc with a task, not a desire for completeness. Lead with the thing they
came to do, in the order they will do it:
- **A reference** lists what exists: every flag, every field, with its default and its type.
- **A guide** walks one path to one outcome. Resist documenting every branch — link instead.
- **A README** orients in sixty seconds: what it is, install, the first command that works.
## Show, then say
A working example beats a paragraph about one. Every command in the doc must be one you
ran, with its real output. A snippet that was never executed is a bug waiting for a reader.
## Match the house style
Read the neighbouring docs first: their heading depth, their code-fence language tags,
their tone. A doc that reads foreign is a doc nobody trusts enough to maintain.
## Keep it true over time
Document the stable contract, not the current implementation, unless the point is the
implementation. The fewer specifics a doc pins down, the less it rots.
+40
View File
@@ -0,0 +1,40 @@
---
name: frontend
description: Build or fix a web UI. Use when working on components, state, rendering performance, forms, or anything the user sees and interacts with in a browser.
---
# Frontend
The user's experience is the metric. Fast, clear, and forgiving beats clever.
## State lives as low as it can
Lift state only as high as the components that share it. Global state for something two siblings
need is re-render and complexity for everything. Server data is not client state — cache it with
the data layer rather than duplicating it into a store you must keep in sync by hand.
## Rendering is the usual bottleneck
Before optimising, find what re-renders. A component that re-renders on every parent render
because of an inline object or function prop is the common case. Memoize the expensive subtree,
not everything — `useMemo` and `useCallback` have a cost too, and slapping them everywhere is
its own slowdown.
## Forms respect the user
- Validate on blur or submit, not on every keystroke, and show the message at the field.
- Never clear a form on an error. The user's input is the most expensive thing on the page.
- Disable the submit while submitting, and say what is happening. A double-submitted form is a
duplicate record.
## Accessibility is not a later pass
Semantic HTML first: a `<button>` that looks like a button beats a `<div>` with a click
handler. Keyboard-reachable everything, visible focus, labels on inputs, alt text that conveys
the point not the pixels. Colour is never the only carrier of meaning.
## Measure what the user feels
Load: get the first meaningful paint and the time-to-interactive down before micro-tuning.
Bundle: split the route nobody opens, lazy-load the heavy component. A Lighthouse number is a
proxy; the goal is that it never feels slow.
+41
View File
@@ -0,0 +1,41 @@
---
name: git-workflow
description: Work with branches, rebases, merges, and history. Use when untangling a branch, preparing a PR, deciding rebase vs merge, or recovering from a git mistake.
---
# Git workflow
History is a communication tool. Write it for the person who reads it in six months — usually you.
## One branch, one purpose
A branch that does two things produces a PR that can only be reviewed as all-or-nothing and
reverted only whole. Keep it small and single-purpose; open the second thing as its own branch.
## Rebase to clean up, merge to preserve
- Rebase your own unpushed work freely: it makes a linear, readable history.
- Never rebase a branch others have pulled — it rewrites commits they have, and the next pull
becomes a mess. Merge shared branches instead.
- Interactive rebase before opening the PR: squash the "fix typo" and "wip" commits into the
change they belong to. The PR should read as a series of intentional steps, not a diary.
## Recover without panic
- `git reflog` finds almost anything you "lost": the branch you deleted, the commit you reset
away. Nothing committed is truly gone for ~30 days.
- A bad merge: `git merge --abort`. A bad rebase: `git rebase --abort`. Both stop cleanly
rather than pushing forward into a worse state.
- Committed to the wrong branch: `git reset --soft` to keep the work, switch, recommit.
## The commit message is the review's first page
Subject under 70 chars, imperative, says what changed. Body explains *why* when it is not
obvious. A reviewer who cannot tell why a change exists from its message will ask, or worse,
approve without understanding.
## Read the conflict, do not guess
On a conflict, open the file and understand both sides before resolving. Taking "ours" or
"theirs" wholesale because it is faster is how a resolved conflict silently drops someone's
work.
+41
View File
@@ -0,0 +1,41 @@
---
name: i18n
description: Internationalise or localise a product. Use when extracting strings for translation, formatting dates and numbers for a locale, handling pluralisation, or fixing layout that breaks in another language.
---
# Internationalisation
Hard-coded English is a bug in every other language. Externalise strings and never assume a
grammar.
## Every user-facing string is a key
No string in the UI lives in code; it lives in a message catalogue under a key. Concatenating
translated fragments is the classic bug: "You have " + n + " messages" cannot be reordered for
a language whose grammar puts the number elsewhere. Use a format with named placeholders:
`{count, plural, ...}`, translated as a whole.
## Pluralisation and gender are not English
Languages have one, two, several, or no plural forms, with rules that do not map to "1 vs other".
Use the ICU plural machinery of your i18n library and let the translator fill in every form the
locale needs. The same goes for gendered agreement.
## Format dates, numbers, and currencies by locale
Never `dd/mm/yyyy` by hand: `03/04/2025` is March 4th to one user and April 3rd to another.
Use the platform's locale-aware formatter (`Intl.DateTimeFormat`, `Intl.NumberFormat`).
Store and transmit ISO 8601 / UTC; format for display only.
## Layout breaks in translation
German runs ~30% longer than English; some scripts are right-to-left. Flexible layout, no fixed
widths on translated text, and CSS logical properties (`margin-inline-start` not
`margin-left`) so RTL mirrors correctly. Test with a pseudo-locale that lengthens and accents
every string to find the overflows before a translator does.
## Sort and search correctly
String order is locale-dependent: `ä` sorts with `a` in German, after `z` in Swedish. Use
locale-aware collation (`Intl.Collator` or the database's) rather than byte order, and normalise
Unicode before comparing, because the same character has more than one byte representation.
+40
View File
@@ -0,0 +1,40 @@
---
name: incident
description: Respond to a production incident. Use when something is down, degraded, or misbehaving in production and must be diagnosed and mitigated under time pressure.
---
# Incident response
Mitigate first, diagnose second. Restore service, then find out why.
## Confirm and scope before touching anything
What is actually broken, for whom, since when? Check the signal, not the report: the dashboard,
the error rate, the health endpoint. A wrong scope sends you chasing a symptom. State the impact
plainly in one line before you start changing things.
## Recent change is the prime suspect
Most incidents follow a deploy, a config change, a flag flip, or a scaling event. What changed
in the window before it broke? Check the deploy log and the diff. The fastest fix is usually to
undo the last change, not to understand it.
## Mitigate, then understand
- Roll back the deploy, flip the flag off, fail over, scale up, restart the wedged process —
whichever restores service fastest, even if you do not yet know the root cause.
- A mitigation you can reverse beats a perfect diagnosis that takes an hour. Note what you did so
it can be undone or made permanent later.
- Do not deploy an unreviewed "fix" into the fire; it adds a second change to a system already
misbehaving.
## Preserve evidence before it rotates away
Capture the logs, the error, the relevant metrics, a snapshot of the state — before a restart or
a rollback destroys it. You will want it for the postmortem, and it may be the only copy.
## Communicate and follow up
Say what is broken, what you are doing, and when the next update is — to whoever is affected,
in plain language, on a schedule. Afterwards: write the timeline, the root cause, and the
follow-ups that stop it recurring. An incident with no follow-up is a loan against the next one.
+42
View File
@@ -0,0 +1,42 @@
---
name: logging
description: Add or improve logging and observability. Use when debugging in production, adding structured logs, choosing log levels, or making a system traceable.
---
# Logging
Logs are how you debug a system you cannot attach a debugger to. Write them for the 3am
incident, not the happy path.
## Structure over prose
Emit fields, not sentences: `{ user: id, action: "checkout", ms: 142, ok: false }`, not
`"User checked out"`. Structured logs are searchable and aggregable; a sentence is neither.
One event, one line, one level.
## Levels are a contract
- `error` — something is broken and someone should look. Not "a user gave bad input".
- `warn` — unexpected but handled; worth a glance.
- `info` — the meaningful state transitions: started, finished, the decision made. Sparse.
- `debug` — everything you might want while diagnosing, off in production.
A log at the wrong level trains people to ignore the right one. If everything is `error`,
nothing is.
## Log the decision points, not every line
At a boundary — a request in, a call out, a branch taken — log what was decided and the inputs
that decided it, with a correlation id that follows the request across services. You should be
able to trace one request end to end from the id alone.
## Never log a secret
No passwords, tokens, session ids, full card numbers, or personal data beyond what policy
allows. Redact at the point of logging, not by hoping a downstream filter catches it. A secret
in a log aggregator is a secret to rotate.
## Measure, do not just log
For anything with a latency or a rate, a metric answers "is it slow?" faster than a thousand
log lines. Logs explain *why*; metrics tell you *that* something is wrong in the first place.
+55
View File
@@ -0,0 +1,55 @@
---
name: migrate
description: Upgrade a dependency, framework, or language version across a codebase. Use when a major version bump, a deprecation, or a breaking API change has to be applied.
---
# Migration
The failure mode is a half-applied migration: it compiles, most tests pass, and one code
path still uses the old API.
## Read the changelog before the code
Find what actually broke. A major version usually has a migration guide; read it and list
the changes that apply to this codebase specifically. Below 1.0, treat a minor bump as
breaking — semver promises nothing there.
## Find every call site before changing one
Grep for the old API across the whole repository, including tests, scripts, config, CI
workflows, Dockerfiles, and documentation. A version literal pinned in a workflow while the
manifest says something else is a split-brain deploy.
Write the list down with `todo_write`. The list is the migration; the edits are mechanical.
## Change in one shape
Apply the same transformation everywhere rather than improving each site as you pass
through it. A migration mixed with refactoring cannot be reviewed, and cannot be reverted
if the upgrade turns out to be wrong.
`apply_patch` is the tool for this: one atomic patch across the files that must land
together.
## Verify at the boundary that broke
Type checks catch signature changes and miss behaviour changes — the two ways a migration
actually breaks you. Run the tests, then actually *use* the thing that was upgraded: start
the server, run the CLI, execute the query, hit the endpoint. A green suite over an
untested upgrade path proves only that the suite did not cover it.
Pay special attention to silent behaviour changes: a default that flipped, a deprecated
call that still runs but does something subtly different, an error type that changed shape.
These compile, pass type checks, and still break production.
## Never hand-merge a lockfile
On a conflict, take either side whole and regenerate with the package manager. The resolver
owns that file; a hand-merge is a split-brain dependency tree that installs differently on
every machine.
## Report
The version before and after, every file class touched, what you verified by running, the
behaviour changes you checked by hand, and anything the changelog said applies that you
deliberately did not do — with the reason.
+42
View File
@@ -0,0 +1,42 @@
---
name: onboarding
description: Orient in an unfamiliar codebase. Use when dropped into a new project and asked to understand it, or when writing the docs that help a newcomer get productive.
---
# Onboarding to a codebase
Understand the running system before the source. The goal is a correct mental model, not to have
read every file.
## Get it running first
Build it, run it, run the tests. A project you can execute you can interrogate; one you have
only read you can only guess at. The README and the `package.json`/`Makefile` scripts tell you
the intended commands; if they do not work, that is your first finding.
## Trace one request end to end
Pick the central thing the system does and follow it: the entry point, the route or main, the
handler, the data out and back. One full path teaches you the architecture faster than reading
any single module. Note the layers you cross — that is the system's real structure.
## Read the structure, not the files
- The directory layout names the major components and their boundaries.
- The dependency manifest names the frameworks and the big choices already made.
- The tests show what the code is supposed to do, often better than the code does.
- `git log` on a core file shows what changes often and why — the living parts versus the
stable ones.
## Map the seams
Where does data enter and leave (HTTP, a queue, a file)? Where is state kept (a database,
memory, a cache)? Where are the trust boundaries? Those are the places bugs and features both
live. You do not need to know every file; you need to know where a change of a given kind would
go.
## Ask the codebase, then a person
Grep and the outline/symbol tools answer most "where is X" faster than reading. When genuinely
stuck on *why* something exists — that is a question for a person or the history, not more
reading.
+42
View File
@@ -0,0 +1,42 @@
---
name: optimize-sql
description: Diagnose and fix a slow SQL query. Use when a query is slow, a page makes too many queries, or an execution plan needs reading.
---
# SQL optimisation
Read the plan before changing anything. `EXPLAIN` (or `EXPLAIN ANALYZE`) tells you what the
database actually does; guessing at it is how you add an index that helps nothing.
## Read the plan for the expensive node
Find the node with the highest cost: a sequential scan over a large table, a nested loop over
many rows, a sort that spills to disk. Optimise that node. A plan with ten cheap nodes and one
expensive one has exactly one thing to fix.
## The index that matches the query
- Index the columns in the `WHERE` and `JOIN` clauses, and for a sort, the `ORDER BY`.
- A composite index `(a, b, c)` serves a leftmost prefix: `a`, `a,b`, `a,b,c` — not
`b` alone. Order the columns by the equality filters first, then the range, then the sort.
- A covering index includes every column the query reads, so the table is never touched. That
is the fastest a read gets.
## Write the query so the index is usable
- `WHERE lower(email) = ...` cannot use a plain index on `email`; either store it lowered or
use a functional index. A function on the column defeats the index.
- Leading `LIKE '%x'` cannot use a B-tree index; `LIKE 'x%'` can.
- `OR` across different columns often defeats an index; `UNION` of two indexed queries can be
faster.
## Kill the N+1 first
Before any index: if the page runs one query per row, that is the fix. Batch with
`WHERE id IN (...)` or a join. Ten queries become one beats ten individually-fast queries.
## Measure the change
`EXPLAIN ANALYZE` before and after, against realistic data volume. An index that helps a
100-row table may not justify its write cost at 100 million. Report the timing you actually
measured, not the improvement you expected.
+40
View File
@@ -0,0 +1,40 @@
---
name: perf-frontend
description: Make a web page faster. Use when a page loads slowly, feels janky, fails Core Web Vitals, or ships too much JavaScript.
---
# Frontend performance
Measure the user's experience first: a Lighthouse lab score and, better, real-user data. Optimise
the metric that is actually failing, not the one easiest to move.
## The vitals and what drives them
- **LCP** (largest contentful paint) — almost always the hero image or a web font. Preload it,
size it correctly, serve it in a modern format, and do not let render-blocking resources delay it.
- **INP / responsiveness** — long tasks on the main thread. Break up work, defer non-urgent JS,
and keep event handlers fast. A click that responds in 50ms feels instant; 300ms feels broken.
- **CLS** (layout shift) — images and embeds without dimensions, late-injected banners, web fonts
swapping. Reserve the space before the content arrives.
## Ship less JavaScript
The bundle is usually the problem. Route-level code splitting so a page loads only what it needs,
lazy-load the heavy below-the-fold component, and audit the dependency tree for a large library
imported for one function. Removing 100KB of JS beats most micro-optimisations.
## Network discipline
- Cache static assets with long, content-hashed lifetimes; the second visit should cost almost
nothing.
- Compress (brotli/gzip) and serve images at the size they are displayed, responsive `srcset`,
not a 3000px original in a 300px slot.
- Fetch in parallel, not in waterfalls: start independent requests together, and preload the
critical few.
## Change one thing, measure it
Take a baseline (the metric, the page, the device class), make one change, re-measure on the same
setup. Two changes at once and you do not know which paid. Report the before/after you actually
measured, on a realistic device and connection — a developer's fast laptop and fiber hides what a
mid-range phone on 4G feels.
+49
View File
@@ -0,0 +1,49 @@
---
name: perf
description: Make something faster, or find out why it is slow. Use when a command, request, test suite, or build takes longer than it should.
---
# Performance
Measure first. A change made without a number before it is a guess with extra steps, and
most "optimisations" made on a guess make the code worse and no faster.
## Get a number
Time the actual operation, not a proxy for it: `time`, the framework's own timing output,
or a loop around the slow call with a timestamp either side. Use realistic input — a fast
result on a tiny fixture tells you nothing about the production case. Record the baseline
with `remember` so the comparison survives compaction, and run it enough times that a
warm cache and jitter do not fool you.
If you cannot measure it, say so and stop. Optimising an unmeasured path is how a codebase
accumulates complexity that buys nothing.
## Find where the time goes
- **Wall-clock dominated by one call?** Look there and nowhere else. The biggest node is
the only one worth touching.
- **Spread evenly?** Suspect the loop around it: an O(n²) walk, a query per row, a file
read per iteration, an allocation per element.
- **Idle time?** It is waiting, not computing: a sequential chain of independent awaits, an
unpooled connection, a contended lock, a slow remote call.
The usual culprits, in the order they actually appear: N+1 queries, work repeated inside a
loop that could be hoisted, a missing index, sequential awaits that could run together,
reading a whole file to use one line, and re-parsing something that could be parsed once.
## Change one thing
One change, then re-measure on the same setup. Two changes together and you do not know
which one paid — and one of them may have cost. If the number did not move, revert the
change; an optimisation that does not measure is just complexity.
## Stop when it is fast enough
State the target before you start: "the test suite under a minute", "the endpoint under
200ms". Past the target, further work is complexity with no user on the other end of it.
## Report
Baseline, the change, the new number, and what you deliberately did not do. A 40% win with
one line changed is a better report than a 45% win that restructured a module.
+38
View File
@@ -0,0 +1,38 @@
---
name: plan
description: Break a non-trivial task into an ordered, verifiable sequence before writing code. Use when a request is large, spans several files, or its steps depend on each other.
---
# Planning
A plan that cannot be checked is a wish. Every step ends in something you can run.
## Understand before you sequence
Read enough to know the real shape of the work: the entry point, the data's path, the
module that owns the behaviour. A plan made from filenames alone reorders itself the
moment you open the first file. Grep the actual call sites; do not plan around a guess.
## Order by dependency, not by file
A step may depend on another's output: a type before its callers, a schema before its
migration, a test helper before the tests that use it. Sequence so nothing references
what does not exist yet. If two steps are independent, say so — the order between them
is free and you may take the cheaper one first to derisk the rest.
## One step, one verifiable outcome
Each step names the command that proves it done: a test that passes, a build that
compiles, a script that runs. "Wire it up" is not a step. Write the list with
`todo_write`, then work it in order, marking done immediately — not in a batch at the end.
## Keep it small
The plan is a scaffold, not the building. If a step grows past "change these few files",
split it. If the task turns out smaller than it looked, drop the remaining steps and say
why rather than inventing work to fill them.
## Replan when the ground moves
New information that changes the order or the scope is a reason to rewrite the list, not
to push through it. A stale plan followed faithfully is worse than no plan.
+42
View File
@@ -0,0 +1,42 @@
---
name: readme
description: Write or fix a project README. Use when creating a README, when a new user cannot get the project running from it, or when it has drifted from the code.
---
# README
A README has sixty seconds to answer: what is this, do I want it, and how do I run it. Everything
else is secondary to those three.
## The first screen answers three questions
1. **What it is** in one or two sentences, concrete about the problem it solves — not "a modern
solution" but "a CLI that lints Terraform plans against your org's policies".
2. **Install** — the one command that gets it.
3. **The first thing that works** — the minimal command or snippet that produces visible output.
If a new user cannot get a win in two minutes, most leave.
## Verify every command
Run each command in the README against a clean environment and paste its real output. The most
common README defect is an install or quickstart that no longer works because the code moved and
the doc did not. If you cannot run it, do not write it.
## Structure for scanning
After the quickstart, in the order a new user needs them: features as a short list of what it
does (not how), the common tasks as copy-paste examples, configuration as a table of options with
defaults, then links to deeper docs. Headings let a reader jump; a wall of prose gets skimmed
past the thing they needed.
## Show, do not tell
A three-line example of real use beats a paragraph describing capability. Show the input and the
output. A screenshot or asciinema of the actual tool running is worth a hundred adjectives —
include one if the tool has any visual surface.
## Keep it true
Document the stable interface, not this week's implementation, or the README rots. Re-read it on
every release: a README that contradicts the current version is worse than a short one, because
it actively misleads.
+44
View File
@@ -0,0 +1,44 @@
---
name: refactor
description: Restructure code without changing behaviour. Use when asked to refactor, clean up, extract, or reorganise.
---
# Refactoring
Behaviour must not change. That is the whole constraint — every other goal (clarity,
structure, naming) is subordinate to it. The moment behaviour changes, you are no longer
refactoring, you are editing, and the safety argument below stops holding.
## Establish the safety net first
Run the existing tests and record that they pass — with `remember`, so the baseline
survives compaction. If the code has no tests, write one that pins current behaviour,
*including the ugly parts*: the odd return value, the quirk callers depend on. You are not
judging the behaviour, you are freezing it. Refactoring untested code is not refactoring;
it is rewriting, and it belongs under the edit workflow with its own verification.
## Then move in small steps
One transformation at a time, tests green between each. Rename, then extract, then move —
not all three in one edit. The mechanical refactorings are the safe ones: rename, extract
function, inline, move. Compose them. A large refactor that fails leaves you unable to
tell which of five steps broke it; a small one that fails tells you exactly which.
After each step, run the tests, not just the typechecker. Types catch signature drift;
they do not catch a reordered conditional or a dropped early return.
## What not to do
- Do not fix bugs while refactoring. Note them, finish the refactor green, then fix in a
separate change — otherwise a regression could be either the refactor or the fix.
- Do not add abstraction for a single caller. Duplication beats a premature interface;
the third caller is when the abstraction earns its name.
- Do not widen the scope. The request was this code, not its neighbours. A refactor that
"while we're here" touches five more files is five more files of unreviewable risk.
- Do not change public API unless asked; if it must change, say so first and update every
caller in the same change.
## Done means
Tests pass, behaviour is identical, and the diff is smaller than the reader feared. If the
diff is larger than the code it moved, you abstracted too early — put it back.
+40
View File
@@ -0,0 +1,40 @@
---
name: release
description: Cut a release: versioning, changelogs, tagging, publishing. Use when asked to release, bump a version, write release notes, or fix a broken publish.
---
# Release
A release is a promise that a specific, identified state of the code works. Make it
reproducible or do not make it.
## Version says what changed
Semver: breaking is a major, a feature is a minor, a fix is a patch. The number is a message to
whoever upgrades, not a marketing choice. Below 1.0, say so plainly — semver promises nothing
and the version should not pretend otherwise.
## The changelog is for the upgrader
- Group by what the reader must do: breaking changes and required actions first, then features,
then fixes.
- Write it as "you can now X" or "Y no longer Z", from the user's side, not the commit's. A
changelog that is a git log is a changelog nobody reads.
- Every breaking change names the migration: what to change to keep working.
## Verify before you tag
The release candidate builds clean from a fresh checkout, the tests pass, and the version string
in the source matches the tag you are about to push. A version/tag mismatch published is the
kind of thing that ships "0.4" labelled as "0.3" forever.
## Tag the commit, publish the artifact
Tag the exact commit that was verified, and build the artifact from that tag — not from a
working tree that has since moved. The tag is immutable; never move it to a different commit.
If a release is wrong, cut a new one with a new number; do not quietly re-tag.
## If it goes wrong
Have the rollback ready before you need it: the previous artifact still available, the deploy
reversible. A bad release is fixed forward with a patch release, not by deleting the evidence.
+48
View File
@@ -0,0 +1,48 @@
---
name: review
description: Review a diff or a file for defects. Use when asked to review, critique, or check code before it ships.
---
# Code review
Severity order. Do not lead with style — a review that opens on naming while a real bug
sits three lines down has failed at its one job.
1. **Incorrect behaviour** — wrong result, wrong edge case, wrong state after failure.
2. **Missing validation at trust boundaries** — user input, network responses, file contents,
anything crossing a process line. Internal calls need no defensive checks.
3. **Security** — injection, path traversal, secrets in logs or errors, missing authz.
4. **Resource handling** — unclosed handles, unbounded growth, unawaited promises.
5. **Clarity** — only when it will cause a future defect.
## How to read the change
- Read the diff against its intent. Does it actually do what the title/commit says? A
correct-looking diff that solves the wrong problem is the most expensive approval.
- Read the *deleted* lines as carefully as the added ones. Behaviour is often lost in a
removal, and diffs render deletions quietly.
- Follow each new call one level into the callee. The assumption that breaks it is usually
one level down, invisible in the diff itself.
## For each finding
State file and line, the concrete failure (what input makes it break, or why it always
breaks), and the change. Show the fix as code when it is short. "This could be a problem"
without a path to a real input is noise; either trace it or drop it.
Order findings by severity and lead with the worst. Skip anything a formatter would fix.
Skip preference. If a choice is defensible, leave it — a review is not a place to impose
your style on code that works.
## Say when it is fine
A review that invents problems to look thorough is worse than a short one. If the change
is correct, say so plainly and stop. "Looks correct, and here is what I checked" is a
complete and useful review.
## Verify, do not assume
Read the surrounding code before calling something a bug. A "missing" null check often
exists one level up; a "redundant" guard often covers a caller you have not seen. Run the
tests or write the failing input if that is what settles it. A finding you verified is
worth ten you suspected.
+53
View File
@@ -0,0 +1,53 @@
---
name: security
description: Review code for security defects, or write code that handles untrusted input. Use when touching authentication, user input, file paths, shell commands, SQL, or anything reachable from the network.
---
# Security
Find the trust boundary first. Everything crossing it is hostile until parsed.
## The boundaries in most codebases
- Request bodies, query strings, headers, cookies.
- File contents and filenames, including paths a user supplied.
- Environment variables in a multi-tenant deployment.
- Anything a model or a third-party API returned.
Inside a boundary, values are already validated and re-checking them is noise. At the
boundary, nothing is optional.
## What to look for, in order
1. **Injection.** String-built SQL, shell commands assembled from input, `eval`, template
rendering with user data as the template rather than the data. The fix is parameters and
argument arrays, never escaping.
2. **Missing authorisation.** An endpoint that checks *who* you are but not *what* you may
touch. Look for an id taken from the request and used without an ownership check.
3. **Path traversal.** `../` in anything joined onto a filesystem root. Resolve, then verify
the result is still inside the root — a prefix check on the raw input misses
`a/../../secret`.
4. **Secrets in the wrong place.** Keys in source, in logs, in error messages, in a commit.
A secret that reached a log is a secret to rotate.
5. **Server-side request forgery.** A URL from input, fetched. Block private and loopback
addresses by *resolved* address, and re-check every redirect hop.
6. **Weak crypto and hand-rolled auth.** Homemade token formats, `Math.random` for anything
security-bearing, comparisons on secrets that are not constant time.
## Verify the path before reporting
Trace each candidate from an attacker-controlled value to the sink before you name it. A
"this could be unsafe" without that path is noise that buries the real finding. If you
cannot construct the malicious input that reaches the sink, either keep looking or say
plainly that you could not confirm it.
Do not fix a symptom at one caller when the sink is shared. Grep every caller and fix the
seam once — a sanitiser at one of five call sites is four open holes and one false sense
of safety.
## Reporting
File, line, the path from input to sink, a concrete payload, and the fix. Rank by
exploitability: a reachable injection beats a theoretical weakness in dead code. Say
plainly when a thing that looks dangerous is actually fine, and why — a reviewer's
confidence in the clean parts is worth as much as a finding.
+50
View File
@@ -0,0 +1,50 @@
---
name: test
description: Write or repair tests. Use when adding coverage, fixing a flaky test, or asked how something should be tested.
---
# Testing
A test earns its place by failing when the code is wrong. A test that cannot fail — or
that passes regardless — is not a test, it is overhead with a green checkmark.
## Match the project
Read two existing test files first. Use their runner, their assertion style, their file
layout, their naming, their way of building fixtures. A test that looks foreign is a test
nobody maintains, and an unmaintained test is deleted the first time it goes red.
## Test behaviour, not implementation
Assert on what a caller observes: the return value, the written file, the emitted event,
the status code. A test that reaches into private state or mocks a collaborator's
internals breaks on every refactor while catching nothing real. If you cannot say what
the caller sees, you are testing the how, and the how is allowed to change.
Cover, in this order of value:
- **The failure** — the bad input, the missing file, the null. Failure cases catch more
real defects than happy paths, because most code is written for the happy path first.
- **The boundaries** — empty, one, the maximum, off-by-one at each edge.
- **The normal case** — one, to prove the wiring works at all.
## Never do this
- Do not assert what the code currently returns without knowing it is correct — that pins
the bug into the suite and calls it a specification.
- Do not weaken an assertion to make a test pass. If it fails, either the code or the
expectation is wrong; find out which before you touch either.
- Do not delete a failing test to go green. It is telling you something; listen.
- Do not test the framework or the library. Your code is the subject; their code has its
own suite.
## Flaky tests
A test that passes alone and fails in a suite is a shared-state problem: a global, a temp
directory, a port, an unawaited promise, leftover data, or ordering. Find which — run it
repeatedly and in isolation to confirm, then remove the shared state. Do not add a retry:
a retried flake is a real intermittent bug you have decided to stop hearing about.
## Verify
Run the test and watch it fail before the fix, pass after. A test you never saw fail is
not known to test anything.
+40
View File
@@ -0,0 +1,40 @@
---
name: ux-copy
description: Write user-interface text: labels, errors, empty states, onboarding. Use when wording a button, an error message, a confirmation, or any text the user reads in the product.
---
# UX copy
Interface text is part of the interface. Clear, short, and honest beats clever.
## Lead with what the user cares about
- Buttons say the action and the object: "Save changes", not "OK". "Delete project", not "Yes".
- Headings say the outcome or the thing, not the category: "Your invoices" over "Billing section".
- Front-load the information word. Users scan the first two words; "3 errors found" scans, "Found
3 errors" buries it.
## Error messages say what happened and what to do
Never blame, never jargon, never just a code. "Couldn't save because you're offline. Your changes
are kept — try again when you're back." The pattern is: what went wrong, whether their work is
safe, the one action to take. An error that only says "Something went wrong" makes the user do
the debugging.
## Empty states teach, they do not apologise
An empty screen is a chance to say what goes here and how to start: "No projects yet. Create your
first to see it here." plus the button. "Nothing to display" wastes the moment the user is most
receptive to guidance.
## Be honest about destructive and irreversible actions
A confirmation names exactly what will happen and that it cannot be undone: "Delete 'Invoices
2024'? This permanently removes 3,120 records and cannot be undone." The destructive button
repeats the verb: "Delete", never a bare "Confirm" that could mean anything.
## Consistent words for consistent things
Pick one term per concept and use it everywhere — if it is a "project" in one place it is not a
"workspace" in another. Sentence case for labels, no exclamation marks, no "please", no "simply".
The tone is a competent colleague, not a marketing page.
+53
View File
@@ -0,0 +1,53 @@
---
name: verify
description: Confirm a change actually works by using it, not by reading it. Use before reporting a task complete, or when asked whether something works.
---
# Verification
A green test suite says the tests pass. It does not say the feature works. The two are
different claims, and only one of them is what the user asked for.
## Run the artifact, not the source
Build it and use it the way a user would, end to end:
- **CLI** — build the binary and run it. Happy path, bad input, `--help`. Read the actual
output, not the output you expected.
- **HTTP service** — start it and `curl` the endpoint. Check the status line and the body,
not just that it returned something.
- **Library** — write a throwaway script that imports and calls the new code end to end,
the way a consumer would.
- **Script or job** — run it against real input and inspect what it produced.
Delete the throwaway afterwards. A verification script left behind becomes clutter the
next person trips over.
## What counts as evidence
Command output you actually saw. Paste the relevant lines, not a summary of them — a
summary hides the one line that mattered.
These are not evidence:
- "The tests pass" for a change the tests do not cover.
- "The types check" for anything about runtime behaviour.
- "It compiles" for anything about correctness.
- "It should work now" for anything at all.
## Check the failure path too
Feed it the input you expect to be rejected and confirm it is rejected, with a message
that says why. Then the edge case at the boundary. A feature that works only on correct
input is half-built, and the half that is missing is the half users hit first.
## Report what you did not verify
Say plainly what you could not run and why: a missing credential, a service you cannot
start, a platform you are not on. An honest gap is useful — the reader can fill it. A
claim that hides one is a bug you just shipped in prose.
## When verification fails
The defect is yours to fix in this turn. Do not report the task complete with a note that
it did not work — that is a failure report, not a completion.
+22 -1
View File
@@ -141,11 +141,17 @@ let counter = 0;
*/
export function createTaskTool(opts: {
model: LanguageModel;
/** Cheaper model for `explore`, which is search rather than reasoning. Defaults to `model`. */
subagentModel?: LanguageModel;
/** Its id, so the parent can price the subagent's spend separately. */
subagentModelId?: string;
cwd?: string;
maxSteps?: number;
report?: SubagentReporter;
/** Parent-owned approval for a worker's gated calls. Omit to disable `worker`. */
approve?: SubagentApproval;
/** Records a finished run's token use, so /cost can split subagent from parent spend. */
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
}) {
const canWrite = opts.approve !== undefined;
@@ -184,10 +190,15 @@ export function createTaskTool(opts: {
let steps = 0;
let text = '';
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
try {
// `explore` is search, not reasoning, so it runs on the cheaper model when
// one is configured. `review` and `worker` keep the parent's: they judge
// and they change, both of which want the full model.
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
const result = streamText({
model: opts.model,
model,
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
messages: [{ role: 'user', content: prompt }],
tools: TOOLS[flavour],
@@ -230,6 +241,13 @@ export function createTaskTool(opts: {
throw part.error instanceof Error ? part.error : new Error(message);
}
}
try {
const usage = await result.usage;
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
} catch {
// A run that errored before producing usage has nothing to account for.
}
} catch (e) {
const message = e instanceof Error ? e.message : String(e);
report?.({ type: 'error', id, message });
@@ -238,6 +256,9 @@ export function createTaskTool(opts: {
const trimmed = text.trim();
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
// Settled after the stream closes; a failed run reports nothing rather than
// a half count. The parent prices these against the subagent's own model id.
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
return trimmed || 'Subagent returned no findings.';
},
});
+17
View File
@@ -0,0 +1,17 @@
/**
* Lets TypeScript resolve Bun's raw-text imports of `.md` files.
*
* `import x from './file.md' with { type: 'text' }` returns the file's contents as
* a string; tsc does not know that without a module declaration. Bun handles the
* actual loading (and inlines it into a compiled binary); this only teaches the
* typechecker the shape.
*/
declare module '*.md' {
const content: string;
export default content;
}
declare module '*.png' {
const content: ArrayBuffer;
export default content;
}
+414
View File
@@ -0,0 +1,414 @@
import { tool } from 'ai';
import { stat } from 'node:fs/promises';
import { resolve } from 'node:path';
import { z } from 'zod';
import { jail, posix, walk } from './ignore';
import { git } from './tools-git';
/**
* The second batch of built-in tools, kept out of tools.ts so that file stays
* reviewable. Four families:
*
* edit precise line-level edits that need no full-file rewrite
* inspect filesystem navigation and metadata
* git ext read-only git queries beyond the core five (argv-spawned, no shell)
* code structured reads of source and environment
*
* Every write goes through `jail`, every read honours .gitignore through `walk`,
* and every git call spawns the binary with a fixed argv — the same rules as the
* core tools, so the approval and guard model needs nothing new.
*/
const MAX_OUTPUT = 30_000;
const cap = (s: string) =>
s.length <= MAX_OUTPUT ? s : `${s.slice(0, MAX_OUTPUT)}\n... [truncated ${s.length - MAX_OUTPUT} chars]`;
const lines = (text: string) => text.split('\n');
async function readLines(path: string): Promise<{ abs: string; lines: string[] }> {
const abs = jail(path);
const file = Bun.file(abs);
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
return { abs, lines: lines(await file.text()) };
}
// ---------------------------------------------------------------------------
// edit
// ---------------------------------------------------------------------------
export const insertLinesTool = tool({
description:
'Insert lines at a 1-based position in a file, pushing the rest down. Cheaper and safer than a rewrite for adding a block in the middle.',
inputSchema: z.object({
path: z.string(),
line: z.number().int().min(1).describe('Insert before this 1-based line; one past the end appends'),
text: z.string().describe('The lines to insert'),
}),
execute: async ({ path, line, text }) => {
const { abs, lines: cur } = await readLines(path);
if (line > cur.length + 1) throw new Error(`line ${line} is past the end of ${path} (${cur.length} lines)`);
cur.splice(line - 1, 0, ...lines(text));
await Bun.write(abs, cur.join('\n'));
return `Inserted ${lines(text).length} line(s) at ${path}:${line}`;
},
});
export const deleteLinesTool = tool({
description: 'Delete an inclusive range of lines from a file. Refuses to delete the whole file; use delete_file for that.',
inputSchema: z.object({
path: z.string(),
start: z.number().int().min(1),
end: z.number().int().min(1),
}),
execute: async ({ path, start, end }) => {
if (end < start) throw new Error('end must be >= start');
const { abs, lines: cur } = await readLines(path);
if (end > cur.length) throw new Error(`end ${end} is past the end of ${path} (${cur.length} lines)`);
if (start === 1 && end === cur.length) throw new Error('that deletes the whole file; use delete_file instead');
cur.splice(start - 1, end - start + 1);
await Bun.write(abs, cur.join('\n'));
return `Deleted lines ${start}-${end} from ${path}`;
},
});
export const replaceLinesTool = tool({
description: 'Replace an inclusive range of lines with new text, in one write.',
inputSchema: z.object({
path: z.string(),
start: z.number().int().min(1),
end: z.number().int().min(1),
text: z.string().describe('Replacement content for the range'),
}),
execute: async ({ path, start, end, text }) => {
if (end < start) throw new Error('end must be >= start');
const { abs, lines: cur } = await readLines(path);
if (end > cur.length) throw new Error(`end ${end} is past the end of ${path} (${cur.length} lines)`);
cur.splice(start - 1, end - start + 1, ...lines(text));
await Bun.write(abs, cur.join('\n'));
return `Replaced lines ${start}-${end} in ${path}`;
},
});
export const appendFileTool = tool({
description: 'Append text to the end of a file without reading the whole thing into the edit.',
inputSchema: z.object({ path: z.string(), text: z.string() }),
execute: async ({ path, text }) => {
const { abs, lines: cur } = await readLines(path);
await Bun.write(abs, `${cur.join('\n').replace(/\n?$/, '\n')}${text.replace(/\n?$/, '')}\n`);
return `Appended ${lines(text).length} line(s) to ${path}`;
},
});
export const prependFileTool = tool({
description: 'Prepend text to the start of a file, e.g. a license header or an import block.',
inputSchema: z.object({ path: z.string(), text: z.string() }),
execute: async ({ path, text }) => {
const { abs, lines: cur } = await readLines(path);
await Bun.write(abs, `${text.replace(/\n?$/, '\n')}${cur.join('\n')}`);
return `Prepended ${lines(text).length} line(s) to ${path}`;
},
});
export const countLinesTool = tool({
description: 'Count lines in one file, or per file across a glob. A quick size read before deciding to open something large.',
inputSchema: z.object({
path: z.string().optional().describe('One file. Omit to use pattern instead'),
pattern: z.string().optional().describe('Glob, e.g. "src/**/*.ts", to count many files'),
}),
execute: async ({ path, pattern }) => {
if (!path && !pattern) throw new Error('pass a path or a pattern');
const out: string[] = [];
const glob = pattern ? new Bun.Glob(pattern) : undefined;
for await (const rel of walk({})) {
if (path && rel !== posix(path)) continue;
if (glob && !glob.match(rel)) continue;
const abs = resolve(process.cwd(), rel);
try {
const n = (await Bun.file(abs).text()).split('\n').length;
out.push(`${n}\t${rel}`);
} catch {
continue;
}
if (out.length >= 500) break;
}
return out.length ? cap(out.join('\n')) : 'No matching text files.';
},
});
// ---------------------------------------------------------------------------
// inspect
// ---------------------------------------------------------------------------
const MAX_TREE = 400;
export const treeTool = tool({
description:
'Indented directory tree from a path, honouring .gitignore, with directories first. Faster to scan than list_dir for a broad shape.',
inputSchema: z.object({
path: z.string().optional().describe('Start directory, default the workspace root'),
depth: z.number().int().min(1).max(6).optional().describe('Default 3'),
}),
execute: async ({ path = '.', depth = 3 }) => {
const prefix = path === '.' ? '' : `${posix(path).replace(/\/$/, '')}/`;
const rows: { rel: string; depth: number; dir: boolean }[] = [];
for await (const rel of walk({})) {
if (prefix && !rel.startsWith(prefix)) continue;
const rest = prefix ? rel.slice(prefix.length) : rel;
const parts = rest.split('/');
if (parts.length > depth) continue;
for (let d = 1; d <= parts.length; d++) {
const ancestor = parts.slice(0, d).join('/');
if (!rows.some((r) => r.rel === ancestor)) rows.push({ rel: ancestor, depth: d, dir: d < parts.length });
}
if (rows.length >= MAX_TREE) break;
}
rows.sort((a, b) => a.rel.localeCompare(b.rel));
const out = rows.map((r) => `${' '.repeat(r.depth - 1)}${r.rel.split('/').at(-1)}${r.dir ? '/' : ''}`);
return out.length ? cap((prefix ? `${prefix.replace(/\/$/, '')}/\n` : './\n') + out.join('\n')) : `Nothing under ${path}.`;
},
});
export const fileInfoTool = tool({
description: 'Metadata for one file: size, line count, modified time, and whether it is text or binary.',
inputSchema: z.object({ path: z.string() }),
execute: async ({ path }) => {
const abs = jail(path);
let entry: Awaited<ReturnType<typeof stat>>;
try {
entry = await stat(abs);
} catch {
throw new Error(`No such file: ${path}`);
}
if (entry.isDirectory()) return `${path}: directory`;
const bytes = new Uint8Array(await Bun.file(abs).slice(0, 8192).arrayBuffer());
const binary = bytes.includes(0);
const linesN = binary ? undefined : (await Bun.file(abs).text()).split('\n').length;
return `${path}: ${entry.size} bytes${linesN === undefined ? '' : `, ${linesN} lines`}, ${binary ? 'binary' : 'text'}, modified ${entry.mtime.toISOString()}`;
},
});
export const findFilesTool = tool({
description: 'Find files whose *name* contains a substring (not a glob), e.g. "auth" or ".test.". Honours .gitignore.',
inputSchema: z.object({
name: z.string().describe('Substring to match against the filename'),
limit: z.number().int().min(1).optional().describe('Default 100'),
}),
execute: async ({ name, limit = 100 }) => {
const needle = name.toLowerCase();
const hits: string[] = [];
for await (const rel of walk({})) {
if ((rel.split('/').at(-1) ?? '').toLowerCase().includes(needle)) hits.push(rel);
if (hits.length >= limit) break;
}
return hits.length ? cap(hits.join('\n')) : `No files matching "${name}".`;
},
});
export const recentFilesTool = tool({
description: 'Files modified most recently, newest first. Orient in a tree you did not write, or find what a tool just touched.',
inputSchema: z.object({ limit: z.number().int().min(1).optional().describe('Default 20') }),
execute: async ({ limit = 20 }) => {
const seen: { rel: string; mtime: number }[] = [];
for await (const rel of walk({})) {
try {
const s = await stat(resolve(process.cwd(), rel));
seen.push({ rel, mtime: s.mtimeMs });
} catch {
continue;
}
}
seen.sort((a, b) => b.mtime - a.mtime);
const out = seen.slice(0, limit).map((s) => `${new Date(s.mtime).toISOString().slice(0, 19).replace('T', ' ')} ${s.rel}`);
return out.length ? cap(out.join('\n')) : 'No files found.';
},
});
export const changedFilesTool = tool({
description: 'Files git reports as modified, staged, or untracked — the working-tree delta at a glance, without a full status.',
inputSchema: z.object({}),
execute: async () => {
const result = await git(['status', '--porcelain'], process.cwd());
if (!result.ok) throw new Error(result.message);
const out = result.stdout
.split('\n')
.filter(Boolean)
.map((l) => `${l.slice(0, 2).trim() || ' '} ${posix(l.slice(3))}`);
return out.length ? cap(out.join('\n')) : 'Working tree clean.';
},
});
// ---------------------------------------------------------------------------
// git ext (read-only, argv-spawned)
// ---------------------------------------------------------------------------
const gitRun = async (args: string[], empty: string): Promise<string> => {
const result = await git(args, process.cwd());
if (!result.ok) throw new Error(result.message);
return cap(result.stdout.trim() || empty);
};
export const gitLogFileTool = tool({
description: 'Commits that touched one file, newest first, with hash, date, and subject.',
inputSchema: z.object({
path: z.string(),
limit: z.number().int().min(1).optional().describe('Default 15'),
}),
execute: async ({ path, limit = 15 }) =>
gitRun(['log', `--max-count=${limit}`, '--pretty=format:%h %ad %s', '--date=short', '--', posix(path)], 'No history for that file.'),
});
export const gitDiffCommitsTool = tool({
description: 'Diff between two refs (branches, tags, or commits), optionally limited to one path.',
inputSchema: z.object({
from: z.string().describe('Base ref'),
to: z.string().describe('Target ref'),
path: z.string().optional().describe('Limit the diff to this file'),
}),
execute: async ({ from, to, path }) =>
gitRun(['diff', `${from}...${to}`, ...(path ? ['--', posix(path)] : [])], `No differences between ${from} and ${to}.`),
});
export const gitShowFileTool = tool({
description: 'The contents of a file at a ref, e.g. what auth.ts looked like at HEAD~3 or on main.',
inputSchema: z.object({
ref: z.string().describe('Branch, tag, or commit'),
path: z.string(),
}),
execute: async ({ ref, path }) => gitRun(['show', `${ref}:${posix(path)}`], `No ${path} at ${ref}.`),
});
export const gitCurrentBranchTool = tool({
description: 'The current branch, plus its upstream and ahead/behind count when one is set.',
inputSchema: z.object({}),
execute: async () => gitRun(['status', '--short', '--branch'], 'no commits yet'),
});
export const gitChangedInRefTool = tool({
description: 'Files changed between a ref and the working tree, name only.',
inputSchema: z.object({ ref: z.string().describe('Compare the working tree against this ref, e.g. main') }),
execute: async ({ ref }) => gitRun(['diff', '--name-only', ref], `No changes against ${ref}.`),
});
// ---------------------------------------------------------------------------
// code
// ---------------------------------------------------------------------------
export const outlineTool = tool({
description:
'Top-level declarations of a source file — functions, classes, types, exports — as a compact structural map. Read this before opening a large file.',
inputSchema: z.object({ path: z.string() }),
execute: async ({ path }) => {
const abs = jail(path);
const file = Bun.file(abs);
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
const decl = /^\s*(export\s+(default\s+)?)?(async\s+)?(function|class|interface|type|enum|const|let|var|def|func|fn|struct|impl|trait|pub)\b/;
const out: string[] = [];
(await file.text()).split('\n').forEach((l, i) => {
if (decl.test(l)) out.push(`${i + 1}: ${l.trim().slice(0, 120)}`);
});
return out.length ? cap(out.join('\n')) : `No top-level declarations found in ${path}.`;
},
});
export const readSymbolTool = tool({
description: 'The full body of one top-level definition (function, class, type) from a file, by name.',
inputSchema: z.object({
path: z.string(),
name: z.string().describe('The identifier to extract'),
}),
execute: async ({ path, name }) => {
const abs = jail(path);
const file = Bun.file(abs);
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
const src = (await file.text()).split('\n');
const start = src.findIndex((l) => new RegExp(`\\b${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`).test(l) && !/^\s*(\/\/|#)/.test(l));
if (start === -1) throw new Error(`No definition of "${name}" found in ${path}`);
// Walk forward until the indentation returns to the declaration's level, which
// is the end of the block for brace and indentation languages alike.
const indent = /^(\s*)/.exec(src[start]!)![1]!.length;
let end = start;
for (let i = start + 1; i < src.length; i++) {
const l = src[i]!;
if (l.trim() === '') continue;
if (/^(\s*)/.exec(l)![1]!.length <= indent && l.trim() !== '}' && l.trim() !== '};') break;
end = i;
}
return cap(src.slice(start, end + 1).map((l, i) => `${start + i + 1}: ${l}`).join('\n'));
},
});
export const envInfoTool = tool({
description: 'Platform, shell, runtimes, and package managers present, so commands are written for what is actually installed.',
inputSchema: z.object({}),
execute: async () => {
const probe = async (bin: string, args: string[]) => {
try {
const proc = Bun.spawn([bin, ...args], { stdout: 'pipe', stderr: 'ignore' });
const out = await new Response(proc.stdout).text();
await proc.exited;
return out.trim().split('\n')[0] ?? 'present';
} catch {
return undefined;
}
};
const rows = [`platform: ${process.platform} ${process.arch}`, `cwd: ${process.cwd()}`];
for (const [label, bin, args] of [
['node', 'node', ['--version']],
['bun', 'bun', ['--version']],
['git', 'git', ['--version']],
['npm', 'npm', ['--version']],
['python', 'python', ['--version']],
['rg', 'rg', ['--version']],
] as const) {
const v = await probe(bin, [...args]);
if (v) rows.push(`${label}: ${v}`);
}
return rows.join('\n');
},
});
export const countTokensTool = tool({
description: 'Estimate the token cost of a file or a string before sending it to the model (~4 chars per token).',
inputSchema: z.object({
path: z.string().optional().describe('A file to measure'),
text: z.string().optional().describe('Or a string to measure'),
}),
execute: async ({ path, text }) => {
let content = text;
if (content === undefined) {
if (!path) throw new Error('pass a path or text');
const abs = jail(path);
const file = Bun.file(abs);
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
content = await file.text();
}
const chars = content.length;
return `${path ?? 'input'}: ${chars} chars, ~${Math.round(chars / 4)} tokens`;
},
});
/** The 20, registered by name for the tools map and the `extra` tool set. */
export const extraTools = {
insert_lines: insertLinesTool,
delete_lines: deleteLinesTool,
replace_lines: replaceLinesTool,
append_file: appendFileTool,
prepend_file: prependFileTool,
count_lines: countLinesTool,
tree: treeTool,
file_info: fileInfoTool,
find_files: findFilesTool,
recent_files: recentFilesTool,
changed_files: changedFilesTool,
git_log_file: gitLogFileTool,
git_diff_commits: gitDiffCommitsTool,
git_show_file: gitShowFileTool,
git_current_branch: gitCurrentBranchTool,
git_changed_in_ref: gitChangedInRefTool,
outline: outlineTool,
read_symbol: readSymbolTool,
env_info: envInfoTool,
count_tokens: countTokensTool,
};
export const EXTRA_TOOL_NAMES = Object.keys(extraTools);
+115 -1
View File
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
import { join, resolve } from 'node:path';
import { z } from 'zod';
import { jail, posix, walk } from './ignore';
import { EXTRA_TOOL_NAMES, extraTools } from './tools-extra';
import { GIT_TOOL_NAMES, gitTools } from './tools-git';
import { NET_TOOL_NAMES, netTools } from './tools-net';
@@ -734,6 +735,114 @@ export const deleteFileTool = tool({
},
});
/**
* Definition patterns for `find_symbol`, keyed loosely by language.
*
* Each entry matches the line where a symbol of that shape is *introduced* — a
* declaration, not a use — so the agent can jump to a definition instead of
* reading whole files to find it. `name` is interpolated escaped, so a symbol
* that is a regex metacharacter cannot break the pattern.
*/
const SYMBOL_PATTERNS: { re: (name: string) => string }[] = [
// JS/TS: function foo(, const foo =, class foo, foo(, export ... foo
{ re: (n) => `^(export\\s+)?(async\\s+)?(function\\s+${n}|(const|let|var)\\s+${n}\\s*=|class\\s+${n}\\b|interface\\s+${n}\\b|type\\s+${n}\\b|enum\\s+${n}\\b)` },
// Python: def foo(, class foo
{ re: (n) => `^(async\\s+)?(def\\s+${n}\\s*\\(|class\\s+${n}\\b)` },
// Go/Rust/Java-ish: func foo(, fn foo(, struct foo
{ re: (n) => `^(pub\\s+)?(func\\s+(\\(.*\\)\\s*)?${n}\\s*\\(|fn\\s+${n}\\s*\\(|struct\\s+${n}\\b|impl\\s+${n}\\b)` },
];
const escapeRe = (s: string) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
const MAX_SYMBOL_HITS = 40;
export const findSymbolTool = tool({
description:
'Locate where a function, class, type, or constant is *defined*, across JS/TS, Python, Go, and Rust. ' +
'Returns path:line hits. Faster and more precise than grep for "where is X declared", because it matches ' +
'declarations rather than every use.',
inputSchema: z.object({
name: z.string().describe('The exact identifier to find, e.g. parseConfig'),
include: z.string().optional().describe('Glob limiting which files are searched, default "**/*"'),
}),
execute: async ({ name, include = '**/*' }) => {
const trimmed = name.trim();
if (!trimmed) throw new Error('a symbol name is required');
const n = escapeRe(trimmed);
const glob = new Bun.Glob(include);
const hits: string[] = [];
for await (const rel of walk({})) {
if (!glob.match(rel)) continue;
const abs = resolve(process.cwd(), rel);
let lines: string[];
try {
if (await isBinary(abs)) continue;
lines = (await Bun.file(abs).text()).split('\n');
} catch {
continue;
}
for (let i = 0; i < lines.length; i++) {
const line = lines[i] ?? '';
if (line.trimStart().startsWith('//') || line.trimStart().startsWith('#')) continue;
if (SYMBOL_PATTERNS.some((p) => new RegExp(p.re(n)).test(line))) {
hits.push(`${rel}:${i + 1}: ${line.trim().slice(0, 160)}`);
break; // one declaration per file is the useful answer; more is noise.
}
if (hits.length >= MAX_SYMBOL_HITS) break;
}
if (hits.length >= MAX_SYMBOL_HITS) break;
}
return hits.length ? cap(hits.join('\n')) : `No definition of "${trimmed}" found.`;
},
});
/**
* A dotted-path lookup into a JSON document, so a large manifest, lockfile, or
* config can be read one value at a time instead of entering the context whole.
* `a.b.0.c` walks objects and arrays; a missing segment reports the path that
* resolved, so a wrong key is diagnosable rather than a bare "undefined".
*/
export const jsonQueryTool = tool({
description:
'Read one value out of a JSON file by dotted path (e.g. "scripts.build" or "dependencies.react"). ' +
'Use it on large manifests and configs instead of reading the whole file into context.',
inputSchema: z.object({
path: z.string().describe('JSON file, relative to the workspace root'),
query: z.string().describe('Dotted path into the document, e.g. "scripts.build". Array indexes are numeric segments.'),
}),
execute: async ({ path, query }) => {
const abs = jail(path);
const file = Bun.file(abs);
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
let doc: unknown;
try {
doc = JSON.parse(await file.text());
} catch (e) {
throw new Error(`${path} is not valid JSON: ${(e as Error).message}`);
}
let node: unknown = doc;
const walked: string[] = [];
for (const seg of query.split('.').filter(Boolean)) {
if (node === null || typeof node !== 'object') {
throw new Error(`"${walked.join('.') || '(root)'}" is ${node === null ? 'null' : typeof node}, not an object; cannot read "${seg}"`);
}
const record = node as Record<string, unknown>;
if (!(seg in record)) {
const keys = Object.keys(record).slice(0, 12).join(', ');
throw new Error(`no key "${seg}" under "${walked.join('.') || '(root)'}". Keys here: ${keys}${Object.keys(record).length > 12 ? ', …' : ''}`);
}
node = record[seg];
walked.push(seg);
}
const rendered = typeof node === 'string' ? node : JSON.stringify(node, null, 2);
return cap(`${query} = ${rendered}`);
},
});
export const tools = {
read_file: readFileTool,
read_many_files: readManyFilesTool,
@@ -746,9 +855,12 @@ export const tools = {
list_dir: listDirTool,
glob: globTool,
grep: grepTool,
find_symbol: findSymbolTool,
json_query: jsonQueryTool,
bash: bashTool,
...gitTools,
...netTools,
...extraTools,
};
/**
@@ -765,6 +877,8 @@ export const tools = {
export const TOOL_SETS = {
core: ['read_file', 'write_file', 'edit_file', 'glob', 'grep', 'bash'],
'edit-plus': ['multi_edit', 'list_dir', 'read_many_files', 'apply_patch', 'move_file', 'delete_file'],
nav: ['find_symbol', 'json_query'],
extra: EXTRA_TOOL_NAMES,
git: GIT_TOOL_NAMES,
net: NET_TOOL_NAMES,
} as const satisfies Record<string, readonly string[]>;
@@ -776,7 +890,7 @@ export const TOOL_SET_NAMES = Object.keys(TOOL_SETS) as ToolSetName[];
export const isToolSetName = (v: string): v is ToolSetName => (TOOL_SET_NAMES as string[]).includes(v);
/** Sets offered when the config says nothing. `net` is opt-in. */
export const DEFAULT_TOOL_SETS: ToolSetName[] = ['core', 'edit-plus', 'git'];
export const DEFAULT_TOOL_SETS: ToolSetName[] = ['core', 'edit-plus', 'nav', 'extra', 'git'];
/** Which set a tool came from, for `/tools`. Session, plugin, and MCP tools have none. */
export function toolSetOf(name: string): ToolSetName | undefined {
+123 -34
View File
@@ -1,6 +1,7 @@
import { Box, Static, Text, useApp, useInput, useStdout } from 'ink';
import React, { useCallback, useEffect, useRef, useState } from 'react';
import { parseCommand, matchCommands } from '../commands';
import { expandCommand, type CustomCommand } from '../custom-commands';
import { THINKING_LEVELS, VARIANTS } from '../agents';
import { completePath, matchPaths, pathToken } from '../complete';
import type { Config } from '../config';
@@ -19,6 +20,8 @@ import {
OutputPanel,
QueuePanel,
RegistryPanel,
Footer,
InputStatus,
StatusBar,
SubagentPanel,
ThinkingPanel,
@@ -32,6 +35,7 @@ import {
import { CommandMenu, InstallConfirm, Picker } from './Pickers';
import { contextPanel, costPanel, todosPanel, toolsPanel } from './panel-bodies';
import { PromptInput } from './PromptInput';
import { accent, glyph } from './theme';
import { nextKey, resultSummary, toolDetail, withResult, type Line, type NewLine } from './transcript';
export { createApprovalBridge, createNoticeBus, createSubagentBus, applySubagentEvent };
@@ -40,6 +44,8 @@ export type { ApprovalBridge, NoticeBus, SubagentBus };
/** Everything the slash commands need from the outside world. */
export type AppHooks = {
sessionId: string;
/** Session title for the welcome dashboard; absent in tests. */
title?: string;
config: () => Config;
switchModel: (id: string) => string;
switchAgent: (name: string) => string;
@@ -59,6 +65,8 @@ export type AppHooks = {
instructionFiles: () => string[];
/** Ignore-aware workspace paths for `@` completion, loaded on first use. */
listPaths: () => Promise<string[]>;
/** Custom slash commands from markdown files, for the menu and the parser. */
customCommands?: () => readonly CustomCommand[];
/** Registry index, installed set, and the install/remove actions. */
registry: {
list: () => Promise<RegistryRow[]>;
@@ -85,6 +93,8 @@ export function App({
session,
bridge,
header,
headerNode,
version,
hooks,
notices,
askBridge,
@@ -94,6 +104,10 @@ export function App({
session: Session;
bridge: ApprovalBridge;
header: string;
/** Rich welcome screen; when present it replaces the plain `header` string. */
headerNode?: React.ReactNode;
/** Build version, shown in the welcome dashboard's meta panel. */
version?: string;
hooks: AppHooks;
notices?: NoticeBus;
askBridge?: AskBridge;
@@ -101,7 +115,19 @@ export function App({
needsProvider?: boolean;
}) {
const { exit } = useApp();
const { write } = useStdout();
const { write, stdout } = useStdout();
// The footer splits hints left from context/cost right, and the input box and
// dashboards lay out against the real terminal width, so it is tracked and
// kept current on resize rather than read once.
const [termWidth, setTermWidth] = useState(stdout?.columns ?? 80);
useEffect(() => {
if (!stdout) return;
const onResize = () => setTermWidth(stdout.columns ?? 80);
stdout.on('resize', onResize);
return () => {
stdout.off('resize', onResize);
};
}, [stdout]);
const [history, setHistory] = useState<Line[]>([]);
const [draft, setDraft] = useState('');
const [live, setLive] = useState('');
@@ -141,7 +167,7 @@ export function App({
const modal =
pending !== undefined || asking !== undefined || onboarding || installing !== undefined || addingMcp;
const anyPicker = modelPicker !== undefined || agentPicker || thinkPicker;
const matches = matchCommands(draft);
const matches = matchCommands(draft, hooks.customCommands?.() ?? []);
const menuOpen = matches.length > 0 && !menuDismissed && !busy && !modal && !anyPicker && !panel;
const highlighted = matches[Math.min(menuIndex, matches.length - 1)];
@@ -439,7 +465,7 @@ export function App({
// Enter on an open menu runs the highlighted entry, so `/mo` + enter works.
const chosen = menuOpen && highlighted ? `/${highlighted.name}` : raw;
const action = parseCommand(chosen);
const action = parseCommand(chosen, hooks.customCommands?.() ?? []);
switch (action.type) {
case 'none':
@@ -495,6 +521,7 @@ export function App({
model: hooks.config().model,
agent: hooks.agentName(),
thinking: hooks.thinkingLevel(),
...(hooks.config().subagentModel ? { subagentModel: hooks.config().subagentModel! } : {}),
}),
);
return;
@@ -610,7 +637,8 @@ export function App({
push({ kind: 'user', text: chosen.trim() });
setWorking(true);
try {
push({ kind: 'info', text: await hooks.summarizeMemory() }); } catch (e) {
push({ kind: 'info', text: await hooks.summarizeMemory() });
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
@@ -681,6 +709,37 @@ export function App({
setRecall((h) => (h.at(-1) === action.text ? h : [...h, action.text]));
await runTurn(action.text);
return;
case 'custom': {
const typed = chosen.trim();
push({ kind: 'user', text: typed });
setWorking(true);
try {
// A command may pin an agent; it runs the prompt under that variant
// and restores afterwards, so one command does not leak its agent into
// the rest of the session.
const previous = hooks.agentName();
if (action.command.agent && action.command.agent !== previous) {
try {
hooks.switchAgent(action.command.agent);
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
}
const prompt = await expandCommand(action.command, action.args);
await runTurn(prompt);
if (action.command.agent && action.command.agent !== previous) {
try {
hooks.switchAgent(previous);
} catch {
// Restoring the agent is best-effort; the next /agent sets it explicitly.
}
}
} catch (e) {
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
}
setWorking(false);
return;
}
}
},
[exit, highlighted, hooks, menuOpen, push, runTurn, session, setWorking, unconfigured, write],
@@ -695,13 +754,24 @@ export function App({
<Static items={history}>
{(line) => (
<Box key={line.key} flexDirection="column" marginBottom={1}>
{line.kind === 'user' && <Text color="cyan">{`> ${line.text}`}</Text>}
{line.kind === 'assistant' && <Markdown text={line.text} />}
{line.kind === 'user' && (
<Text color={accent.user} bold>
{`${glyph.user} ${line.text}`}
</Text>
)}
{line.kind === 'assistant' && (
<Box>
<Text color={accent.ok}>{`${glyph.assistant} `}</Text>
<Box flexGrow={1} flexDirection="column">
<Markdown text={line.text} />
</Box>
</Box>
)}
{line.kind === 'tool' && (
<Box flexDirection="column">
<Box>
<Text color={line.ok ? 'magenta' : 'red'}>{line.ok ? '*' : 'x'} </Text>
<Text color={line.ok ? 'magenta' : 'red'} bold>
<Text color={line.ok ? accent.tool : accent.err}>{line.ok ? glyph.toolOk : glyph.toolErr} </Text>
<Text color={line.ok ? accent.tool : accent.err} bold>
{line.name}
</Text>
{line.detail[0] !== undefined && <Text dimColor>{` ${line.detail[0]}`}</Text>}
@@ -712,23 +782,26 @@ export function App({
</Text>
))}
{line.result !== undefined && line.result.length > 0 && (
<Text color={line.ok ? undefined : 'red'} dimColor={line.ok}>
{` ${line.ok ? '->' : 'x'} ${line.result}`}
<Text color={line.ok ? undefined : accent.err} dimColor={line.ok}>
{` ${line.ok ? glyph.result : glyph.err} ${line.result}`}
</Text>
)}
</Box>
)}
{line.kind === 'info' && <Text dimColor>{line.text}</Text>}
{line.kind === 'error' && <Text color="red">error: {line.text}</Text>}
{line.kind === 'info' && <Text dimColor>{`${glyph.info} ${line.text}`}</Text>}
{line.kind === 'error' && <Text color={accent.err}>{`${glyph.err} ${line.text}`}</Text>}
</Box>
)}
</Static>
{history.length === 0 && (
<Box marginBottom={1}>
<Text dimColor>{header}</Text>
</Box>
)}
{history.length === 0 &&
(headerNode !== undefined ? (
headerNode
) : (
<Box flexDirection="column" marginBottom={1}>
<Text dimColor>{header}</Text>
</Box>
))}
{agents.length > 0 && <SubagentPanel agents={agents} />}
@@ -869,19 +942,37 @@ export function App({
)}
{!modal && !anyPicker && (
<Box flexDirection="column">
<Box flexDirection="column" marginTop={1} width={termWidth}>
<QueuePanel prompts={queue} />
<Box>
<Text color="cyan">{'> '}</Text>
<PromptInput
key={inputGeneration}
value={draft}
initialCursor={inputCursor}
onChange={onDraftChange}
onSubmit={submit}
history={recall}
onKey={handleInputKey}
placeholder={busy ? 'type to queue for the next turn...' : 'ask shiro-neko... (/ commands, @ files)'}
<Box
flexDirection="column"
width={termWidth}
borderStyle="round"
borderColor={accent.mute}
borderLeftColor={busy ? accent.warn : accent.user}
paddingLeft={1}
paddingRight={1}
>
<Box>
<Text color={accent.user} bold>
{`${glyph.user} `}
</Text>
<PromptInput
key={inputGeneration}
value={draft}
initialCursor={inputCursor}
onChange={onDraftChange}
onSubmit={submit}
history={recall}
onKey={handleInputKey}
placeholder={busy ? 'type to queue for the next turn…' : 'ask shiro-neko… (/ commands, @ files)'}
/>
</Box>
<InputStatus
agent={hooks.agentName()}
model={hooks.config().model}
right={`${hooks.thinkingLevel()} ${glyph.info} ${session.activeTools().length} tools`}
width={termWidth - 6}
/>
</Box>
{fileOpen ? (
@@ -894,17 +985,15 @@ export function App({
) : (
menuOpen && <CommandMenu matches={matches} index={Math.min(menuIndex, matches.length - 1)} />
)}
<StatusBar
model={hooks.config().model}
agent={hooks.agentName()}
thinking={hooks.thinkingLevel()}
<Footer
busy={busy}
width={termWidth}
contextTokens={session.estimatedTokens()}
contextLimit={session.compactThreshold()}
cost={(() => {
const spend = costOf(hooks.config().model, session.inputTokens, session.outputTokens);
return spend === undefined ? 'unpriced' : formatUsd(spend);
})()}
toolCount={session.activeTools().length}
/>
</Box>
)}
+20 -7
View File
@@ -2,6 +2,7 @@ import { Box, Text, useInput } from 'ink';
import React from 'react';
import type { ApprovalDecision, ApprovalRequest } from '../session';
import { Diff } from './Diff';
import { accent, glyph } from './theme';
import { toolDetail } from './transcript';
export type Pending = { req: ApprovalRequest; resolve: (d: ApprovalDecision) => void };
@@ -85,9 +86,9 @@ export function Approval({ pending }: { pending: Pending }) {
const grant = req.suggestedPattern === '*' ? req.toolName : `${req.toolName} ${req.suggestedPattern}`;
return (
<Box flexDirection="column" borderStyle="round" borderColor="yellow" paddingX={1}>
<Text color="yellow" bold>
{reason(req)}
<Box flexDirection="column" borderStyle="round" borderColor={accent.warn} paddingX={1}>
<Text color={accent.warn} bold>
{`${glyph.warn} ${reason(req)}`}
</Text>
{req.repeated && <Text dimColor>allowed by the rules, but this is the third identical call this turn</Text>}
{req.subagent && !req.repeated && (
@@ -97,10 +98,22 @@ export function Approval({ pending }: { pending: Pending }) {
<Text dimColor>{`matched ${req.toolName}: "${req.matchedPattern}"`}</Text>
)}
<ApprovalDetail name={req.toolName} input={req.input} />
<Text>
<Text color="green">y</Text> allow once | <Text color="green">a</Text> always allow {grant} |{' '}
<Text color="red">n</Text> deny
</Text>
<Box marginTop={1}>
<Text>
<Text color={accent.ok} bold>
y
</Text>
<Text dimColor>{` allow once ${glyph.sep} `}</Text>
<Text color={accent.ok} bold>
a
</Text>
<Text dimColor>{` always allow ${grant} ${glyph.sep} `}</Text>
<Text color={accent.err} bold>
n
</Text>
<Text dimColor> deny</Text>
</Text>
</Box>
</Box>
);
}
+9 -4
View File
@@ -4,6 +4,7 @@ import React, { useState } from 'react';
import type { AskRequest } from '../ask';
import { InlineMarkdown } from './Markdown';
import { PromptInput } from './PromptInput';
import { accent, glyph } from './theme';
export type AskPending = { req: AskRequest; resolve: (answers: string[] | undefined) => void };
@@ -70,8 +71,8 @@ export function AskPanel({ pending }: { pending: AskPending }) {
const detailOf = (label: string) => (options ?? []).find((o) => o.label === label)?.detail;
return (
<Box flexDirection="column" borderStyle="double" borderColor="yellow" paddingX={1}>
<Text color="yellow" bold>
<Box flexDirection="column" borderStyle="double" borderColor={accent.warn} paddingX={1}>
<Text color={accent.warn} bold>
shiro is asking
</Text>
<Box marginBottom={1}>
@@ -80,7 +81,9 @@ export function AskPanel({ pending }: { pending: AskPending }) {
{typing ? (
<Box>
<Text color="yellow">{'> '}</Text>
<Text color={accent.warn} bold>
{`${glyph.user} `}
</Text>
<PromptInput
value={draft}
onChange={setDraft}
@@ -107,7 +110,9 @@ export function AskPanel({ pending }: { pending: AskPending }) {
/>
{draft.length > 0 && <Text dimColor>{draft}</Text>}
<Text dimColor>
{multiple ? 'space/enter toggles, pick submit when done' : 'enter to choose'} | esc to skip
{multiple
? `space/enter toggles, pick submit when done ${glyph.sep} esc to skip`
: `enter to choose ${glyph.sep} esc to skip`}
</Text>
</Box>
)}
+6 -5
View File
@@ -1,5 +1,6 @@
import { Box, Text } from 'ink';
import React from 'react';
import { accent, glyph } from './theme';
export type DiffLine = { kind: 'context' | 'add' | 'remove'; text: string; at: number };
@@ -97,21 +98,21 @@ export function Diff({ before, after, path }: { before: string; after: string; p
<Box flexDirection="column">
{path && (
<Text>
<Text bold>{path}</Text> <Text color="green">+{added}</Text> <Text color="red">-{removed}</Text>
<Text bold>{path}</Text> <Text color={accent.ok}>+{added}</Text> <Text color={accent.err}>-{removed}</Text>
</Text>
)}
{shown.map((line, i) =>
line.kind === 'gap' ? (
<Text key={i} dimColor>{` ... lines ${line.from}-${line.to} unchanged`}</Text>
<Text key={i} dimColor>{` ${glyph.fold} lines ${line.from}-${line.to} unchanged`}</Text>
) : (
<Text
key={i}
color={line.kind === 'add' ? 'green' : line.kind === 'remove' ? 'red' : undefined}
color={line.kind === 'add' ? accent.ok : line.kind === 'remove' ? accent.err : undefined}
dimColor={line.kind === 'context'}
>{` ${String(line.at).padStart(3)} ${line.kind === 'add' ? '+' : line.kind === 'remove' ? '-' : ' '} ${line.text}`}</Text>
>{` ${String(line.at).padStart(3)} ${line.kind === 'add' ? '+' : line.kind === 'remove' ? '-' : glyph.sep} ${line.text}`}</Text>
),
)}
{hidden > 0 && <Text dimColor>{` ... ${hidden} more diff lines`}</Text>}
{hidden > 0 && <Text dimColor>{` ${glyph.fold} ${hidden} more diff lines`}</Text>}
</Box>
);
}
+92
View File
@@ -0,0 +1,92 @@
import { Box, Text } from 'ink';
import React from 'react';
import { SidePanel } from './Panels';
import { accent, glyph } from './theme';
export type HeaderFact = {
label: string;
value: string;
/** Defaults to the quiet metadata colour; set for anything the user must not miss. */
tone?: 'warn' | 'err' | 'ok' | 'info';
};
const TONE_COLOR: Record<NonNullable<HeaderFact['tone']>, string> = {
warn: accent.warn,
err: accent.err,
ok: accent.ok,
info: accent.info,
};
/**
* The welcome dashboard, shown once before the first turn, in OpenCode's grammar.
*
* The session is introduced by a banner (what this conversation is), then the
* environment is grouped into a labelled panel so the eye scans one label rather
* than a wall of text. A final meta bar carries cwd and version, the two facts a
* bug report needs. Facts that demand attention — a missing provider, a failed
* plugin, `--yolo` — are lifted out of the quiet layer with colour, because a
* warning rendered dim is a warning nobody reads.
*/
export function Header({
version,
provider,
model,
sessionId,
cwd,
title,
facts,
}: {
version: string;
provider?: string;
model?: string;
sessionId: string;
cwd: string;
title?: string;
facts: readonly HeaderFact[];
}) {
return (
<Box flexDirection="column" marginBottom={1}>
<SidePanel label={title ?? 'new session'} tone={accent.user}>
<Box>
{model ? (
<>
<Text color={accent.ok}>{`${glyph.ok} `}</Text>
<Text bold>
{provider ? `${provider}/` : ''}
{model}
</Text>
<Text dimColor>{` ${glyph.sep} session ${sessionId}`}</Text>
</>
) : (
<>
<Text color={accent.warn}>{`${glyph.warn} `}</Text>
<Text color={accent.warn}>no provider configured</Text>
<Text dimColor>{` ${glyph.sep} run /provider to begin`}</Text>
</>
)}
</Box>
</SidePanel>
{facts.length > 0 && (
<SidePanel label="environment" tone={accent.ok}>
{facts.map((f, i) => (
<Box key={i}>
<Text dimColor>{f.label.padEnd(14)}</Text>
<Text color={f.tone ? TONE_COLOR[f.tone] : undefined} dimColor={f.tone === undefined}>
{f.value}
</Text>
</Box>
))}
</SidePanel>
)}
<Box paddingX={1}>
<Text dimColor>{cwd}</Text>
<Text dimColor>{` ${glyph.sep} `}</Text>
<Text dimColor>{`shiro-neko ${version}`}</Text>
<Text dimColor>{` ${glyph.sep} `}</Text>
<Text color={accent.user}>/help for commands</Text>
</Box>
</Box>
);
}
+211 -62
View File
@@ -4,28 +4,30 @@ import React from 'react';
import { TODO_MARK, type Todo } from '../notebook';
import type { SubagentKind } from '../subagent';
import { InlineMarkdown } from './Markdown';
import { accent, glyph, meter } from './theme';
const STATUS_COLOR: Record<Todo['status'], string | undefined> = {
pending: undefined,
in_progress: 'cyan',
done: 'green',
blocked: 'red',
in_progress: accent.user,
done: accent.ok,
blocked: accent.err,
};
/** Task list with a progress bar, shown above the input while a list exists. */
export function TodoPanel({ todos, width = 40 }: { todos: Todo[]; width?: number }) {
/** Task list with a progress meter, shown above the input while a list exists. */
export function TodoPanel({ todos, width = 24 }: { todos: Todo[]; width?: number }) {
const done = todos.filter((t) => t.status === 'done').length;
const blocked = todos.filter((t) => t.status === 'blocked').length;
const filled = todos.length === 0 ? 0 : Math.round((done / todos.length) * width);
const ratio = todos.length === 0 ? 0 : done / todos.length;
return (
<Box flexDirection="column" borderStyle="round" borderColor="gray" paddingX={1} marginBottom={1}>
<Box flexDirection="column" borderStyle="round" borderColor={accent.mute} paddingX={1} marginBottom={1}>
<Box>
<Text bold>tasks </Text>
<Text color="green">{'#'.repeat(filled)}</Text>
<Text dimColor>{'.'.repeat(Math.max(0, width - filled))}</Text>
<Text bold color={accent.user}>
tasks{' '}
</Text>
<Text color={accent.ok}>{meter(ratio, width)}</Text>
<Text dimColor>{` ${done}/${todos.length}`}</Text>
{blocked > 0 && <Text color="red">{` ${blocked} blocked`}</Text>}
{blocked > 0 && <Text color={accent.err}>{` ${glyph.blocked} ${blocked} blocked`}</Text>}
</Box>
{todos.map((t, i) => (
<Box key={i}>
@@ -60,7 +62,7 @@ export type SubagentView = {
const KIND_LABEL: Record<SubagentKind, string> = { explore: 'explore', review: 'review', worker: 'worker' };
/** A worker can write, so its panel entry has to be distinguishable at a glance. */
const KIND_COLOUR: Record<SubagentKind, string> = { explore: 'cyan', review: 'blue', worker: 'yellow' };
const KIND_COLOUR: Record<SubagentKind, string> = { explore: accent.user, review: accent.info, worker: accent.warn };
/**
* Live view of delegated work.
@@ -74,33 +76,35 @@ export function SubagentPanel({ agents, steps = 4 }: { agents: SubagentView[]; s
if (agents.length === 0) return null;
return (
<Box flexDirection="column" borderStyle="round" borderColor="magenta" paddingX={1} marginBottom={1}>
<Box flexDirection="column" borderStyle="round" borderColor={accent.tool} paddingX={1} marginBottom={1}>
{agents.map((a) => (
<Box key={a.id} flexDirection="column">
<Box>
{a.status === 'running' ? (
<Text color="magenta">
<Text color={accent.tool}>
<Spinner type="dots" />
</Text>
) : (
<Text color={a.status === 'done' ? 'green' : 'red'}>{a.status === 'done' ? '*' : 'x'}</Text>
<Text color={a.status === 'done' ? accent.ok : accent.err}>
{a.status === 'done' ? glyph.ok : glyph.err}
</Text>
)}
<Text bold color={KIND_COLOUR[a.kind]}>{` ${KIND_LABEL[a.kind]}`}</Text>
{a.kind === 'worker' && <Text color="yellow">{' (writes)'}</Text>}
<Text>{`: ${a.description}`}</Text>
{a.kind === 'worker' && <Text color={accent.warn}>{' (writes)'}</Text>}
<Text>{` ${glyph.sep} ${a.description}`}</Text>
<Text dimColor>{` ${a.steps.length} step${a.steps.length === 1 ? '' : 's'}`}</Text>
</Box>
{a.steps.slice(-steps).map((s, i) => (
<Box key={i} flexDirection="column">
<Text dimColor>{` ${s.tool}(${s.summary.slice(0, 58)})`}</Text>
{s.outcome !== undefined && (
<Text color={s.ok === false ? 'red' : undefined} dimColor={s.ok !== false}>
{` ${s.ok === false ? 'x' : '->'} ${s.outcome}`}
<Text color={s.ok === false ? accent.err : undefined} dimColor={s.ok !== false}>
{` ${s.ok === false ? glyph.err : glyph.result} ${s.outcome}`}
</Text>
)}
</Box>
))}
{a.error && <Text color="red">{` ${a.error}`}</Text>}
{a.error && <Text color={accent.err}>{` ${a.error}`}</Text>}
</Box>
))}
</Box>
@@ -123,10 +127,10 @@ const SLOW_AFTER = 10;
*/
export function Working({ seconds }: { seconds: number }) {
return (
<Text color="yellow">
<Text color={accent.warn}>
<Spinner type="dots" />{' '}
<Text dimColor>
{seconds >= SLOW_AFTER ? `working ${elapsed(seconds)}... esc to interrupt` : 'working... esc to interrupt'}
{seconds >= SLOW_AFTER ? `working ${elapsed(seconds)} ${glyph.sep} esc to interrupt` : `working ${glyph.sep} esc to interrupt`}
</Text>
</Text>
);
@@ -142,7 +146,7 @@ export function OutputPanel({ text, lines = 8 }: { text: string; lines?: number
.slice(-lines)
.map((l, i) => (
<Text key={i} dimColor>
{` | ${l}`}
{` ${glyph.sep} ${l}`}
</Text>
))}
</Box>
@@ -161,10 +165,10 @@ export function ActiveTool({ name, detail = [] }: { name: string; detail?: reado
return (
<Box flexDirection="column">
<Box>
<Text color="magenta">
<Text color={accent.tool}>
<Spinner type="dots" />
</Text>
<Text bold>{` ${name}`}</Text>
<Text bold color={accent.tool}>{` ${name}`}</Text>
{detail[0] !== undefined && <Text dimColor>{` ${detail[0]}`}</Text>}
</Box>
{detail.slice(1, 6).map((d, i) => (
@@ -172,7 +176,7 @@ export function ActiveTool({ name, detail = [] }: { name: string; detail?: reado
{` ${d}`}
</Text>
))}
{detail.length > 6 && <Text dimColor>{` ... ${detail.length - 6} more`}</Text>}
{detail.length > 6 && <Text dimColor>{` ${glyph.fold} ${detail.length - 6} more`}</Text>}
</Box>
);
}
@@ -189,12 +193,12 @@ export function ThinkingPanel({ text, expanded, lines = 8 }: { text: string; exp
const tokens = Math.round(text.length / 4);
if (!expanded) {
return <Text dimColor>{`thinking... ~${tokens} tokens ctrl-r to expand`}</Text>;
return <Text dimColor>{`thinking ${glyph.fold} ~${tokens} tokens ${glyph.sep} ctrl-r to expand`}</Text>;
}
return (
<Box flexDirection="column" marginBottom={1}>
<Text dimColor>{`thinking ~${tokens} tokens ctrl-r to collapse`}</Text>
<Text dimColor>{`thinking ~${tokens} tokens ${glyph.sep} ctrl-r to collapse`}</Text>
{text
.split('\n')
.slice(-lines)
@@ -212,10 +216,10 @@ export function QueuePanel({ prompts }: { prompts: readonly string[] }) {
if (prompts.length === 0) return null;
return (
<Box flexDirection="column">
<Text color="cyan">{`queued: ${prompts.length}`}</Text>
<Text color={accent.user}>{`${glyph.queued} queued: ${prompts.length}`}</Text>
{prompts.map((p, i) => (
<Text key={i} dimColor>
{` ${i + 1}. ${p.length > 70 ? `${p.slice(0, 70)}...` : p}`}
{` ${i + 1}. ${p.length > 70 ? `${p.slice(0, 70)}…` : p}`}
</Text>
))}
</Box>
@@ -237,7 +241,7 @@ export function FileMenu({
if (loading) {
return (
<Box marginTop={1}>
<Text dimColor>indexing files...</Text>
<Text dimColor>indexing files…</Text>
</Box>
);
}
@@ -253,17 +257,25 @@ export function FileMenu({
return (
<Box flexDirection="column" marginTop={1}>
{paths.map((p, i) => (
<Text key={p} color={i === index ? 'cyan' : undefined} dimColor={i !== index}>
{i === index ? '> ' : ' '}
<Text key={p} color={i === index ? accent.user : undefined} dimColor={i !== index}>
{i === index ? `${glyph.user} ` : ' '}
{p}
</Text>
))}
<Text dimColor>up/down move | tab or enter insert | esc dismiss</Text>
<Text dimColor>{`↑↓ move ${glyph.sep} tab/enter insert ${glyph.sep} esc dismiss`}</Text>
</Box>
);
}
/** Status line under the transcript: model, agent, thinking, context, spend. */
/**
* The persistent status line under the input.
*
* Three glance groups separated by quiet pipes: what is running (model, agent,
* thinking), how full the context is (a meter plus a percentage), and what it has
* cost. The context meter is the one piece of live information that changes
* mid-session, so it is the only element with a bar; everything else stays text so
* the bar is the thing the eye lands on.
*/
export function StatusBar({
model,
agent,
@@ -286,18 +298,144 @@ export function StatusBar({
// Amber from two thirds, red once compaction is imminent: the point is to warn
// before a turn silently loses its history, not after. Past 90 the colour is
// backed by words, because a reader watching the transcript is not watching this.
const contextColor = pct === undefined ? undefined : pct >= 90 ? 'red' : pct >= 66 ? 'yellow' : undefined;
const contextColor = pct === undefined ? undefined : pct >= 90 ? accent.err : pct >= 66 ? accent.warn : undefined;
const sep = <Text dimColor>{` ${glyph.sep} `}</Text>;
return (
<Box>
<Text dimColor>{`${model} `}</Text>
<Text color="cyan">{agent}</Text>
<Text dimColor>{`/${thinking} ${toolCount} tools `}</Text>
<Text dimColor>{model}</Text>
{sep}
<Text color={accent.user}>{agent}</Text>
<Text dimColor>{`/${thinking}`}</Text>
{sep}
<Text dimColor>{`${toolCount} tools`}</Text>
{sep}
<Text color={contextColor} dimColor={contextColor === undefined}>
{pct === undefined ? `~${contextTokens} ctx` : `${pct}% ctx`}
{pct === undefined ? `~${contextTokens} ctx` : `${meter(pct / 100, 8)} ${pct}%`}
</Text>
{pct !== undefined && pct >= 90 && <Text color="red">{' compacting soon'}</Text>}
<Text dimColor>{` ${cost}`}</Text>
{pct !== undefined && pct >= 90 && <Text color={accent.err}>{' compacting soon'}</Text>}
{sep}
<Text dimColor>{cost}</Text>
</Box>
);
}
/**
* The one-line identity row inside the input box, OpenCode-style.
*
* Agent and model are the two things a prompt is answered *by*, so they sit
* directly under the cursor rather than in a footer the eye has to travel to.
* `right` carries the quieter facts (tool count, thinking level) pushed to the
* far edge of the box, so the row reads as two anchored groups.
*/
export function InputStatus({
agent,
model,
right,
width,
}: {
agent: string;
model: string;
right?: string;
width: number;
}) {
const left = ` ${agent} · ${model}`;
const rightText = right ? `${right} ` : '';
const gap = Math.max(1, width - left.length - rightText.length);
return (
<Box>
<Text color={accent.user} bold>
{agent}
</Text>
<Text dimColor>{` ${glyph.info} ${model}`}</Text>
<Text dimColor>{' '.repeat(gap)}</Text>
{right !== undefined && <Text dimColor>{right}</Text>}
<Text> </Text>
</Box>
);
}
/**
* The full-width footer beneath the input, OpenCode-style.
*
* Key hints live on the left (what you can press), the live numbers on the right
* (context and cost). Spreading them to opposite edges means neither has to
* compete for the same glance, and the meter stays the most saturated thing on
* the line so the eye finds it first when a turn runs long.
*/
export function Footer({
busy,
contextTokens,
contextLimit,
cost,
width,
}: {
busy: boolean;
contextTokens: number;
contextLimit?: number;
cost: string;
width: number;
}) {
const pct = contextLimit ? Math.min(100, Math.round((contextTokens / contextLimit) * 100)) : undefined;
const contextColor = pct === undefined ? undefined : pct >= 90 ? accent.err : pct >= 66 ? accent.warn : undefined;
const hint = busy ? `${glyph.bullet} esc interrupt` : `/ commands ${glyph.sep} @ files ${glyph.sep} ↑ history`;
const right =
pct === undefined ? `~${contextTokens} ctx ${glyph.sep} ${cost}` : `${meter(pct / 100, 8)} ${pct}% ${glyph.sep} ${cost}`;
const gap = Math.max(1, width - hint.length - right.length - 1);
return (
<Box>
<Text dimColor>{hint}</Text>
<Text>{' '.repeat(gap)}</Text>
<Text color={contextColor} dimColor={contextColor === undefined}>
{right}
</Text>
{pct !== undefined && pct >= 90 && <Text color={accent.err}>{' compacting soon'}</Text>}
</Box>
);
}
/**
* A bordered panel with a coloured label header, OpenCode's sidebar grammar.
*
* Used for the welcome dashboard's grouped facts (capabilities, project, meta).
* The label carries the accent; the rows stay quiet, so a stack of these reads
* as labelled groups rather than as more transcript.
*/
export function SidePanel({
label,
children,
width,
tone = accent.ok,
}: {
label: string;
children: React.ReactNode;
width?: number;
tone?: string;
}) {
return (
<Box
flexDirection="column"
borderStyle="round"
borderColor={accent.mute}
paddingX={1}
marginBottom={1}
{...(width !== undefined ? { width } : {})}
>
<Text color={tone} bold>
{label}
</Text>
{children}
</Box>
);
}
/** One `name value` row inside a SidePanel, name padded so a stack aligns. */
export function SideRow({ name, value, nameWidth = 12, dim = true }: { name: string; value: string; nameWidth?: number; dim?: boolean }) {
return (
<Box>
<Text dimColor={dim}>{name.padEnd(nameWidth)}</Text>
<Text>{value}</Text>
</Box>
);
}
@@ -307,17 +445,19 @@ export type PanelLine = { label: string; value: string };
/** Bordered popup for a command's output, e.g. /skills or /cost. */
export function InfoPanel({ title, hint, lines }: { title: string; hint?: string; lines: PanelLine[] | string }) {
return (
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1} marginBottom={1}>
<Text color="cyan" bold>
{title}
</Text>
{hint && <Text dimColor>{hint}</Text>}
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1} marginBottom={1}>
<Box>
<Text color={accent.user} bold>
{title}
</Text>
{hint && <Text dimColor>{` ${hint}`}</Text>}
</Box>
{typeof lines === 'string' ? (
<InlineMarkdown text={lines} />
) : (
lines.map((l, i) => (
<Box key={i}>
<Text color="gray">{l.label.padEnd(14)}</Text>
<Text color={accent.mute}>{l.label.padEnd(14)}</Text>
<Text>{l.value}</Text>
</Box>
))
@@ -335,7 +475,7 @@ export type RegistryRow = {
installed?: boolean;
};
const KIND_COLOR: Record<RegistryRow['kind'], string> = { skill: 'green', plugin: 'magenta' };
const KIND_COLOR: Record<RegistryRow['kind'], string> = { skill: accent.ok, plugin: accent.tool };
/**
* The registry index as a table.
@@ -359,20 +499,22 @@ export function RegistryPanel({
const width = Math.min(22, Math.max(...rows.map((r) => r.name.length)) + 1);
return (
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1} marginBottom={1}>
<Text color="cyan" bold>
{title}
</Text>
{hint && <Text dimColor>{hint}</Text>}
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1} marginBottom={1}>
<Box>
<Text color={accent.user} bold>
{title}
</Text>
{hint && <Text dimColor>{` ${hint}`}</Text>}
</Box>
{rows.map((r) => (
<Box key={`${r.kind}:${r.name}`}>
<Text color={KIND_COLOR[r.kind]}>{r.kind === 'skill' ? 'S' : 'P'} </Text>
<Text bold>{r.name.padEnd(width)}</Text>
<Text dimColor>{r.description.length > 58 ? `${r.description.slice(0, 58)}...` : r.description}</Text>
{r.installed && <Text color="green">{' installed'}</Text>}
<Text dimColor>{r.description.length > 58 ? `${r.description.slice(0, 58)}…` : r.description}</Text>
{r.installed && <Text color={accent.ok}>{` ${glyph.ok} installed`}</Text>}
</Box>
))}
<Text dimColor>{'S skill P plugin | /registry add <name> | esc to dismiss'}</Text>
<Text dimColor>{`S skill P plugin ${glyph.sep} /registry add <name> ${glyph.sep} esc to dismiss`}</Text>
</Box>
);
}
@@ -402,8 +544,8 @@ export function InstallPrompt({
const hidden = body.length - shown.length;
return (
<Box flexDirection="column" borderStyle="round" borderColor="yellow" paddingX={1}>
<Text color="yellow" bold>
<Box flexDirection="column" borderStyle="round" borderColor={accent.warn} paddingX={1}>
<Text color={accent.warn} bold>
{`install ${kind} "${name}"?`}
</Text>
<Text dimColor>{url}</Text>
@@ -413,16 +555,23 @@ export function InstallPrompt({
{` ${l}`}
</Text>
))}
{hidden > 0 && <Text dimColor>{` ... ${hidden} more lines`}</Text>}
{hidden > 0 && <Text dimColor>{` ${glyph.fold} ${hidden} more lines`}</Text>}
</Box>
<Box marginTop={1} flexDirection="column">
<Text color="yellow">
<Text color={accent.warn}>
{kind === 'skill'
? 'A skill is instructions the agent follows. This text joins your system prompt.'
: 'A plugin adds refusal rules. It is data, not code: nothing here is executed.'}
</Text>
<Text>
<Text color="green">y</Text> install | <Text color="red">n</Text> cancel
<Text color={accent.ok} bold>
y
</Text>
<Text dimColor>{` install ${glyph.sep} `}</Text>
<Text color={accent.err} bold>
n
</Text>
<Text dimColor> cancel</Text>
</Text>
</Box>
</Box>
+14 -11
View File
@@ -3,6 +3,7 @@ import SelectInput from 'ink-select-input';
import React from 'react';
import type { CommandSpec } from '../commands';
import { InstallPrompt, type RegistryRow } from './Panels';
import { accent, glyph } from './theme';
/** The `/` menu, narrowing as the name is typed. */
export function CommandMenu({ matches, index }: { matches: readonly CommandSpec[]; index: number }) {
@@ -11,14 +12,14 @@ export function CommandMenu({ matches, index }: { matches: readonly CommandSpec[
<Box flexDirection="column" marginTop={1}>
{matches.map((c, i) => (
<Box key={c.name}>
<Text color={i === index ? 'cyan' : undefined}>{i === index ? '> ' : ' '}</Text>
<Text color={i === index ? 'cyan' : undefined} bold={i === index}>
<Text color={i === index ? accent.user : undefined}>{i === index ? `${glyph.user} ` : ' '}</Text>
<Text color={i === index ? accent.user : undefined} bold={i === index}>
{`/${c.name}${c.arg ? ` ${c.arg}` : ''}`.padEnd(width)}
</Text>
<Text dimColor>{c.summary}</Text>
</Box>
))}
<Text dimColor>up/down move | tab complete | enter run | esc dismiss</Text>
<Text dimColor>{`↑↓ move ${glyph.sep} tab complete ${glyph.sep} enter run ${glyph.sep} esc dismiss`}</Text>
</Box>
);
}
@@ -63,13 +64,13 @@ export function Frame({
children: React.ReactNode;
}) {
return (
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
<Text color="cyan" bold>
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1}>
<Text color={accent.user} bold>
{title}
</Text>
{hint && <Text dimColor>{hint}</Text>}
{warning && <Text color="yellow">could not list models: {warning}</Text>}
{error && <Text color="red">{error}</Text>}
{warning && <Text color={accent.warn}>could not list models: {warning}</Text>}
{error && <Text color={accent.err}>{error}</Text>}
{children}
</Box>
);
@@ -79,7 +80,7 @@ export function Frame({
export function Row({ label, children }: { label: string; children: React.ReactNode }) {
return (
<Box>
<Text color="cyan">{label}: </Text>
<Text color={accent.user}>{label}: </Text>
{children}
</Box>
);
@@ -109,11 +110,13 @@ export function Picker({
onSelect: (value: string) => void;
}) {
return (
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
<Text color="cyan" bold>
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1}>
<Text color={accent.user} bold>
{title}
</Text>
<Text dimColor>{hint ? `${hint} - enter to select, esc to cancel` : 'enter to select, esc to cancel'}</Text>
<Text dimColor>
{hint ? `${hint} ${glyph.sep} enter to select ${glyph.sep} esc to cancel` : `enter to select ${glyph.sep} esc to cancel`}
</Text>
<SelectInput
items={options.map((o) => ({ key: o.value, label: o.label, value: o.value }))}
limit={limit}
+28 -12
View File
@@ -30,20 +30,36 @@ export function toolsPanel(session: Session): Panel {
export function costPanel(
session: Session,
info: { sessionId: string; model: string; agent: string; thinking: string },
info: { sessionId: string; model: string; agent: string; thinking: string; subagentModel?: string },
): Panel {
const spend = costOf(info.model, session.inputTokens, session.outputTokens);
return {
title: 'cost',
hint: `session ${info.sessionId}`,
body: [
`- model: \`${info.model}\``,
`- billed: ${session.inputTokens} in / ${session.outputTokens} out`,
`- spend: ${spend === undefined ? 'unpriced model' : formatUsd(spend)}`,
`- context: ~${session.estimatedTokens()} tokens`,
`- agent: \`${info.agent}\` thinking \`${info.thinking}\``,
].join('\n'),
};
const lines = [
`- model: \`${info.model}\``,
`- billed: ${session.inputTokens} in / ${session.outputTokens} out`,
`- spend: ${spend === undefined ? 'unpriced model' : formatUsd(spend)}`,
];
// Subagent spend is priced against its own model id, which may be the cheaper
// one, so it is reported as its own line rather than folded into the parent's.
if (session.subagentInputTokens + session.subagentOutputTokens > 0) {
const subModel = info.subagentModel ?? info.model;
const subSpend = costOf(subModel, session.subagentInputTokens, session.subagentOutputTokens);
lines.push(
`- subagents: ${session.subagentInputTokens} in / ${session.subagentOutputTokens} out (\`${subModel}\`)${
subSpend === undefined ? '' : ` - ${formatUsd(subSpend)}`
}`,
);
}
const ceiling = session.spend();
if (ceiling.ceiling !== undefined) {
lines.push(
`- ceiling: ${ceiling.usd === undefined ? 'unpriced' : formatUsd(ceiling.usd)} of ${formatUsd(ceiling.ceiling)}`,
);
}
lines.push(`- context: ~${session.estimatedTokens()} tokens`, `- agent: \`${info.agent}\` thinking \`${info.thinking}\``);
return { title: 'cost', hint: `session ${info.sessionId}`, body: lines.join('\n') };
}
export function contextPanel(files: readonly string[]): Panel {

Some files were not shown because too many files have changed in this diff Show More