release 1.0.0: cost control, 41 tools, 29 skills, custom commands, auto-load
This commit is contained in:
+186
@@ -0,0 +1,186 @@
|
||||
import { tool, type ToolSet } from 'ai';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { jail } from './ignore';
|
||||
import { manifestToPlugin, parseManifest, type PluginManifest } from './registry';
|
||||
import type { Plugin } from './plugins';
|
||||
|
||||
/**
|
||||
* Auto-registration and auto-loading of external skills, tools, and plugins.
|
||||
*
|
||||
* Everything here is *data*, never code — the same rule the registry enforces.
|
||||
* An external tool is a bounded manifest (a shell template through the guard, an
|
||||
* HTTP fetch, or a file read), an external plugin a refusal manifest, an external
|
||||
* skill a markdown body. Loading arbitrary code from disk would let an entry read
|
||||
* every file the agent can read and lie about what it blocks, so it is not offered.
|
||||
*
|
||||
* Directories, later shadowing earlier by name:
|
||||
* ~/.shiro-neko/{tools,plugins,skills} (user)
|
||||
* .shiro/{tools,plugins,skills} (project)
|
||||
* Skills already load through skills.ts; this module adds tools and plugins and
|
||||
* the one place cli turns them all on.
|
||||
*/
|
||||
|
||||
const home = () => process.env['SHIRO_HOME'] ?? homedir();
|
||||
|
||||
export type LoadError = { name: string; message: string };
|
||||
|
||||
function dirs(kind: 'tools' | 'plugins' | 'skills', cwd: string): string[] {
|
||||
return [join(home(), '.shiro-neko', kind), join(cwd, '.shiro', kind)];
|
||||
}
|
||||
|
||||
async function scan(dir: string, ext: string): Promise<string[]> {
|
||||
const files: string[] = [];
|
||||
try {
|
||||
for await (const f of new Bun.Glob(`*.${ext}`).scan({ cwd: dir, onlyFiles: true })) files.push(f);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
return files.sort();
|
||||
}
|
||||
|
||||
const MAX_PATTERN = 200;
|
||||
const nameSchema = z.string().min(1).max(40).regex(/^[a-z0-9][a-z0-9-_]*$/i);
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// External tools, as bounded manifests.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Three kinds of tool, each with a ceiling on what it can do. None runs arbitrary
|
||||
* code: `shell` interpolates a fixed template and runs it through the guard and
|
||||
* the platform shell, `http` fetches a fixed URL, `read` returns a fixed file's
|
||||
* contents (jailed to the workspace). The input is a single optional `arg` string
|
||||
* substituted into a `{arg}` placeholder, so a manifest cannot take structure it
|
||||
* was not declared for.
|
||||
*/
|
||||
const toolManifestSchema = z.object({
|
||||
name: nameSchema,
|
||||
description: z.string().min(1).max(300),
|
||||
kind: z.enum(['shell', 'http', 'read']),
|
||||
/** The template with an optional `{arg}` placeholder. */
|
||||
command: z.string().max(500).optional(),
|
||||
url: z.string().max(500).optional(),
|
||||
path: z.string().max(300).optional(),
|
||||
/** Set false to require approval before running. Default true (auto-approved). */
|
||||
autoApprove: z.boolean().optional(),
|
||||
});
|
||||
|
||||
export type ToolManifest = z.infer<typeof toolManifestSchema>;
|
||||
|
||||
export function parseToolManifest(source: string): ToolManifest {
|
||||
let raw: unknown;
|
||||
try {
|
||||
raw = JSON.parse(source);
|
||||
} catch {
|
||||
throw new Error('the tool manifest is not valid JSON');
|
||||
}
|
||||
const parsed = toolManifestSchema.safeParse(raw);
|
||||
if (!parsed.success) {
|
||||
throw new Error(`the tool manifest is malformed: ${parsed.error.issues[0]?.message ?? 'unknown reason'}`);
|
||||
}
|
||||
const m = parsed.data;
|
||||
if (m.kind === 'shell' && !m.command) throw new Error(`shell tool "${m.name}" needs a command template`);
|
||||
if (m.kind === 'http' && !m.url) throw new Error(`http tool "${m.name}" needs a url`);
|
||||
if (m.kind === 'read' && !m.path) throw new Error(`read tool "${m.name}" needs a path`);
|
||||
return m;
|
||||
}
|
||||
|
||||
const MAX_TOOL_OUTPUT = 30_000;
|
||||
const cap = (s: string) => (s.length <= MAX_TOOL_OUTPUT ? s : `${s.slice(0, MAX_TOOL_OUTPUT)}\n... [truncated]`);
|
||||
|
||||
/** The guard an external shell tool runs through, supplied by cli so it shares the real chain. */
|
||||
export type ShellGuard = (command: string) => Promise<string | undefined>;
|
||||
|
||||
/**
|
||||
* A manifest as a live tool. The guard is applied to every `shell` invocation, so
|
||||
* an external tool cannot smuggle a destructive command past the user any more
|
||||
* than a built-in bash call can.
|
||||
*/
|
||||
export function manifestToTool(manifest: ToolManifest, guard: ShellGuard) {
|
||||
const inputSchema = z.object({ arg: z.string().optional().describe('optional argument substituted into {arg}') });
|
||||
const substitute = (template: string, arg: string) => template.replaceAll('{arg}', arg);
|
||||
|
||||
return tool({
|
||||
description: `${manifest.description} (external ${manifest.kind} tool)`,
|
||||
inputSchema,
|
||||
execute: async ({ arg = '' }) => {
|
||||
if (manifest.kind === 'read') {
|
||||
const abs = jail(substitute(manifest.path!, arg));
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`no such file: ${manifest.path}`);
|
||||
return cap(await file.text());
|
||||
}
|
||||
|
||||
if (manifest.kind === 'http') {
|
||||
const url = substitute(manifest.url!, arg);
|
||||
if (!/^https:\/\//i.test(url)) throw new Error(`http tools may only fetch https URLs, got: ${url}`);
|
||||
const res = await fetch(url, { redirect: 'follow', signal: AbortSignal.timeout(20_000) });
|
||||
if (!res.ok) throw new Error(`${url} returned ${res.status}`);
|
||||
return cap(await res.text());
|
||||
}
|
||||
|
||||
const command = substitute(manifest.command!, arg);
|
||||
const blocked = await guard(command);
|
||||
if (blocked) throw new Error(`refused: ${blocked}`);
|
||||
const shell = process.platform === 'win32' ? ['cmd', '/c', command] : ['bash', '-lc', command];
|
||||
const proc = Bun.spawn(shell, { stdout: 'pipe', stderr: 'pipe' });
|
||||
const [out, err, code] = await Promise.all([
|
||||
new Response(proc.stdout).text(),
|
||||
new Response(proc.stderr).text(),
|
||||
proc.exited,
|
||||
]);
|
||||
if (code !== 0) throw new Error(`exited ${code}: ${err.trim().slice(0, 300)}`);
|
||||
return cap(out.trim() || '(no output)');
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export type ExternalTools = { tools: ToolSet; autoApprove: string[]; errors: LoadError[] };
|
||||
|
||||
/** Loads every external tool manifest, project shadowing user by name. Bad files are reported and skipped. */
|
||||
export async function loadExternalTools(cwd: string, guard: ShellGuard): Promise<ExternalTools> {
|
||||
const tools: ToolSet = {};
|
||||
const autoApprove: string[] = [];
|
||||
const errors: LoadError[] = [];
|
||||
|
||||
for (const dir of dirs('tools', cwd)) {
|
||||
for (const file of await scan(dir, 'json')) {
|
||||
const fallback = file.replace(/\.json$/i, '');
|
||||
try {
|
||||
const manifest = parseToolManifest(await Bun.file(join(dir, file)).text());
|
||||
tools[manifest.name] = manifestToTool(manifest, guard);
|
||||
if (manifest.autoApprove !== false) autoApprove.push(manifest.name);
|
||||
} catch (e) {
|
||||
errors.push({ name: fallback, message: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}
|
||||
}
|
||||
return { tools, autoApprove, errors };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// External plugins, as refusal manifests (same shape the registry installs).
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type ExternalPlugins = { plugins: Plugin[]; errors: LoadError[] };
|
||||
|
||||
/** Loads refusal-manifest plugins from disk, merging with any already installed via the registry. */
|
||||
export async function loadExternalPlugins(cwd: string): Promise<ExternalPlugins> {
|
||||
const byName = new Map<string, Plugin>();
|
||||
const errors: LoadError[] = [];
|
||||
|
||||
for (const dir of dirs('plugins', cwd)) {
|
||||
for (const file of await scan(dir, 'json')) {
|
||||
const fallback = file.replace(/\.json$/i, '');
|
||||
try {
|
||||
const manifest: PluginManifest = parseManifest(await Bun.file(join(dir, file)).text());
|
||||
byName.set(manifest.name, manifestToPlugin(manifest));
|
||||
} catch (e) {
|
||||
errors.push({ name: fallback, message: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}
|
||||
}
|
||||
return { plugins: [...byName.values()], errors };
|
||||
}
|
||||
+88
-29
@@ -3,6 +3,7 @@ import { render } from 'ink';
|
||||
import React from 'react';
|
||||
import type { LanguageModel, ModelMessage } from 'ai';
|
||||
import { resolveAgent, VARIANTS, isThinkingLevel, type AgentVariant } from './agents';
|
||||
import { loadExternalPlugins, loadExternalTools } from './autoload';
|
||||
import { configPath, loadConfig, missingKeyMessage, resolveModel, writeConfigFile, type Config } from './config';
|
||||
import type { FallbackEvent } from './fallback';
|
||||
import { farewell } from './farewell';
|
||||
@@ -18,12 +19,14 @@ import { createHost } from './plugins';
|
||||
import { fetchModels, presetById } from './providers';
|
||||
import * as registry from './registry';
|
||||
import { Session } from './session';
|
||||
import { loadCustomCommands } from './custom-commands';
|
||||
import { loadSkills } from './skills';
|
||||
import * as store from './store';
|
||||
import { createTaskTool, type SubagentApproval } from './subagent';
|
||||
import { VERSION, versionLine } from './version';
|
||||
import { createAskBridge } from './ui/Ask';
|
||||
import { App, createApprovalBridge, createNoticeBus, createSubagentBus, type AppHooks } from './ui/App';
|
||||
import { Header, type HeaderFact } from './ui/Header';
|
||||
import type { RegistryRow as AppRegistryRow } from './ui/Panels';
|
||||
|
||||
// SDK warnings go straight to stderr, which tears up the Ink render.
|
||||
@@ -154,6 +157,7 @@ if (resumeArg) {
|
||||
const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers);
|
||||
const instructions = has('--no-instructions') ? [] : await loadInstructions();
|
||||
const skills = has('--no-skills') ? [] : await loadSkills();
|
||||
const customCommands = await loadCustomCommands();
|
||||
const promptHistory = await store.loadHistory();
|
||||
|
||||
const installedPlugins = has('--no-plugins') ? { plugins: [], errors: [] } : await registry.loadInstalledPlugins();
|
||||
@@ -202,9 +206,29 @@ const enabledPlugins = has('--no-plugins') ? [] : (cfg.plugins ?? DEFAULT_ENABLE
|
||||
const pluginErrors = enabledPlugins
|
||||
.filter((name) => !BUILTIN_PLUGINS.some((p) => p.name === name))
|
||||
.map((name) => ({ plugin: name, message: 'no such plugin' }));
|
||||
|
||||
// External skills, tools, and plugins auto-load from ~/.shiro-neko/<kind> and
|
||||
// .shiro/<kind>. All are data, never code; a bad file is reported, not fatal.
|
||||
const externalPlugins = has('--no-plugins') ? { plugins: [], errors: [] } : await loadExternalPlugins(process.cwd());
|
||||
|
||||
const plugins = createHost(
|
||||
[...BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)), ...installedPlugins.plugins],
|
||||
[...pluginErrors, ...installedPlugins.errors],
|
||||
[
|
||||
...BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)),
|
||||
...installedPlugins.plugins,
|
||||
...externalPlugins.plugins,
|
||||
],
|
||||
[
|
||||
...pluginErrors,
|
||||
...installedPlugins.errors,
|
||||
...externalPlugins.errors.map((e) => ({ plugin: e.name, message: e.message })),
|
||||
],
|
||||
);
|
||||
|
||||
// External shell tools run through the same guard chain as a built-in bash call,
|
||||
// so an installed tool cannot do what the agent itself may not. Late-bound because
|
||||
// the host above is what runs the chain.
|
||||
const externalTools = await loadExternalTools(process.cwd(), async (command) =>
|
||||
plugins.guard({ toolName: 'bash', input: { command }, cwd: process.cwd() }),
|
||||
);
|
||||
|
||||
const memory = has('--no-memory') ? undefined : new Memory(process.cwd(), languageModel);
|
||||
@@ -261,8 +285,23 @@ const subagentGate: SubagentApproval = (req) => {
|
||||
return approveSubagent(req);
|
||||
};
|
||||
|
||||
// A subagent doing search rather than reasoning can run on a cheaper model.
|
||||
// It resolves against the same provider and key, so a configured `subagentModel`
|
||||
// never needs a second credential.
|
||||
const subagentModel =
|
||||
cfg.subagentModel && cfg.subagentModel !== cfg.model && cfg.apiKey
|
||||
? resolveModel({ ...cfg, model: cfg.subagentModel }, reportFallback)
|
||||
: (languageModel ?? unconfiguredModel);
|
||||
|
||||
// Late-bound like `approveSubagent`: the task tool is built into `extraTools`
|
||||
// before the Session that owns the spend ledger exists, so the usage callback is
|
||||
// wired after construction.
|
||||
let recordSubagent: (usage: { inputTokens: number; outputTokens: number }) => void = () => {};
|
||||
|
||||
const session = new Session({
|
||||
model: languageModel ?? unconfiguredModel,
|
||||
modelId: cfg.model,
|
||||
...(cfg.subagentModel ? { subagentModelId: cfg.subagentModel } : {}),
|
||||
askApproval: bridge.ask,
|
||||
yolo,
|
||||
instructions,
|
||||
@@ -276,8 +315,10 @@ const session = new Session({
|
||||
...(memory ? { memory } : {}),
|
||||
...(record.notebook ? { notebook: record.notebook } : {}),
|
||||
...(cfg.maxRetries !== undefined ? { maxRetries: cfg.maxRetries } : {}),
|
||||
...(cfg.maxSpendUsd !== undefined ? { maxSpendUsd: cfg.maxSpendUsd } : {}),
|
||||
extraTools: {
|
||||
...(mcp?.tools ?? {}),
|
||||
...externalTools.tools,
|
||||
git_commit_message: createCommitMessageTool({
|
||||
model: languageModel ?? unconfiguredModel,
|
||||
...(headless ? {} : { cwd: process.cwd() }),
|
||||
@@ -287,6 +328,9 @@ const session = new Session({
|
||||
: {
|
||||
task: createTaskTool({
|
||||
model: languageModel ?? unconfiguredModel,
|
||||
subagentModel,
|
||||
subagentModelId: cfg.subagentModel,
|
||||
onUsage: (u) => recordSubagent(u),
|
||||
...(headless ? {} : { report: subagents.emit }),
|
||||
// A worker's writes go through the parent's rules and the parent's
|
||||
// prompt. Headless has nobody to answer, so `worker` is withheld there
|
||||
@@ -295,7 +339,7 @@ const session = new Session({
|
||||
}),
|
||||
}),
|
||||
},
|
||||
autoApprove: ['task', 'git_commit_message'],
|
||||
autoApprove: ['task', 'git_commit_message', ...externalTools.autoApprove],
|
||||
messages: [...record.messages],
|
||||
onChange: (messages) => {
|
||||
// Debounced so a long tool loop does not hit the disk on every step.
|
||||
@@ -305,6 +349,7 @@ const session = new Session({
|
||||
});
|
||||
|
||||
approveSubagent = session.approveForSubagent();
|
||||
recordSubagent = (u) => session.recordSubagentUsage(u);
|
||||
|
||||
async function shutdown(code: number): Promise<never> {
|
||||
clearTimeout(saveTimer);
|
||||
@@ -344,6 +389,7 @@ const hooks: AppHooks = {
|
||||
for await (const rel of walk({ limit: 5000 })) found.push(rel);
|
||||
return found;
|
||||
},
|
||||
customCommands: () => customCommands,
|
||||
registry: {
|
||||
list: async () => {
|
||||
const entries = await registry.fetchIndex(cfg.registryUrl);
|
||||
@@ -548,35 +594,46 @@ const hooks: AppHooks = {
|
||||
},
|
||||
};
|
||||
|
||||
const header = [
|
||||
needsProvider
|
||||
? `shiro-neko ${VERSION} no provider configured`
|
||||
: `shiro-neko ${VERSION} ${cfg.provider}/${record.model} session ${record.id.slice(0, 8)}`,
|
||||
`agent: ${agentVariant.name} thinking: ${agentVariant.thinking}`,
|
||||
`cwd: ${process.cwd()}`,
|
||||
restored ? `resumed ${record.messages.length} messages` : undefined,
|
||||
// The welcome dashboard's environment facts, in scan order. Anything that should
|
||||
// stop the user — a failed plugin, `--yolo`, a missing key — is given a tone so it
|
||||
// lifts out of the quiet metadata rather than blending into it.
|
||||
const facts: HeaderFact[] = [
|
||||
{ label: 'agent', value: `${agentVariant.name} thinking ${agentVariant.thinking}` },
|
||||
restored ? { label: 'resumed', value: `${record.messages.length} messages` } : undefined,
|
||||
instructions.length > 0
|
||||
? `instructions: ${instructions.map((i) => i.path.split(/[\\/]/).at(-1)).join(', ')}`
|
||||
: 'no AGENTS.md found - /init writes one',
|
||||
skills.length > 0 ? `skills: ${skills.map((s) => s.name).join(', ')}` : undefined,
|
||||
plugins.plugins.length > 0 ? `plugins: ${plugins.plugins.map((p) => p.name).join(', ')}` : undefined,
|
||||
...plugins.errors.map((e) => `plugin ${e.plugin}: ${e.message}`),
|
||||
memory && memory.all().length > 0 ? `memory: ${memory.all().length} notes about this project` : undefined,
|
||||
mcp && Object.keys(mcp.tools).length > 0 ? `mcp: ${Object.keys(mcp.tools).length} tools` : undefined,
|
||||
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
|
||||
? `mcp: ${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`
|
||||
? { label: 'instructions', value: instructions.map((i) => i.path.split(/[\\/]/).at(-1)!).join(', ') }
|
||||
: { label: 'instructions', value: 'none - /init writes an AGENTS.md', tone: 'info' },
|
||||
skills.length > 0 ? { label: 'skills', value: skills.map((s) => s.name).join(', ') } : undefined,
|
||||
plugins.plugins.length > 0
|
||||
? { label: 'plugins', value: plugins.plugins.map((p) => p.name).join(', ') }
|
||||
: undefined,
|
||||
...(mcp?.errors ?? []).map((e) => `mcp ${e.server} failed: ${e.message}`),
|
||||
...plugins.errors.map((e) => ({ label: 'plugin error', value: `${e.plugin}: ${e.message}`, tone: 'err' as const })),
|
||||
memory && memory.all().length > 0
|
||||
? { label: 'memory', value: `${memory.all().length} notes about this project` }
|
||||
: undefined,
|
||||
mcp && Object.keys(mcp.tools).length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).length} tools` } : undefined,
|
||||
!mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0
|
||||
? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const }
|
||||
: undefined,
|
||||
...(mcp?.errors ?? []).map((e) => ({ label: 'mcp error', value: `${e.server}: ${e.message}`, tone: 'err' as const })),
|
||||
yolo
|
||||
? 'approvals: OFF (--yolo), but deny rules and the guard still apply'
|
||||
? { label: 'approvals', value: 'OFF (--yolo) - deny rules and the guard still apply', tone: 'warn' as const }
|
||||
: cfg.permission
|
||||
? `approvals: rules for ${Object.keys(cfg.permission).join(', ')}, defaults elsewhere`
|
||||
: 'approvals: ask for write_file, edit_file, multi_edit, apply_patch, move_file, delete_file, bash, web_fetch, mcp__*',
|
||||
cfg.toolSets ? `tool sets: core, ${cfg.toolSets.join(', ')}` : undefined,
|
||||
'/help for commands',
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join('\n');
|
||||
? { label: 'approvals', value: `rules for ${Object.keys(cfg.permission).join(', ')}, defaults elsewhere` }
|
||||
: { label: 'approvals', value: 'ask for writes, bash, web_fetch, mcp' },
|
||||
cfg.toolSets ? { label: 'tool sets', value: `core, ${cfg.toolSets.join(', ')}` } : undefined,
|
||||
].filter((f): f is HeaderFact => f !== undefined);
|
||||
|
||||
const headerNode = (
|
||||
<Header
|
||||
version={VERSION}
|
||||
{...(needsProvider ? {} : { provider: cfg.provider, model: record.model })}
|
||||
sessionId={record.id.slice(0, 8)}
|
||||
cwd={process.cwd()}
|
||||
title={restored ? record.title : undefined}
|
||||
facts={facts}
|
||||
/>
|
||||
);
|
||||
|
||||
// ctrl-c has to reach the App: with a command running it kills that command and
|
||||
// keeps the turn. Ink's own handler would exit the process before we saw the key.
|
||||
@@ -584,7 +641,9 @@ const app = render(
|
||||
<App
|
||||
session={session}
|
||||
bridge={bridge}
|
||||
header={header}
|
||||
header=""
|
||||
headerNode={headerNode}
|
||||
version={VERSION}
|
||||
hooks={hooks}
|
||||
notices={notices}
|
||||
askBridge={askBridge}
|
||||
|
||||
+18
-6
@@ -1,3 +1,5 @@
|
||||
import type { CustomCommand } from './custom-commands';
|
||||
|
||||
export type CommandAction =
|
||||
| { type: 'none' }
|
||||
| { type: 'prompt'; text: string }
|
||||
@@ -24,6 +26,8 @@ export type CommandAction =
|
||||
| { type: 'info'; text: string }
|
||||
| { type: 'model'; model: string }
|
||||
| { type: 'resume'; id: string }
|
||||
/** A custom command from a markdown file, expanded against its arguments. */
|
||||
| { type: 'custom'; command: CustomCommand; args: string[] }
|
||||
| { type: 'unknown'; name: string };
|
||||
|
||||
export type CommandSpec = {
|
||||
@@ -81,11 +85,12 @@ export const HELP = [
|
||||
* An exact name sorts first so pressing enter on `/model` cannot run `/models`.
|
||||
* Aliases stay hidden to keep the list short.
|
||||
*/
|
||||
export function matchCommands(input: string): CommandSpec[] {
|
||||
export function matchCommands(input: string, custom: readonly CustomCommand[] = []): CommandSpec[] {
|
||||
if (!input.startsWith('/')) return [];
|
||||
const typed = input.slice(1).toLowerCase();
|
||||
if (typed.includes(' ')) return [];
|
||||
const hits = COMMANDS.filter((c) => c.name.startsWith(typed));
|
||||
const customSpecs: CommandSpec[] = custom.map((c) => ({ name: c.name, summary: c.description }));
|
||||
const hits = [...COMMANDS, ...customSpecs].filter((c) => c.name.startsWith(typed));
|
||||
const exact = hits.findIndex((c) => c.name === typed);
|
||||
return exact > 0 ? [hits[exact]!, ...hits.filter((_, i) => i !== exact)] : hits;
|
||||
}
|
||||
@@ -156,8 +161,13 @@ function parseMcp(arg: string): CommandAction {
|
||||
}
|
||||
}
|
||||
|
||||
/** Pure parser: no IO, so the TUI and headless mode share one definition. */
|
||||
export function parseCommand(raw: string): CommandAction {
|
||||
/**
|
||||
* Pure parser: no IO, so the TUI and headless mode share one definition.
|
||||
*
|
||||
* Custom commands are consulted only after every built-in name misses, so a
|
||||
* markdown file can add a command but never shadow one that ships with the binary.
|
||||
*/
|
||||
export function parseCommand(raw: string, custom: readonly CustomCommand[] = []): CommandAction {
|
||||
const input = raw.trim();
|
||||
if (!input) return { type: 'none' };
|
||||
if (!input.startsWith('/')) return { type: 'prompt', text: input };
|
||||
@@ -215,7 +225,9 @@ export function parseCommand(raw: string): CommandAction {
|
||||
return arg ? { type: 'model', model: arg } : { type: 'models' };
|
||||
case 'resume':
|
||||
return arg ? { type: 'resume', id: arg } : { type: 'info', text: 'usage: /resume <session-id>' };
|
||||
default:
|
||||
return { type: 'unknown', name };
|
||||
default: {
|
||||
const cmd = custom.find((c) => c.name === name);
|
||||
return cmd ? { type: 'custom', command: cmd, args: arg ? arg.split(/\s+/) : [] } : { type: 'unknown', name };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,6 +20,10 @@ export type Config = {
|
||||
presetId?: string;
|
||||
/** Retries per model call for transient failures. SDK default is 2. */
|
||||
maxRetries?: number;
|
||||
/** USD ceiling for a session's spend: warn at 80%, refuse the next turn at 100%. */
|
||||
maxSpendUsd?: number;
|
||||
/** Model id for subagents; omit to share the parent's. */
|
||||
subagentModel?: string;
|
||||
/** Default agent variant name. */
|
||||
agent?: string;
|
||||
/** Default thinking level. */
|
||||
@@ -90,6 +94,8 @@ export async function loadConfig(): Promise<Config> {
|
||||
apiKey: process.env['SHIRO_API_KEY'] ?? file.apiKey ?? process.env[ENV_KEY[provider]],
|
||||
...(file.presetId ? { presetId: file.presetId } : {}),
|
||||
...(file.maxRetries !== undefined ? { maxRetries: file.maxRetries } : {}),
|
||||
...(typeof file.maxSpendUsd === 'number' && file.maxSpendUsd > 0 ? { maxSpendUsd: file.maxSpendUsd } : {}),
|
||||
...(file.subagentModel ? { subagentModel: file.subagentModel } : {}),
|
||||
...(file.agent ? { agent: file.agent } : {}),
|
||||
...(file.thinking ? { thinking: file.thinking } : {}),
|
||||
...(Array.isArray(file.plugins) ? { plugins: file.plugins } : {}),
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { guardPlugin } from './plugins-builtin';
|
||||
|
||||
/**
|
||||
* Custom slash commands read from markdown files.
|
||||
*
|
||||
* `.shiro/commands/<name>.md` in the project and `~/.shiro-neko/commands/<name>.md`
|
||||
* for the user. The filename is the command; the body becomes the prompt. A project
|
||||
* command shadows a user command of the same name, so a repo can specialise a
|
||||
* personal default.
|
||||
*/
|
||||
export type CustomCommand = {
|
||||
name: string;
|
||||
/** One-line summary for the `/` menu, from frontmatter or the first body line. */
|
||||
description: string;
|
||||
/** Agent to run it under, when frontmatter sets one. */
|
||||
agent?: string;
|
||||
/** The prompt template, before substitution. */
|
||||
body: string;
|
||||
origin: 'project' | 'user';
|
||||
path: string;
|
||||
};
|
||||
|
||||
const MAX_BODY = 20_000;
|
||||
|
||||
/** Reads frontmatter `description` and `agent`; everything after the `---` fence is the prompt. */
|
||||
function parse(name: string, source: string, origin: CustomCommand['origin'], path: string): CustomCommand | undefined {
|
||||
const match = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/.exec(source.trimStart());
|
||||
const meta: Record<string, string> = {};
|
||||
let body = source;
|
||||
if (match) {
|
||||
for (const line of match[1]!.split(/\r?\n/)) {
|
||||
const kv = /^([A-Za-z_-]+)\s*:\s*(.*)$/.exec(line.trim());
|
||||
if (kv) meta[kv[1]!.toLowerCase()] = kv[2]!.replace(/^["']|["']$/g, '').trim();
|
||||
}
|
||||
body = match[2]!;
|
||||
}
|
||||
const trimmed = body.trim().slice(0, MAX_BODY);
|
||||
if (!trimmed) return undefined;
|
||||
const description = meta['description'] ?? trimmed.split('\n').find((l) => l.trim().length > 0)?.trim().slice(0, 60) ?? name;
|
||||
return {
|
||||
name,
|
||||
description,
|
||||
...(meta['agent'] ? { agent: meta['agent'] } : {}),
|
||||
body: trimmed,
|
||||
origin,
|
||||
path,
|
||||
};
|
||||
}
|
||||
|
||||
function commandDirs(cwd: string): { dir: string; origin: CustomCommand['origin'] }[] {
|
||||
const home = join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko');
|
||||
return [
|
||||
{ dir: join(home, 'commands'), origin: 'user' },
|
||||
{ dir: join(cwd, '.shiro', 'commands'), origin: 'project' },
|
||||
];
|
||||
}
|
||||
|
||||
/** Loads every custom command, project shadowing user by name. A file that fails to parse is skipped. */
|
||||
export async function loadCustomCommands(cwd = process.cwd()): Promise<CustomCommand[]> {
|
||||
const byName = new Map<string, CustomCommand>();
|
||||
for (const { dir, origin } of commandDirs(cwd)) {
|
||||
let files: string[] = [];
|
||||
try {
|
||||
for await (const f of new Bun.Glob('*.md').scan({ cwd: dir, onlyFiles: true })) files.push(f);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const file of files.sort()) {
|
||||
const name = file.replace(/\.md$/i, '');
|
||||
if (!/^[a-z0-9][a-z0-9-_]*$/i.test(name)) continue;
|
||||
const path = join(dir, file);
|
||||
try {
|
||||
const cmd = parse(name, await Bun.file(path).text(), origin, path);
|
||||
if (cmd) byName.set(cmd.name, cmd);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name));
|
||||
}
|
||||
|
||||
/** Runs a `` !`cmd` `` substitution through the guard before executing it. */
|
||||
async function runSubstitution(command: string): Promise<string> {
|
||||
const blocked = await guardPlugin.beforeToolCall!({ toolName: 'bash', input: { command }, cwd: process.cwd() });
|
||||
if (blocked) throw new Error(`shell substitution refused: ${blocked}`);
|
||||
|
||||
// The same shell bash uses, so a substitution and a bash call agree on syntax.
|
||||
const shell = process.platform === 'win32' ? ['cmd', '/c', command] : ['bash', '-lc', command];
|
||||
const proc = Bun.spawn(shell, { stdout: 'pipe', stderr: 'pipe' });
|
||||
const [out, err, code] = await Promise.all([
|
||||
new Response(proc.stdout).text(),
|
||||
new Response(proc.stderr).text(),
|
||||
proc.exited,
|
||||
]);
|
||||
if (code !== 0) throw new Error(`shell substitution \`!${command}\` exited ${code}: ${err.trim().slice(0, 200)}`);
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Expands a command's body against the arguments it was typed with.
|
||||
*
|
||||
* `$ARGUMENTS` is the whole argument string, `$1`, `$2`, … the positionals, and
|
||||
* `` !`cmd` `` runs a shell command and inlines its output — each such command
|
||||
* passed through the guard first, so a custom command cannot smuggle a destructive
|
||||
* call past the user the way a plain bash call cannot.
|
||||
*/
|
||||
export async function expandCommand(cmd: CustomCommand, args: string[]): Promise<string> {
|
||||
let out = cmd.body;
|
||||
out = out.replaceAll('$ARGUMENTS', args.join(' '));
|
||||
out = out.replace(/\$(\d+)/g, (_, i) => args[Number(i) - 1] ?? '');
|
||||
|
||||
const substitutions = [...out.matchAll(/!`([^`]+)`/g)];
|
||||
for (const m of substitutions) {
|
||||
const value = await runSubstitution(m[1]!);
|
||||
out = out.replace(m[0], value);
|
||||
}
|
||||
return out.trim();
|
||||
}
|
||||
+133
-3
@@ -218,6 +218,122 @@ export const formatPlugin: Plugin = {
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
* Bash command patterns a guard refuses, shared by several small plugins.
|
||||
*
|
||||
* Each plugin owns one concern so it can be toggled alone; they are data (a name,
|
||||
* a pattern list, an appendix), never code beyond the matcher they all share.
|
||||
*/
|
||||
const bashRefusal = (patterns: { re: RegExp; why: string }[]) => {
|
||||
return ({ toolName, input }: Parameters<NonNullable<Plugin['beforeToolCall']>>[0]) => {
|
||||
if (toolName !== 'bash') return undefined;
|
||||
const command = String((input as { command?: unknown } | null)?.command ?? '');
|
||||
if (!command) return undefined;
|
||||
for (const { re, why } of patterns) {
|
||||
if (re.test(command)) return `refusing "${command.slice(0, 120)}" (${why}). Run it yourself if it is really needed.`;
|
||||
}
|
||||
return undefined;
|
||||
};
|
||||
};
|
||||
|
||||
export const noForcePushPlugin: Plugin = {
|
||||
name: 'no-force-push',
|
||||
description: 'refuses any push that rewrites remote history',
|
||||
appendix: 'The no-force-push plugin refuses force pushes. Ask the user to run one by hand if it is truly intended.',
|
||||
beforeToolCall: bashRefusal([
|
||||
{ re: /\bgit\s+push\b[^|]*(--force\b|--force-with-lease\b|\s-f\b)/, why: 'rewrites remote history' },
|
||||
{ re: /\bgit\s+push\b[^|]*\s+\+/, why: 'a force push via refspec' },
|
||||
]),
|
||||
};
|
||||
|
||||
export const noMainCommitPlugin: Plugin = {
|
||||
name: 'no-main-commit',
|
||||
description: 'refuses to commit directly to main or master',
|
||||
appendix: 'The no-main-commit plugin refuses to commit to main/master. Create a branch and commit there instead.',
|
||||
beforeToolCall: bashRefusal([
|
||||
{ re: /\bgit\s+(commit|merge)\b[^|]*\b(main|master)\b/, why: 'touches the default branch directly' },
|
||||
{ re: /\bgit\s+checkout\s+(main|master)\b[^|]*&&[^|]*\bcommit\b/, why: 'commits on the default branch' },
|
||||
]),
|
||||
};
|
||||
|
||||
export const noRootPlugin: Plugin = {
|
||||
name: 'no-root',
|
||||
description: 'refuses commands run with sudo or as an elevated shell',
|
||||
appendix: 'The no-root plugin refuses sudo and elevation. Nothing the agent does should need it; ask the user to run it themselves.',
|
||||
beforeToolCall: bashRefusal([
|
||||
{ re: /(^|\s)sudo\b/, why: 'elevated privileges' },
|
||||
{ re: /\brunas\b|\bStart-Process\b[^|]*-Verb\s+RunAs/i, why: 'an elevated process' },
|
||||
]),
|
||||
};
|
||||
|
||||
export const noNetPipePlugin: Plugin = {
|
||||
name: 'no-net-pipe',
|
||||
description: 'refuses to execute anything downloaded straight into a shell',
|
||||
appendix: 'The no-net-pipe plugin refuses piping a download into an interpreter. Download, review the file, then run it.',
|
||||
beforeToolCall: bashRefusal([
|
||||
{ re: /\b(curl|wget)\b[^|]*\|\s*(ba|z|k)?sh\b|\b(curl|wget)\b[^|]*\|\s*(node|python|ruby|perl|bun)\b/i, why: 'executes a download unseen' },
|
||||
{ re: /\biex\b|\bInvoke-Expression\b[^|]*\b(iwr|Invoke-WebRequest|curl)\b/i, why: 'executes a download unseen' },
|
||||
]),
|
||||
};
|
||||
|
||||
export const noGitConfigPlugin: Plugin = {
|
||||
name: 'no-git-config',
|
||||
description: 'refuses to change git configuration or global state',
|
||||
appendix: 'The no-git-config plugin refuses to edit git config. Tell the user the exact config change to make themselves.',
|
||||
beforeToolCall: bashRefusal([
|
||||
{ re: /\bgit\s+config\b[^|]*(--global|--system)/, why: 'changes global git configuration' },
|
||||
{ re: /\bgit\s+config\b[^|]*(user\.(name|email)|core\.(sshCommand|editor|pager))\s+\S/, why: 'changes how git identifies or runs' },
|
||||
]),
|
||||
};
|
||||
|
||||
export const noEnvWritePlugin: Plugin = {
|
||||
name: 'no-env-write',
|
||||
description: 'refuses to print or export secrets into the shell environment',
|
||||
appendix: 'The no-env-write plugin refuses to export or echo credentials into the environment. The user sets their own secrets.',
|
||||
beforeToolCall: bashRefusal([
|
||||
{ re: /\b(export|setx?)\s+[A-Z_]*(KEY|TOKEN|SECRET|PASSWORD|PASSWD)\s*=/i, why: 'writes a credential into the environment' },
|
||||
{ re: /\becho\b[^|]*\b(api[_-]?key|secret|token|password)\b[^|]*>>?\s*\S/i, why: 'writes a credential to a file' },
|
||||
]),
|
||||
};
|
||||
|
||||
export const conventionalCommitPlugin: Plugin = {
|
||||
name: 'conventional-commit',
|
||||
description: 'nudges commit messages toward the conventional format',
|
||||
appendix:
|
||||
'The conventional-commit plugin is advisory: write commit subjects as type(scope): summary, e.g. ' +
|
||||
'`fix(auth): reject expired tokens`. Types: feat, fix, docs, style, refactor, perf, test, build, ci, chore, revert.',
|
||||
};
|
||||
|
||||
export const testsFirstPlugin: Plugin = {
|
||||
name: 'tests-first',
|
||||
description: 'reminds the agent to pin behaviour with a failing test before fixing',
|
||||
appendix:
|
||||
'The tests-first plugin is advisory: for a bug, write or find the test that reproduces it before changing code. ' +
|
||||
'Watch it fail, then fix, then watch it pass. A fix without a failing-then-passing test is unverified.',
|
||||
};
|
||||
|
||||
export const smallDiffsPlugin: Plugin = {
|
||||
name: 'small-diffs',
|
||||
description: 'reminds the agent to keep a change focused on one thing',
|
||||
appendix:
|
||||
'The small-diffs plugin is advisory: one change does one thing. Do not tidy, rename, or reformat outside the ' +
|
||||
'task. A diff that is hard to review is usually two diffs wearing one coat — split it.',
|
||||
};
|
||||
|
||||
export const confirmDeletePlugin: Plugin = {
|
||||
name: 'confirm-delete',
|
||||
description: 'refuses delete calls that name broad or ambiguous paths',
|
||||
appendix: 'The confirm-delete plugin refuses deletes that name a directory or a wildcard. Delete one explicit file at a time.',
|
||||
beforeToolCall: ({ toolName, input }) => {
|
||||
if (toolName !== 'delete_file') return undefined;
|
||||
const path = String((input as { path?: unknown } | null)?.path ?? '');
|
||||
if (/[*?[\]]/.test(path) || path.endsWith('/') || path === '.' || path === '') {
|
||||
return `refusing to delete "${path}" (ambiguous or broad). Delete one explicit file.`;
|
||||
}
|
||||
return undefined;
|
||||
},
|
||||
};
|
||||
|
||||
export const BUILTIN_PLUGINS: Plugin[] = [
|
||||
guardPlugin,
|
||||
secretsPlugin,
|
||||
@@ -225,15 +341,29 @@ export const BUILTIN_PLUGINS: Plugin[] = [
|
||||
bellPlugin,
|
||||
timePlugin,
|
||||
formatPlugin,
|
||||
noForcePushPlugin,
|
||||
noMainCommitPlugin,
|
||||
noRootPlugin,
|
||||
noNetPipePlugin,
|
||||
noGitConfigPlugin,
|
||||
noEnvWritePlugin,
|
||||
conventionalCommitPlugin,
|
||||
testsFirstPlugin,
|
||||
smallDiffsPlugin,
|
||||
confirmDeletePlugin,
|
||||
];
|
||||
|
||||
/**
|
||||
* Enabled unless the config turns them off.
|
||||
*
|
||||
* `guard`, `secrets`, and `protect` are refusals, so they are on: a user who has to
|
||||
* opt into a safety check does not have it. `bell` and `format` both act on their
|
||||
* own — one makes noise, the other writes files — so they are opt-in.
|
||||
* opt into a safety check does not have it. The four narrow safety refusals
|
||||
* (`no-force-push`, `no-net-pipe`, `no-root`, `no-env-write`) are on for the same
|
||||
* reason — each blocks a single irreversible class of mistake. `bell` and `format`
|
||||
* act on their own, and the advisory/opinionated plugins (`no-main-commit`,
|
||||
* `conventional-commit`, `tests-first`, `small-diffs`, `confirm-delete`,
|
||||
* `no-git-config`) encode a workflow preference, so all of those are opt-in.
|
||||
*/
|
||||
export const DEFAULT_ENABLED = ['guard', 'secrets', 'protect', 'time'];
|
||||
export const DEFAULT_ENABLED = ['guard', 'secrets', 'protect', 'time', 'no-force-push', 'no-net-pipe', 'no-root', 'no-env-write'];
|
||||
|
||||
export { DESTRUCTIVE, SECRET_PATHS, PROTECTED_PATHS };
|
||||
|
||||
+49
-4
@@ -43,6 +43,28 @@ const TOOL_DOCS: ToolDoc[] = [
|
||||
name: 'grep',
|
||||
line: 'search contents. Prefer it over reading many files; scope with include to keep results small.',
|
||||
},
|
||||
{ name: 'find_symbol', line: 'jump to where a function, class, or type is defined. Use it before grep when you want a declaration, not every use.' },
|
||||
{ name: 'json_query', line: 'read one value from a JSON file by dotted path, e.g. scripts.build, instead of reading it whole.' },
|
||||
{ name: 'insert_lines', line: 'insert a block at a line number, pushing the rest down. Cheaper than a rewrite for adding to the middle of a file.' },
|
||||
{ name: 'delete_lines', line: 'delete a line range. Refuses the whole file; use delete_file for that.' },
|
||||
{ name: 'replace_lines', line: 'replace a line range with new text in one write.' },
|
||||
{ name: 'append_file', line: 'add to the end of a file without a full rewrite.' },
|
||||
{ name: 'prepend_file', line: 'add to the top of a file, e.g. a header or an import block.' },
|
||||
{ name: 'count_lines', line: 'line counts for one file or a glob. A size read before opening something large.' },
|
||||
{ name: 'tree', line: 'indented directory tree, ignore-aware. Scan a broad shape faster than list_dir.' },
|
||||
{ name: 'file_info', line: 'size, line count, modified time, text or binary, for one file.' },
|
||||
{ name: 'find_files', line: 'find files whose name contains a substring, e.g. "auth". Not a glob.' },
|
||||
{ name: 'recent_files', line: 'files modified most recently. Find what a tool just touched.' },
|
||||
{ name: 'changed_files', line: 'the working-tree delta git reports, at a glance.' },
|
||||
{ name: 'git_log_file', line: 'commits that touched one file, newest first.' },
|
||||
{ name: 'git_diff_commits', line: 'diff between two refs, optionally one path.' },
|
||||
{ name: 'git_show_file', line: 'a file\'s contents at a ref, e.g. auth.ts at HEAD~3.' },
|
||||
{ name: 'git_current_branch', line: 'current branch with upstream and ahead/behind.' },
|
||||
{ name: 'git_changed_in_ref', line: 'files changed between a ref and the working tree, names only.' },
|
||||
{ name: 'outline', line: 'top-level declarations of a source file. Read it before opening a large file.' },
|
||||
{ name: 'read_symbol', line: 'the full body of one definition by name.' },
|
||||
{ name: 'env_info', line: 'platform, shell, and which runtimes are installed, before writing a command.' },
|
||||
{ name: 'count_tokens', line: 'estimate the token cost of a file or string before sending it.' },
|
||||
{
|
||||
name: 'edit_file',
|
||||
line: 'oldString must match byte-for-byte including indentation, and be unique. Include surrounding lines to disambiguate. Prefer several small edits over one large rewrite.',
|
||||
@@ -143,6 +165,7 @@ export function systemPrompt(parts: PromptParts): string {
|
||||
|
||||
const toolNames = availableTools ?? TOOL_DOCS.map((d) => d.name);
|
||||
const canRun = toolNames.includes('bash');
|
||||
const canDelegate = toolNames.includes('task');
|
||||
const approvalTools = toolNames.filter((name) =>
|
||||
['write_file', 'edit_file', 'multi_edit', 'apply_patch', 'move_file', 'delete_file', 'bash', 'web_fetch'].includes(
|
||||
name,
|
||||
@@ -150,19 +173,35 @@ export function systemPrompt(parts: PromptParts): string {
|
||||
);
|
||||
|
||||
const workflow = [
|
||||
'- Read before you write. Ground every claim about the code in something you actually opened.',
|
||||
'- Make the smallest change that solves the task. A bugfix diff contains only the bug.',
|
||||
'- Read before you write. Ground every claim about the code in something you actually opened. Never describe code you have not read.',
|
||||
'- Make the smallest change that solves the task. A bugfix diff contains only the bug; a feature diff contains only the feature.',
|
||||
'- Match the existing style, libraries, and conventions. Sample a neighbouring file before inventing a pattern.',
|
||||
approvalTools.length > 0
|
||||
? `- ${approvalTools.join(', ')} need the user to approve each call. If one is denied, stop and ask what to do instead of working around it.`
|
||||
: '- You have no tools that change anything this turn. Investigate and report; do not describe edits as if you had made them.',
|
||||
canRun
|
||||
? "- After changing code, verify it: run the project's build or tests. \"Should work\" is not verification."
|
||||
? "- After changing code, verify it: run the project's build or tests. \"Should work\" is not verification; output you saw is."
|
||||
: '- You cannot run commands this turn, so say what should be run to verify rather than claiming it passes.',
|
||||
'- When something fails twice, stop and re-read the error literally. Check that the code you think is running is the code that is running.',
|
||||
].join('\n');
|
||||
|
||||
// The failure loop is its own block so a stuck model has a procedure, not a vague
|
||||
// instruction to "try harder". Written as discrete steps because a model in a loop
|
||||
// needs an exit, not encouragement.
|
||||
const recovery = [
|
||||
'- Fail once: read the error literally and fix the thing it names, not the thing you expected.',
|
||||
'- Fail twice on the same attempt: stop. Confirm the code running is the code you think — right file, fresh build, no stale cache or shadowed import.',
|
||||
'- Fail three times: change strategy, not parameters. Reproduce smaller, print the value at the failure point, or ask. Do not re-run the same call hoping for a different result.',
|
||||
].join('\n');
|
||||
|
||||
const delegation = canDelegate
|
||||
? `- Delegate with task for a search across many files or a self-contained change you need not watch. Its prompt must stand alone — it sees none of this conversation. Keep work you must supervise in your own turn.`
|
||||
: '';
|
||||
|
||||
const workflow2 = [
|
||||
canAsk
|
||||
? '- Ask rather than guess when two readings of the request lead to different work. Decide small things yourself and say what you assumed.'
|
||||
: '- No one can answer a question this run. Decide yourself and state the assumption plainly.',
|
||||
'- Long sessions compact as context fills. Record what stays true with remember; restate the goal on a long task.',
|
||||
].join('\n');
|
||||
|
||||
return `You are Shiro Neko, a coding agent working in the user's terminal.
|
||||
@@ -178,6 +217,12 @@ ${renderTools(toolNames)}
|
||||
How to work
|
||||
${workflow}
|
||||
|
||||
When something fails
|
||||
${recovery}
|
||||
${delegation ? `\nDelegating\n${delegation}\n` : ''}
|
||||
Working with the user
|
||||
${workflow2}
|
||||
|
||||
How to reply
|
||||
- Lead with the outcome. The user wants to know what happened, not what you are about to do.
|
||||
- No preamble, no restating the task, no summary of your own summary.
|
||||
|
||||
@@ -15,6 +15,7 @@ import type { Memory } from './memory';
|
||||
import { Notebook, type NotebookState } from './notebook';
|
||||
import { Permissions, type PermissionConfig } from './permission';
|
||||
import type { PluginHost } from './plugins';
|
||||
import { costOf, formatUsd } from './pricing';
|
||||
import { systemPrompt } from './prompt';
|
||||
import { detachProviderItems, pruneToFit } from './prune';
|
||||
import { createSkillTool, renderSkills, type Skill } from './skills';
|
||||
@@ -53,10 +54,16 @@ export type AgentEvent =
|
||||
|
||||
export type SessionOptions = {
|
||||
model: LanguageModel;
|
||||
/** Model id, for pricing the session's spend against the ceiling. */
|
||||
modelId?: string;
|
||||
/** Subagent model id, when it differs; its spend prices against this. */
|
||||
subagentModelId?: string;
|
||||
askApproval: (req: ApprovalRequest) => Promise<ApprovalDecision>;
|
||||
yolo?: boolean;
|
||||
cwd?: string;
|
||||
maxSteps?: number;
|
||||
/** USD ceiling: warn at 80%, refuse the next turn at 100%. */
|
||||
maxSpendUsd?: number;
|
||||
/** MCP and subagent tools merged on top of the built-ins. */
|
||||
extraTools?: ToolSet;
|
||||
/** Tool sets offered this session; omit for all of them. `core` is always on. */
|
||||
@@ -116,6 +123,9 @@ export class Session {
|
||||
readonly notebook: Notebook;
|
||||
inputTokens = 0;
|
||||
outputTokens = 0;
|
||||
/** Subagent token use, priced against the subagent's own model id in /cost. */
|
||||
subagentInputTokens = 0;
|
||||
subagentOutputTokens = 0;
|
||||
private model: LanguageModel;
|
||||
private variant: AgentVariant;
|
||||
private readonly permissions: Permissions;
|
||||
@@ -123,6 +133,8 @@ export class Session {
|
||||
private readonly seen = new Map<string, number>();
|
||||
/** One stale-item repair per turn, so a repeating 404 cannot loop the run. */
|
||||
private staleItemsRepaired = false;
|
||||
/** The 80% spend warning is shown once, not on every turn past the line. */
|
||||
private warnedSpend = false;
|
||||
private controller: AbortController | undefined;
|
||||
|
||||
constructor(private readonly opts: SessionOptions) {
|
||||
@@ -213,10 +225,19 @@ export class Session {
|
||||
this.messages.length = 0;
|
||||
this.inputTokens = 0;
|
||||
this.outputTokens = 0;
|
||||
this.subagentInputTokens = 0;
|
||||
this.subagentOutputTokens = 0;
|
||||
this.warnedSpend = false;
|
||||
this.notebook.clear();
|
||||
this.opts.onChange?.(this.messages);
|
||||
}
|
||||
|
||||
/** A subagent's finished run, folded into the session's spend and the /cost split. */
|
||||
recordSubagentUsage(usage: { inputTokens: number; outputTokens: number }): void {
|
||||
this.subagentInputTokens += usage.inputTokens;
|
||||
this.subagentOutputTokens += usage.outputTokens;
|
||||
}
|
||||
|
||||
replace(messages: ModelMessage[]): void {
|
||||
this.messages.length = 0;
|
||||
this.messages.push(...messages);
|
||||
@@ -236,6 +257,27 @@ export class Session {
|
||||
return this.opts.compactThreshold ?? DEFAULT_COMPACT_THRESHOLD;
|
||||
}
|
||||
|
||||
/**
|
||||
* The session's spend so far and the configured ceiling, for the UI's status
|
||||
* and the refuse-the-next-turn check. Unpriced models report no spend: a
|
||||
* ceiling cannot be enforced against a model we cannot price.
|
||||
*/
|
||||
spend(): { usd?: number; ceiling?: number; overWarn: boolean; overLimit: boolean } {
|
||||
const ceiling = this.opts.maxSpendUsd;
|
||||
const parent = costOf(this.opts.modelId ?? '', this.inputTokens, this.outputTokens);
|
||||
const sub =
|
||||
this.subagentInputTokens + this.subagentOutputTokens > 0
|
||||
? costOf(this.opts.subagentModelId ?? this.opts.modelId ?? '', this.subagentInputTokens, this.subagentOutputTokens)
|
||||
: 0;
|
||||
// Spend is only knowable when every part is priced; an unpriced piece means
|
||||
// the total is a lower bound, so the ceiling is not enforced against it.
|
||||
const usd = parent === undefined || sub === undefined ? undefined : parent + sub;
|
||||
if (ceiling === undefined || usd === undefined) {
|
||||
return { ...(usd !== undefined ? { usd } : {}), ...(ceiling !== undefined ? { ceiling } : {}), overWarn: false, overLimit: false };
|
||||
}
|
||||
return { usd, ceiling, overWarn: usd >= ceiling * 0.8, overLimit: usd >= ceiling };
|
||||
}
|
||||
|
||||
private systemFor(): string {
|
||||
return systemPrompt({
|
||||
cwd: this.opts.cwd ?? process.cwd(),
|
||||
@@ -337,6 +379,21 @@ export class Session {
|
||||
}
|
||||
|
||||
async *send(userText: string): AsyncGenerator<AgentEvent> {
|
||||
// The ceiling is checked before the model is: a turn started past the limit
|
||||
// would spend money the caller said not to. An unpriced model cannot be
|
||||
// measured, so it is never refused here — the ceiling simply cannot see it.
|
||||
const spend = this.spend();
|
||||
if (spend.overLimit) {
|
||||
yield {
|
||||
type: 'error',
|
||||
error: new Error(
|
||||
`spend ceiling reached: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used. Raise maxSpendUsd or start a new session.`,
|
||||
),
|
||||
};
|
||||
yield { type: 'done' };
|
||||
return;
|
||||
}
|
||||
|
||||
this.messages.push({ role: 'user', content: userText });
|
||||
this.opts.onChange?.(this.messages);
|
||||
this.controller = new AbortController();
|
||||
@@ -528,6 +585,16 @@ export class Session {
|
||||
const usage = await result.usage;
|
||||
this.inputTokens += usage.inputTokens ?? 0;
|
||||
this.outputTokens += usage.outputTokens ?? 0;
|
||||
// Warn as the ceiling comes into view, once, so a long session is not
|
||||
// surprised by a refusal it never saw coming.
|
||||
const spend = this.spend();
|
||||
if (spend.overWarn && !this.warnedSpend) {
|
||||
this.warnedSpend = true;
|
||||
yield {
|
||||
type: 'notice',
|
||||
text: `approaching spend ceiling: ${formatUsd(spend.usd ?? 0)} of ${formatUsd(spend.ceiling ?? 0)} used`,
|
||||
};
|
||||
}
|
||||
yield { type: 'done', inputTokens: usage.inputTokens, outputTokens: usage.outputTokens };
|
||||
return;
|
||||
}
|
||||
|
||||
+64
-414
@@ -1,420 +1,70 @@
|
||||
/**
|
||||
* Skills bundled with the binary.
|
||||
*
|
||||
* These are string constants rather than files on disk because `bun build --compile`
|
||||
* only embeds modules reachable through imports; a directory of .md files would be
|
||||
* missing from the shipped binary.
|
||||
* Each skill is a Markdown file in `src/skills-md/`, loaded here as a raw-text import.
|
||||
* The `.md` file is the single source of truth — frontmatter and body in proper
|
||||
* Markdown — so skills are edited and reviewed as Markdown, not as escaped strings
|
||||
* inside TypeScript. Bun inlines every text import into the compiled binary, so the
|
||||
* folder ships with `bun build --compile` exactly as the old string constants did.
|
||||
*/
|
||||
import accessibility from './skills-md/accessibility.md' with { type: 'text' };
|
||||
import apiDesign from './skills-md/api-design.md' with { type: 'text' };
|
||||
import ciCd from './skills-md/ci-cd.md' with { type: 'text' };
|
||||
import commit from './skills-md/commit.md' with { type: 'text' };
|
||||
import data from './skills-md/data.md' with { type: 'text' };
|
||||
import db from './skills-md/db.md' with { type: 'text' };
|
||||
import debug from './skills-md/debug.md' with { type: 'text' };
|
||||
import deps from './skills-md/deps.md' with { type: 'text' };
|
||||
import docker from './skills-md/docker.md' with { type: 'text' };
|
||||
import docs from './skills-md/docs.md' with { type: 'text' };
|
||||
import frontend from './skills-md/frontend.md' with { type: 'text' };
|
||||
import gitWorkflow from './skills-md/git-workflow.md' with { type: 'text' };
|
||||
import i18n from './skills-md/i18n.md' with { type: 'text' };
|
||||
import incident from './skills-md/incident.md' with { type: 'text' };
|
||||
import logging from './skills-md/logging.md' with { type: 'text' };
|
||||
import migrate from './skills-md/migrate.md' with { type: 'text' };
|
||||
import onboarding from './skills-md/onboarding.md' with { type: 'text' };
|
||||
import optimizeSql from './skills-md/optimize-sql.md' with { type: 'text' };
|
||||
import perf from './skills-md/perf.md' with { type: 'text' };
|
||||
import perfFrontend from './skills-md/perf-frontend.md' with { type: 'text' };
|
||||
import plan from './skills-md/plan.md' with { type: 'text' };
|
||||
import readme from './skills-md/readme.md' with { type: 'text' };
|
||||
import refactor from './skills-md/refactor.md' with { type: 'text' };
|
||||
import release from './skills-md/release.md' with { type: 'text' };
|
||||
import review from './skills-md/review.md' with { type: 'text' };
|
||||
import security from './skills-md/security.md' with { type: 'text' };
|
||||
import test from './skills-md/test.md' with { type: 'text' };
|
||||
import uxCopy from './skills-md/ux-copy.md' with { type: 'text' };
|
||||
import verify from './skills-md/verify.md' with { type: 'text' };
|
||||
|
||||
export const BUILTIN_SKILLS: { name: string; source: string }[] = [
|
||||
{
|
||||
name: 'debug',
|
||||
source: `---
|
||||
name: debug
|
||||
description: Track down a bug whose cause is not obvious. Use when a test fails for unclear reasons, behaviour differs between environments, or an earlier fix did not hold.
|
||||
---
|
||||
|
||||
# Debugging
|
||||
|
||||
Do not guess. A guess that happens to work leaves the real cause in place.
|
||||
|
||||
## Reproduce first
|
||||
|
||||
Find the smallest command that shows the failure and record it with \`remember\`. If you
|
||||
cannot reproduce it, say so and ask what the user did differently — do not proceed on a
|
||||
hypothesis you cannot test.
|
||||
|
||||
## Three hypotheses, then evidence
|
||||
|
||||
Write down at least three causes that would produce this exact symptom. Rank them by how
|
||||
cheap they are to disprove, then disprove them in that order. State which one you are
|
||||
testing before you test it.
|
||||
|
||||
Evidence means observed output: a log line, a failing assertion, a value printed at the
|
||||
point of failure. "It should be X" is not evidence.
|
||||
|
||||
## Bisect when the space is large
|
||||
|
||||
- Recent regression: check what changed last.
|
||||
- Unclear layer: assert the value at each boundary until one is wrong.
|
||||
- Intermittent: run it in a loop and capture the failing case, do not reason about it abstractly.
|
||||
|
||||
## Fix the cause
|
||||
|
||||
Once you know the cause, fix that and nothing else. Do not tidy surrounding code in the
|
||||
same change — a bugfix diff should contain only the bug.
|
||||
|
||||
Write a test that fails before the fix and passes after. If you cannot express the bug as
|
||||
a test, say why.
|
||||
|
||||
## After two failed attempts
|
||||
|
||||
Stop. Re-read the error text literally, character by character. Check your assumption
|
||||
about which code is actually running: the wrong file, a stale build, a shadowed import,
|
||||
or a cached dependency accounts for most "impossible" bugs.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'review',
|
||||
source: `---
|
||||
name: review
|
||||
description: Review a diff or a file for defects. Use when asked to review, critique, or check code before it ships.
|
||||
---
|
||||
|
||||
# Code review
|
||||
|
||||
Severity order. Do not lead with style.
|
||||
|
||||
1. **Incorrect behaviour** — wrong result, wrong edge case, wrong state after failure.
|
||||
2. **Missing validation at trust boundaries** — user input, network responses, file contents,
|
||||
anything crossing a process line. Internal calls need no defensive checks.
|
||||
3. **Security** — injection, path traversal, secrets in logs or errors, missing authz.
|
||||
4. **Resource handling** — unclosed handles, unbounded growth, unawaited promises.
|
||||
5. **Clarity** — only when it will cause a future defect.
|
||||
|
||||
## For each finding
|
||||
|
||||
State file and line, what breaks, and the change. Show the fix as code when it is short.
|
||||
|
||||
Skip anything a formatter would fix. Skip preference. If a choice is defensible, leave it.
|
||||
|
||||
## Say when it is fine
|
||||
|
||||
A review that invents problems to look thorough is worse than a short one. If the change
|
||||
is correct, say so and stop.
|
||||
|
||||
## Verify, do not assume
|
||||
|
||||
Read the surrounding code before calling something a bug. A "missing" null check often
|
||||
exists one level up. Run the tests if that is what settles it.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'refactor',
|
||||
source: `---
|
||||
name: refactor
|
||||
description: Restructure code without changing behaviour. Use when asked to refactor, clean up, extract, or reorganise.
|
||||
---
|
||||
|
||||
# Refactoring
|
||||
|
||||
Behaviour must not change. That is the whole constraint.
|
||||
|
||||
## Establish the safety net first
|
||||
|
||||
Run the existing tests and record that they pass. If the code has no tests, write one that
|
||||
pins current behaviour — including the ugly parts — before touching anything. Refactoring
|
||||
untested code is rewriting it.
|
||||
|
||||
## Then move in small steps
|
||||
|
||||
One transformation at a time, tests green between each. Rename, then extract, then move —
|
||||
not all three in one edit. A large refactor that fails leaves you unable to tell which step
|
||||
broke it.
|
||||
|
||||
## What not to do
|
||||
|
||||
- Do not fix bugs while refactoring. Note them, finish, fix separately.
|
||||
- Do not add abstraction for a single caller. Duplication beats a premature interface.
|
||||
- Do not widen the scope. The request was this code, not its neighbours.
|
||||
- Do not change public API unless asked; if it must change, say so first.
|
||||
|
||||
## Done means
|
||||
|
||||
Tests pass, behaviour is identical, and the diff is smaller than the reader feared.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'test',
|
||||
source: `---
|
||||
name: test
|
||||
description: Write or repair tests. Use when adding coverage, fixing a flaky test, or asked how something should be tested.
|
||||
---
|
||||
|
||||
# Testing
|
||||
|
||||
A test earns its place by failing when the code is wrong.
|
||||
|
||||
## Match the project
|
||||
|
||||
Read two existing test files first. Use their runner, their assertion style, their file
|
||||
layout, their naming. A test that looks foreign is a test nobody maintains.
|
||||
|
||||
## Test behaviour, not implementation
|
||||
|
||||
Assert on what a caller observes. A test that reaches into private state breaks on every
|
||||
refactor and catches nothing.
|
||||
|
||||
Cover: the normal case, the boundaries, and the failure. Failure cases catch more real
|
||||
defects than happy paths.
|
||||
|
||||
## Never do this
|
||||
|
||||
- Do not assert what the code currently returns without knowing it is correct — that pins
|
||||
the bug.
|
||||
- Do not weaken an assertion to make a test pass. If it fails, either the code or the
|
||||
expectation is wrong; find out which.
|
||||
- Do not delete a failing test. It is telling you something.
|
||||
|
||||
## Flaky tests
|
||||
|
||||
A test that passes alone and fails in a suite is a shared-state problem: a global, a
|
||||
temp directory, a port, an unawaited promise, or ordering. Find which, do not add a retry.
|
||||
|
||||
## Verify
|
||||
|
||||
Run the test and watch it fail before the fix, pass after. A test you never saw fail is
|
||||
not known to work.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'verify',
|
||||
source: `---
|
||||
name: verify
|
||||
description: Confirm a change actually works by using it, not by reading it. Use before reporting a task complete, or when asked whether something works.
|
||||
---
|
||||
|
||||
# Verification
|
||||
|
||||
A green test suite says the tests pass. It does not say the feature works.
|
||||
|
||||
## Run the artifact, not the source
|
||||
|
||||
Build it and use it the way a user would:
|
||||
|
||||
- **CLI** — build the binary and run it. Happy path, bad input, \`--help\`. Read the output.
|
||||
- **HTTP service** — start it and \`curl\` the endpoint. Check the status and the body.
|
||||
- **Library** — write a throwaway script that imports and calls the new code end to end.
|
||||
- **Script or job** — run it against real input and inspect what it produced.
|
||||
|
||||
Delete the throwaway afterwards.
|
||||
|
||||
## What counts as evidence
|
||||
|
||||
Command output you actually saw. Paste the relevant lines, not a summary of them.
|
||||
|
||||
These are not evidence:
|
||||
|
||||
- "The tests pass" for a change tests do not cover.
|
||||
- "The types check" for anything about runtime behaviour.
|
||||
- "It should work now" for anything at all.
|
||||
|
||||
## Check the failure path too
|
||||
|
||||
Feed it the input you expect to be rejected and confirm it is rejected, with a message
|
||||
that says why. A feature that works only on correct input is half-built.
|
||||
|
||||
## Report what you did not verify
|
||||
|
||||
Say plainly what you could not run and why: a missing credential, a service you cannot
|
||||
start, a platform you are not on. An honest gap is useful; a claim that hides one is not.
|
||||
|
||||
## When verification fails
|
||||
|
||||
The defect is yours to fix in this turn. Do not report the task complete with a note that
|
||||
it did not work.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'commit',
|
||||
source: `---
|
||||
name: commit
|
||||
description: Stage and commit work. Use when asked to commit, or to split existing changes into commits.
|
||||
---
|
||||
|
||||
# Committing
|
||||
|
||||
Never commit unless the user asked. If it is unclear whether they did, ask.
|
||||
|
||||
## Look before you stage
|
||||
|
||||
\`git_status\` and \`git_diff\` first. You are looking for two things:
|
||||
|
||||
1. Changes that are not yours. Another agent or the user may share this worktree, and
|
||||
\`git add .\` takes their half-finished work with yours.
|
||||
2. Files that should never be committed: \`.env\`, credentials, keys, large build output,
|
||||
anything a \`.gitignore\` rule was supposed to catch and did not. Flag these to the user
|
||||
rather than committing them.
|
||||
|
||||
Stage the specific paths you changed. \`git add .\` is how unrelated work ends up in a
|
||||
commit that then has to be reverted whole.
|
||||
|
||||
## One commit, one reason
|
||||
|
||||
If the diff does two unrelated things, make two commits. A commit that both fixes a bug and
|
||||
renames a module cannot be reverted, cherry-picked, or bisected usefully.
|
||||
|
||||
## The message
|
||||
|
||||
Match the repository's existing style — read \`git_log\` before writing one. Failing that:
|
||||
|
||||
- A subject line under 70 characters, imperative, saying what changed.
|
||||
- A body explaining *why*, when the reason is not obvious from the diff. Wrap at 72.
|
||||
- No "as requested", no restating the diff line by line, no emoji unless the repo uses them.
|
||||
|
||||
## Do not
|
||||
|
||||
- Do not \`--amend\` a commit that has been pushed. Write a new one.
|
||||
- Do not \`--no-verify\`. If a hook rejects the commit, the hook found something.
|
||||
- Do not \`git push\` unless asked, and never force-push without being asked explicitly.
|
||||
- Do not commit and then immediately fix it up with a second commit. Get it right, or say
|
||||
what is wrong.
|
||||
|
||||
## After committing
|
||||
|
||||
Report the short hash and the subject. If a hook rewrote files, say so and confirm the
|
||||
final state is what was intended.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'security',
|
||||
source: `---
|
||||
name: security
|
||||
description: Review code for security defects, or write code that handles untrusted input. Use when touching authentication, user input, file paths, shell commands, SQL, or anything reachable from the network.
|
||||
---
|
||||
|
||||
# Security
|
||||
|
||||
Find the trust boundary first. Everything crossing it is hostile until parsed.
|
||||
|
||||
## The boundaries in most codebases
|
||||
|
||||
- Request bodies, query strings, headers, cookies.
|
||||
- File contents and filenames, including paths a user supplied.
|
||||
- Environment variables in a multi-tenant deployment.
|
||||
- Anything a model or a third-party API returned.
|
||||
|
||||
Inside a boundary, values are already validated and re-checking them is noise. At the
|
||||
boundary, nothing is optional.
|
||||
|
||||
## What to look for, in order
|
||||
|
||||
1. **Injection.** String-built SQL, shell commands assembled from input, \`eval\`, template
|
||||
rendering with user data as the template rather than the data. The fix is parameters and
|
||||
argument arrays, never escaping.
|
||||
2. **Missing authorisation.** An endpoint that checks *who* you are but not *what* you may
|
||||
touch. Look for an id taken from the request and used without an ownership check.
|
||||
3. **Path traversal.** \`../\` in anything joined onto a filesystem root. Resolve, then verify
|
||||
the result is still inside the root — a prefix check on the raw input misses
|
||||
\`a/../../secret\`.
|
||||
4. **Secrets in the wrong place.** Keys in source, in logs, in error messages, in a commit.
|
||||
A secret that reached a log is a secret to rotate.
|
||||
5. **Server-side request forgery.** A URL from input, fetched. Block private and loopback
|
||||
addresses by *resolved* address, and re-check every redirect hop.
|
||||
6. **Weak crypto and hand-rolled auth.** Homemade token formats, \`Math.random\` for anything
|
||||
security-bearing, comparisons on secrets that are not constant time.
|
||||
|
||||
## What not to do
|
||||
|
||||
Do not report a finding you cannot trace to a concrete input. "This could be unsafe" without
|
||||
a path from an attacker-controlled value to the sink is noise that buries the real one.
|
||||
|
||||
Do not fix a symptom at one caller when the sink is shared. Grep every caller and fix the
|
||||
seam once.
|
||||
|
||||
## Reporting
|
||||
|
||||
File, line, the path from input to sink, and the fix. Say plainly when a thing that looks
|
||||
dangerous is actually fine, and why — a reviewer's confidence is worth as much as a finding.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'perf',
|
||||
source: `---
|
||||
name: perf
|
||||
description: Make something faster, or find out why it is slow. Use when a command, request, test suite, or build takes longer than it should.
|
||||
---
|
||||
|
||||
# Performance
|
||||
|
||||
Measure first. A change made without a number before it is a guess with extra steps.
|
||||
|
||||
## Get a number
|
||||
|
||||
Time the actual operation, not a proxy for it. \`time\`, the framework's own timing output,
|
||||
or a loop around the slow call with a timestamp either side. Record the baseline with
|
||||
\`remember\` so the comparison survives compaction.
|
||||
|
||||
If you cannot measure it, say so and stop. Optimising an unmeasured path is how a codebase
|
||||
accumulates complexity that buys nothing.
|
||||
|
||||
## Find where the time goes
|
||||
|
||||
- **Wall-clock dominated by one call?** Look there and nowhere else.
|
||||
- **Spread evenly?** Suspect the loop around it: an O(n²) walk, a query per row, a file read
|
||||
per iteration.
|
||||
- **Idle time?** It is waiting: a sequential chain of independent awaits, an unpooled
|
||||
connection, a lock.
|
||||
|
||||
The usual culprits, in the order they actually appear: N+1 queries, work repeated inside a
|
||||
loop that could be hoisted, a missing index, sequential awaits that could run together,
|
||||
reading a whole file to use one line, and re-parsing something that could be parsed once.
|
||||
|
||||
## Change one thing
|
||||
|
||||
One change, then re-measure. Two changes together and you do not know which one paid — and
|
||||
one of them may have cost.
|
||||
|
||||
## Stop when it is fast enough
|
||||
|
||||
State the target before you start: "the test suite under a minute", "the endpoint under
|
||||
200ms". Past the target, further work is complexity with no user on the other end of it.
|
||||
|
||||
## Report
|
||||
|
||||
Baseline, change, new number, and what you did not do. A 40% win with one line changed is a
|
||||
better report than a 45% win that restructured a module.
|
||||
`,
|
||||
},
|
||||
{
|
||||
name: 'migrate',
|
||||
source: `---
|
||||
name: migrate
|
||||
description: Upgrade a dependency, framework, or language version across a codebase. Use when a major version bump, a deprecation, or a breaking API change has to be applied.
|
||||
---
|
||||
|
||||
# Migration
|
||||
|
||||
The failure mode is a half-applied migration: it compiles, most tests pass, and one code
|
||||
path still uses the old API.
|
||||
|
||||
## Read the changelog before the code
|
||||
|
||||
Find what actually broke. A major version usually has a migration guide; read it and list
|
||||
the changes that apply to this codebase specifically. Below 1.0, treat a minor bump as
|
||||
breaking — semver promises nothing there.
|
||||
|
||||
## Find every call site before changing one
|
||||
|
||||
Grep for the old API across the whole repository, including tests, scripts, config, CI
|
||||
workflows, Dockerfiles, and documentation. A version literal pinned in a workflow while the
|
||||
manifest says something else is a split-brain deploy.
|
||||
|
||||
Write the list down with \`todo_write\`. The list is the migration; the edits are mechanical.
|
||||
|
||||
## Change in one shape
|
||||
|
||||
Apply the same transformation everywhere rather than improving each site as you pass
|
||||
through it. A migration mixed with refactoring cannot be reviewed, and cannot be reverted
|
||||
if the upgrade turns out to be wrong.
|
||||
|
||||
\`apply_patch\` is the tool for this: one atomic patch across the files that must land
|
||||
together.
|
||||
|
||||
## Verify at the boundary that broke
|
||||
|
||||
Type checks catch signature changes and miss behaviour changes. Run the tests, then actually
|
||||
use the thing that was upgraded: start the server, run the CLI, execute the query. A green
|
||||
suite over an untested upgrade path proves the suite did not cover it.
|
||||
|
||||
## Never hand-merge a lockfile
|
||||
|
||||
On a conflict, take either side whole and regenerate with the package manager. The resolver
|
||||
owns that file.
|
||||
|
||||
## Report
|
||||
|
||||
The version before and after, every file class touched, what you verified by running, and
|
||||
anything the changelog said applies that you deliberately did not do.
|
||||
`,
|
||||
},
|
||||
{ name: 'accessibility', source: accessibility },
|
||||
{ name: 'api-design', source: apiDesign },
|
||||
{ name: 'ci-cd', source: ciCd },
|
||||
{ name: 'commit', source: commit },
|
||||
{ name: 'data', source: data },
|
||||
{ name: 'db', source: db },
|
||||
{ name: 'debug', source: debug },
|
||||
{ name: 'deps', source: deps },
|
||||
{ name: 'docker', source: docker },
|
||||
{ name: 'docs', source: docs },
|
||||
{ name: 'frontend', source: frontend },
|
||||
{ name: 'git-workflow', source: gitWorkflow },
|
||||
{ name: 'i18n', source: i18n },
|
||||
{ name: 'incident', source: incident },
|
||||
{ name: 'logging', source: logging },
|
||||
{ name: 'migrate', source: migrate },
|
||||
{ name: 'onboarding', source: onboarding },
|
||||
{ name: 'optimize-sql', source: optimizeSql },
|
||||
{ name: 'perf', source: perf },
|
||||
{ name: 'perf-frontend', source: perfFrontend },
|
||||
{ name: 'plan', source: plan },
|
||||
{ name: 'readme', source: readme },
|
||||
{ name: 'refactor', source: refactor },
|
||||
{ name: 'release', source: release },
|
||||
{ name: 'review', source: review },
|
||||
{ name: 'security', source: security },
|
||||
{ name: 'test', source: test },
|
||||
{ name: 'ux-copy', source: uxCopy },
|
||||
{ name: 'verify', source: verify },
|
||||
];
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
---
|
||||
name: accessibility
|
||||
description: Make a UI accessible. Use when adding a feature that must work with a keyboard or screen reader, fixing contrast or focus issues, or reviewing for WCAG.
|
||||
---
|
||||
|
||||
# Accessibility
|
||||
|
||||
Accessibility is usability for everyone, including people using a keyboard, a screen reader, a
|
||||
magnifier, or a noisy display. Build it in, not on.
|
||||
|
||||
## Semantic HTML does the heavy lifting
|
||||
|
||||
A `<button>`, `<a>`, `<input>`, `<nav>`, `<main>` carries behaviour and meaning for free
|
||||
that a `<div>` with a click handler does not. Reach for the native element first; add ARIA only
|
||||
when no native element fits. The first rule of ARIA is do not use ARIA if a native element exists.
|
||||
|
||||
## Keyboard is the baseline
|
||||
|
||||
- Every interactive element is reachable and operable with Tab and Enter/Space alone.
|
||||
- A visible focus indicator on everything — never `outline: none` without a replacement.
|
||||
- Logical tab order following the visual order, and focus managed into and out of modals,
|
||||
menus, and dialogs (trapped while open, returned to the trigger on close).
|
||||
|
||||
## Screen readers hear structure
|
||||
|
||||
- Headings in order (`h1` once, then down a level at a time) so the page has a navigable outline.
|
||||
- Every `<input>` has a `<label>`; every icon-only button has an accessible name; every image
|
||||
has alt text that conveys its point (or empty alt when it is purely decorative).
|
||||
- Dynamic changes announce themselves: a toast, an error, a loaded region uses a live region so
|
||||
it is heard, not just seen.
|
||||
|
||||
## Contrast and meaning
|
||||
|
||||
Text meets 4.5:1 against its background (3:1 for large text). Colour is never the only carrier
|
||||
of meaning — pair it with an icon, a label, or a pattern. A red-only "error" is invisible to a
|
||||
colour-blind user.
|
||||
|
||||
## Test it the way it is used
|
||||
|
||||
Tab through the whole flow. Turn on a screen reader and listen. Zoom to 200% and 400%. Run an
|
||||
automated checker for the mechanical half — then do the manual half it cannot cover, because
|
||||
most accessibility failures are not machine-detectable.
|
||||
@@ -0,0 +1,45 @@
|
||||
---
|
||||
name: api-design
|
||||
description: Design or revise an HTTP or library API. Use when adding an endpoint, shaping request/response bodies, naming resources, or reviewing an API for consistency.
|
||||
---
|
||||
|
||||
# API design
|
||||
|
||||
An API is a contract. Every choice is a promise you cannot take back without a major version.
|
||||
|
||||
## Resource before action
|
||||
|
||||
Name things, not verbs. `POST /users` to create, not `POST /createUser`. The URL is the
|
||||
noun; the method is the verb. When you reach for a verb in the path, that is a sign the
|
||||
resource is missing — `POST /users/:id/deactivations` reads better than `/deactivateUser`
|
||||
when the operation has state of its own.
|
||||
|
||||
## Shape the body for the reader
|
||||
|
||||
- Field names are `snake_case` or `camelCase`, picked once for the whole API. A body that
|
||||
mixes both is a body nobody documented.
|
||||
- Return the object, not a wrapper, unless the wrapper carries something: `{ "user": {...} }`
|
||||
only when there is also pagination, a cursor, or an error envelope.
|
||||
- Errors have a stable shape: a machine-readable `code`, a human `message`, and the field
|
||||
that failed. A client should never have to parse the message.
|
||||
|
||||
## Status codes mean something
|
||||
|
||||
- `201` for a created resource, with the resource in the body.
|
||||
- `204` for success with nothing to return.
|
||||
- `400` for a body that failed validation, `401` unauthenticated, `403` authenticated but
|
||||
not allowed, `404` not found or not allowed to know, `409` a conflict with current state,
|
||||
`422` well-formed but semantically wrong.
|
||||
- Never `200` with an error in the body. A client checking only the status will treat it as
|
||||
success.
|
||||
|
||||
## Idempotency and safety
|
||||
|
||||
GET, PUT, DELETE must be safe to retry: same request, same state. POST is not. If a client can
|
||||
double-submit, provide an idempotency key or a natural unique constraint, and say which.
|
||||
|
||||
## Version when you must, not before
|
||||
|
||||
Add fields freely; removing or renaming is a break. If you are not yet committed, say so with
|
||||
a `beta` or `v0` marker rather than locking a shape you have not used. Document the contract
|
||||
you guarantee, not the implementation that happens to produce it.
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
name: ci-cd
|
||||
description: Write or repair CI/CD pipelines and workflow files. Use when a build fails in CI but not locally, when adding a workflow, or when caching, matrix, or deploy steps need design.
|
||||
---
|
||||
|
||||
# CI/CD
|
||||
|
||||
CI is a second machine that does not have your setup. "Works on my machine" means the pipeline
|
||||
is missing something your machine has.
|
||||
|
||||
## Reproduce the environment, not the symptom
|
||||
|
||||
When CI fails and local passes, the difference is the environment: the toolchain version, an
|
||||
uncommitted file, a cache, an env var, the OS. Diff those before touching the code. Read the
|
||||
failing log literally — the first error, not the last, which is usually a downstream echo.
|
||||
|
||||
## Pin everything that can move
|
||||
|
||||
- Toolchain versions (`node`, `bun`, `python`), action versions, base images. `latest`
|
||||
is a build that breaks on a day you did nothing.
|
||||
- Lockfiles go in the repo and the install respects them (`--frozen-lockfile`, `ci`). An
|
||||
install that re-resolves in CI is a different build from the one you tested.
|
||||
|
||||
## Cache the expensive, deterministic part
|
||||
|
||||
Dependencies are the cache; build output usually is not. Key the cache on the lockfile hash so
|
||||
a changed dependency invalidates it. A cache that is too broad serves stale artifacts; too
|
||||
narrow saves nothing.
|
||||
|
||||
## Fail fast, in the right order
|
||||
|
||||
Cheap checks first: lint and typecheck before the test matrix, tests before the deploy. A
|
||||
pipeline that deploys before it verifies publishes the bug it was built to catch.
|
||||
|
||||
## Secrets and deploys
|
||||
|
||||
Secrets live in the CI secret store, never in the file, and are masked in logs. A deploy step
|
||||
is gated: on a tag, on a protected branch, on a manual approval — never on every push. Assume
|
||||
every log line is public and write the pipeline accordingly.
|
||||
@@ -0,0 +1,54 @@
|
||||
---
|
||||
name: commit
|
||||
description: Stage and commit work. Use when asked to commit, or to split existing changes into commits.
|
||||
---
|
||||
|
||||
# Committing
|
||||
|
||||
Never commit unless the user asked. If it is unclear whether they did, ask. A commit is a
|
||||
durable statement about shared history, not a save-point.
|
||||
|
||||
## Look before you stage
|
||||
|
||||
`git_status` and `git_diff` first — read the whole diff you are about to commit. You are
|
||||
looking for three things:
|
||||
|
||||
1. **Changes that are not yours.** Another agent or the user may share this worktree, and
|
||||
`git add .` takes their half-finished work with yours.
|
||||
2. **Files that should never be committed:** `.env`, credentials, keys, large build
|
||||
output, anything a `.gitignore` rule was supposed to catch and did not. Flag these to
|
||||
the user rather than committing them — a committed secret is a secret to rotate.
|
||||
3. **Your own accidents:** debug prints, commented-out code, a stray `TODO`, a file you
|
||||
opened and saved by mistake. Revert them before staging, not in a follow-up commit.
|
||||
|
||||
Stage the specific paths you changed. `git add .` is how unrelated work ends up in a
|
||||
commit that then has to be reverted whole.
|
||||
|
||||
## One commit, one reason
|
||||
|
||||
If the diff does two unrelated things, make two commits. A commit that both fixes a bug
|
||||
and renames a module cannot be reverted, cherry-picked, or bisected usefully. Each commit
|
||||
should pass the tests on its own — a series of broken commits defeats `git bisect`.
|
||||
|
||||
## The message
|
||||
|
||||
Match the repository's existing style — read `git_log` before writing one. Failing that:
|
||||
|
||||
- A subject line under 70 characters, imperative mood, saying what changed: "Fix off-by-one
|
||||
in pagination", not "fixed a bug" or "changes".
|
||||
- A body explaining *why* when the reason is not obvious from the diff. Wrap at 72.
|
||||
- No "as requested", no restating the diff line by line, no emoji unless the repo uses them,
|
||||
no sign-off noise the repo does not already use.
|
||||
|
||||
## Do not
|
||||
|
||||
- Do not `--amend` a commit that has been pushed. Write a new one.
|
||||
- Do not `--no-verify`. If a hook rejects the commit, the hook found something — read it.
|
||||
- Do not `git push` unless asked, and never force-push without being asked explicitly.
|
||||
- Do not commit and then immediately fix it up with a second commit. Get it right, or say
|
||||
what is wrong.
|
||||
|
||||
## After committing
|
||||
|
||||
Report the short hash and the subject. If a hook rewrote files, say so, and confirm the
|
||||
final state — `git_status` again — is what was intended.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: data
|
||||
description: Process, validate, or transform data. Use when parsing files, cleaning datasets, designing a data pipeline, or debugging a transform that produces wrong output.
|
||||
---
|
||||
|
||||
# Data
|
||||
|
||||
Bad data fails silently and far away from where it entered. Validate at the boundary, keep the
|
||||
raw, and make every transform checkable.
|
||||
|
||||
## Validate at the boundary
|
||||
|
||||
Parse and validate when data enters the system, not when it is used. A schema check at the edge
|
||||
turns a corrupt record into a clear rejection; skipping it turns the same record into a wrong
|
||||
answer three layers later. Reject loudly, with the record and the reason — never coerce and
|
||||
carry on.
|
||||
|
||||
## Keep the raw
|
||||
|
||||
Store the untransformed input alongside the derived. When a transform turns out to be wrong,
|
||||
the raw lets you recompute; without it, the information is gone. Derived data is rebuildable;
|
||||
source data is not.
|
||||
|
||||
## Transformations are pure and tested
|
||||
|
||||
A transform takes input and returns output with no hidden state, so it can be tested on a
|
||||
fixture and re-run safely. Test the edge cases that actually occur in data: the empty field,
|
||||
the wrong type, the unexpected null, the duplicate, the encoding that is not UTF-8.
|
||||
|
||||
## Duplicates, nulls, and ranges are the usual corruption
|
||||
|
||||
Check for: unexpected duplicates on a key, nulls where a value is required, values outside a
|
||||
sane range (a negative age, a date in the future), and referential breaks (an id pointing at
|
||||
nothing). These four catch most real-world data problems before they reach a report.
|
||||
|
||||
## Idempotent pipelines
|
||||
|
||||
A step that can be re-run without duplicating or corrupting its output is a step you can retry
|
||||
after a failure. Key on a stable id and upsert rather than blind-insert. A pipeline you cannot
|
||||
safely re-run is a pipeline you will one day have to fix by hand at 2am.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: db
|
||||
description: Design schemas, write migrations, or fix query and data problems. Use when adding a table, writing a migration, debugging a slow query, or choosing keys and indexes.
|
||||
---
|
||||
|
||||
# Databases
|
||||
|
||||
The schema is the hardest thing to change in the whole system. Design it for the queries, not
|
||||
the object model.
|
||||
|
||||
## Keys and constraints are the real schema
|
||||
|
||||
- Every table has a primary key; prefer a surrogate `id` unless a natural key is truly stable.
|
||||
- Foreign keys and `NOT NULL` are not optional decoration — they are the constraints that stop
|
||||
bad data at the door instead of in application code six months later.
|
||||
- Unique constraints belong on the thing that must be unique (email, slug), enforced by the
|
||||
database, not by a check-then-insert that races.
|
||||
|
||||
## Migrations are one-way and additive where possible
|
||||
|
||||
- Never edit a migration that has run anywhere. Add a new one.
|
||||
- Destructive changes (drop column, rename, change type) are two migrations: add the new shape,
|
||||
deploy code that writes both, then remove the old in a later release. A single migration that
|
||||
renames a column breaks every old copy of the app still running.
|
||||
- Test a migration against real data volume. `ALTER` on ten rows is instant; on ten million it
|
||||
locks the table.
|
||||
|
||||
## Indexes follow the queries
|
||||
|
||||
Index the columns you filter and join on, in the order the query uses them. A composite index
|
||||
`(a, b)` serves `WHERE a` and `WHERE a, b` but not `WHERE b` alone. Read the query plan
|
||||
(`EXPLAIN`) before adding one — a guess is an index that costs writes and serves nothing.
|
||||
|
||||
## The N+1 is the default bug
|
||||
|
||||
A query per row in a loop is the most common database performance defect. Fetch the set with a
|
||||
join or a batched `WHERE id IN (...)`. If a page does one query per item, that is the fix
|
||||
before any caching.
|
||||
@@ -0,0 +1,63 @@
|
||||
---
|
||||
name: debug
|
||||
description: Track down a bug whose cause is not obvious. Use when a test fails for unclear reasons, behaviour differs between environments, or an earlier fix did not hold.
|
||||
---
|
||||
|
||||
# Debugging
|
||||
|
||||
Do not guess. A guess that happens to work leaves the real cause in place, and it will
|
||||
fire again — usually in production, usually at a worse time.
|
||||
|
||||
## Reproduce first
|
||||
|
||||
Find the smallest command that shows the failure and record it with `remember`. If you
|
||||
cannot reproduce it, say so and ask what the user did differently — do not proceed on a
|
||||
hypothesis you cannot test.
|
||||
|
||||
Shrink the reproduction until it is minimal: one input, one call, one assertion. Every
|
||||
moving part you leave in is a place the bug can hide. A reproduction that takes thirty
|
||||
steps will not get run often enough to confirm the fix.
|
||||
|
||||
## Three hypotheses, then evidence
|
||||
|
||||
Write down at least three causes that would produce this exact symptom — not "the code is
|
||||
wrong" but specific mechanisms: "the offset is off by one when the page is empty", "the
|
||||
cache is read before the write lands". Rank them by how cheap they are to disprove, then
|
||||
disprove them in that order. State which one you are testing before you test it.
|
||||
|
||||
Evidence means observed output: a log line, a failing assertion, a value printed at the
|
||||
point of failure. "It should be X" is not evidence. When the evidence contradicts your
|
||||
favoured hypothesis, the hypothesis is wrong — do not explain the evidence away.
|
||||
|
||||
## Localise before you fix
|
||||
|
||||
Assert the value at each boundary until one is wrong. The bug lives between the last
|
||||
boundary where the value is right and the first where it is wrong. Fixing before you have
|
||||
that bracket means editing the wrong place and learning nothing.
|
||||
|
||||
## Bisect when the space is large
|
||||
|
||||
- Recent regression: `git bisect` or read what changed last. The bug arrived in a commit;
|
||||
find which one.
|
||||
- Unclear layer: assert the value at each boundary until one is wrong.
|
||||
- Intermittent: run it in a loop and capture the failing case with full logging. Do not
|
||||
reason about a race abstractly — make it happen on demand, then it is no longer
|
||||
intermittent.
|
||||
|
||||
## Fix the cause, not the symptom
|
||||
|
||||
Once you know the cause, fix that and nothing else. Do not tidy surrounding code in the
|
||||
same change — a bugfix diff should contain only the bug, so it can be reverted whole if it
|
||||
is wrong.
|
||||
|
||||
Write a test that fails before the fix and passes after. Watch it fail first; a test you
|
||||
never saw fail proves nothing. If you cannot express the bug as a test, say why — and say
|
||||
what you ran instead to confirm the fix.
|
||||
|
||||
## After two failed attempts
|
||||
|
||||
Stop. Re-read the error text literally, character by character — most "impossible" bugs
|
||||
are a misread message. Then check your assumption about which code is actually running:
|
||||
the wrong file, a stale build, a shadowed import, a cached dependency, or an env var that
|
||||
differs from your shell. Verify by printing something at the point you *think* executes;
|
||||
if it does not print, that is your answer.
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
name: deps
|
||||
description: Manage dependencies: choosing, adding, updating, or removing them. Use when evaluating a library, resolving a version conflict, pruning unused deps, or hardening the supply chain.
|
||||
---
|
||||
|
||||
# Dependencies
|
||||
|
||||
Every dependency is code you did not write but now maintain. Add deliberately, prune regularly.
|
||||
|
||||
## Choose on maintenance, not features
|
||||
|
||||
Before adding: is it actively maintained (recent commits, responsive issues), widely used, and
|
||||
small enough to be worth it? A dependency that saves a day and is abandoned in a year costs a
|
||||
week. For something small and stable, a dozen lines in your own codebase often beats a package.
|
||||
|
||||
## Pin and lock
|
||||
|
||||
Exact versions in the manifest for anything that matters, a lockfile committed, and installs
|
||||
that respect it. A `^` range means your build tomorrow differs from your build today. The
|
||||
lockfile is the build's memory; do not delete it to "fix" a conflict — resolve the conflict.
|
||||
|
||||
## Update on a schedule, read the changelog
|
||||
|
||||
Routine small updates beat a yearly painful one. For a major bump: read the changelog and the
|
||||
migration guide, find every call site of the changed API, and apply one shape of change (see
|
||||
the migrate skill). Update one thing at a time so a regression has an obvious cause.
|
||||
|
||||
## Know your transitive tree
|
||||
|
||||
A direct dependency drags in dozens of transitive ones. Audit the tree for: known
|
||||
vulnerabilities (`audit`/SCA tooling), abandoned packages deep in it, and duplicate copies of
|
||||
the same library at different versions bloating the bundle. Remove what you no longer use — an
|
||||
unused dependency is attack surface and install time for nothing.
|
||||
|
||||
## Supply chain is a trust decision
|
||||
|
||||
A package runs its install scripts with your permissions. Prefer packages with provenance and a
|
||||
reproducible build, be wary of sudden ownership transfers, and pin so a hijacked publish does
|
||||
not reach you automatically. The lockfile is also your audit trail of exactly what shipped.
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
name: docker
|
||||
description: Write or fix Dockerfiles and container setups. Use when an image is too large, a build is slow, a container will not start, or layering and caching need design.
|
||||
---
|
||||
|
||||
# Docker
|
||||
|
||||
An image is a build artifact. Small, reproducible, and boring is the goal.
|
||||
|
||||
## Layer cache is the whole speed game
|
||||
|
||||
Order instructions from least to most frequently changed: base image, then dependency
|
||||
manifests, then `install`, then source copy, then build. Copying `.` before installing
|
||||
dependencies means every code change re-runs the install — the single most common Dockerfile
|
||||
mistake.
|
||||
|
||||
## Small images, on purpose
|
||||
|
||||
- Use multi-stage builds: build in a full toolchain stage, copy only the artifact into a slim
|
||||
runtime stage. The compiler does not ship to production.
|
||||
- Pick a slim or distroless base unless you need the tooling. Alpine is small but musl breaks
|
||||
some binaries; know why you chose it.
|
||||
- One `RUN` with `&&` for related steps, cleaning up in the same layer — a separate `RUN rm`
|
||||
does not shrink the image, the data is still in the earlier layer.
|
||||
|
||||
## The container is not a VM
|
||||
|
||||
- One process per container, as PID 1, so signals work. Use an init if the app spawns children.
|
||||
- Do not run as root. Add a user and `USER` it.
|
||||
- Read-only filesystem where possible; write to a mounted volume for anything that must persist.
|
||||
Nothing in the image is writable state.
|
||||
|
||||
## .dockerignore is as important as the Dockerfile
|
||||
|
||||
Exclude `.git`, `node_modules`, build output, and any secret file. A context that sends the
|
||||
whole repo is slow, and a secret copied into an image layer is a secret to rotate.
|
||||
|
||||
## Healthcheck and logs
|
||||
|
||||
The process logs to stdout/stderr, never to a file inside the container — the runtime collects
|
||||
it. Add a `HEALTHCHECK` that proves the service answers, not just that the process exists.
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
name: docs
|
||||
description: Write or update documentation, READMEs, and guides. Use when asked to document a feature, write usage docs, or bring docs back in line with the code.
|
||||
---
|
||||
|
||||
# Documentation
|
||||
|
||||
Docs lie by omission. Write only what you have verified in the code.
|
||||
|
||||
## Ground every claim in the source
|
||||
|
||||
Before documenting a behaviour, read it. A flag, a default, an error message — open the
|
||||
code and quote what it actually does, not what the name suggests. The most damaging doc
|
||||
line is the confident one that was true two versions ago. If the code and the existing
|
||||
docs disagree, the code is right; say so and fix the doc.
|
||||
|
||||
## Answer the reader's actual question
|
||||
|
||||
A reader opens a doc with a task, not a desire for completeness. Lead with the thing they
|
||||
came to do, in the order they will do it:
|
||||
|
||||
- **A reference** lists what exists: every flag, every field, with its default and its type.
|
||||
- **A guide** walks one path to one outcome. Resist documenting every branch — link instead.
|
||||
- **A README** orients in sixty seconds: what it is, install, the first command that works.
|
||||
|
||||
## Show, then say
|
||||
|
||||
A working example beats a paragraph about one. Every command in the doc must be one you
|
||||
ran, with its real output. A snippet that was never executed is a bug waiting for a reader.
|
||||
|
||||
## Match the house style
|
||||
|
||||
Read the neighbouring docs first: their heading depth, their code-fence language tags,
|
||||
their tone. A doc that reads foreign is a doc nobody trusts enough to maintain.
|
||||
|
||||
## Keep it true over time
|
||||
|
||||
Document the stable contract, not the current implementation, unless the point is the
|
||||
implementation. The fewer specifics a doc pins down, the less it rots.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: frontend
|
||||
description: Build or fix a web UI. Use when working on components, state, rendering performance, forms, or anything the user sees and interacts with in a browser.
|
||||
---
|
||||
|
||||
# Frontend
|
||||
|
||||
The user's experience is the metric. Fast, clear, and forgiving beats clever.
|
||||
|
||||
## State lives as low as it can
|
||||
|
||||
Lift state only as high as the components that share it. Global state for something two siblings
|
||||
need is re-render and complexity for everything. Server data is not client state — cache it with
|
||||
the data layer rather than duplicating it into a store you must keep in sync by hand.
|
||||
|
||||
## Rendering is the usual bottleneck
|
||||
|
||||
Before optimising, find what re-renders. A component that re-renders on every parent render
|
||||
because of an inline object or function prop is the common case. Memoize the expensive subtree,
|
||||
not everything — `useMemo` and `useCallback` have a cost too, and slapping them everywhere is
|
||||
its own slowdown.
|
||||
|
||||
## Forms respect the user
|
||||
|
||||
- Validate on blur or submit, not on every keystroke, and show the message at the field.
|
||||
- Never clear a form on an error. The user's input is the most expensive thing on the page.
|
||||
- Disable the submit while submitting, and say what is happening. A double-submitted form is a
|
||||
duplicate record.
|
||||
|
||||
## Accessibility is not a later pass
|
||||
|
||||
Semantic HTML first: a `<button>` that looks like a button beats a `<div>` with a click
|
||||
handler. Keyboard-reachable everything, visible focus, labels on inputs, alt text that conveys
|
||||
the point not the pixels. Colour is never the only carrier of meaning.
|
||||
|
||||
## Measure what the user feels
|
||||
|
||||
Load: get the first meaningful paint and the time-to-interactive down before micro-tuning.
|
||||
Bundle: split the route nobody opens, lazy-load the heavy component. A Lighthouse number is a
|
||||
proxy; the goal is that it never feels slow.
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
name: git-workflow
|
||||
description: Work with branches, rebases, merges, and history. Use when untangling a branch, preparing a PR, deciding rebase vs merge, or recovering from a git mistake.
|
||||
---
|
||||
|
||||
# Git workflow
|
||||
|
||||
History is a communication tool. Write it for the person who reads it in six months — usually you.
|
||||
|
||||
## One branch, one purpose
|
||||
|
||||
A branch that does two things produces a PR that can only be reviewed as all-or-nothing and
|
||||
reverted only whole. Keep it small and single-purpose; open the second thing as its own branch.
|
||||
|
||||
## Rebase to clean up, merge to preserve
|
||||
|
||||
- Rebase your own unpushed work freely: it makes a linear, readable history.
|
||||
- Never rebase a branch others have pulled — it rewrites commits they have, and the next pull
|
||||
becomes a mess. Merge shared branches instead.
|
||||
- Interactive rebase before opening the PR: squash the "fix typo" and "wip" commits into the
|
||||
change they belong to. The PR should read as a series of intentional steps, not a diary.
|
||||
|
||||
## Recover without panic
|
||||
|
||||
- `git reflog` finds almost anything you "lost": the branch you deleted, the commit you reset
|
||||
away. Nothing committed is truly gone for ~30 days.
|
||||
- A bad merge: `git merge --abort`. A bad rebase: `git rebase --abort`. Both stop cleanly
|
||||
rather than pushing forward into a worse state.
|
||||
- Committed to the wrong branch: `git reset --soft` to keep the work, switch, recommit.
|
||||
|
||||
## The commit message is the review's first page
|
||||
|
||||
Subject under 70 chars, imperative, says what changed. Body explains *why* when it is not
|
||||
obvious. A reviewer who cannot tell why a change exists from its message will ask, or worse,
|
||||
approve without understanding.
|
||||
|
||||
## Read the conflict, do not guess
|
||||
|
||||
On a conflict, open the file and understand both sides before resolving. Taking "ours" or
|
||||
"theirs" wholesale because it is faster is how a resolved conflict silently drops someone's
|
||||
work.
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
name: i18n
|
||||
description: Internationalise or localise a product. Use when extracting strings for translation, formatting dates and numbers for a locale, handling pluralisation, or fixing layout that breaks in another language.
|
||||
---
|
||||
|
||||
# Internationalisation
|
||||
|
||||
Hard-coded English is a bug in every other language. Externalise strings and never assume a
|
||||
grammar.
|
||||
|
||||
## Every user-facing string is a key
|
||||
|
||||
No string in the UI lives in code; it lives in a message catalogue under a key. Concatenating
|
||||
translated fragments is the classic bug: "You have " + n + " messages" cannot be reordered for
|
||||
a language whose grammar puts the number elsewhere. Use a format with named placeholders:
|
||||
`{count, plural, ...}`, translated as a whole.
|
||||
|
||||
## Pluralisation and gender are not English
|
||||
|
||||
Languages have one, two, several, or no plural forms, with rules that do not map to "1 vs other".
|
||||
Use the ICU plural machinery of your i18n library and let the translator fill in every form the
|
||||
locale needs. The same goes for gendered agreement.
|
||||
|
||||
## Format dates, numbers, and currencies by locale
|
||||
|
||||
Never `dd/mm/yyyy` by hand: `03/04/2025` is March 4th to one user and April 3rd to another.
|
||||
Use the platform's locale-aware formatter (`Intl.DateTimeFormat`, `Intl.NumberFormat`).
|
||||
Store and transmit ISO 8601 / UTC; format for display only.
|
||||
|
||||
## Layout breaks in translation
|
||||
|
||||
German runs ~30% longer than English; some scripts are right-to-left. Flexible layout, no fixed
|
||||
widths on translated text, and CSS logical properties (`margin-inline-start` not
|
||||
`margin-left`) so RTL mirrors correctly. Test with a pseudo-locale that lengthens and accents
|
||||
every string to find the overflows before a translator does.
|
||||
|
||||
## Sort and search correctly
|
||||
|
||||
String order is locale-dependent: `ä` sorts with `a` in German, after `z` in Swedish. Use
|
||||
locale-aware collation (`Intl.Collator` or the database's) rather than byte order, and normalise
|
||||
Unicode before comparing, because the same character has more than one byte representation.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: incident
|
||||
description: Respond to a production incident. Use when something is down, degraded, or misbehaving in production and must be diagnosed and mitigated under time pressure.
|
||||
---
|
||||
|
||||
# Incident response
|
||||
|
||||
Mitigate first, diagnose second. Restore service, then find out why.
|
||||
|
||||
## Confirm and scope before touching anything
|
||||
|
||||
What is actually broken, for whom, since when? Check the signal, not the report: the dashboard,
|
||||
the error rate, the health endpoint. A wrong scope sends you chasing a symptom. State the impact
|
||||
plainly in one line before you start changing things.
|
||||
|
||||
## Recent change is the prime suspect
|
||||
|
||||
Most incidents follow a deploy, a config change, a flag flip, or a scaling event. What changed
|
||||
in the window before it broke? Check the deploy log and the diff. The fastest fix is usually to
|
||||
undo the last change, not to understand it.
|
||||
|
||||
## Mitigate, then understand
|
||||
|
||||
- Roll back the deploy, flip the flag off, fail over, scale up, restart the wedged process —
|
||||
whichever restores service fastest, even if you do not yet know the root cause.
|
||||
- A mitigation you can reverse beats a perfect diagnosis that takes an hour. Note what you did so
|
||||
it can be undone or made permanent later.
|
||||
- Do not deploy an unreviewed "fix" into the fire; it adds a second change to a system already
|
||||
misbehaving.
|
||||
|
||||
## Preserve evidence before it rotates away
|
||||
|
||||
Capture the logs, the error, the relevant metrics, a snapshot of the state — before a restart or
|
||||
a rollback destroys it. You will want it for the postmortem, and it may be the only copy.
|
||||
|
||||
## Communicate and follow up
|
||||
|
||||
Say what is broken, what you are doing, and when the next update is — to whoever is affected,
|
||||
in plain language, on a schedule. Afterwards: write the timeline, the root cause, and the
|
||||
follow-ups that stop it recurring. An incident with no follow-up is a loan against the next one.
|
||||
@@ -0,0 +1,42 @@
|
||||
---
|
||||
name: logging
|
||||
description: Add or improve logging and observability. Use when debugging in production, adding structured logs, choosing log levels, or making a system traceable.
|
||||
---
|
||||
|
||||
# Logging
|
||||
|
||||
Logs are how you debug a system you cannot attach a debugger to. Write them for the 3am
|
||||
incident, not the happy path.
|
||||
|
||||
## Structure over prose
|
||||
|
||||
Emit fields, not sentences: `{ user: id, action: "checkout", ms: 142, ok: false }`, not
|
||||
`"User checked out"`. Structured logs are searchable and aggregable; a sentence is neither.
|
||||
One event, one line, one level.
|
||||
|
||||
## Levels are a contract
|
||||
|
||||
- `error` — something is broken and someone should look. Not "a user gave bad input".
|
||||
- `warn` — unexpected but handled; worth a glance.
|
||||
- `info` — the meaningful state transitions: started, finished, the decision made. Sparse.
|
||||
- `debug` — everything you might want while diagnosing, off in production.
|
||||
|
||||
A log at the wrong level trains people to ignore the right one. If everything is `error`,
|
||||
nothing is.
|
||||
|
||||
## Log the decision points, not every line
|
||||
|
||||
At a boundary — a request in, a call out, a branch taken — log what was decided and the inputs
|
||||
that decided it, with a correlation id that follows the request across services. You should be
|
||||
able to trace one request end to end from the id alone.
|
||||
|
||||
## Never log a secret
|
||||
|
||||
No passwords, tokens, session ids, full card numbers, or personal data beyond what policy
|
||||
allows. Redact at the point of logging, not by hoping a downstream filter catches it. A secret
|
||||
in a log aggregator is a secret to rotate.
|
||||
|
||||
## Measure, do not just log
|
||||
|
||||
For anything with a latency or a rate, a metric answers "is it slow?" faster than a thousand
|
||||
log lines. Logs explain *why*; metrics tell you *that* something is wrong in the first place.
|
||||
@@ -0,0 +1,55 @@
|
||||
---
|
||||
name: migrate
|
||||
description: Upgrade a dependency, framework, or language version across a codebase. Use when a major version bump, a deprecation, or a breaking API change has to be applied.
|
||||
---
|
||||
|
||||
# Migration
|
||||
|
||||
The failure mode is a half-applied migration: it compiles, most tests pass, and one code
|
||||
path still uses the old API.
|
||||
|
||||
## Read the changelog before the code
|
||||
|
||||
Find what actually broke. A major version usually has a migration guide; read it and list
|
||||
the changes that apply to this codebase specifically. Below 1.0, treat a minor bump as
|
||||
breaking — semver promises nothing there.
|
||||
|
||||
## Find every call site before changing one
|
||||
|
||||
Grep for the old API across the whole repository, including tests, scripts, config, CI
|
||||
workflows, Dockerfiles, and documentation. A version literal pinned in a workflow while the
|
||||
manifest says something else is a split-brain deploy.
|
||||
|
||||
Write the list down with `todo_write`. The list is the migration; the edits are mechanical.
|
||||
|
||||
## Change in one shape
|
||||
|
||||
Apply the same transformation everywhere rather than improving each site as you pass
|
||||
through it. A migration mixed with refactoring cannot be reviewed, and cannot be reverted
|
||||
if the upgrade turns out to be wrong.
|
||||
|
||||
`apply_patch` is the tool for this: one atomic patch across the files that must land
|
||||
together.
|
||||
|
||||
## Verify at the boundary that broke
|
||||
|
||||
Type checks catch signature changes and miss behaviour changes — the two ways a migration
|
||||
actually breaks you. Run the tests, then actually *use* the thing that was upgraded: start
|
||||
the server, run the CLI, execute the query, hit the endpoint. A green suite over an
|
||||
untested upgrade path proves only that the suite did not cover it.
|
||||
|
||||
Pay special attention to silent behaviour changes: a default that flipped, a deprecated
|
||||
call that still runs but does something subtly different, an error type that changed shape.
|
||||
These compile, pass type checks, and still break production.
|
||||
|
||||
## Never hand-merge a lockfile
|
||||
|
||||
On a conflict, take either side whole and regenerate with the package manager. The resolver
|
||||
owns that file; a hand-merge is a split-brain dependency tree that installs differently on
|
||||
every machine.
|
||||
|
||||
## Report
|
||||
|
||||
The version before and after, every file class touched, what you verified by running, the
|
||||
behaviour changes you checked by hand, and anything the changelog said applies that you
|
||||
deliberately did not do — with the reason.
|
||||
@@ -0,0 +1,42 @@
|
||||
---
|
||||
name: onboarding
|
||||
description: Orient in an unfamiliar codebase. Use when dropped into a new project and asked to understand it, or when writing the docs that help a newcomer get productive.
|
||||
---
|
||||
|
||||
# Onboarding to a codebase
|
||||
|
||||
Understand the running system before the source. The goal is a correct mental model, not to have
|
||||
read every file.
|
||||
|
||||
## Get it running first
|
||||
|
||||
Build it, run it, run the tests. A project you can execute you can interrogate; one you have
|
||||
only read you can only guess at. The README and the `package.json`/`Makefile` scripts tell you
|
||||
the intended commands; if they do not work, that is your first finding.
|
||||
|
||||
## Trace one request end to end
|
||||
|
||||
Pick the central thing the system does and follow it: the entry point, the route or main, the
|
||||
handler, the data out and back. One full path teaches you the architecture faster than reading
|
||||
any single module. Note the layers you cross — that is the system's real structure.
|
||||
|
||||
## Read the structure, not the files
|
||||
|
||||
- The directory layout names the major components and their boundaries.
|
||||
- The dependency manifest names the frameworks and the big choices already made.
|
||||
- The tests show what the code is supposed to do, often better than the code does.
|
||||
- `git log` on a core file shows what changes often and why — the living parts versus the
|
||||
stable ones.
|
||||
|
||||
## Map the seams
|
||||
|
||||
Where does data enter and leave (HTTP, a queue, a file)? Where is state kept (a database,
|
||||
memory, a cache)? Where are the trust boundaries? Those are the places bugs and features both
|
||||
live. You do not need to know every file; you need to know where a change of a given kind would
|
||||
go.
|
||||
|
||||
## Ask the codebase, then a person
|
||||
|
||||
Grep and the outline/symbol tools answer most "where is X" faster than reading. When genuinely
|
||||
stuck on *why* something exists — that is a question for a person or the history, not more
|
||||
reading.
|
||||
@@ -0,0 +1,42 @@
|
||||
---
|
||||
name: optimize-sql
|
||||
description: Diagnose and fix a slow SQL query. Use when a query is slow, a page makes too many queries, or an execution plan needs reading.
|
||||
---
|
||||
|
||||
# SQL optimisation
|
||||
|
||||
Read the plan before changing anything. `EXPLAIN` (or `EXPLAIN ANALYZE`) tells you what the
|
||||
database actually does; guessing at it is how you add an index that helps nothing.
|
||||
|
||||
## Read the plan for the expensive node
|
||||
|
||||
Find the node with the highest cost: a sequential scan over a large table, a nested loop over
|
||||
many rows, a sort that spills to disk. Optimise that node. A plan with ten cheap nodes and one
|
||||
expensive one has exactly one thing to fix.
|
||||
|
||||
## The index that matches the query
|
||||
|
||||
- Index the columns in the `WHERE` and `JOIN` clauses, and for a sort, the `ORDER BY`.
|
||||
- A composite index `(a, b, c)` serves a leftmost prefix: `a`, `a,b`, `a,b,c` — not
|
||||
`b` alone. Order the columns by the equality filters first, then the range, then the sort.
|
||||
- A covering index includes every column the query reads, so the table is never touched. That
|
||||
is the fastest a read gets.
|
||||
|
||||
## Write the query so the index is usable
|
||||
|
||||
- `WHERE lower(email) = ...` cannot use a plain index on `email`; either store it lowered or
|
||||
use a functional index. A function on the column defeats the index.
|
||||
- Leading `LIKE '%x'` cannot use a B-tree index; `LIKE 'x%'` can.
|
||||
- `OR` across different columns often defeats an index; `UNION` of two indexed queries can be
|
||||
faster.
|
||||
|
||||
## Kill the N+1 first
|
||||
|
||||
Before any index: if the page runs one query per row, that is the fix. Batch with
|
||||
`WHERE id IN (...)` or a join. Ten queries become one beats ten individually-fast queries.
|
||||
|
||||
## Measure the change
|
||||
|
||||
`EXPLAIN ANALYZE` before and after, against realistic data volume. An index that helps a
|
||||
100-row table may not justify its write cost at 100 million. Report the timing you actually
|
||||
measured, not the improvement you expected.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: perf-frontend
|
||||
description: Make a web page faster. Use when a page loads slowly, feels janky, fails Core Web Vitals, or ships too much JavaScript.
|
||||
---
|
||||
|
||||
# Frontend performance
|
||||
|
||||
Measure the user's experience first: a Lighthouse lab score and, better, real-user data. Optimise
|
||||
the metric that is actually failing, not the one easiest to move.
|
||||
|
||||
## The vitals and what drives them
|
||||
|
||||
- **LCP** (largest contentful paint) — almost always the hero image or a web font. Preload it,
|
||||
size it correctly, serve it in a modern format, and do not let render-blocking resources delay it.
|
||||
- **INP / responsiveness** — long tasks on the main thread. Break up work, defer non-urgent JS,
|
||||
and keep event handlers fast. A click that responds in 50ms feels instant; 300ms feels broken.
|
||||
- **CLS** (layout shift) — images and embeds without dimensions, late-injected banners, web fonts
|
||||
swapping. Reserve the space before the content arrives.
|
||||
|
||||
## Ship less JavaScript
|
||||
|
||||
The bundle is usually the problem. Route-level code splitting so a page loads only what it needs,
|
||||
lazy-load the heavy below-the-fold component, and audit the dependency tree for a large library
|
||||
imported for one function. Removing 100KB of JS beats most micro-optimisations.
|
||||
|
||||
## Network discipline
|
||||
|
||||
- Cache static assets with long, content-hashed lifetimes; the second visit should cost almost
|
||||
nothing.
|
||||
- Compress (brotli/gzip) and serve images at the size they are displayed, responsive `srcset`,
|
||||
not a 3000px original in a 300px slot.
|
||||
- Fetch in parallel, not in waterfalls: start independent requests together, and preload the
|
||||
critical few.
|
||||
|
||||
## Change one thing, measure it
|
||||
|
||||
Take a baseline (the metric, the page, the device class), make one change, re-measure on the same
|
||||
setup. Two changes at once and you do not know which paid. Report the before/after you actually
|
||||
measured, on a realistic device and connection — a developer's fast laptop and fiber hides what a
|
||||
mid-range phone on 4G feels.
|
||||
@@ -0,0 +1,49 @@
|
||||
---
|
||||
name: perf
|
||||
description: Make something faster, or find out why it is slow. Use when a command, request, test suite, or build takes longer than it should.
|
||||
---
|
||||
|
||||
# Performance
|
||||
|
||||
Measure first. A change made without a number before it is a guess with extra steps, and
|
||||
most "optimisations" made on a guess make the code worse and no faster.
|
||||
|
||||
## Get a number
|
||||
|
||||
Time the actual operation, not a proxy for it: `time`, the framework's own timing output,
|
||||
or a loop around the slow call with a timestamp either side. Use realistic input — a fast
|
||||
result on a tiny fixture tells you nothing about the production case. Record the baseline
|
||||
with `remember` so the comparison survives compaction, and run it enough times that a
|
||||
warm cache and jitter do not fool you.
|
||||
|
||||
If you cannot measure it, say so and stop. Optimising an unmeasured path is how a codebase
|
||||
accumulates complexity that buys nothing.
|
||||
|
||||
## Find where the time goes
|
||||
|
||||
- **Wall-clock dominated by one call?** Look there and nowhere else. The biggest node is
|
||||
the only one worth touching.
|
||||
- **Spread evenly?** Suspect the loop around it: an O(n²) walk, a query per row, a file
|
||||
read per iteration, an allocation per element.
|
||||
- **Idle time?** It is waiting, not computing: a sequential chain of independent awaits, an
|
||||
unpooled connection, a contended lock, a slow remote call.
|
||||
|
||||
The usual culprits, in the order they actually appear: N+1 queries, work repeated inside a
|
||||
loop that could be hoisted, a missing index, sequential awaits that could run together,
|
||||
reading a whole file to use one line, and re-parsing something that could be parsed once.
|
||||
|
||||
## Change one thing
|
||||
|
||||
One change, then re-measure on the same setup. Two changes together and you do not know
|
||||
which one paid — and one of them may have cost. If the number did not move, revert the
|
||||
change; an optimisation that does not measure is just complexity.
|
||||
|
||||
## Stop when it is fast enough
|
||||
|
||||
State the target before you start: "the test suite under a minute", "the endpoint under
|
||||
200ms". Past the target, further work is complexity with no user on the other end of it.
|
||||
|
||||
## Report
|
||||
|
||||
Baseline, the change, the new number, and what you deliberately did not do. A 40% win with
|
||||
one line changed is a better report than a 45% win that restructured a module.
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
name: plan
|
||||
description: Break a non-trivial task into an ordered, verifiable sequence before writing code. Use when a request is large, spans several files, or its steps depend on each other.
|
||||
---
|
||||
|
||||
# Planning
|
||||
|
||||
A plan that cannot be checked is a wish. Every step ends in something you can run.
|
||||
|
||||
## Understand before you sequence
|
||||
|
||||
Read enough to know the real shape of the work: the entry point, the data's path, the
|
||||
module that owns the behaviour. A plan made from filenames alone reorders itself the
|
||||
moment you open the first file. Grep the actual call sites; do not plan around a guess.
|
||||
|
||||
## Order by dependency, not by file
|
||||
|
||||
A step may depend on another's output: a type before its callers, a schema before its
|
||||
migration, a test helper before the tests that use it. Sequence so nothing references
|
||||
what does not exist yet. If two steps are independent, say so — the order between them
|
||||
is free and you may take the cheaper one first to derisk the rest.
|
||||
|
||||
## One step, one verifiable outcome
|
||||
|
||||
Each step names the command that proves it done: a test that passes, a build that
|
||||
compiles, a script that runs. "Wire it up" is not a step. Write the list with
|
||||
`todo_write`, then work it in order, marking done immediately — not in a batch at the end.
|
||||
|
||||
## Keep it small
|
||||
|
||||
The plan is a scaffold, not the building. If a step grows past "change these few files",
|
||||
split it. If the task turns out smaller than it looked, drop the remaining steps and say
|
||||
why rather than inventing work to fill them.
|
||||
|
||||
## Replan when the ground moves
|
||||
|
||||
New information that changes the order or the scope is a reason to rewrite the list, not
|
||||
to push through it. A stale plan followed faithfully is worse than no plan.
|
||||
@@ -0,0 +1,42 @@
|
||||
---
|
||||
name: readme
|
||||
description: Write or fix a project README. Use when creating a README, when a new user cannot get the project running from it, or when it has drifted from the code.
|
||||
---
|
||||
|
||||
# README
|
||||
|
||||
A README has sixty seconds to answer: what is this, do I want it, and how do I run it. Everything
|
||||
else is secondary to those three.
|
||||
|
||||
## The first screen answers three questions
|
||||
|
||||
1. **What it is** in one or two sentences, concrete about the problem it solves — not "a modern
|
||||
solution" but "a CLI that lints Terraform plans against your org's policies".
|
||||
2. **Install** — the one command that gets it.
|
||||
3. **The first thing that works** — the minimal command or snippet that produces visible output.
|
||||
If a new user cannot get a win in two minutes, most leave.
|
||||
|
||||
## Verify every command
|
||||
|
||||
Run each command in the README against a clean environment and paste its real output. The most
|
||||
common README defect is an install or quickstart that no longer works because the code moved and
|
||||
the doc did not. If you cannot run it, do not write it.
|
||||
|
||||
## Structure for scanning
|
||||
|
||||
After the quickstart, in the order a new user needs them: features as a short list of what it
|
||||
does (not how), the common tasks as copy-paste examples, configuration as a table of options with
|
||||
defaults, then links to deeper docs. Headings let a reader jump; a wall of prose gets skimmed
|
||||
past the thing they needed.
|
||||
|
||||
## Show, do not tell
|
||||
|
||||
A three-line example of real use beats a paragraph describing capability. Show the input and the
|
||||
output. A screenshot or asciinema of the actual tool running is worth a hundred adjectives —
|
||||
include one if the tool has any visual surface.
|
||||
|
||||
## Keep it true
|
||||
|
||||
Document the stable interface, not this week's implementation, or the README rots. Re-read it on
|
||||
every release: a README that contradicts the current version is worse than a short one, because
|
||||
it actively misleads.
|
||||
@@ -0,0 +1,44 @@
|
||||
---
|
||||
name: refactor
|
||||
description: Restructure code without changing behaviour. Use when asked to refactor, clean up, extract, or reorganise.
|
||||
---
|
||||
|
||||
# Refactoring
|
||||
|
||||
Behaviour must not change. That is the whole constraint — every other goal (clarity,
|
||||
structure, naming) is subordinate to it. The moment behaviour changes, you are no longer
|
||||
refactoring, you are editing, and the safety argument below stops holding.
|
||||
|
||||
## Establish the safety net first
|
||||
|
||||
Run the existing tests and record that they pass — with `remember`, so the baseline
|
||||
survives compaction. If the code has no tests, write one that pins current behaviour,
|
||||
*including the ugly parts*: the odd return value, the quirk callers depend on. You are not
|
||||
judging the behaviour, you are freezing it. Refactoring untested code is not refactoring;
|
||||
it is rewriting, and it belongs under the edit workflow with its own verification.
|
||||
|
||||
## Then move in small steps
|
||||
|
||||
One transformation at a time, tests green between each. Rename, then extract, then move —
|
||||
not all three in one edit. The mechanical refactorings are the safe ones: rename, extract
|
||||
function, inline, move. Compose them. A large refactor that fails leaves you unable to
|
||||
tell which of five steps broke it; a small one that fails tells you exactly which.
|
||||
|
||||
After each step, run the tests, not just the typechecker. Types catch signature drift;
|
||||
they do not catch a reordered conditional or a dropped early return.
|
||||
|
||||
## What not to do
|
||||
|
||||
- Do not fix bugs while refactoring. Note them, finish the refactor green, then fix in a
|
||||
separate change — otherwise a regression could be either the refactor or the fix.
|
||||
- Do not add abstraction for a single caller. Duplication beats a premature interface;
|
||||
the third caller is when the abstraction earns its name.
|
||||
- Do not widen the scope. The request was this code, not its neighbours. A refactor that
|
||||
"while we're here" touches five more files is five more files of unreviewable risk.
|
||||
- Do not change public API unless asked; if it must change, say so first and update every
|
||||
caller in the same change.
|
||||
|
||||
## Done means
|
||||
|
||||
Tests pass, behaviour is identical, and the diff is smaller than the reader feared. If the
|
||||
diff is larger than the code it moved, you abstracted too early — put it back.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: release
|
||||
description: Cut a release: versioning, changelogs, tagging, publishing. Use when asked to release, bump a version, write release notes, or fix a broken publish.
|
||||
---
|
||||
|
||||
# Release
|
||||
|
||||
A release is a promise that a specific, identified state of the code works. Make it
|
||||
reproducible or do not make it.
|
||||
|
||||
## Version says what changed
|
||||
|
||||
Semver: breaking is a major, a feature is a minor, a fix is a patch. The number is a message to
|
||||
whoever upgrades, not a marketing choice. Below 1.0, say so plainly — semver promises nothing
|
||||
and the version should not pretend otherwise.
|
||||
|
||||
## The changelog is for the upgrader
|
||||
|
||||
- Group by what the reader must do: breaking changes and required actions first, then features,
|
||||
then fixes.
|
||||
- Write it as "you can now X" or "Y no longer Z", from the user's side, not the commit's. A
|
||||
changelog that is a git log is a changelog nobody reads.
|
||||
- Every breaking change names the migration: what to change to keep working.
|
||||
|
||||
## Verify before you tag
|
||||
|
||||
The release candidate builds clean from a fresh checkout, the tests pass, and the version string
|
||||
in the source matches the tag you are about to push. A version/tag mismatch published is the
|
||||
kind of thing that ships "0.4" labelled as "0.3" forever.
|
||||
|
||||
## Tag the commit, publish the artifact
|
||||
|
||||
Tag the exact commit that was verified, and build the artifact from that tag — not from a
|
||||
working tree that has since moved. The tag is immutable; never move it to a different commit.
|
||||
If a release is wrong, cut a new one with a new number; do not quietly re-tag.
|
||||
|
||||
## If it goes wrong
|
||||
|
||||
Have the rollback ready before you need it: the previous artifact still available, the deploy
|
||||
reversible. A bad release is fixed forward with a patch release, not by deleting the evidence.
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
name: review
|
||||
description: Review a diff or a file for defects. Use when asked to review, critique, or check code before it ships.
|
||||
---
|
||||
|
||||
# Code review
|
||||
|
||||
Severity order. Do not lead with style — a review that opens on naming while a real bug
|
||||
sits three lines down has failed at its one job.
|
||||
|
||||
1. **Incorrect behaviour** — wrong result, wrong edge case, wrong state after failure.
|
||||
2. **Missing validation at trust boundaries** — user input, network responses, file contents,
|
||||
anything crossing a process line. Internal calls need no defensive checks.
|
||||
3. **Security** — injection, path traversal, secrets in logs or errors, missing authz.
|
||||
4. **Resource handling** — unclosed handles, unbounded growth, unawaited promises.
|
||||
5. **Clarity** — only when it will cause a future defect.
|
||||
|
||||
## How to read the change
|
||||
|
||||
- Read the diff against its intent. Does it actually do what the title/commit says? A
|
||||
correct-looking diff that solves the wrong problem is the most expensive approval.
|
||||
- Read the *deleted* lines as carefully as the added ones. Behaviour is often lost in a
|
||||
removal, and diffs render deletions quietly.
|
||||
- Follow each new call one level into the callee. The assumption that breaks it is usually
|
||||
one level down, invisible in the diff itself.
|
||||
|
||||
## For each finding
|
||||
|
||||
State file and line, the concrete failure (what input makes it break, or why it always
|
||||
breaks), and the change. Show the fix as code when it is short. "This could be a problem"
|
||||
without a path to a real input is noise; either trace it or drop it.
|
||||
|
||||
Order findings by severity and lead with the worst. Skip anything a formatter would fix.
|
||||
Skip preference. If a choice is defensible, leave it — a review is not a place to impose
|
||||
your style on code that works.
|
||||
|
||||
## Say when it is fine
|
||||
|
||||
A review that invents problems to look thorough is worse than a short one. If the change
|
||||
is correct, say so plainly and stop. "Looks correct, and here is what I checked" is a
|
||||
complete and useful review.
|
||||
|
||||
## Verify, do not assume
|
||||
|
||||
Read the surrounding code before calling something a bug. A "missing" null check often
|
||||
exists one level up; a "redundant" guard often covers a caller you have not seen. Run the
|
||||
tests or write the failing input if that is what settles it. A finding you verified is
|
||||
worth ten you suspected.
|
||||
@@ -0,0 +1,53 @@
|
||||
---
|
||||
name: security
|
||||
description: Review code for security defects, or write code that handles untrusted input. Use when touching authentication, user input, file paths, shell commands, SQL, or anything reachable from the network.
|
||||
---
|
||||
|
||||
# Security
|
||||
|
||||
Find the trust boundary first. Everything crossing it is hostile until parsed.
|
||||
|
||||
## The boundaries in most codebases
|
||||
|
||||
- Request bodies, query strings, headers, cookies.
|
||||
- File contents and filenames, including paths a user supplied.
|
||||
- Environment variables in a multi-tenant deployment.
|
||||
- Anything a model or a third-party API returned.
|
||||
|
||||
Inside a boundary, values are already validated and re-checking them is noise. At the
|
||||
boundary, nothing is optional.
|
||||
|
||||
## What to look for, in order
|
||||
|
||||
1. **Injection.** String-built SQL, shell commands assembled from input, `eval`, template
|
||||
rendering with user data as the template rather than the data. The fix is parameters and
|
||||
argument arrays, never escaping.
|
||||
2. **Missing authorisation.** An endpoint that checks *who* you are but not *what* you may
|
||||
touch. Look for an id taken from the request and used without an ownership check.
|
||||
3. **Path traversal.** `../` in anything joined onto a filesystem root. Resolve, then verify
|
||||
the result is still inside the root — a prefix check on the raw input misses
|
||||
`a/../../secret`.
|
||||
4. **Secrets in the wrong place.** Keys in source, in logs, in error messages, in a commit.
|
||||
A secret that reached a log is a secret to rotate.
|
||||
5. **Server-side request forgery.** A URL from input, fetched. Block private and loopback
|
||||
addresses by *resolved* address, and re-check every redirect hop.
|
||||
6. **Weak crypto and hand-rolled auth.** Homemade token formats, `Math.random` for anything
|
||||
security-bearing, comparisons on secrets that are not constant time.
|
||||
|
||||
## Verify the path before reporting
|
||||
|
||||
Trace each candidate from an attacker-controlled value to the sink before you name it. A
|
||||
"this could be unsafe" without that path is noise that buries the real finding. If you
|
||||
cannot construct the malicious input that reaches the sink, either keep looking or say
|
||||
plainly that you could not confirm it.
|
||||
|
||||
Do not fix a symptom at one caller when the sink is shared. Grep every caller and fix the
|
||||
seam once — a sanitiser at one of five call sites is four open holes and one false sense
|
||||
of safety.
|
||||
|
||||
## Reporting
|
||||
|
||||
File, line, the path from input to sink, a concrete payload, and the fix. Rank by
|
||||
exploitability: a reachable injection beats a theoretical weakness in dead code. Say
|
||||
plainly when a thing that looks dangerous is actually fine, and why — a reviewer's
|
||||
confidence in the clean parts is worth as much as a finding.
|
||||
@@ -0,0 +1,50 @@
|
||||
---
|
||||
name: test
|
||||
description: Write or repair tests. Use when adding coverage, fixing a flaky test, or asked how something should be tested.
|
||||
---
|
||||
|
||||
# Testing
|
||||
|
||||
A test earns its place by failing when the code is wrong. A test that cannot fail — or
|
||||
that passes regardless — is not a test, it is overhead with a green checkmark.
|
||||
|
||||
## Match the project
|
||||
|
||||
Read two existing test files first. Use their runner, their assertion style, their file
|
||||
layout, their naming, their way of building fixtures. A test that looks foreign is a test
|
||||
nobody maintains, and an unmaintained test is deleted the first time it goes red.
|
||||
|
||||
## Test behaviour, not implementation
|
||||
|
||||
Assert on what a caller observes: the return value, the written file, the emitted event,
|
||||
the status code. A test that reaches into private state or mocks a collaborator's
|
||||
internals breaks on every refactor while catching nothing real. If you cannot say what
|
||||
the caller sees, you are testing the how, and the how is allowed to change.
|
||||
|
||||
Cover, in this order of value:
|
||||
- **The failure** — the bad input, the missing file, the null. Failure cases catch more
|
||||
real defects than happy paths, because most code is written for the happy path first.
|
||||
- **The boundaries** — empty, one, the maximum, off-by-one at each edge.
|
||||
- **The normal case** — one, to prove the wiring works at all.
|
||||
|
||||
## Never do this
|
||||
|
||||
- Do not assert what the code currently returns without knowing it is correct — that pins
|
||||
the bug into the suite and calls it a specification.
|
||||
- Do not weaken an assertion to make a test pass. If it fails, either the code or the
|
||||
expectation is wrong; find out which before you touch either.
|
||||
- Do not delete a failing test to go green. It is telling you something; listen.
|
||||
- Do not test the framework or the library. Your code is the subject; their code has its
|
||||
own suite.
|
||||
|
||||
## Flaky tests
|
||||
|
||||
A test that passes alone and fails in a suite is a shared-state problem: a global, a temp
|
||||
directory, a port, an unawaited promise, leftover data, or ordering. Find which — run it
|
||||
repeatedly and in isolation to confirm, then remove the shared state. Do not add a retry:
|
||||
a retried flake is a real intermittent bug you have decided to stop hearing about.
|
||||
|
||||
## Verify
|
||||
|
||||
Run the test and watch it fail before the fix, pass after. A test you never saw fail is
|
||||
not known to test anything.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
name: ux-copy
|
||||
description: Write user-interface text: labels, errors, empty states, onboarding. Use when wording a button, an error message, a confirmation, or any text the user reads in the product.
|
||||
---
|
||||
|
||||
# UX copy
|
||||
|
||||
Interface text is part of the interface. Clear, short, and honest beats clever.
|
||||
|
||||
## Lead with what the user cares about
|
||||
|
||||
- Buttons say the action and the object: "Save changes", not "OK". "Delete project", not "Yes".
|
||||
- Headings say the outcome or the thing, not the category: "Your invoices" over "Billing section".
|
||||
- Front-load the information word. Users scan the first two words; "3 errors found" scans, "Found
|
||||
3 errors" buries it.
|
||||
|
||||
## Error messages say what happened and what to do
|
||||
|
||||
Never blame, never jargon, never just a code. "Couldn't save because you're offline. Your changes
|
||||
are kept — try again when you're back." The pattern is: what went wrong, whether their work is
|
||||
safe, the one action to take. An error that only says "Something went wrong" makes the user do
|
||||
the debugging.
|
||||
|
||||
## Empty states teach, they do not apologise
|
||||
|
||||
An empty screen is a chance to say what goes here and how to start: "No projects yet. Create your
|
||||
first to see it here." plus the button. "Nothing to display" wastes the moment the user is most
|
||||
receptive to guidance.
|
||||
|
||||
## Be honest about destructive and irreversible actions
|
||||
|
||||
A confirmation names exactly what will happen and that it cannot be undone: "Delete 'Invoices
|
||||
2024'? This permanently removes 3,120 records and cannot be undone." The destructive button
|
||||
repeats the verb: "Delete", never a bare "Confirm" that could mean anything.
|
||||
|
||||
## Consistent words for consistent things
|
||||
|
||||
Pick one term per concept and use it everywhere — if it is a "project" in one place it is not a
|
||||
"workspace" in another. Sentence case for labels, no exclamation marks, no "please", no "simply".
|
||||
The tone is a competent colleague, not a marketing page.
|
||||
@@ -0,0 +1,53 @@
|
||||
---
|
||||
name: verify
|
||||
description: Confirm a change actually works by using it, not by reading it. Use before reporting a task complete, or when asked whether something works.
|
||||
---
|
||||
|
||||
# Verification
|
||||
|
||||
A green test suite says the tests pass. It does not say the feature works. The two are
|
||||
different claims, and only one of them is what the user asked for.
|
||||
|
||||
## Run the artifact, not the source
|
||||
|
||||
Build it and use it the way a user would, end to end:
|
||||
|
||||
- **CLI** — build the binary and run it. Happy path, bad input, `--help`. Read the actual
|
||||
output, not the output you expected.
|
||||
- **HTTP service** — start it and `curl` the endpoint. Check the status line and the body,
|
||||
not just that it returned something.
|
||||
- **Library** — write a throwaway script that imports and calls the new code end to end,
|
||||
the way a consumer would.
|
||||
- **Script or job** — run it against real input and inspect what it produced.
|
||||
|
||||
Delete the throwaway afterwards. A verification script left behind becomes clutter the
|
||||
next person trips over.
|
||||
|
||||
## What counts as evidence
|
||||
|
||||
Command output you actually saw. Paste the relevant lines, not a summary of them — a
|
||||
summary hides the one line that mattered.
|
||||
|
||||
These are not evidence:
|
||||
|
||||
- "The tests pass" for a change the tests do not cover.
|
||||
- "The types check" for anything about runtime behaviour.
|
||||
- "It compiles" for anything about correctness.
|
||||
- "It should work now" for anything at all.
|
||||
|
||||
## Check the failure path too
|
||||
|
||||
Feed it the input you expect to be rejected and confirm it is rejected, with a message
|
||||
that says why. Then the edge case at the boundary. A feature that works only on correct
|
||||
input is half-built, and the half that is missing is the half users hit first.
|
||||
|
||||
## Report what you did not verify
|
||||
|
||||
Say plainly what you could not run and why: a missing credential, a service you cannot
|
||||
start, a platform you are not on. An honest gap is useful — the reader can fill it. A
|
||||
claim that hides one is a bug you just shipped in prose.
|
||||
|
||||
## When verification fails
|
||||
|
||||
The defect is yours to fix in this turn. Do not report the task complete with a note that
|
||||
it did not work — that is a failure report, not a completion.
|
||||
+22
-1
@@ -141,11 +141,17 @@ let counter = 0;
|
||||
*/
|
||||
export function createTaskTool(opts: {
|
||||
model: LanguageModel;
|
||||
/** Cheaper model for `explore`, which is search rather than reasoning. Defaults to `model`. */
|
||||
subagentModel?: LanguageModel;
|
||||
/** Its id, so the parent can price the subagent's spend separately. */
|
||||
subagentModelId?: string;
|
||||
cwd?: string;
|
||||
maxSteps?: number;
|
||||
report?: SubagentReporter;
|
||||
/** Parent-owned approval for a worker's gated calls. Omit to disable `worker`. */
|
||||
approve?: SubagentApproval;
|
||||
/** Records a finished run's token use, so /cost can split subagent from parent spend. */
|
||||
onUsage?: (usage: { kind: SubagentKind; inputTokens: number; outputTokens: number }) => void;
|
||||
}) {
|
||||
const canWrite = opts.approve !== undefined;
|
||||
|
||||
@@ -184,10 +190,15 @@ export function createTaskTool(opts: {
|
||||
|
||||
let steps = 0;
|
||||
let text = '';
|
||||
let usedTokens: { inputTokens: number; outputTokens: number } | undefined;
|
||||
|
||||
try {
|
||||
// `explore` is search, not reasoning, so it runs on the cheaper model when
|
||||
// one is configured. `review` and `worker` keep the parent's: they judge
|
||||
// and they change, both of which want the full model.
|
||||
const model = flavour === 'explore' ? (opts.subagentModel ?? opts.model) : opts.model;
|
||||
const result = streamText({
|
||||
model: opts.model,
|
||||
model,
|
||||
system: PROMPTS[flavour](opts.cwd ?? process.cwd()),
|
||||
messages: [{ role: 'user', content: prompt }],
|
||||
tools: TOOLS[flavour],
|
||||
@@ -230,6 +241,13 @@ export function createTaskTool(opts: {
|
||||
throw part.error instanceof Error ? part.error : new Error(message);
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const usage = await result.usage;
|
||||
usedTokens = { inputTokens: usage.inputTokens ?? 0, outputTokens: usage.outputTokens ?? 0 };
|
||||
} catch {
|
||||
// A run that errored before producing usage has nothing to account for.
|
||||
}
|
||||
} catch (e) {
|
||||
const message = e instanceof Error ? e.message : String(e);
|
||||
report?.({ type: 'error', id, message });
|
||||
@@ -238,6 +256,9 @@ export function createTaskTool(opts: {
|
||||
|
||||
const trimmed = text.trim();
|
||||
report?.({ type: 'end', id, ok: trimmed.length > 0, steps });
|
||||
// Settled after the stream closes; a failed run reports nothing rather than
|
||||
// a half count. The parent prices these against the subagent's own model id.
|
||||
if (usedTokens) opts.onUsage?.({ kind: flavour, ...usedTokens });
|
||||
return trimmed || 'Subagent returned no findings.';
|
||||
},
|
||||
});
|
||||
|
||||
Vendored
+17
@@ -0,0 +1,17 @@
|
||||
/**
|
||||
* Lets TypeScript resolve Bun's raw-text imports of `.md` files.
|
||||
*
|
||||
* `import x from './file.md' with { type: 'text' }` returns the file's contents as
|
||||
* a string; tsc does not know that without a module declaration. Bun handles the
|
||||
* actual loading (and inlines it into a compiled binary); this only teaches the
|
||||
* typechecker the shape.
|
||||
*/
|
||||
declare module '*.md' {
|
||||
const content: string;
|
||||
export default content;
|
||||
}
|
||||
|
||||
declare module '*.png' {
|
||||
const content: ArrayBuffer;
|
||||
export default content;
|
||||
}
|
||||
@@ -0,0 +1,414 @@
|
||||
import { tool } from 'ai';
|
||||
import { stat } from 'node:fs/promises';
|
||||
import { resolve } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { jail, posix, walk } from './ignore';
|
||||
import { git } from './tools-git';
|
||||
|
||||
/**
|
||||
* The second batch of built-in tools, kept out of tools.ts so that file stays
|
||||
* reviewable. Four families:
|
||||
*
|
||||
* edit precise line-level edits that need no full-file rewrite
|
||||
* inspect filesystem navigation and metadata
|
||||
* git ext read-only git queries beyond the core five (argv-spawned, no shell)
|
||||
* code structured reads of source and environment
|
||||
*
|
||||
* Every write goes through `jail`, every read honours .gitignore through `walk`,
|
||||
* and every git call spawns the binary with a fixed argv — the same rules as the
|
||||
* core tools, so the approval and guard model needs nothing new.
|
||||
*/
|
||||
|
||||
const MAX_OUTPUT = 30_000;
|
||||
const cap = (s: string) =>
|
||||
s.length <= MAX_OUTPUT ? s : `${s.slice(0, MAX_OUTPUT)}\n... [truncated ${s.length - MAX_OUTPUT} chars]`;
|
||||
|
||||
const lines = (text: string) => text.split('\n');
|
||||
|
||||
async function readLines(path: string): Promise<{ abs: string; lines: string[] }> {
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
return { abs, lines: lines(await file.text()) };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// edit
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const insertLinesTool = tool({
|
||||
description:
|
||||
'Insert lines at a 1-based position in a file, pushing the rest down. Cheaper and safer than a rewrite for adding a block in the middle.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
line: z.number().int().min(1).describe('Insert before this 1-based line; one past the end appends'),
|
||||
text: z.string().describe('The lines to insert'),
|
||||
}),
|
||||
execute: async ({ path, line, text }) => {
|
||||
const { abs, lines: cur } = await readLines(path);
|
||||
if (line > cur.length + 1) throw new Error(`line ${line} is past the end of ${path} (${cur.length} lines)`);
|
||||
cur.splice(line - 1, 0, ...lines(text));
|
||||
await Bun.write(abs, cur.join('\n'));
|
||||
return `Inserted ${lines(text).length} line(s) at ${path}:${line}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const deleteLinesTool = tool({
|
||||
description: 'Delete an inclusive range of lines from a file. Refuses to delete the whole file; use delete_file for that.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
start: z.number().int().min(1),
|
||||
end: z.number().int().min(1),
|
||||
}),
|
||||
execute: async ({ path, start, end }) => {
|
||||
if (end < start) throw new Error('end must be >= start');
|
||||
const { abs, lines: cur } = await readLines(path);
|
||||
if (end > cur.length) throw new Error(`end ${end} is past the end of ${path} (${cur.length} lines)`);
|
||||
if (start === 1 && end === cur.length) throw new Error('that deletes the whole file; use delete_file instead');
|
||||
cur.splice(start - 1, end - start + 1);
|
||||
await Bun.write(abs, cur.join('\n'));
|
||||
return `Deleted lines ${start}-${end} from ${path}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const replaceLinesTool = tool({
|
||||
description: 'Replace an inclusive range of lines with new text, in one write.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
start: z.number().int().min(1),
|
||||
end: z.number().int().min(1),
|
||||
text: z.string().describe('Replacement content for the range'),
|
||||
}),
|
||||
execute: async ({ path, start, end, text }) => {
|
||||
if (end < start) throw new Error('end must be >= start');
|
||||
const { abs, lines: cur } = await readLines(path);
|
||||
if (end > cur.length) throw new Error(`end ${end} is past the end of ${path} (${cur.length} lines)`);
|
||||
cur.splice(start - 1, end - start + 1, ...lines(text));
|
||||
await Bun.write(abs, cur.join('\n'));
|
||||
return `Replaced lines ${start}-${end} in ${path}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const appendFileTool = tool({
|
||||
description: 'Append text to the end of a file without reading the whole thing into the edit.',
|
||||
inputSchema: z.object({ path: z.string(), text: z.string() }),
|
||||
execute: async ({ path, text }) => {
|
||||
const { abs, lines: cur } = await readLines(path);
|
||||
await Bun.write(abs, `${cur.join('\n').replace(/\n?$/, '\n')}${text.replace(/\n?$/, '')}\n`);
|
||||
return `Appended ${lines(text).length} line(s) to ${path}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const prependFileTool = tool({
|
||||
description: 'Prepend text to the start of a file, e.g. a license header or an import block.',
|
||||
inputSchema: z.object({ path: z.string(), text: z.string() }),
|
||||
execute: async ({ path, text }) => {
|
||||
const { abs, lines: cur } = await readLines(path);
|
||||
await Bun.write(abs, `${text.replace(/\n?$/, '\n')}${cur.join('\n')}`);
|
||||
return `Prepended ${lines(text).length} line(s) to ${path}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const countLinesTool = tool({
|
||||
description: 'Count lines in one file, or per file across a glob. A quick size read before deciding to open something large.',
|
||||
inputSchema: z.object({
|
||||
path: z.string().optional().describe('One file. Omit to use pattern instead'),
|
||||
pattern: z.string().optional().describe('Glob, e.g. "src/**/*.ts", to count many files'),
|
||||
}),
|
||||
execute: async ({ path, pattern }) => {
|
||||
if (!path && !pattern) throw new Error('pass a path or a pattern');
|
||||
const out: string[] = [];
|
||||
const glob = pattern ? new Bun.Glob(pattern) : undefined;
|
||||
for await (const rel of walk({})) {
|
||||
if (path && rel !== posix(path)) continue;
|
||||
if (glob && !glob.match(rel)) continue;
|
||||
const abs = resolve(process.cwd(), rel);
|
||||
try {
|
||||
const n = (await Bun.file(abs).text()).split('\n').length;
|
||||
out.push(`${n}\t${rel}`);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (out.length >= 500) break;
|
||||
}
|
||||
return out.length ? cap(out.join('\n')) : 'No matching text files.';
|
||||
},
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// inspect
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const MAX_TREE = 400;
|
||||
|
||||
export const treeTool = tool({
|
||||
description:
|
||||
'Indented directory tree from a path, honouring .gitignore, with directories first. Faster to scan than list_dir for a broad shape.',
|
||||
inputSchema: z.object({
|
||||
path: z.string().optional().describe('Start directory, default the workspace root'),
|
||||
depth: z.number().int().min(1).max(6).optional().describe('Default 3'),
|
||||
}),
|
||||
execute: async ({ path = '.', depth = 3 }) => {
|
||||
const prefix = path === '.' ? '' : `${posix(path).replace(/\/$/, '')}/`;
|
||||
const rows: { rel: string; depth: number; dir: boolean }[] = [];
|
||||
for await (const rel of walk({})) {
|
||||
if (prefix && !rel.startsWith(prefix)) continue;
|
||||
const rest = prefix ? rel.slice(prefix.length) : rel;
|
||||
const parts = rest.split('/');
|
||||
if (parts.length > depth) continue;
|
||||
for (let d = 1; d <= parts.length; d++) {
|
||||
const ancestor = parts.slice(0, d).join('/');
|
||||
if (!rows.some((r) => r.rel === ancestor)) rows.push({ rel: ancestor, depth: d, dir: d < parts.length });
|
||||
}
|
||||
if (rows.length >= MAX_TREE) break;
|
||||
}
|
||||
rows.sort((a, b) => a.rel.localeCompare(b.rel));
|
||||
const out = rows.map((r) => `${' '.repeat(r.depth - 1)}${r.rel.split('/').at(-1)}${r.dir ? '/' : ''}`);
|
||||
return out.length ? cap((prefix ? `${prefix.replace(/\/$/, '')}/\n` : './\n') + out.join('\n')) : `Nothing under ${path}.`;
|
||||
},
|
||||
});
|
||||
|
||||
export const fileInfoTool = tool({
|
||||
description: 'Metadata for one file: size, line count, modified time, and whether it is text or binary.',
|
||||
inputSchema: z.object({ path: z.string() }),
|
||||
execute: async ({ path }) => {
|
||||
const abs = jail(path);
|
||||
let entry: Awaited<ReturnType<typeof stat>>;
|
||||
try {
|
||||
entry = await stat(abs);
|
||||
} catch {
|
||||
throw new Error(`No such file: ${path}`);
|
||||
}
|
||||
if (entry.isDirectory()) return `${path}: directory`;
|
||||
const bytes = new Uint8Array(await Bun.file(abs).slice(0, 8192).arrayBuffer());
|
||||
const binary = bytes.includes(0);
|
||||
const linesN = binary ? undefined : (await Bun.file(abs).text()).split('\n').length;
|
||||
return `${path}: ${entry.size} bytes${linesN === undefined ? '' : `, ${linesN} lines`}, ${binary ? 'binary' : 'text'}, modified ${entry.mtime.toISOString()}`;
|
||||
},
|
||||
});
|
||||
|
||||
export const findFilesTool = tool({
|
||||
description: 'Find files whose *name* contains a substring (not a glob), e.g. "auth" or ".test.". Honours .gitignore.',
|
||||
inputSchema: z.object({
|
||||
name: z.string().describe('Substring to match against the filename'),
|
||||
limit: z.number().int().min(1).optional().describe('Default 100'),
|
||||
}),
|
||||
execute: async ({ name, limit = 100 }) => {
|
||||
const needle = name.toLowerCase();
|
||||
const hits: string[] = [];
|
||||
for await (const rel of walk({})) {
|
||||
if ((rel.split('/').at(-1) ?? '').toLowerCase().includes(needle)) hits.push(rel);
|
||||
if (hits.length >= limit) break;
|
||||
}
|
||||
return hits.length ? cap(hits.join('\n')) : `No files matching "${name}".`;
|
||||
},
|
||||
});
|
||||
|
||||
export const recentFilesTool = tool({
|
||||
description: 'Files modified most recently, newest first. Orient in a tree you did not write, or find what a tool just touched.',
|
||||
inputSchema: z.object({ limit: z.number().int().min(1).optional().describe('Default 20') }),
|
||||
execute: async ({ limit = 20 }) => {
|
||||
const seen: { rel: string; mtime: number }[] = [];
|
||||
for await (const rel of walk({})) {
|
||||
try {
|
||||
const s = await stat(resolve(process.cwd(), rel));
|
||||
seen.push({ rel, mtime: s.mtimeMs });
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
seen.sort((a, b) => b.mtime - a.mtime);
|
||||
const out = seen.slice(0, limit).map((s) => `${new Date(s.mtime).toISOString().slice(0, 19).replace('T', ' ')} ${s.rel}`);
|
||||
return out.length ? cap(out.join('\n')) : 'No files found.';
|
||||
},
|
||||
});
|
||||
|
||||
export const changedFilesTool = tool({
|
||||
description: 'Files git reports as modified, staged, or untracked — the working-tree delta at a glance, without a full status.',
|
||||
inputSchema: z.object({}),
|
||||
execute: async () => {
|
||||
const result = await git(['status', '--porcelain'], process.cwd());
|
||||
if (!result.ok) throw new Error(result.message);
|
||||
const out = result.stdout
|
||||
.split('\n')
|
||||
.filter(Boolean)
|
||||
.map((l) => `${l.slice(0, 2).trim() || ' '} ${posix(l.slice(3))}`);
|
||||
return out.length ? cap(out.join('\n')) : 'Working tree clean.';
|
||||
},
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// git ext (read-only, argv-spawned)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const gitRun = async (args: string[], empty: string): Promise<string> => {
|
||||
const result = await git(args, process.cwd());
|
||||
if (!result.ok) throw new Error(result.message);
|
||||
return cap(result.stdout.trim() || empty);
|
||||
};
|
||||
|
||||
export const gitLogFileTool = tool({
|
||||
description: 'Commits that touched one file, newest first, with hash, date, and subject.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
limit: z.number().int().min(1).optional().describe('Default 15'),
|
||||
}),
|
||||
execute: async ({ path, limit = 15 }) =>
|
||||
gitRun(['log', `--max-count=${limit}`, '--pretty=format:%h %ad %s', '--date=short', '--', posix(path)], 'No history for that file.'),
|
||||
});
|
||||
|
||||
export const gitDiffCommitsTool = tool({
|
||||
description: 'Diff between two refs (branches, tags, or commits), optionally limited to one path.',
|
||||
inputSchema: z.object({
|
||||
from: z.string().describe('Base ref'),
|
||||
to: z.string().describe('Target ref'),
|
||||
path: z.string().optional().describe('Limit the diff to this file'),
|
||||
}),
|
||||
execute: async ({ from, to, path }) =>
|
||||
gitRun(['diff', `${from}...${to}`, ...(path ? ['--', posix(path)] : [])], `No differences between ${from} and ${to}.`),
|
||||
});
|
||||
|
||||
export const gitShowFileTool = tool({
|
||||
description: 'The contents of a file at a ref, e.g. what auth.ts looked like at HEAD~3 or on main.',
|
||||
inputSchema: z.object({
|
||||
ref: z.string().describe('Branch, tag, or commit'),
|
||||
path: z.string(),
|
||||
}),
|
||||
execute: async ({ ref, path }) => gitRun(['show', `${ref}:${posix(path)}`], `No ${path} at ${ref}.`),
|
||||
});
|
||||
|
||||
export const gitCurrentBranchTool = tool({
|
||||
description: 'The current branch, plus its upstream and ahead/behind count when one is set.',
|
||||
inputSchema: z.object({}),
|
||||
execute: async () => gitRun(['status', '--short', '--branch'], 'no commits yet'),
|
||||
});
|
||||
|
||||
export const gitChangedInRefTool = tool({
|
||||
description: 'Files changed between a ref and the working tree, name only.',
|
||||
inputSchema: z.object({ ref: z.string().describe('Compare the working tree against this ref, e.g. main') }),
|
||||
execute: async ({ ref }) => gitRun(['diff', '--name-only', ref], `No changes against ${ref}.`),
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// code
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const outlineTool = tool({
|
||||
description:
|
||||
'Top-level declarations of a source file — functions, classes, types, exports — as a compact structural map. Read this before opening a large file.',
|
||||
inputSchema: z.object({ path: z.string() }),
|
||||
execute: async ({ path }) => {
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
const decl = /^\s*(export\s+(default\s+)?)?(async\s+)?(function|class|interface|type|enum|const|let|var|def|func|fn|struct|impl|trait|pub)\b/;
|
||||
const out: string[] = [];
|
||||
(await file.text()).split('\n').forEach((l, i) => {
|
||||
if (decl.test(l)) out.push(`${i + 1}: ${l.trim().slice(0, 120)}`);
|
||||
});
|
||||
return out.length ? cap(out.join('\n')) : `No top-level declarations found in ${path}.`;
|
||||
},
|
||||
});
|
||||
|
||||
export const readSymbolTool = tool({
|
||||
description: 'The full body of one top-level definition (function, class, type) from a file, by name.',
|
||||
inputSchema: z.object({
|
||||
path: z.string(),
|
||||
name: z.string().describe('The identifier to extract'),
|
||||
}),
|
||||
execute: async ({ path, name }) => {
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
const src = (await file.text()).split('\n');
|
||||
const start = src.findIndex((l) => new RegExp(`\\b${name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`).test(l) && !/^\s*(\/\/|#)/.test(l));
|
||||
if (start === -1) throw new Error(`No definition of "${name}" found in ${path}`);
|
||||
// Walk forward until the indentation returns to the declaration's level, which
|
||||
// is the end of the block for brace and indentation languages alike.
|
||||
const indent = /^(\s*)/.exec(src[start]!)![1]!.length;
|
||||
let end = start;
|
||||
for (let i = start + 1; i < src.length; i++) {
|
||||
const l = src[i]!;
|
||||
if (l.trim() === '') continue;
|
||||
if (/^(\s*)/.exec(l)![1]!.length <= indent && l.trim() !== '}' && l.trim() !== '};') break;
|
||||
end = i;
|
||||
}
|
||||
return cap(src.slice(start, end + 1).map((l, i) => `${start + i + 1}: ${l}`).join('\n'));
|
||||
},
|
||||
});
|
||||
|
||||
export const envInfoTool = tool({
|
||||
description: 'Platform, shell, runtimes, and package managers present, so commands are written for what is actually installed.',
|
||||
inputSchema: z.object({}),
|
||||
execute: async () => {
|
||||
const probe = async (bin: string, args: string[]) => {
|
||||
try {
|
||||
const proc = Bun.spawn([bin, ...args], { stdout: 'pipe', stderr: 'ignore' });
|
||||
const out = await new Response(proc.stdout).text();
|
||||
await proc.exited;
|
||||
return out.trim().split('\n')[0] ?? 'present';
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
};
|
||||
const rows = [`platform: ${process.platform} ${process.arch}`, `cwd: ${process.cwd()}`];
|
||||
for (const [label, bin, args] of [
|
||||
['node', 'node', ['--version']],
|
||||
['bun', 'bun', ['--version']],
|
||||
['git', 'git', ['--version']],
|
||||
['npm', 'npm', ['--version']],
|
||||
['python', 'python', ['--version']],
|
||||
['rg', 'rg', ['--version']],
|
||||
] as const) {
|
||||
const v = await probe(bin, [...args]);
|
||||
if (v) rows.push(`${label}: ${v}`);
|
||||
}
|
||||
return rows.join('\n');
|
||||
},
|
||||
});
|
||||
|
||||
export const countTokensTool = tool({
|
||||
description: 'Estimate the token cost of a file or a string before sending it to the model (~4 chars per token).',
|
||||
inputSchema: z.object({
|
||||
path: z.string().optional().describe('A file to measure'),
|
||||
text: z.string().optional().describe('Or a string to measure'),
|
||||
}),
|
||||
execute: async ({ path, text }) => {
|
||||
let content = text;
|
||||
if (content === undefined) {
|
||||
if (!path) throw new Error('pass a path or text');
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
content = await file.text();
|
||||
}
|
||||
const chars = content.length;
|
||||
return `${path ?? 'input'}: ${chars} chars, ~${Math.round(chars / 4)} tokens`;
|
||||
},
|
||||
});
|
||||
|
||||
/** The 20, registered by name for the tools map and the `extra` tool set. */
|
||||
export const extraTools = {
|
||||
insert_lines: insertLinesTool,
|
||||
delete_lines: deleteLinesTool,
|
||||
replace_lines: replaceLinesTool,
|
||||
append_file: appendFileTool,
|
||||
prepend_file: prependFileTool,
|
||||
count_lines: countLinesTool,
|
||||
tree: treeTool,
|
||||
file_info: fileInfoTool,
|
||||
find_files: findFilesTool,
|
||||
recent_files: recentFilesTool,
|
||||
changed_files: changedFilesTool,
|
||||
git_log_file: gitLogFileTool,
|
||||
git_diff_commits: gitDiffCommitsTool,
|
||||
git_show_file: gitShowFileTool,
|
||||
git_current_branch: gitCurrentBranchTool,
|
||||
git_changed_in_ref: gitChangedInRefTool,
|
||||
outline: outlineTool,
|
||||
read_symbol: readSymbolTool,
|
||||
env_info: envInfoTool,
|
||||
count_tokens: countTokensTool,
|
||||
};
|
||||
|
||||
export const EXTRA_TOOL_NAMES = Object.keys(extraTools);
|
||||
+115
-1
@@ -3,6 +3,7 @@ import { stat } from 'node:fs/promises';
|
||||
import { join, resolve } from 'node:path';
|
||||
import { z } from 'zod';
|
||||
import { jail, posix, walk } from './ignore';
|
||||
import { EXTRA_TOOL_NAMES, extraTools } from './tools-extra';
|
||||
import { GIT_TOOL_NAMES, gitTools } from './tools-git';
|
||||
import { NET_TOOL_NAMES, netTools } from './tools-net';
|
||||
|
||||
@@ -734,6 +735,114 @@ export const deleteFileTool = tool({
|
||||
},
|
||||
});
|
||||
|
||||
/**
|
||||
* Definition patterns for `find_symbol`, keyed loosely by language.
|
||||
*
|
||||
* Each entry matches the line where a symbol of that shape is *introduced* — a
|
||||
* declaration, not a use — so the agent can jump to a definition instead of
|
||||
* reading whole files to find it. `name` is interpolated escaped, so a symbol
|
||||
* that is a regex metacharacter cannot break the pattern.
|
||||
*/
|
||||
const SYMBOL_PATTERNS: { re: (name: string) => string }[] = [
|
||||
// JS/TS: function foo(, const foo =, class foo, foo(, export ... foo
|
||||
{ re: (n) => `^(export\\s+)?(async\\s+)?(function\\s+${n}|(const|let|var)\\s+${n}\\s*=|class\\s+${n}\\b|interface\\s+${n}\\b|type\\s+${n}\\b|enum\\s+${n}\\b)` },
|
||||
// Python: def foo(, class foo
|
||||
{ re: (n) => `^(async\\s+)?(def\\s+${n}\\s*\\(|class\\s+${n}\\b)` },
|
||||
// Go/Rust/Java-ish: func foo(, fn foo(, struct foo
|
||||
{ re: (n) => `^(pub\\s+)?(func\\s+(\\(.*\\)\\s*)?${n}\\s*\\(|fn\\s+${n}\\s*\\(|struct\\s+${n}\\b|impl\\s+${n}\\b)` },
|
||||
];
|
||||
|
||||
const escapeRe = (s: string) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
|
||||
const MAX_SYMBOL_HITS = 40;
|
||||
|
||||
export const findSymbolTool = tool({
|
||||
description:
|
||||
'Locate where a function, class, type, or constant is *defined*, across JS/TS, Python, Go, and Rust. ' +
|
||||
'Returns path:line hits. Faster and more precise than grep for "where is X declared", because it matches ' +
|
||||
'declarations rather than every use.',
|
||||
inputSchema: z.object({
|
||||
name: z.string().describe('The exact identifier to find, e.g. parseConfig'),
|
||||
include: z.string().optional().describe('Glob limiting which files are searched, default "**/*"'),
|
||||
}),
|
||||
execute: async ({ name, include = '**/*' }) => {
|
||||
const trimmed = name.trim();
|
||||
if (!trimmed) throw new Error('a symbol name is required');
|
||||
const n = escapeRe(trimmed);
|
||||
const glob = new Bun.Glob(include);
|
||||
|
||||
const hits: string[] = [];
|
||||
for await (const rel of walk({})) {
|
||||
if (!glob.match(rel)) continue;
|
||||
const abs = resolve(process.cwd(), rel);
|
||||
let lines: string[];
|
||||
try {
|
||||
if (await isBinary(abs)) continue;
|
||||
lines = (await Bun.file(abs).text()).split('\n');
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const line = lines[i] ?? '';
|
||||
if (line.trimStart().startsWith('//') || line.trimStart().startsWith('#')) continue;
|
||||
if (SYMBOL_PATTERNS.some((p) => new RegExp(p.re(n)).test(line))) {
|
||||
hits.push(`${rel}:${i + 1}: ${line.trim().slice(0, 160)}`);
|
||||
break; // one declaration per file is the useful answer; more is noise.
|
||||
}
|
||||
if (hits.length >= MAX_SYMBOL_HITS) break;
|
||||
}
|
||||
if (hits.length >= MAX_SYMBOL_HITS) break;
|
||||
}
|
||||
return hits.length ? cap(hits.join('\n')) : `No definition of "${trimmed}" found.`;
|
||||
},
|
||||
});
|
||||
|
||||
/**
|
||||
* A dotted-path lookup into a JSON document, so a large manifest, lockfile, or
|
||||
* config can be read one value at a time instead of entering the context whole.
|
||||
* `a.b.0.c` walks objects and arrays; a missing segment reports the path that
|
||||
* resolved, so a wrong key is diagnosable rather than a bare "undefined".
|
||||
*/
|
||||
export const jsonQueryTool = tool({
|
||||
description:
|
||||
'Read one value out of a JSON file by dotted path (e.g. "scripts.build" or "dependencies.react"). ' +
|
||||
'Use it on large manifests and configs instead of reading the whole file into context.',
|
||||
inputSchema: z.object({
|
||||
path: z.string().describe('JSON file, relative to the workspace root'),
|
||||
query: z.string().describe('Dotted path into the document, e.g. "scripts.build". Array indexes are numeric segments.'),
|
||||
}),
|
||||
execute: async ({ path, query }) => {
|
||||
const abs = jail(path);
|
||||
const file = Bun.file(abs);
|
||||
if (!(await file.exists())) throw new Error(`No such file: ${path}`);
|
||||
|
||||
let doc: unknown;
|
||||
try {
|
||||
doc = JSON.parse(await file.text());
|
||||
} catch (e) {
|
||||
throw new Error(`${path} is not valid JSON: ${(e as Error).message}`);
|
||||
}
|
||||
|
||||
let node: unknown = doc;
|
||||
const walked: string[] = [];
|
||||
for (const seg of query.split('.').filter(Boolean)) {
|
||||
if (node === null || typeof node !== 'object') {
|
||||
throw new Error(`"${walked.join('.') || '(root)'}" is ${node === null ? 'null' : typeof node}, not an object; cannot read "${seg}"`);
|
||||
}
|
||||
const record = node as Record<string, unknown>;
|
||||
if (!(seg in record)) {
|
||||
const keys = Object.keys(record).slice(0, 12).join(', ');
|
||||
throw new Error(`no key "${seg}" under "${walked.join('.') || '(root)'}". Keys here: ${keys}${Object.keys(record).length > 12 ? ', …' : ''}`);
|
||||
}
|
||||
node = record[seg];
|
||||
walked.push(seg);
|
||||
}
|
||||
|
||||
const rendered = typeof node === 'string' ? node : JSON.stringify(node, null, 2);
|
||||
return cap(`${query} = ${rendered}`);
|
||||
},
|
||||
});
|
||||
|
||||
export const tools = {
|
||||
read_file: readFileTool,
|
||||
read_many_files: readManyFilesTool,
|
||||
@@ -746,9 +855,12 @@ export const tools = {
|
||||
list_dir: listDirTool,
|
||||
glob: globTool,
|
||||
grep: grepTool,
|
||||
find_symbol: findSymbolTool,
|
||||
json_query: jsonQueryTool,
|
||||
bash: bashTool,
|
||||
...gitTools,
|
||||
...netTools,
|
||||
...extraTools,
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -765,6 +877,8 @@ export const tools = {
|
||||
export const TOOL_SETS = {
|
||||
core: ['read_file', 'write_file', 'edit_file', 'glob', 'grep', 'bash'],
|
||||
'edit-plus': ['multi_edit', 'list_dir', 'read_many_files', 'apply_patch', 'move_file', 'delete_file'],
|
||||
nav: ['find_symbol', 'json_query'],
|
||||
extra: EXTRA_TOOL_NAMES,
|
||||
git: GIT_TOOL_NAMES,
|
||||
net: NET_TOOL_NAMES,
|
||||
} as const satisfies Record<string, readonly string[]>;
|
||||
@@ -776,7 +890,7 @@ export const TOOL_SET_NAMES = Object.keys(TOOL_SETS) as ToolSetName[];
|
||||
export const isToolSetName = (v: string): v is ToolSetName => (TOOL_SET_NAMES as string[]).includes(v);
|
||||
|
||||
/** Sets offered when the config says nothing. `net` is opt-in. */
|
||||
export const DEFAULT_TOOL_SETS: ToolSetName[] = ['core', 'edit-plus', 'git'];
|
||||
export const DEFAULT_TOOL_SETS: ToolSetName[] = ['core', 'edit-plus', 'nav', 'extra', 'git'];
|
||||
|
||||
/** Which set a tool came from, for `/tools`. Session, plugin, and MCP tools have none. */
|
||||
export function toolSetOf(name: string): ToolSetName | undefined {
|
||||
|
||||
+123
-34
@@ -1,6 +1,7 @@
|
||||
import { Box, Static, Text, useApp, useInput, useStdout } from 'ink';
|
||||
import React, { useCallback, useEffect, useRef, useState } from 'react';
|
||||
import { parseCommand, matchCommands } from '../commands';
|
||||
import { expandCommand, type CustomCommand } from '../custom-commands';
|
||||
import { THINKING_LEVELS, VARIANTS } from '../agents';
|
||||
import { completePath, matchPaths, pathToken } from '../complete';
|
||||
import type { Config } from '../config';
|
||||
@@ -19,6 +20,8 @@ import {
|
||||
OutputPanel,
|
||||
QueuePanel,
|
||||
RegistryPanel,
|
||||
Footer,
|
||||
InputStatus,
|
||||
StatusBar,
|
||||
SubagentPanel,
|
||||
ThinkingPanel,
|
||||
@@ -32,6 +35,7 @@ import {
|
||||
import { CommandMenu, InstallConfirm, Picker } from './Pickers';
|
||||
import { contextPanel, costPanel, todosPanel, toolsPanel } from './panel-bodies';
|
||||
import { PromptInput } from './PromptInput';
|
||||
import { accent, glyph } from './theme';
|
||||
import { nextKey, resultSummary, toolDetail, withResult, type Line, type NewLine } from './transcript';
|
||||
|
||||
export { createApprovalBridge, createNoticeBus, createSubagentBus, applySubagentEvent };
|
||||
@@ -40,6 +44,8 @@ export type { ApprovalBridge, NoticeBus, SubagentBus };
|
||||
/** Everything the slash commands need from the outside world. */
|
||||
export type AppHooks = {
|
||||
sessionId: string;
|
||||
/** Session title for the welcome dashboard; absent in tests. */
|
||||
title?: string;
|
||||
config: () => Config;
|
||||
switchModel: (id: string) => string;
|
||||
switchAgent: (name: string) => string;
|
||||
@@ -59,6 +65,8 @@ export type AppHooks = {
|
||||
instructionFiles: () => string[];
|
||||
/** Ignore-aware workspace paths for `@` completion, loaded on first use. */
|
||||
listPaths: () => Promise<string[]>;
|
||||
/** Custom slash commands from markdown files, for the menu and the parser. */
|
||||
customCommands?: () => readonly CustomCommand[];
|
||||
/** Registry index, installed set, and the install/remove actions. */
|
||||
registry: {
|
||||
list: () => Promise<RegistryRow[]>;
|
||||
@@ -85,6 +93,8 @@ export function App({
|
||||
session,
|
||||
bridge,
|
||||
header,
|
||||
headerNode,
|
||||
version,
|
||||
hooks,
|
||||
notices,
|
||||
askBridge,
|
||||
@@ -94,6 +104,10 @@ export function App({
|
||||
session: Session;
|
||||
bridge: ApprovalBridge;
|
||||
header: string;
|
||||
/** Rich welcome screen; when present it replaces the plain `header` string. */
|
||||
headerNode?: React.ReactNode;
|
||||
/** Build version, shown in the welcome dashboard's meta panel. */
|
||||
version?: string;
|
||||
hooks: AppHooks;
|
||||
notices?: NoticeBus;
|
||||
askBridge?: AskBridge;
|
||||
@@ -101,7 +115,19 @@ export function App({
|
||||
needsProvider?: boolean;
|
||||
}) {
|
||||
const { exit } = useApp();
|
||||
const { write } = useStdout();
|
||||
const { write, stdout } = useStdout();
|
||||
// The footer splits hints left from context/cost right, and the input box and
|
||||
// dashboards lay out against the real terminal width, so it is tracked and
|
||||
// kept current on resize rather than read once.
|
||||
const [termWidth, setTermWidth] = useState(stdout?.columns ?? 80);
|
||||
useEffect(() => {
|
||||
if (!stdout) return;
|
||||
const onResize = () => setTermWidth(stdout.columns ?? 80);
|
||||
stdout.on('resize', onResize);
|
||||
return () => {
|
||||
stdout.off('resize', onResize);
|
||||
};
|
||||
}, [stdout]);
|
||||
const [history, setHistory] = useState<Line[]>([]);
|
||||
const [draft, setDraft] = useState('');
|
||||
const [live, setLive] = useState('');
|
||||
@@ -141,7 +167,7 @@ export function App({
|
||||
const modal =
|
||||
pending !== undefined || asking !== undefined || onboarding || installing !== undefined || addingMcp;
|
||||
const anyPicker = modelPicker !== undefined || agentPicker || thinkPicker;
|
||||
const matches = matchCommands(draft);
|
||||
const matches = matchCommands(draft, hooks.customCommands?.() ?? []);
|
||||
const menuOpen = matches.length > 0 && !menuDismissed && !busy && !modal && !anyPicker && !panel;
|
||||
const highlighted = matches[Math.min(menuIndex, matches.length - 1)];
|
||||
|
||||
@@ -439,7 +465,7 @@ export function App({
|
||||
|
||||
// Enter on an open menu runs the highlighted entry, so `/mo` + enter works.
|
||||
const chosen = menuOpen && highlighted ? `/${highlighted.name}` : raw;
|
||||
const action = parseCommand(chosen);
|
||||
const action = parseCommand(chosen, hooks.customCommands?.() ?? []);
|
||||
|
||||
switch (action.type) {
|
||||
case 'none':
|
||||
@@ -495,6 +521,7 @@ export function App({
|
||||
model: hooks.config().model,
|
||||
agent: hooks.agentName(),
|
||||
thinking: hooks.thinkingLevel(),
|
||||
...(hooks.config().subagentModel ? { subagentModel: hooks.config().subagentModel! } : {}),
|
||||
}),
|
||||
);
|
||||
return;
|
||||
@@ -610,7 +637,8 @@ export function App({
|
||||
push({ kind: 'user', text: chosen.trim() });
|
||||
setWorking(true);
|
||||
try {
|
||||
push({ kind: 'info', text: await hooks.summarizeMemory() }); } catch (e) {
|
||||
push({ kind: 'info', text: await hooks.summarizeMemory() });
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
@@ -681,6 +709,37 @@ export function App({
|
||||
setRecall((h) => (h.at(-1) === action.text ? h : [...h, action.text]));
|
||||
await runTurn(action.text);
|
||||
return;
|
||||
case 'custom': {
|
||||
const typed = chosen.trim();
|
||||
push({ kind: 'user', text: typed });
|
||||
setWorking(true);
|
||||
try {
|
||||
// A command may pin an agent; it runs the prompt under that variant
|
||||
// and restores afterwards, so one command does not leak its agent into
|
||||
// the rest of the session.
|
||||
const previous = hooks.agentName();
|
||||
if (action.command.agent && action.command.agent !== previous) {
|
||||
try {
|
||||
hooks.switchAgent(action.command.agent);
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
}
|
||||
const prompt = await expandCommand(action.command, action.args);
|
||||
await runTurn(prompt);
|
||||
if (action.command.agent && action.command.agent !== previous) {
|
||||
try {
|
||||
hooks.switchAgent(previous);
|
||||
} catch {
|
||||
// Restoring the agent is best-effort; the next /agent sets it explicitly.
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
push({ kind: 'error', text: e instanceof Error ? e.message : String(e) });
|
||||
}
|
||||
setWorking(false);
|
||||
return;
|
||||
}
|
||||
}
|
||||
},
|
||||
[exit, highlighted, hooks, menuOpen, push, runTurn, session, setWorking, unconfigured, write],
|
||||
@@ -695,13 +754,24 @@ export function App({
|
||||
<Static items={history}>
|
||||
{(line) => (
|
||||
<Box key={line.key} flexDirection="column" marginBottom={1}>
|
||||
{line.kind === 'user' && <Text color="cyan">{`> ${line.text}`}</Text>}
|
||||
{line.kind === 'assistant' && <Markdown text={line.text} />}
|
||||
{line.kind === 'user' && (
|
||||
<Text color={accent.user} bold>
|
||||
{`${glyph.user} ${line.text}`}
|
||||
</Text>
|
||||
)}
|
||||
{line.kind === 'assistant' && (
|
||||
<Box>
|
||||
<Text color={accent.ok}>{`${glyph.assistant} `}</Text>
|
||||
<Box flexGrow={1} flexDirection="column">
|
||||
<Markdown text={line.text} />
|
||||
</Box>
|
||||
</Box>
|
||||
)}
|
||||
{line.kind === 'tool' && (
|
||||
<Box flexDirection="column">
|
||||
<Box>
|
||||
<Text color={line.ok ? 'magenta' : 'red'}>{line.ok ? '*' : 'x'} </Text>
|
||||
<Text color={line.ok ? 'magenta' : 'red'} bold>
|
||||
<Text color={line.ok ? accent.tool : accent.err}>{line.ok ? glyph.toolOk : glyph.toolErr} </Text>
|
||||
<Text color={line.ok ? accent.tool : accent.err} bold>
|
||||
{line.name}
|
||||
</Text>
|
||||
{line.detail[0] !== undefined && <Text dimColor>{` ${line.detail[0]}`}</Text>}
|
||||
@@ -712,23 +782,26 @@ export function App({
|
||||
</Text>
|
||||
))}
|
||||
{line.result !== undefined && line.result.length > 0 && (
|
||||
<Text color={line.ok ? undefined : 'red'} dimColor={line.ok}>
|
||||
{` ${line.ok ? '->' : 'x'} ${line.result}`}
|
||||
<Text color={line.ok ? undefined : accent.err} dimColor={line.ok}>
|
||||
{` ${line.ok ? glyph.result : glyph.err} ${line.result}`}
|
||||
</Text>
|
||||
)}
|
||||
</Box>
|
||||
)}
|
||||
{line.kind === 'info' && <Text dimColor>{line.text}</Text>}
|
||||
{line.kind === 'error' && <Text color="red">error: {line.text}</Text>}
|
||||
{line.kind === 'info' && <Text dimColor>{`${glyph.info} ${line.text}`}</Text>}
|
||||
{line.kind === 'error' && <Text color={accent.err}>{`${glyph.err} ${line.text}`}</Text>}
|
||||
</Box>
|
||||
)}
|
||||
</Static>
|
||||
|
||||
{history.length === 0 && (
|
||||
<Box marginBottom={1}>
|
||||
<Text dimColor>{header}</Text>
|
||||
</Box>
|
||||
)}
|
||||
{history.length === 0 &&
|
||||
(headerNode !== undefined ? (
|
||||
headerNode
|
||||
) : (
|
||||
<Box flexDirection="column" marginBottom={1}>
|
||||
<Text dimColor>{header}</Text>
|
||||
</Box>
|
||||
))}
|
||||
|
||||
{agents.length > 0 && <SubagentPanel agents={agents} />}
|
||||
|
||||
@@ -869,19 +942,37 @@ export function App({
|
||||
)}
|
||||
|
||||
{!modal && !anyPicker && (
|
||||
<Box flexDirection="column">
|
||||
<Box flexDirection="column" marginTop={1} width={termWidth}>
|
||||
<QueuePanel prompts={queue} />
|
||||
<Box>
|
||||
<Text color="cyan">{'> '}</Text>
|
||||
<PromptInput
|
||||
key={inputGeneration}
|
||||
value={draft}
|
||||
initialCursor={inputCursor}
|
||||
onChange={onDraftChange}
|
||||
onSubmit={submit}
|
||||
history={recall}
|
||||
onKey={handleInputKey}
|
||||
placeholder={busy ? 'type to queue for the next turn...' : 'ask shiro-neko... (/ commands, @ files)'}
|
||||
<Box
|
||||
flexDirection="column"
|
||||
width={termWidth}
|
||||
borderStyle="round"
|
||||
borderColor={accent.mute}
|
||||
borderLeftColor={busy ? accent.warn : accent.user}
|
||||
paddingLeft={1}
|
||||
paddingRight={1}
|
||||
>
|
||||
<Box>
|
||||
<Text color={accent.user} bold>
|
||||
{`${glyph.user} `}
|
||||
</Text>
|
||||
<PromptInput
|
||||
key={inputGeneration}
|
||||
value={draft}
|
||||
initialCursor={inputCursor}
|
||||
onChange={onDraftChange}
|
||||
onSubmit={submit}
|
||||
history={recall}
|
||||
onKey={handleInputKey}
|
||||
placeholder={busy ? 'type to queue for the next turn…' : 'ask shiro-neko… (/ commands, @ files)'}
|
||||
/>
|
||||
</Box>
|
||||
<InputStatus
|
||||
agent={hooks.agentName()}
|
||||
model={hooks.config().model}
|
||||
right={`${hooks.thinkingLevel()} ${glyph.info} ${session.activeTools().length} tools`}
|
||||
width={termWidth - 6}
|
||||
/>
|
||||
</Box>
|
||||
{fileOpen ? (
|
||||
@@ -894,17 +985,15 @@ export function App({
|
||||
) : (
|
||||
menuOpen && <CommandMenu matches={matches} index={Math.min(menuIndex, matches.length - 1)} />
|
||||
)}
|
||||
<StatusBar
|
||||
model={hooks.config().model}
|
||||
agent={hooks.agentName()}
|
||||
thinking={hooks.thinkingLevel()}
|
||||
<Footer
|
||||
busy={busy}
|
||||
width={termWidth}
|
||||
contextTokens={session.estimatedTokens()}
|
||||
contextLimit={session.compactThreshold()}
|
||||
cost={(() => {
|
||||
const spend = costOf(hooks.config().model, session.inputTokens, session.outputTokens);
|
||||
return spend === undefined ? 'unpriced' : formatUsd(spend);
|
||||
})()}
|
||||
toolCount={session.activeTools().length}
|
||||
/>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
+20
-7
@@ -2,6 +2,7 @@ import { Box, Text, useInput } from 'ink';
|
||||
import React from 'react';
|
||||
import type { ApprovalDecision, ApprovalRequest } from '../session';
|
||||
import { Diff } from './Diff';
|
||||
import { accent, glyph } from './theme';
|
||||
import { toolDetail } from './transcript';
|
||||
|
||||
export type Pending = { req: ApprovalRequest; resolve: (d: ApprovalDecision) => void };
|
||||
@@ -85,9 +86,9 @@ export function Approval({ pending }: { pending: Pending }) {
|
||||
const grant = req.suggestedPattern === '*' ? req.toolName : `${req.toolName} ${req.suggestedPattern}`;
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="yellow" paddingX={1}>
|
||||
<Text color="yellow" bold>
|
||||
{reason(req)}
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.warn} paddingX={1}>
|
||||
<Text color={accent.warn} bold>
|
||||
{`${glyph.warn} ${reason(req)}`}
|
||||
</Text>
|
||||
{req.repeated && <Text dimColor>allowed by the rules, but this is the third identical call this turn</Text>}
|
||||
{req.subagent && !req.repeated && (
|
||||
@@ -97,10 +98,22 @@ export function Approval({ pending }: { pending: Pending }) {
|
||||
<Text dimColor>{`matched ${req.toolName}: "${req.matchedPattern}"`}</Text>
|
||||
)}
|
||||
<ApprovalDetail name={req.toolName} input={req.input} />
|
||||
<Text>
|
||||
<Text color="green">y</Text> allow once | <Text color="green">a</Text> always allow {grant} |{' '}
|
||||
<Text color="red">n</Text> deny
|
||||
</Text>
|
||||
<Box marginTop={1}>
|
||||
<Text>
|
||||
<Text color={accent.ok} bold>
|
||||
y
|
||||
</Text>
|
||||
<Text dimColor>{` allow once ${glyph.sep} `}</Text>
|
||||
<Text color={accent.ok} bold>
|
||||
a
|
||||
</Text>
|
||||
<Text dimColor>{` always allow ${grant} ${glyph.sep} `}</Text>
|
||||
<Text color={accent.err} bold>
|
||||
n
|
||||
</Text>
|
||||
<Text dimColor> deny</Text>
|
||||
</Text>
|
||||
</Box>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
+9
-4
@@ -4,6 +4,7 @@ import React, { useState } from 'react';
|
||||
import type { AskRequest } from '../ask';
|
||||
import { InlineMarkdown } from './Markdown';
|
||||
import { PromptInput } from './PromptInput';
|
||||
import { accent, glyph } from './theme';
|
||||
|
||||
export type AskPending = { req: AskRequest; resolve: (answers: string[] | undefined) => void };
|
||||
|
||||
@@ -70,8 +71,8 @@ export function AskPanel({ pending }: { pending: AskPending }) {
|
||||
const detailOf = (label: string) => (options ?? []).find((o) => o.label === label)?.detail;
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="double" borderColor="yellow" paddingX={1}>
|
||||
<Text color="yellow" bold>
|
||||
<Box flexDirection="column" borderStyle="double" borderColor={accent.warn} paddingX={1}>
|
||||
<Text color={accent.warn} bold>
|
||||
shiro is asking
|
||||
</Text>
|
||||
<Box marginBottom={1}>
|
||||
@@ -80,7 +81,9 @@ export function AskPanel({ pending }: { pending: AskPending }) {
|
||||
|
||||
{typing ? (
|
||||
<Box>
|
||||
<Text color="yellow">{'> '}</Text>
|
||||
<Text color={accent.warn} bold>
|
||||
{`${glyph.user} `}
|
||||
</Text>
|
||||
<PromptInput
|
||||
value={draft}
|
||||
onChange={setDraft}
|
||||
@@ -107,7 +110,9 @@ export function AskPanel({ pending }: { pending: AskPending }) {
|
||||
/>
|
||||
{draft.length > 0 && <Text dimColor>{draft}</Text>}
|
||||
<Text dimColor>
|
||||
{multiple ? 'space/enter toggles, pick submit when done' : 'enter to choose'} | esc to skip
|
||||
{multiple
|
||||
? `space/enter toggles, pick submit when done ${glyph.sep} esc to skip`
|
||||
: `enter to choose ${glyph.sep} esc to skip`}
|
||||
</Text>
|
||||
</Box>
|
||||
)}
|
||||
|
||||
+6
-5
@@ -1,5 +1,6 @@
|
||||
import { Box, Text } from 'ink';
|
||||
import React from 'react';
|
||||
import { accent, glyph } from './theme';
|
||||
|
||||
export type DiffLine = { kind: 'context' | 'add' | 'remove'; text: string; at: number };
|
||||
|
||||
@@ -97,21 +98,21 @@ export function Diff({ before, after, path }: { before: string; after: string; p
|
||||
<Box flexDirection="column">
|
||||
{path && (
|
||||
<Text>
|
||||
<Text bold>{path}</Text> <Text color="green">+{added}</Text> <Text color="red">-{removed}</Text>
|
||||
<Text bold>{path}</Text> <Text color={accent.ok}>+{added}</Text> <Text color={accent.err}>-{removed}</Text>
|
||||
</Text>
|
||||
)}
|
||||
{shown.map((line, i) =>
|
||||
line.kind === 'gap' ? (
|
||||
<Text key={i} dimColor>{` ... lines ${line.from}-${line.to} unchanged`}</Text>
|
||||
<Text key={i} dimColor>{` ${glyph.fold} lines ${line.from}-${line.to} unchanged`}</Text>
|
||||
) : (
|
||||
<Text
|
||||
key={i}
|
||||
color={line.kind === 'add' ? 'green' : line.kind === 'remove' ? 'red' : undefined}
|
||||
color={line.kind === 'add' ? accent.ok : line.kind === 'remove' ? accent.err : undefined}
|
||||
dimColor={line.kind === 'context'}
|
||||
>{` ${String(line.at).padStart(3)} ${line.kind === 'add' ? '+' : line.kind === 'remove' ? '-' : ' '} ${line.text}`}</Text>
|
||||
>{` ${String(line.at).padStart(3)} ${line.kind === 'add' ? '+' : line.kind === 'remove' ? '-' : glyph.sep} ${line.text}`}</Text>
|
||||
),
|
||||
)}
|
||||
{hidden > 0 && <Text dimColor>{` ... ${hidden} more diff lines`}</Text>}
|
||||
{hidden > 0 && <Text dimColor>{` ${glyph.fold} ${hidden} more diff lines`}</Text>}
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
import { Box, Text } from 'ink';
|
||||
import React from 'react';
|
||||
import { SidePanel } from './Panels';
|
||||
import { accent, glyph } from './theme';
|
||||
|
||||
export type HeaderFact = {
|
||||
label: string;
|
||||
value: string;
|
||||
/** Defaults to the quiet metadata colour; set for anything the user must not miss. */
|
||||
tone?: 'warn' | 'err' | 'ok' | 'info';
|
||||
};
|
||||
|
||||
const TONE_COLOR: Record<NonNullable<HeaderFact['tone']>, string> = {
|
||||
warn: accent.warn,
|
||||
err: accent.err,
|
||||
ok: accent.ok,
|
||||
info: accent.info,
|
||||
};
|
||||
|
||||
/**
|
||||
* The welcome dashboard, shown once before the first turn, in OpenCode's grammar.
|
||||
*
|
||||
* The session is introduced by a banner (what this conversation is), then the
|
||||
* environment is grouped into a labelled panel so the eye scans one label rather
|
||||
* than a wall of text. A final meta bar carries cwd and version, the two facts a
|
||||
* bug report needs. Facts that demand attention — a missing provider, a failed
|
||||
* plugin, `--yolo` — are lifted out of the quiet layer with colour, because a
|
||||
* warning rendered dim is a warning nobody reads.
|
||||
*/
|
||||
export function Header({
|
||||
version,
|
||||
provider,
|
||||
model,
|
||||
sessionId,
|
||||
cwd,
|
||||
title,
|
||||
facts,
|
||||
}: {
|
||||
version: string;
|
||||
provider?: string;
|
||||
model?: string;
|
||||
sessionId: string;
|
||||
cwd: string;
|
||||
title?: string;
|
||||
facts: readonly HeaderFact[];
|
||||
}) {
|
||||
return (
|
||||
<Box flexDirection="column" marginBottom={1}>
|
||||
<SidePanel label={title ?? 'new session'} tone={accent.user}>
|
||||
<Box>
|
||||
{model ? (
|
||||
<>
|
||||
<Text color={accent.ok}>{`${glyph.ok} `}</Text>
|
||||
<Text bold>
|
||||
{provider ? `${provider}/` : ''}
|
||||
{model}
|
||||
</Text>
|
||||
<Text dimColor>{` ${glyph.sep} session ${sessionId}`}</Text>
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<Text color={accent.warn}>{`${glyph.warn} `}</Text>
|
||||
<Text color={accent.warn}>no provider configured</Text>
|
||||
<Text dimColor>{` ${glyph.sep} run /provider to begin`}</Text>
|
||||
</>
|
||||
)}
|
||||
</Box>
|
||||
</SidePanel>
|
||||
|
||||
{facts.length > 0 && (
|
||||
<SidePanel label="environment" tone={accent.ok}>
|
||||
{facts.map((f, i) => (
|
||||
<Box key={i}>
|
||||
<Text dimColor>{f.label.padEnd(14)}</Text>
|
||||
<Text color={f.tone ? TONE_COLOR[f.tone] : undefined} dimColor={f.tone === undefined}>
|
||||
{f.value}
|
||||
</Text>
|
||||
</Box>
|
||||
))}
|
||||
</SidePanel>
|
||||
)}
|
||||
|
||||
<Box paddingX={1}>
|
||||
<Text dimColor>{cwd}</Text>
|
||||
<Text dimColor>{` ${glyph.sep} `}</Text>
|
||||
<Text dimColor>{`shiro-neko ${version}`}</Text>
|
||||
<Text dimColor>{` ${glyph.sep} `}</Text>
|
||||
<Text color={accent.user}>/help for commands</Text>
|
||||
</Box>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
+211
-62
@@ -4,28 +4,30 @@ import React from 'react';
|
||||
import { TODO_MARK, type Todo } from '../notebook';
|
||||
import type { SubagentKind } from '../subagent';
|
||||
import { InlineMarkdown } from './Markdown';
|
||||
import { accent, glyph, meter } from './theme';
|
||||
|
||||
const STATUS_COLOR: Record<Todo['status'], string | undefined> = {
|
||||
pending: undefined,
|
||||
in_progress: 'cyan',
|
||||
done: 'green',
|
||||
blocked: 'red',
|
||||
in_progress: accent.user,
|
||||
done: accent.ok,
|
||||
blocked: accent.err,
|
||||
};
|
||||
|
||||
/** Task list with a progress bar, shown above the input while a list exists. */
|
||||
export function TodoPanel({ todos, width = 40 }: { todos: Todo[]; width?: number }) {
|
||||
/** Task list with a progress meter, shown above the input while a list exists. */
|
||||
export function TodoPanel({ todos, width = 24 }: { todos: Todo[]; width?: number }) {
|
||||
const done = todos.filter((t) => t.status === 'done').length;
|
||||
const blocked = todos.filter((t) => t.status === 'blocked').length;
|
||||
const filled = todos.length === 0 ? 0 : Math.round((done / todos.length) * width);
|
||||
const ratio = todos.length === 0 ? 0 : done / todos.length;
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="gray" paddingX={1} marginBottom={1}>
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.mute} paddingX={1} marginBottom={1}>
|
||||
<Box>
|
||||
<Text bold>tasks </Text>
|
||||
<Text color="green">{'#'.repeat(filled)}</Text>
|
||||
<Text dimColor>{'.'.repeat(Math.max(0, width - filled))}</Text>
|
||||
<Text bold color={accent.user}>
|
||||
tasks{' '}
|
||||
</Text>
|
||||
<Text color={accent.ok}>{meter(ratio, width)}</Text>
|
||||
<Text dimColor>{` ${done}/${todos.length}`}</Text>
|
||||
{blocked > 0 && <Text color="red">{` ${blocked} blocked`}</Text>}
|
||||
{blocked > 0 && <Text color={accent.err}>{` ${glyph.blocked} ${blocked} blocked`}</Text>}
|
||||
</Box>
|
||||
{todos.map((t, i) => (
|
||||
<Box key={i}>
|
||||
@@ -60,7 +62,7 @@ export type SubagentView = {
|
||||
const KIND_LABEL: Record<SubagentKind, string> = { explore: 'explore', review: 'review', worker: 'worker' };
|
||||
|
||||
/** A worker can write, so its panel entry has to be distinguishable at a glance. */
|
||||
const KIND_COLOUR: Record<SubagentKind, string> = { explore: 'cyan', review: 'blue', worker: 'yellow' };
|
||||
const KIND_COLOUR: Record<SubagentKind, string> = { explore: accent.user, review: accent.info, worker: accent.warn };
|
||||
|
||||
/**
|
||||
* Live view of delegated work.
|
||||
@@ -74,33 +76,35 @@ export function SubagentPanel({ agents, steps = 4 }: { agents: SubagentView[]; s
|
||||
if (agents.length === 0) return null;
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="magenta" paddingX={1} marginBottom={1}>
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.tool} paddingX={1} marginBottom={1}>
|
||||
{agents.map((a) => (
|
||||
<Box key={a.id} flexDirection="column">
|
||||
<Box>
|
||||
{a.status === 'running' ? (
|
||||
<Text color="magenta">
|
||||
<Text color={accent.tool}>
|
||||
<Spinner type="dots" />
|
||||
</Text>
|
||||
) : (
|
||||
<Text color={a.status === 'done' ? 'green' : 'red'}>{a.status === 'done' ? '*' : 'x'}</Text>
|
||||
<Text color={a.status === 'done' ? accent.ok : accent.err}>
|
||||
{a.status === 'done' ? glyph.ok : glyph.err}
|
||||
</Text>
|
||||
)}
|
||||
<Text bold color={KIND_COLOUR[a.kind]}>{` ${KIND_LABEL[a.kind]}`}</Text>
|
||||
{a.kind === 'worker' && <Text color="yellow">{' (writes)'}</Text>}
|
||||
<Text>{`: ${a.description}`}</Text>
|
||||
{a.kind === 'worker' && <Text color={accent.warn}>{' (writes)'}</Text>}
|
||||
<Text>{` ${glyph.sep} ${a.description}`}</Text>
|
||||
<Text dimColor>{` ${a.steps.length} step${a.steps.length === 1 ? '' : 's'}`}</Text>
|
||||
</Box>
|
||||
{a.steps.slice(-steps).map((s, i) => (
|
||||
<Box key={i} flexDirection="column">
|
||||
<Text dimColor>{` ${s.tool}(${s.summary.slice(0, 58)})`}</Text>
|
||||
{s.outcome !== undefined && (
|
||||
<Text color={s.ok === false ? 'red' : undefined} dimColor={s.ok !== false}>
|
||||
{` ${s.ok === false ? 'x' : '->'} ${s.outcome}`}
|
||||
<Text color={s.ok === false ? accent.err : undefined} dimColor={s.ok !== false}>
|
||||
{` ${s.ok === false ? glyph.err : glyph.result} ${s.outcome}`}
|
||||
</Text>
|
||||
)}
|
||||
</Box>
|
||||
))}
|
||||
{a.error && <Text color="red">{` ${a.error}`}</Text>}
|
||||
{a.error && <Text color={accent.err}>{` ${a.error}`}</Text>}
|
||||
</Box>
|
||||
))}
|
||||
</Box>
|
||||
@@ -123,10 +127,10 @@ const SLOW_AFTER = 10;
|
||||
*/
|
||||
export function Working({ seconds }: { seconds: number }) {
|
||||
return (
|
||||
<Text color="yellow">
|
||||
<Text color={accent.warn}>
|
||||
<Spinner type="dots" />{' '}
|
||||
<Text dimColor>
|
||||
{seconds >= SLOW_AFTER ? `working ${elapsed(seconds)}... esc to interrupt` : 'working... esc to interrupt'}
|
||||
{seconds >= SLOW_AFTER ? `working ${elapsed(seconds)} ${glyph.sep} esc to interrupt` : `working ${glyph.sep} esc to interrupt`}
|
||||
</Text>
|
||||
</Text>
|
||||
);
|
||||
@@ -142,7 +146,7 @@ export function OutputPanel({ text, lines = 8 }: { text: string; lines?: number
|
||||
.slice(-lines)
|
||||
.map((l, i) => (
|
||||
<Text key={i} dimColor>
|
||||
{` | ${l}`}
|
||||
{` ${glyph.sep} ${l}`}
|
||||
</Text>
|
||||
))}
|
||||
</Box>
|
||||
@@ -161,10 +165,10 @@ export function ActiveTool({ name, detail = [] }: { name: string; detail?: reado
|
||||
return (
|
||||
<Box flexDirection="column">
|
||||
<Box>
|
||||
<Text color="magenta">
|
||||
<Text color={accent.tool}>
|
||||
<Spinner type="dots" />
|
||||
</Text>
|
||||
<Text bold>{` ${name}`}</Text>
|
||||
<Text bold color={accent.tool}>{` ${name}`}</Text>
|
||||
{detail[0] !== undefined && <Text dimColor>{` ${detail[0]}`}</Text>}
|
||||
</Box>
|
||||
{detail.slice(1, 6).map((d, i) => (
|
||||
@@ -172,7 +176,7 @@ export function ActiveTool({ name, detail = [] }: { name: string; detail?: reado
|
||||
{` ${d}`}
|
||||
</Text>
|
||||
))}
|
||||
{detail.length > 6 && <Text dimColor>{` ... ${detail.length - 6} more`}</Text>}
|
||||
{detail.length > 6 && <Text dimColor>{` ${glyph.fold} ${detail.length - 6} more`}</Text>}
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
@@ -189,12 +193,12 @@ export function ThinkingPanel({ text, expanded, lines = 8 }: { text: string; exp
|
||||
const tokens = Math.round(text.length / 4);
|
||||
|
||||
if (!expanded) {
|
||||
return <Text dimColor>{`thinking... ~${tokens} tokens ctrl-r to expand`}</Text>;
|
||||
return <Text dimColor>{`thinking ${glyph.fold} ~${tokens} tokens ${glyph.sep} ctrl-r to expand`}</Text>;
|
||||
}
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" marginBottom={1}>
|
||||
<Text dimColor>{`thinking ~${tokens} tokens ctrl-r to collapse`}</Text>
|
||||
<Text dimColor>{`thinking ~${tokens} tokens ${glyph.sep} ctrl-r to collapse`}</Text>
|
||||
{text
|
||||
.split('\n')
|
||||
.slice(-lines)
|
||||
@@ -212,10 +216,10 @@ export function QueuePanel({ prompts }: { prompts: readonly string[] }) {
|
||||
if (prompts.length === 0) return null;
|
||||
return (
|
||||
<Box flexDirection="column">
|
||||
<Text color="cyan">{`queued: ${prompts.length}`}</Text>
|
||||
<Text color={accent.user}>{`${glyph.queued} queued: ${prompts.length}`}</Text>
|
||||
{prompts.map((p, i) => (
|
||||
<Text key={i} dimColor>
|
||||
{` ${i + 1}. ${p.length > 70 ? `${p.slice(0, 70)}...` : p}`}
|
||||
{` ${i + 1}. ${p.length > 70 ? `${p.slice(0, 70)}…` : p}`}
|
||||
</Text>
|
||||
))}
|
||||
</Box>
|
||||
@@ -237,7 +241,7 @@ export function FileMenu({
|
||||
if (loading) {
|
||||
return (
|
||||
<Box marginTop={1}>
|
||||
<Text dimColor>indexing files...</Text>
|
||||
<Text dimColor>indexing files…</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
@@ -253,17 +257,25 @@ export function FileMenu({
|
||||
return (
|
||||
<Box flexDirection="column" marginTop={1}>
|
||||
{paths.map((p, i) => (
|
||||
<Text key={p} color={i === index ? 'cyan' : undefined} dimColor={i !== index}>
|
||||
{i === index ? '> ' : ' '}
|
||||
<Text key={p} color={i === index ? accent.user : undefined} dimColor={i !== index}>
|
||||
{i === index ? `${glyph.user} ` : ' '}
|
||||
{p}
|
||||
</Text>
|
||||
))}
|
||||
<Text dimColor>up/down move | tab or enter insert | esc dismiss</Text>
|
||||
<Text dimColor>{`↑↓ move ${glyph.sep} tab/enter insert ${glyph.sep} esc dismiss`}</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
/** Status line under the transcript: model, agent, thinking, context, spend. */
|
||||
/**
|
||||
* The persistent status line under the input.
|
||||
*
|
||||
* Three glance groups separated by quiet pipes: what is running (model, agent,
|
||||
* thinking), how full the context is (a meter plus a percentage), and what it has
|
||||
* cost. The context meter is the one piece of live information that changes
|
||||
* mid-session, so it is the only element with a bar; everything else stays text so
|
||||
* the bar is the thing the eye lands on.
|
||||
*/
|
||||
export function StatusBar({
|
||||
model,
|
||||
agent,
|
||||
@@ -286,18 +298,144 @@ export function StatusBar({
|
||||
// Amber from two thirds, red once compaction is imminent: the point is to warn
|
||||
// before a turn silently loses its history, not after. Past 90 the colour is
|
||||
// backed by words, because a reader watching the transcript is not watching this.
|
||||
const contextColor = pct === undefined ? undefined : pct >= 90 ? 'red' : pct >= 66 ? 'yellow' : undefined;
|
||||
const contextColor = pct === undefined ? undefined : pct >= 90 ? accent.err : pct >= 66 ? accent.warn : undefined;
|
||||
const sep = <Text dimColor>{` ${glyph.sep} `}</Text>;
|
||||
|
||||
return (
|
||||
<Box>
|
||||
<Text dimColor>{`${model} `}</Text>
|
||||
<Text color="cyan">{agent}</Text>
|
||||
<Text dimColor>{`/${thinking} ${toolCount} tools `}</Text>
|
||||
<Text dimColor>{model}</Text>
|
||||
{sep}
|
||||
<Text color={accent.user}>{agent}</Text>
|
||||
<Text dimColor>{`/${thinking}`}</Text>
|
||||
{sep}
|
||||
<Text dimColor>{`${toolCount} tools`}</Text>
|
||||
{sep}
|
||||
<Text color={contextColor} dimColor={contextColor === undefined}>
|
||||
{pct === undefined ? `~${contextTokens} ctx` : `${pct}% ctx`}
|
||||
{pct === undefined ? `~${contextTokens} ctx` : `${meter(pct / 100, 8)} ${pct}%`}
|
||||
</Text>
|
||||
{pct !== undefined && pct >= 90 && <Text color="red">{' compacting soon'}</Text>}
|
||||
<Text dimColor>{` ${cost}`}</Text>
|
||||
{pct !== undefined && pct >= 90 && <Text color={accent.err}>{' compacting soon'}</Text>}
|
||||
{sep}
|
||||
<Text dimColor>{cost}</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* The one-line identity row inside the input box, OpenCode-style.
|
||||
*
|
||||
* Agent and model are the two things a prompt is answered *by*, so they sit
|
||||
* directly under the cursor rather than in a footer the eye has to travel to.
|
||||
* `right` carries the quieter facts (tool count, thinking level) pushed to the
|
||||
* far edge of the box, so the row reads as two anchored groups.
|
||||
*/
|
||||
export function InputStatus({
|
||||
agent,
|
||||
model,
|
||||
right,
|
||||
width,
|
||||
}: {
|
||||
agent: string;
|
||||
model: string;
|
||||
right?: string;
|
||||
width: number;
|
||||
}) {
|
||||
const left = ` ${agent} · ${model}`;
|
||||
const rightText = right ? `${right} ` : '';
|
||||
const gap = Math.max(1, width - left.length - rightText.length);
|
||||
return (
|
||||
<Box>
|
||||
<Text color={accent.user} bold>
|
||||
{agent}
|
||||
</Text>
|
||||
<Text dimColor>{` ${glyph.info} ${model}`}</Text>
|
||||
<Text dimColor>{' '.repeat(gap)}</Text>
|
||||
{right !== undefined && <Text dimColor>{right}</Text>}
|
||||
<Text> </Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* The full-width footer beneath the input, OpenCode-style.
|
||||
*
|
||||
* Key hints live on the left (what you can press), the live numbers on the right
|
||||
* (context and cost). Spreading them to opposite edges means neither has to
|
||||
* compete for the same glance, and the meter stays the most saturated thing on
|
||||
* the line so the eye finds it first when a turn runs long.
|
||||
*/
|
||||
export function Footer({
|
||||
busy,
|
||||
contextTokens,
|
||||
contextLimit,
|
||||
cost,
|
||||
width,
|
||||
}: {
|
||||
busy: boolean;
|
||||
contextTokens: number;
|
||||
contextLimit?: number;
|
||||
cost: string;
|
||||
width: number;
|
||||
}) {
|
||||
const pct = contextLimit ? Math.min(100, Math.round((contextTokens / contextLimit) * 100)) : undefined;
|
||||
const contextColor = pct === undefined ? undefined : pct >= 90 ? accent.err : pct >= 66 ? accent.warn : undefined;
|
||||
const hint = busy ? `${glyph.bullet} esc interrupt` : `/ commands ${glyph.sep} @ files ${glyph.sep} ↑ history`;
|
||||
const right =
|
||||
pct === undefined ? `~${contextTokens} ctx ${glyph.sep} ${cost}` : `${meter(pct / 100, 8)} ${pct}% ${glyph.sep} ${cost}`;
|
||||
const gap = Math.max(1, width - hint.length - right.length - 1);
|
||||
|
||||
return (
|
||||
<Box>
|
||||
<Text dimColor>{hint}</Text>
|
||||
<Text>{' '.repeat(gap)}</Text>
|
||||
<Text color={contextColor} dimColor={contextColor === undefined}>
|
||||
{right}
|
||||
</Text>
|
||||
{pct !== undefined && pct >= 90 && <Text color={accent.err}>{' compacting soon'}</Text>}
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* A bordered panel with a coloured label header, OpenCode's sidebar grammar.
|
||||
*
|
||||
* Used for the welcome dashboard's grouped facts (capabilities, project, meta).
|
||||
* The label carries the accent; the rows stay quiet, so a stack of these reads
|
||||
* as labelled groups rather than as more transcript.
|
||||
*/
|
||||
export function SidePanel({
|
||||
label,
|
||||
children,
|
||||
width,
|
||||
tone = accent.ok,
|
||||
}: {
|
||||
label: string;
|
||||
children: React.ReactNode;
|
||||
width?: number;
|
||||
tone?: string;
|
||||
}) {
|
||||
return (
|
||||
<Box
|
||||
flexDirection="column"
|
||||
borderStyle="round"
|
||||
borderColor={accent.mute}
|
||||
paddingX={1}
|
||||
marginBottom={1}
|
||||
{...(width !== undefined ? { width } : {})}
|
||||
>
|
||||
<Text color={tone} bold>
|
||||
{label}
|
||||
</Text>
|
||||
{children}
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
|
||||
/** One `name value` row inside a SidePanel, name padded so a stack aligns. */
|
||||
export function SideRow({ name, value, nameWidth = 12, dim = true }: { name: string; value: string; nameWidth?: number; dim?: boolean }) {
|
||||
return (
|
||||
<Box>
|
||||
<Text dimColor={dim}>{name.padEnd(nameWidth)}</Text>
|
||||
<Text>{value}</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
@@ -307,17 +445,19 @@ export type PanelLine = { label: string; value: string };
|
||||
/** Bordered popup for a command's output, e.g. /skills or /cost. */
|
||||
export function InfoPanel({ title, hint, lines }: { title: string; hint?: string; lines: PanelLine[] | string }) {
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1} marginBottom={1}>
|
||||
<Text color="cyan" bold>
|
||||
{title}
|
||||
</Text>
|
||||
{hint && <Text dimColor>{hint}</Text>}
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1} marginBottom={1}>
|
||||
<Box>
|
||||
<Text color={accent.user} bold>
|
||||
{title}
|
||||
</Text>
|
||||
{hint && <Text dimColor>{` ${hint}`}</Text>}
|
||||
</Box>
|
||||
{typeof lines === 'string' ? (
|
||||
<InlineMarkdown text={lines} />
|
||||
) : (
|
||||
lines.map((l, i) => (
|
||||
<Box key={i}>
|
||||
<Text color="gray">{l.label.padEnd(14)}</Text>
|
||||
<Text color={accent.mute}>{l.label.padEnd(14)}</Text>
|
||||
<Text>{l.value}</Text>
|
||||
</Box>
|
||||
))
|
||||
@@ -335,7 +475,7 @@ export type RegistryRow = {
|
||||
installed?: boolean;
|
||||
};
|
||||
|
||||
const KIND_COLOR: Record<RegistryRow['kind'], string> = { skill: 'green', plugin: 'magenta' };
|
||||
const KIND_COLOR: Record<RegistryRow['kind'], string> = { skill: accent.ok, plugin: accent.tool };
|
||||
|
||||
/**
|
||||
* The registry index as a table.
|
||||
@@ -359,20 +499,22 @@ export function RegistryPanel({
|
||||
const width = Math.min(22, Math.max(...rows.map((r) => r.name.length)) + 1);
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1} marginBottom={1}>
|
||||
<Text color="cyan" bold>
|
||||
{title}
|
||||
</Text>
|
||||
{hint && <Text dimColor>{hint}</Text>}
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1} marginBottom={1}>
|
||||
<Box>
|
||||
<Text color={accent.user} bold>
|
||||
{title}
|
||||
</Text>
|
||||
{hint && <Text dimColor>{` ${hint}`}</Text>}
|
||||
</Box>
|
||||
{rows.map((r) => (
|
||||
<Box key={`${r.kind}:${r.name}`}>
|
||||
<Text color={KIND_COLOR[r.kind]}>{r.kind === 'skill' ? 'S' : 'P'} </Text>
|
||||
<Text bold>{r.name.padEnd(width)}</Text>
|
||||
<Text dimColor>{r.description.length > 58 ? `${r.description.slice(0, 58)}...` : r.description}</Text>
|
||||
{r.installed && <Text color="green">{' installed'}</Text>}
|
||||
<Text dimColor>{r.description.length > 58 ? `${r.description.slice(0, 58)}…` : r.description}</Text>
|
||||
{r.installed && <Text color={accent.ok}>{` ${glyph.ok} installed`}</Text>}
|
||||
</Box>
|
||||
))}
|
||||
<Text dimColor>{'S skill P plugin | /registry add <name> | esc to dismiss'}</Text>
|
||||
<Text dimColor>{`S skill P plugin ${glyph.sep} /registry add <name> ${glyph.sep} esc to dismiss`}</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
@@ -402,8 +544,8 @@ export function InstallPrompt({
|
||||
const hidden = body.length - shown.length;
|
||||
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="yellow" paddingX={1}>
|
||||
<Text color="yellow" bold>
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.warn} paddingX={1}>
|
||||
<Text color={accent.warn} bold>
|
||||
{`install ${kind} "${name}"?`}
|
||||
</Text>
|
||||
<Text dimColor>{url}</Text>
|
||||
@@ -413,16 +555,23 @@ export function InstallPrompt({
|
||||
{` ${l}`}
|
||||
</Text>
|
||||
))}
|
||||
{hidden > 0 && <Text dimColor>{` ... ${hidden} more lines`}</Text>}
|
||||
{hidden > 0 && <Text dimColor>{` ${glyph.fold} ${hidden} more lines`}</Text>}
|
||||
</Box>
|
||||
<Box marginTop={1} flexDirection="column">
|
||||
<Text color="yellow">
|
||||
<Text color={accent.warn}>
|
||||
{kind === 'skill'
|
||||
? 'A skill is instructions the agent follows. This text joins your system prompt.'
|
||||
: 'A plugin adds refusal rules. It is data, not code: nothing here is executed.'}
|
||||
</Text>
|
||||
<Text>
|
||||
<Text color="green">y</Text> install | <Text color="red">n</Text> cancel
|
||||
<Text color={accent.ok} bold>
|
||||
y
|
||||
</Text>
|
||||
<Text dimColor>{` install ${glyph.sep} `}</Text>
|
||||
<Text color={accent.err} bold>
|
||||
n
|
||||
</Text>
|
||||
<Text dimColor> cancel</Text>
|
||||
</Text>
|
||||
</Box>
|
||||
</Box>
|
||||
|
||||
+14
-11
@@ -3,6 +3,7 @@ import SelectInput from 'ink-select-input';
|
||||
import React from 'react';
|
||||
import type { CommandSpec } from '../commands';
|
||||
import { InstallPrompt, type RegistryRow } from './Panels';
|
||||
import { accent, glyph } from './theme';
|
||||
|
||||
/** The `/` menu, narrowing as the name is typed. */
|
||||
export function CommandMenu({ matches, index }: { matches: readonly CommandSpec[]; index: number }) {
|
||||
@@ -11,14 +12,14 @@ export function CommandMenu({ matches, index }: { matches: readonly CommandSpec[
|
||||
<Box flexDirection="column" marginTop={1}>
|
||||
{matches.map((c, i) => (
|
||||
<Box key={c.name}>
|
||||
<Text color={i === index ? 'cyan' : undefined}>{i === index ? '> ' : ' '}</Text>
|
||||
<Text color={i === index ? 'cyan' : undefined} bold={i === index}>
|
||||
<Text color={i === index ? accent.user : undefined}>{i === index ? `${glyph.user} ` : ' '}</Text>
|
||||
<Text color={i === index ? accent.user : undefined} bold={i === index}>
|
||||
{`/${c.name}${c.arg ? ` ${c.arg}` : ''}`.padEnd(width)}
|
||||
</Text>
|
||||
<Text dimColor>{c.summary}</Text>
|
||||
</Box>
|
||||
))}
|
||||
<Text dimColor>up/down move | tab complete | enter run | esc dismiss</Text>
|
||||
<Text dimColor>{`↑↓ move ${glyph.sep} tab complete ${glyph.sep} enter run ${glyph.sep} esc dismiss`}</Text>
|
||||
</Box>
|
||||
);
|
||||
}
|
||||
@@ -63,13 +64,13 @@ export function Frame({
|
||||
children: React.ReactNode;
|
||||
}) {
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
|
||||
<Text color="cyan" bold>
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1}>
|
||||
<Text color={accent.user} bold>
|
||||
{title}
|
||||
</Text>
|
||||
{hint && <Text dimColor>{hint}</Text>}
|
||||
{warning && <Text color="yellow">could not list models: {warning}</Text>}
|
||||
{error && <Text color="red">{error}</Text>}
|
||||
{warning && <Text color={accent.warn}>could not list models: {warning}</Text>}
|
||||
{error && <Text color={accent.err}>{error}</Text>}
|
||||
{children}
|
||||
</Box>
|
||||
);
|
||||
@@ -79,7 +80,7 @@ export function Frame({
|
||||
export function Row({ label, children }: { label: string; children: React.ReactNode }) {
|
||||
return (
|
||||
<Box>
|
||||
<Text color="cyan">{label}: </Text>
|
||||
<Text color={accent.user}>{label}: </Text>
|
||||
{children}
|
||||
</Box>
|
||||
);
|
||||
@@ -109,11 +110,13 @@ export function Picker({
|
||||
onSelect: (value: string) => void;
|
||||
}) {
|
||||
return (
|
||||
<Box flexDirection="column" borderStyle="round" borderColor="cyan" paddingX={1}>
|
||||
<Text color="cyan" bold>
|
||||
<Box flexDirection="column" borderStyle="round" borderColor={accent.user} paddingX={1}>
|
||||
<Text color={accent.user} bold>
|
||||
{title}
|
||||
</Text>
|
||||
<Text dimColor>{hint ? `${hint} - enter to select, esc to cancel` : 'enter to select, esc to cancel'}</Text>
|
||||
<Text dimColor>
|
||||
{hint ? `${hint} ${glyph.sep} enter to select ${glyph.sep} esc to cancel` : `enter to select ${glyph.sep} esc to cancel`}
|
||||
</Text>
|
||||
<SelectInput
|
||||
items={options.map((o) => ({ key: o.value, label: o.label, value: o.value }))}
|
||||
limit={limit}
|
||||
|
||||
+28
-12
@@ -30,20 +30,36 @@ export function toolsPanel(session: Session): Panel {
|
||||
|
||||
export function costPanel(
|
||||
session: Session,
|
||||
info: { sessionId: string; model: string; agent: string; thinking: string },
|
||||
info: { sessionId: string; model: string; agent: string; thinking: string; subagentModel?: string },
|
||||
): Panel {
|
||||
const spend = costOf(info.model, session.inputTokens, session.outputTokens);
|
||||
return {
|
||||
title: 'cost',
|
||||
hint: `session ${info.sessionId}`,
|
||||
body: [
|
||||
`- model: \`${info.model}\``,
|
||||
`- billed: ${session.inputTokens} in / ${session.outputTokens} out`,
|
||||
`- spend: ${spend === undefined ? 'unpriced model' : formatUsd(spend)}`,
|
||||
`- context: ~${session.estimatedTokens()} tokens`,
|
||||
`- agent: \`${info.agent}\` thinking \`${info.thinking}\``,
|
||||
].join('\n'),
|
||||
};
|
||||
const lines = [
|
||||
`- model: \`${info.model}\``,
|
||||
`- billed: ${session.inputTokens} in / ${session.outputTokens} out`,
|
||||
`- spend: ${spend === undefined ? 'unpriced model' : formatUsd(spend)}`,
|
||||
];
|
||||
|
||||
// Subagent spend is priced against its own model id, which may be the cheaper
|
||||
// one, so it is reported as its own line rather than folded into the parent's.
|
||||
if (session.subagentInputTokens + session.subagentOutputTokens > 0) {
|
||||
const subModel = info.subagentModel ?? info.model;
|
||||
const subSpend = costOf(subModel, session.subagentInputTokens, session.subagentOutputTokens);
|
||||
lines.push(
|
||||
`- subagents: ${session.subagentInputTokens} in / ${session.subagentOutputTokens} out (\`${subModel}\`)${
|
||||
subSpend === undefined ? '' : ` - ${formatUsd(subSpend)}`
|
||||
}`,
|
||||
);
|
||||
}
|
||||
|
||||
const ceiling = session.spend();
|
||||
if (ceiling.ceiling !== undefined) {
|
||||
lines.push(
|
||||
`- ceiling: ${ceiling.usd === undefined ? 'unpriced' : formatUsd(ceiling.usd)} of ${formatUsd(ceiling.ceiling)}`,
|
||||
);
|
||||
}
|
||||
|
||||
lines.push(`- context: ~${session.estimatedTokens()} tokens`, `- agent: \`${info.agent}\` thinking \`${info.thinking}\``);
|
||||
return { title: 'cost', hint: `session ${info.sessionId}`, body: lines.join('\n') };
|
||||
}
|
||||
|
||||
export function contextPanel(files: readonly string[]): Panel {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user