diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 8cd3300..e28e63b 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -6,7 +6,7 @@ on: workflow_dispatch: inputs: dry_run: - description: Build the artifacts without publishing a release + description: Build and verify the artifacts without publishing a release type: boolean default: true @@ -53,7 +53,11 @@ jobs: publish: needs: build - if: startsWith(github.ref, 'refs/tags/v') + # Tag pushes always publish. A manual dispatch publishes only when dry_run is + # unchecked (false); the default true builds and verifies without a release. + if: >- + startsWith(github.ref, 'refs/tags/v') || + (github.event_name == 'workflow_dispatch' && inputs.dry_run == false) runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..5bdb5c7 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,64 @@ +# shiro-neko + +Agentic coding CLI, built with Bun + TypeScript. The interactive UI is React rendered to the +terminal with Ink; LLM access goes through the Vercel AI SDK (`ai`) with Anthropic, OpenAI, +OpenAI-compatible, and MCP providers. Entry point and only executable is `src/cli.tsx` (bin `shiro`). + +## Commands + +- `bun install --frozen-lockfile` — install deps (CI uses this; lockfile is `bun.lock`) +- `bun run shiro` — run the CLI from source (i.e. `bun run src/cli.tsx`) +- `bun test` — full test suite (`bun:test`, no other runner) +- `bun run typecheck` — `tsc --noEmit`; must pass before committing +- `bun run build` — `bun build --compile` to `dist/shiro` (single native binary) +- `bun run release` — cross-compile all five targets into `dist/release/` +- No linter or formatter is configured; don't invent one. + +CI (`.github/workflows/ci.yml`) runs install → typecheck → test → build on Ubuntu, macOS, and +Windows, pinned to Bun 1.3.14. Everything must be cross-platform: the tools shell out to the +platform shell, and paths in code and tests go through `node:path`, never hardcoded `/`. + +## Layout + +- `src/` — flat modules, one concern per file, lowercase names (`session.ts`, `prune.ts`). + `src/ui/` holds the Ink components, PascalCase (`App.tsx`, `Panels.tsx`). +- `test/` — one `.test.ts` per `src/.ts`; `*.test.tsx` for UI tests via + `ink-testing-library`. `test/helpers.ts` has `testHooks()`, the standard App fixture. +- `docs/` — user-facing docs, one per feature area. +- `scripts/` — `release.ts`, `install.ts` (+ `.sh`/`.ps1` installers). +- `src/version.ts` — hardcoded VERSION; the release workflow fails if it disagrees with the git tag. + +## Conventions + +- ES modules, `verbatimModuleSyntax` on: import types with `import type`. Strict TS with + `noUncheckedIndexedAccess` — indexing gives `T | undefined`, so handle it (`arr[i]!` appears + where provably safe). +- Uses Bun APIs directly (`Bun.file`, `Bun.write`, `Bun.spawn`, `Bun.Glob`) — no fs-extra, no + node shims. File tools read/write through `Bun.*`, not `fs`, where practical. +- Path safety: every user/model-supplied path goes through `jail()` (in `src/ignore.ts`), which + rejects escapes outside `process.cwd()`. Tools resolve paths against `process.cwd()`. +- Error handling: tool `execute` functions return error text to the model or throw `Error` with a + plain message — no error classes, no codes. Storage reads (`store.ts`, `memory.ts`) catch and + degrade to empty rather than throw on corrupt JSON. +- Tools are AI-SDK `tool()` objects with zod `inputSchema`. Any tool that mutates the workspace + must be added to `MUTATING_TOOLS` in `src/tools.ts` — a test in `permission.test.ts` fails + otherwise. The permission system (`src/permission.ts`) matches rules against the tool's subject + (command for `bash`, path for file tools) with glob matching, and is pure/no-IO on purpose. +- Comments explain *why*, often naming the failure being guarded against. Match that style. +- Tests import from `bun:test`, build mock models with `MockLanguageModelV4` + + `simulateReadableStream` from `ai/test`, and each test that touches the filesystem does + `process.chdir()` into a fresh `mkdtemp` dir in `beforeEach` and restores in `afterEach`. + +## Surprising / easy to break + +- `bun test` runs the whole suite including `fallback-live.test.ts`, which spins up real local + HTTP servers via `Bun.serve`, and `commit.test.ts`, which runs real `git` in temp repos — both + need a working network stack and git on PATH. +- Tool output is capped (`MAX_OUTPUT` in `src/tools.ts`); read_file returns NUL-sniffed binary + files as an error. Don't remove these — they stop a model from burning its context. +- Ink renders to the terminal, so library warnings are suppressed at the top of `cli.tsx` + (`AI_SDK_LOG_WARNINGS = false`); anything written to stderr tears the UI. +- Sessions, memory, and history live under `~/.shiro-neko/`, relocatable with `SHIRO_HOME`; + tests depend on that env var to isolate state. Don't resolve the path eagerly at module load — + `store.ts` resolves it per call for this reason. +- Release tags must match `src/version.ts` exactly or `release.ts` stops the build. diff --git a/AUDIT.md b/AUDIT.md new file mode 100644 index 0000000..9e2cec9 --- /dev/null +++ b/AUDIT.md @@ -0,0 +1,201 @@ +# Audit + +Checklist from a full audit of the codebase, run against `main` at `a22d8e1` ("release 0.1.0-beta.5"). +The greps cover every file under `src/`, `test/`, `docs/`, `.github/workflows/`, and `scripts/`. + +Nothing here is a fix — it is a list. Items already tracked in `TODO.md` or `ROADMAP.md` say so; +untracked items are marked **not yet tracked**. + +--- + +## A. Clean findings (verified, no action needed) + +- [x] **No TODO/FIXME/HACK markers in `src/`.** All 26 matches are false positives: + placeholder attributes, the `TODO_MARK` export in `src/notebook.ts` (a literal string + ingredient of the todos feature), and "later" in prose. +- [x] **No `as any` / `@ts-ignore` / `@ts-expect-error` / `@ts-nocheck` / `: any` in `src/`.** + Zero matches. The project's own `docs/development.md` rule is being kept. +- [x] **No silently swallowed errors in `src/`.** Every `catch` was reviewed: + - `src/registry.ts:116,155` — JSON parse failures become descriptive errors. + - `src/registry.ts:120-123,159-162` — zod schema validation with named failure reasons. + - `src/plugins.ts:59-63` — a throwing plugin hook fails **closed** (blocks the call). + - `src/subagent.ts:233-237` — errors are reported and rethrown (never swallowed). + - `src/tools.ts:80-82` — a batch read failure is reported in place, not thrown. + - `src/headless.ts` serialization flattens `Error` before `JSON.stringify` (would emit `{}`). + - `src/ui/App.tsx` — all 14 catch blocks surface the message in the UI. + - `src/plugins-builtin.ts:213-215` — the only quiet `catch`, and it is deliberate, + commented ("a missing binary is not worth interrupting the turn over"). +- [x] **No skipped tests.** No `.skip`, `xit`, or `xdescribe` in `test/` (one match was + `process.exit(` containing "xit("). +- [x] **Registry fetches are size-capped and schema-validated.** `src/registry.ts` caps + content-length and body bytes (`fetchText`, lines 100-108), validates the index with + `indexSchema` (line 120), and regex-validates every plugin manifest pattern (line 167). +- [x] **Version/tag consistency is enforced twice.** `scripts/release.ts` refuses a build when + the tag and `src/version.ts` disagree (line 129), and the release workflow asserts the + built binary prints the expected version (`.github/workflows/release.yml:42-46`). +- [x] **Install scripts match the build targets.** `test/ci.test.ts:85-101` iterates every + `TARGETS` entry from `scripts/release.ts` and asserts the shell/PowerShell installers + fetch exactly those asset names. +- [x] **`.env`/`.pem` are refused on read;** the default permission table + (`src/permission.ts:175-186`) matches the approvals banner in `README.md`. Unknown tools + (MCP, plugins) default to `ask` rather than allow (line 235-236), so `mcp__*` needs no + explicit rule. +- [x] **`--yolo` cannot bypass the guard plugin.** Defaults fold `ask` into `allow` but never + touch `deny` (`src/permission.ts:211-213`), and the guard refuses destructive commands + in `beforeToolCall`, ahead of any approval. +- [x] **Pinned toolchain.** Both workflows pin `bun-version: 1.3.14`, and `test/ci.test.ts:48-52` + fails if the pin ever disagrees with the local `Bun.version`. +- [x] **All three platforms in CI.** `ci.yml` runs the suite on ubuntu, macos, windows + (required — the tools shell out to `rg`, git, and a platform shell). + +--- + +## B. Bugs + +- [ ] **`src/ui/App.tsx:613` — formatting glitch.** The `}` closing the `try` is jammed onto the + same line as the preceding statement: + `push({ kind: 'info', text: await hooks.summarizeMemory() }); } catch (e) {` + Cosmetic only, but it is the kind of blemish left by an unformatted edit and reads as a + slip. **not yet tracked** +- [ ] **`.github/workflows/release.yml` — the `dry_run` input is dead.** `workflow_dispatch` + declares `inputs.dry_run` (default `true`) but no step ever reads it. Nothing consults the + value, so `dry_run=false` changes nothing, and because the `publish` job gates on + `startsWith(github.ref, 'refs/tags/v')`, a manual run can never publish regardless of the + input. Either wire the input into the `publish` `if`, or delete it and let the tag-only + gate be the whole story. **not yet tracked** +- [ ] **`README.md` says "Nineteen built-in tools" — it is now twenty.** `git_commit_message` + (shipped in beta.5 via `src/commit.ts` + `cli.tsx:281`) is a built-in tool, and + `TOOL_SETS.git` carries it (`src/tools-git.ts:189`). The count is one short; + `docs/tools.md` already says "twenty" (line 63), so README is the stale one. + **not yet tracked** + +--- + +## C. Gaps / not implemented (official — tracked in TODO.md or ROADMAP.md) + +### TODO.md "Now" — next up + +- [ ] **Summarize the pruned span.** Compaction drops messages and tells the model nothing, so a + decision from earlier in the session can be contradicted. (TODO.md `## Now`, first item) +- [ ] **A spend ceiling.** `maxSpendUsd` in config, warn at 80%, refuse the next turn at 100%, + headless exits non-zero naming the ceiling. Nothing stops a looping headless run today. + (TODO.md `## Now`) +- [ ] **A cheaper model for subagents.** `subagentModel` in config; an `explore` subagent is + search, not reasoning, and today pays the parent's per-token rate. (TODO.md `## Now`) +- [ ] **Hot-reload an installed entry.** `/registry add` writes the file and says restart; the + skill catalogue and guard chain are assembled at boot. (TODO.md `## Now`) + +### TODO.md "Next" + +- [ ] **MCP without the schema tax.** Twenty MCP tools ≈ 2,750 tokens of schema per request; + `toolSets` does not gate them. Plan: `mcp_list` / `mcp_inspect` / `mcp_call` meta-tools, + prompt names servers not schemas. (TODO.md `## Next`; ROADMAP `## Next` + `## Later`) +- [ ] **Custom commands from a file.** `.shiro/commands/*.md`, `$ARGUMENTS`, `$1`, + `` !`cmd` `` shell substitution with the guard applied. (TODO.md `## Next`; ROADMAP `## Next`) +- [ ] **Derive the tool-name lists.** `TOOL_SETS` and `MUTATING_TOOLS` are hand-maintained; a + tool added to one and forgotten in the other is a silently ungated write. (TODO.md + `## Next`; ROADMAP `## Next` "Derived tool metadata") +- [ ] **Subagent parallelism.** Two independent searches run sequentially; the panel already + renders several agents, the loop does not fan out. (TODO.md `## Next`; ROADMAP `## Later`) +- [ ] **Undo a turn.** `/resume` restores a session but nothing walks one step back; `bash` + effects cannot be snapshotted and the docs would say so. (TODO.md `## Next`; ROADMAP `## Next`) + +### ROADMAP "Next" / "Later" — tracked, not yet scheduled in TODO.cpp-equivalent detail + +- [ ] **Registry trust.** No signatures; `registryUrl` is the whole trust decision. Publisher + keys + pinned digest per entry. (ROADMAP `## Next`; also TODO.md Known rough edges) +- [ ] **Lossless-enough compaction** — same work as "Summarize the pruned span". (ROADMAP `## Next`) +- [ ] **Session branching**, **structured diff review**, **plugin code from disk** (needs a + sandbox story), **prompt caching** (stable prefix vs volatile suffix), **external hooks** + (needs a trust story), **OS-level sandboxing** (Seatbelt/Landlock/Windows equivalent). + (ROADMAP `## Later`) +- [ ] **Deliberately declined** (do not "fix"): web UI, auto-commit, vector search, tool-call + retries, client/server split, LSP integration — all recorded in ROADMAP `## Declined`. + +--- + +## D. Documentation drift (untracked) + +- [ ] **README tool count** — see Bug B.3. **not yet tracked** +- [ ] **TODO.md "Done" is missing the rest of beta.5.** "Kept for one release, then deleted", + but of the beta.5 batch (more tools incl. `git_branch`/`git_commit_message`/ + `move_file`/`delete_file`, the `protect` plugin, `security`/`perf`/`migrate` skills, the + MCP panel wizard, the UI refinements, the farewell message) only "a dead provider item + ends the turn" was checked off. Either the Done list gets the beta.5 items or it gets + rotated, as the file's own rule says. **not yet tracked** +- [ ] **`docs/registry.md` and `docs/headless.md`** are referenced by the README table and both + exist — verified clean, no action. + +--- + +## E. Test-coverage gaps + +- [ ] **`src/cli.tsx` (562 lines) has no unit test.** Nothing in `test/` imports it. Its flag + parsing (`-p`, `--json`, `--yolo`, `--resume`, provider setup, `/provider` wiring) is + exercised only by hand or through `runHeadless` (`test/headless.test.ts`), which bypasses + the argument surface. The largest module in `src/` outside the UI is the least tested one. + **not yet tracked** +- [ ] **`src/ui/Onboard.tsx` has no test.** The provider on-boarding wizard is never rendered in + the suite. **not yet tracked** +- [ ] **`src/ui/PromptInput.tsx` has no test** — the `@` completion input is only covered + indirectly through `App`. (`src/complete.ts` itself is well tested.) **not yet tracked** +- [ ] **`src/ui/panel-bodies.ts` and `src/ui/buses.ts` have no direct tests.** + **not yet tracked** +- [ ] **29 `as any` casts across 17 test files.** The identical mock `usage` object + (`{ inputTokens: {...}, outputTokens: {...} } as any`) is copied verbatim in 7+ UI test + files — a shared typed fixture in `test/helpers.ts` would remove the repetition and the + casts in one move. Production `src/` remains clean; this is test-only debt. **not yet tracked** + +--- + +## F. Maintenance debt + +- [ ] **Pricing table is hand-entered with no source note or date** — `src/pricing.ts:8-22`. + Rates drift; `estimateTokens` also divides JSON length by four (session.ts:90), which is + fine as a compaction threshold but misleads in `/cost`. Both tracked in TODO.md + `## Maintenance`. +- [ ] **`listPaths` walks up to 5000 files once per session** — fine for a repo, wasteful in a + monorepo, never notices a file created after the first `@`. Tracked in TODO.md. +- [ ] **`MUTATING_TOOLS` (tools.ts:799-807) is only used by tests and docs.** The runtime gate + is `DEFAULT_PERMISSIONS` + the unknown-tool `ask` default. Tracked in TODO.md — either + delete it or make `DEFAULT_PERMISSIONS` derive from it. +- [ ] **CI actions are about to leave Node 20.** The last release run annotated that actions on + Node 20 are being forced onto Node 24. `actions/checkout@v4`, `setup-bun@v2`, + `upload-artifact@v4`, `download-artifact@v4` still work, but the major-version bumps will + become the silent fix; watch for the annotation to turn red. **not yet tracked** +- [ ] **Oversized modules.** `src/ui/App.tsx` (863), `src/tools.ts` (717), `src/cli.tsx` (562), + `src/session.ts` (500), `src/ui/Panels.tsx` (398). All of them grew past a comfortable + review size during the beta.5 batch. Not a bug — a "who reads 863 lines" concern. + **not yet tracked** + +--- + +## G. Known rough edges (tracked in TODO.md, reproduced here for the record) + +- [x] `/clear` wipes terminal scrollback (escape sequence takes earlier history with it). +- [x] Memory has no conflict resolution; two contradictory notes both inject. +- [x] Windows `cmd /c` vs `bash -lc` — the prompt names the platform, does not translate. +- [x] An unknown name in `toolSets` is dropped silently — reads as "that set is off". +- [x] Permission rules gate the call, not what it does — no OS sandbox around the shell. +- [x] The reasoning panel is per-turn, not per-step. +- [x] An interrupted command's effects are unknown, and the model is told so. +- [x] `@` completion lists files, not directories. +- [x] An installed skill is a stranger's words in the system prompt; nothing re-checks it later. +- [x] A registry index is trusted for its contents, not its authorship. + +--- + +## Summary + +| Area | Items | +|---|---| +| Bugs | 3 (`App.tsx:613`, dead `dry_run` input, README tool count) | +| Official gaps (Now/Next/Later) | 14 tracked in TODO.md/ROADMAP.md | +| Documentation drift | 2 untracked | +| Test-coverage gaps | 5 (of which `src/cli.tsx` is the significant one) | +| Maintenance debt | 5 (3 tracked, 2 untracked) | +| Known rough edges | 10 (all tracked) | +| Verified clean | 9 areas, including zero `as any` and zero swallowed errors in `src/` | + +The codebase is in good shape for a beta. The three bugs are each one-line fixes; the +coverage gap on `cli.tsx` is the item that will actually bite. \ No newline at end of file diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..3ab05d3 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,89 @@ +# Changelog + +All notable changes to this project are documented here. The format follows +[Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project adheres to +[Semantic Versioning](https://semver.org/spec/v2.0.0.html). + +## [1.0.0] + +The first stable release. Cost control, a larger tool and skill surface, custom slash +commands, auto-loaded extensions, and a redesigned welcome interface, on top of the beta +line's agent loop, approval model, and safety guarantees. + +### Added + +- **Spend ceiling** ("maxSpendUsd" in config). Checked before each turn: past the limit the + model is never called, the turn is refused naming the ceiling, headless exits non-zero, + and it warns once at 80% of the limit. Unpriced models cannot be measured, so the ceiling + does not apply to them. +- **Cheaper subagent model** ("subagentModel" in config). "explore" subagents — search, not + reasoning — resolve against a configured cheaper model while "review" and "worker" keep the + parent's. "/cost" reports subagent spend separately, priced against the subagent's model id. +- **Twenty new built-in tools** (41 total) in a new "extra" tool set, across four families: + - line edits: insert_lines, delete_lines, replace_lines, append_file, prepend_file, count_lines + - filesystem: tree, file_info, find_files, recent_files, changed_files + - git (read-only, argv-spawned): git_log_file, git_diff_commits, git_show_file, git_current_branch, git_changed_in_ref + - code and environment: find_symbol, json_query, outline, read_symbol, env_info, count_tokens +- **Twenty new bundled skills** (29 total), including plan, docs, api-design, ci-cd, db, + docker, frontend, git-workflow, logging, optimize-sql, release, accessibility, data, i18n, + deps, onboarding, ux-copy, readme, incident, perf-frontend. +- **Ten new plugins**, all data. Safety refusals on by default — no-force-push, no-net-pipe, + no-root, no-env-write — and opt-in workflow plugins — no-main-commit, no-git-config, + confirm-delete, conventional-commit, tests-first, small-diffs. +- **Custom slash commands** from Markdown files in ".shiro/commands/" and + "~/.shiro-neko/commands/", with frontmatter "description"/"agent", "$ARGUMENTS" and + positional "$1", and shell substitution passed through the guard. A custom command can + never shadow a built-in. +- **Auto-loaded external extensions** from "~/.shiro-neko/{skills,tools,plugins}" and + ".shiro/{skills,tools,plugins}". All data, never code: tools are bounded manifests (a shell + template through the guard, an HTTPS fetch, or a workspace read), plugins are refusal + manifests. Malformed files are reported and skipped, never fatal. + +### Changed + +- **Skills now live as Markdown files** in "src/skills-md/", embedded into the compiled binary + by Bun text imports, replacing the previous TypeScript string constants. A format test + enforces that each parses with valid frontmatter and a real body. The eleven pre-existing + skills were also deepened. +- **Welcome interface redesigned** into a structured dashboard: a session banner, a grouped + environment panel with attention-worthy facts coloured out of the quiet layer, and a meta + bar. The prompt input sits in a two-tone box with the agent and model row inside it and a + split footer beneath. +- **System prompt advanced** with a discrete failure-recovery loop, a delegation policy, and + compaction awareness. + +### Fixed + +- **Release workflow "dry_run" input is now honoured.** A manual dispatch publishes only when + it is unchecked; tag pushes always publish. Previously the input was declared but never read, + so a manual run could never publish regardless of its value. +- **Documentation drift** corrected across the tool count, the bundled-skill count, the plugin + defaults, and the README quickstart. + +## [0.1.0-beta.5] + +- A dead provider item no longer ends the turn: a 404 naming a missing item rewrites the + history inline and retries once. +- More tools (git_commit_message, git_branch, move_file, delete_file), the "protect" plugin, + the security/perf/migrate skills, the "/mcp add" wizard, and the farewell message. + +## [0.1.0-beta.4] + +- Compaction no longer stops the loop, and is bounded to keep the widest recent tool tail. +- The external registry for skills and plugins. +- Permission rules matched per command and path, replacing the per-tool list. +- apply_patch, web_fetch, and writable worker subagents. + +## [0.1.0-beta.3] + +- Fourteen built-in tools with gateable tool sets. +- Streaming reasoning display, the mid-turn prompt queue, multi_edit, list_dir, read-only git + tools, batch reads, @file completion, and interruptible commands. + +## [0.1.0-beta.1] + +- The core agent loop with SDK-enforced tool approvals, endpoint fallback, and retries. +- The initial tool set, Ink interface, agent variants, skills, plugins, per-project memory, + session persistence, subagents, MCP, and five-platform builds. + +[1.0.0]: https://github.com/zakirkun/shiro-neko/releases/tag/v1.0.0 diff --git a/README.md b/README.md index d5b213e..3e84ed4 100644 --- a/README.md +++ b/README.md @@ -46,11 +46,11 @@ from the models that endpoint actually reports. Settings land in `~/.shiro-neko/config.json`. Run `/provider` any time to change them. ``` -shiro-neko 0.1.0-beta.5 openai/gpt-5 session 0193ab2c +shiro-neko 1.0.0 openai/gpt-5 session 0193ab2c agent: default thinking: medium cwd: /home/you/project -skills: commit, debug, migrate, perf, refactor, review, security, test, verify -plugins: guard, secrets, protect, time +skills: commit, debug, docs, migrate, perf, plan, refactor, review, security, test, verify +plugins: guard, secrets, protect, time, no-force-push, no-net-pipe, no-root, no-env-write approvals: ask for write_file, edit_file, multi_edit, apply_patch, move_file, delete_file, bash, web_fetch, mcp__* /help for commands @@ -113,7 +113,7 @@ record of what it already ran instead of repeating it. **Runs headless.** `shiro -p "review this diff" --json` for scripts and CI. -**Keeps the tool list affordable.** Nineteen built-in tools, grouped into sets. Each costs +**Keeps the tool list affordable.** Forty-one built-in tools, grouped into sets. Each costs about 550 characters of schema on every request, so `{ "toolSets": [] }` trims back to the six core ones and a disabled set reaches neither the wire nor the prompt. @@ -131,6 +131,8 @@ the flags are. | [Skills](docs/skills.md) | the bundled skills, writing your own, why the catalogue is split | | [Plugins](docs/plugins.md) | the interface, the guard and its limits, builtin versus installed | | [Registry](docs/registry.md) | installing external skills and plugins, publishing your own | +| [Custom commands](docs/custom-commands.md) | a Markdown file becomes a slash command, with arguments and shell substitution | +| [Extensions](docs/extensions.md) | auto-loaded external skills, tools, and plugins — data, never code | | [Memory and state](docs/memory.md) | memory, task lists, sessions, compaction and its repair | | [MCP](docs/mcp.md) | connecting servers, namespacing, cost, debugging one | | [Headless mode](docs/headless.md) | `-p`, JSON events, exit codes, CI recipes | @@ -138,6 +140,7 @@ the flags are. | [Development](docs/development.md) | building, testing, adding a tool, releasing | | [Roadmap](ROADMAP.md) | what is next and what has been declined | | [TODO](TODO.md) | the current work list, with known rough edges | +| [Changelog](CHANGELOG.md) | release history, newest first | ## Commands @@ -156,15 +159,18 @@ workspace path. Up and down recall earlier prompts. ## Status -Working: the agent loop, tool approvals, subagents including the gated `worker` kind, skills, -plugins, per-project memory, session persistence, MCP, markdown rendering, headless mode, -five-platform builds, streaming reasoning display, the mid-turn prompt queue, gateable tool -sets, read-only git tools, batch reads, `apply_patch`, `web_fetch`, `@file` completion, -interruptible commands, and the external registry. +Version 1.0 is stable. Working: the agent loop, per-call and per-command tool approvals with a +guard that `--yolo` cannot bypass, subagents including the gated `worker` kind, a spend ceiling +(`maxSpendUsd`) with a cheaper subagent model (`subagentModel`), 41 built-in tools across +gateable sets, 29 bundled skills, built-in and data-only plugins, per-project memory, session +persistence and resume, MCP servers, custom slash commands from markdown files, auto-loaded +external skills/tools/plugins, markdown rendering, headless mode with JSON events for CI, +five-platform builds, streaming reasoning, the mid-turn prompt queue, read-only git tools, +batch reads, `apply_patch`, `web_fetch`, `@file` completion, interruptible commands, and the +external registry. Next up is in [TODO.md](TODO.md); the longer view and what has been declined are in -[ROADMAP.md](ROADMAP.md). The short version of what is missing: a summary of what compaction -discarded, a spend ceiling, and a cheaper model for subagent searches. +[ROADMAP.md](ROADMAP.md); the release history is in [CHANGELOG.md](CHANGELOG.md). ## License diff --git a/ROADMAP.md b/ROADMAP.md index aa971db..720927d 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -189,6 +189,53 @@ each server's live tool count or its connection error, and `/mcp remove` takes o three write `config.json` directly; a new server connects on the next start, because connecting mid-turn would change the tool list under a running request. +### 1.0.0 + +The first stable release. The beta line's architecture held; this release rounds out cost +control, extensibility, and the interface, and hardens the test suite to match. + +**Cost control.** Two halves of one problem, both shipped. A **spend ceiling** (`maxSpendUsd`) +checks before each turn: past the limit the model is never called, the turn is refused naming +the ceiling, headless exits non-zero, and it warns once at 80%. A **cheaper subagent model** +(`subagentModel`) runs `explore` — which is search, not reasoning — on a less expensive model +while `review` and `worker` keep the parent's; `/cost` reports subagent spend as its own line, +priced against the subagent's model id. + +**Tools: 41 built-in.** Twenty new tools in four families, all path-jailed and ignore-aware, +in a new `extra` tool set: precise line edits (`insert_lines`, `delete_lines`, `replace_lines`, +`append_file`, `prepend_file`, `count_lines`), filesystem navigation (`tree`, `file_info`, +`find_files`, `recent_files`, `changed_files`), read-only git extensions (`git_log_file`, +`git_diff_commits`, `git_show_file`, `git_current_branch`, `git_changed_in_ref`), and code and +environment reads (`find_symbol`, `json_query`, `outline`, `read_symbol`, `env_info`, +`count_tokens`). Every git call still spawns the binary with a fixed argv, never a shell. + +**Skills: 29 bundled, as Markdown.** The catalogue grew from nine to twenty-nine and every +skill moved to a single source of truth: a Markdown file in `src/skills-md/`, frontmatter and +body, embedded into the compiled binary by Bun text imports. A format test enforces that each +one parses and carries a real body. + +**Plugins: 10 more, all data.** Six narrow safety refusals (force push, pipe-to-shell, root +elevation, env credential writes, main-branch commits, git config changes) and three advisory +plugins (conventional commits, tests-first, small diffs) plus a delete guard for ambiguous +paths. The safety refusals are on by default for the same reason the guard is; the opinionated +ones are opt-in. + +**Custom slash commands.** A Markdown file in `.shiro/commands/` or `~/.shiro-neko/commands/` +becomes a slash command, with frontmatter `description`/`agent`, `$ARGUMENTS` and `$1` +positionals, and `` !`cmd` `` substitution passed through the guard. A custom command can never +shadow a built-in. + +**Auto-loaded extensions.** External skills, tools, and plugins load from +`~/.shiro-neko//` and `.shiro//` — all data, never code. An external tool is a +bounded manifest (a shell template through the guard, an HTTPS fetch, or a workspace file +read); an external plugin is a refusal manifest. A malformed file is reported and skipped, +never fatal. + +**Interface.** The welcome screen is a structured dashboard — a session banner, a grouped +environment panel with attention-worthy facts lifted out of the quiet layer, and a meta bar — +replacing a wall of dim text. The input sits in an OpenCode-style two-tone box with the +agent·model row inside it and a split footer beneath. + --- ## Next @@ -201,12 +248,6 @@ answer is three meta-tools — `mcp_list`, `mcp_inspect`, `mcp_call` — with th the servers, so a hundred servers cost almost nothing until one is called. Worth keeping direct registration as an option: for a two-tool server the indirection is the more expensive of the two. -### Custom commands from a file - -A markdown file becoming a slash command, with `$ARGUMENTS`, `$1`, `` !`cmd` `` for shell output, -and `@path` for a file. Every comparable CLI has this and none of it is hard; it is missing because -nothing forced the issue. - ### Undo a turn opencode has `/undo` and `/redo`, Claude Code has `/rewind` over file checkpoints. There is @@ -220,12 +261,6 @@ Compaction keeps the model's memory of a turn now, but it still says nothing abo discarded, so the model can contradict its own earlier decision with confidence. A summary of the discarded span costs one cheap call and removes the whole class of problem. -### Cost control - -Two halves of the same problem: an `explore` subagent pays the parent's reasoning rate for -what is really a search, and nothing stops a headless run that loops. A cheaper subagent model -and a per-session ceiling are both small changes on top of the pricing that already exists. - ### Derived tool metadata `TOOL_SETS` and `MUTATING_TOOLS` are hand-maintained lists of tool names. A tool added to one diff --git a/TODO.md b/TODO.md index 50a8c0e..0f3b964 100644 --- a/TODO.md +++ b/TODO.md @@ -19,24 +19,6 @@ confidence. - [ ] Budget it: a summary that grows with the session defeats the point - [ ] Test: a pruned decision is still recoverable from the summary -### A spend ceiling - -A headless run that loops costs real money with nothing to stop it. - -- [ ] `maxSpendUsd` in config, checked after every turn -- [ ] Warn at 80%, refuse to start another turn at 100% -- [ ] Headless exits non-zero with the ceiling named, rather than stopping silently -- [ ] Test: a session past its ceiling refuses the next turn and says why - -### A cheaper model for subagents - -The subagent shares the parent's model. An `explore` run is search, not reasoning, and it -currently pays the parent's per-token rate. - -- [ ] `subagentModel` in config, defaulting to the parent -- [ ] `/cost` separates parent from subagent spend -- [ ] Test: the subagent's calls go to the configured model, the parent's do not - ### Hot-reload an installed entry `/registry add` writes the file and says to restart. The skill catalogue and the guard chain @@ -64,17 +46,6 @@ names only the servers. A hundred servers then cost almost nothing until one is - [ ] Keep per-tool registration as an option: a two-tool server is cheaper registered directly - [ ] Test: a configured server contributes no schema to the request until `mcp_call` -### Custom commands from a file - -Every other CLI in this class has these and they are cheap: a markdown file becomes a slash -command, with `$ARGUMENTS`, `$1`, `` !`cmd` `` for shell output, and `@path` for a file. - -- [ ] `.shiro/commands/*.md` and `~/.shiro-neko/commands/*.md`, name from the filename -- [ ] Frontmatter for `description` and `agent` -- [ ] `$ARGUMENTS` and positional `$1` -- [ ] `` !`cmd` `` substituted before the prompt is sent, with the guard applied to it -- [ ] Test: a command with a shell substitution reaches the model with the output inlined - ### Derive the tool-name lists `TOOL_SETS` and `MUTATING_TOOLS` both list names by hand. A tool added to one and forgotten @@ -151,37 +122,25 @@ Not bugs exactly, but things that will bite someone. ## Done -Kept for one release, then deleted. +Kept for one release, then deleted. The 1.0.0 release batch: -- [x] Reasoning streamed to a collapsed panel, `ctrl-r` to expand, dropped when the turn ends -- [x] The tool in flight named on screen from `tool-input-start` until its result arrives -- [x] Prompts typed during a turn queue and drain in order; `esc` clears the queue -- [x] `toolSets` gating, so a disabled set reaches neither the wire nor the prompt -- [x] `multi_edit`, atomic across several edits to one file -- [x] `list_dir`, ignore-aware and depth-limited -- [x] Read-only git tools: `git_status` `git_diff` `git_log` `git_show` `git_blame` -- [x] Orphaned tool results dropped during pruning, fixing the 400 "No tool call found for - function call output with call_id ..." -- [x] `read_many_files`, concurrent, one labelled block per file, a bad path reported in place -- [x] `@file` completion: picker fed by the ignore-aware walker, tab inserts a relative path -- [x] `ctrl-c` kills the running command and keeps the turn. The kill takes the whole process - tree: killing `cmd /c` alone left the real command holding both pipes open, so the - interrupt appeared to do nothing for 19 seconds -- [x] **Compaction no longer stops the loop.** Pruning used to drop any assistant part whose - reasoning item it removed, which on a reasoning model is every tool call. The model lost - its record of what it had run and re-ran it until the step limit. The repair strips the - provider `itemId` instead of the part, so the same content is sent inline -- [x] **A dead provider item no longer ends the turn.** An `item_reference` resolves only while - the provider still stores that item, so a resumed session could fail on every attempt with - 404 "Item with id 'msg_...' not found". Compaction now sends the history inline, and a 404 - naming a missing item rewrites the history inline and retries once -- [x] `/registry`: browse, search, install, and remove external skills and plugins. Skills are - shown in full before install; plugins are a validated manifest of deny rules, never code -- [x] Context shown as a percentage of the compaction threshold, amber at two thirds, red at 90 -- [x] **Permission rules per command and path**, replacing the per-tool list. `bash` was one - yes/no for `git status` and `rm -rf`, so pressing `a` once removed the gate for both. - Rules match the call's subject, `always` grants a pattern rather than the tool, `.env` and - `.pem` are refused on read, and an identical call repeated three times in a turn asks even - when allowed -- [x] **`web_fetch`**, size-capped HTTP(S) to markdown in the opt-in `net` tool set, with - redirect and private-address checks +- [x] A spend ceiling (`maxSpendUsd`): checked before each turn, refused at 100% naming the + ceiling, warns once at 80%, headless exits non-zero. Unpriced models are not enforced +- [x] A cheaper subagent model (`subagentModel`): `explore` resolves against it, `review` and + `worker` keep the parent's, `/cost` splits subagent spend by model id +- [x] Twenty new built-in tools (41 total) in a new `extra` set: line edits, filesystem + navigation, read-only git extensions, and code/environment reads +- [x] Twenty new bundled skills (29 total) plus the eleven originals deepened; all moved to + `src/skills-md/*.md` as the Markdown source of truth, embedded at build +- [x] Ten new data-only plugins: safety refusals on by default (force push, pipe-to-shell, + root, env credential writes) and opt-in workflow plugins (conventional commit, + tests-first, small diffs, main-branch commits, git config, confirm-delete) +- [x] Custom slash commands from Markdown files, with `$ARGUMENTS`/`$1` and guarded shell + substitution; a custom command never shadows a built-in +- [x] Auto-loaded external skills, tools, and plugins from `~/.shiro-neko/` and + `.shiro/`, all data, never code; a bad file is reported and skipped +- [x] The welcome interface redesigned into a structured dashboard with a session banner, a + grouped environment panel, and a meta bar; the input in a two-tone box with a split footer +- [x] The system prompt advanced: a failure-recovery loop, a delegation policy, compaction awareness +- [x] The release workflow's dead `dry_run` input wired: manual dispatch publishes only when + unchecked, tag pushes always publish diff --git a/bun.lock b/bun.lock index f1833da..4ae3d3c 100644 --- a/bun.lock +++ b/bun.lock @@ -14,11 +14,13 @@ "ink-select-input": "6.2.0", "ink-spinner": "^5.0.0", "ink-text-input": "6.0.0", + "pngjs": "^7.0.0", "react": "19.2.8", "zod": "4.5.4", }, "devDependencies": { "@types/bun": "^1.4.0", + "@types/pngjs": "^6.0.5", "@types/react": "19.2.18", "ink-testing-library": "4.0.0", "react-devtools-core": "^7.0.1", @@ -51,6 +53,8 @@ "@types/node": ["@types/node@26.4.0", "", { "dependencies": { "undici-types": "~8.3.0" } }, "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ=="], + "@types/pngjs": ["@types/pngjs@6.0.5", "", { "dependencies": { "@types/node": "*" } }, "sha512-0k5eKfrA83JOZPppLtS2C7OUtyNAl2wKNxfyYl9Q5g9lPkgBl/9hNyAu6HuEH2J4XmIv2znEpkDd0SaZVxW6iQ=="], + "@types/react": ["@types/react@19.2.18", "", { "dependencies": { "csstype": "^3.2.2" } }, "sha512-AnzbBERsrLKtk2XSfTbYRLjQPdy116Sty4q+T+Bp3IC4l6jNBvreVPAHmpq9qhXQM7CXZPjLVmGMw9sy+hxQ3w=="], "@vercel/oidc": ["@vercel/oidc@3.2.0", "", {}, "sha512-UycprH3T6n3jH0k44NHMa7pnFHGu/N05MjojYr+Mc6I7obkoLIJujSWwin1pCvdy/eOxrI/l3uDLQsmcrOb4ug=="], @@ -131,6 +135,8 @@ "pkce-challenge": ["pkce-challenge@5.0.1", "", {}, "sha512-wQ0b/W4Fr01qtpHlqSqspcj3EhBvimsdh0KlHhH8HRZnMsEa0ea2fTULOXOS9ccQr3om+GcGRk4e+isrZWV8qQ=="], + "pngjs": ["pngjs@7.0.0", "", {}, "sha512-LKWqWJRhstyYo9pGvgor/ivk2w94eSjE3RGVuzLGlr3NmD8bf7RcYGze1mNdEHRP6TRP6rMuDHk5t44hnTRyow=="], + "react": ["react@19.2.8", "", {}, "sha512-PWaYA1L/q9u2u7xYQi+Y3L3Yfnie7XyLeaJICV1MGD6LprsBxcAqGjYyr0eY3p+QdsA+x/Irkt4Qif8D63+Sbw=="], "react-devtools-core": ["react-devtools-core@7.0.1", "", { "dependencies": { "shell-quote": "^1.6.1", "ws": "^7" } }, "sha512-C3yNvRHaizlpiASzy7b9vbnBGLrhvdhl1CbdU6EnZgxPNbai60szdLtl+VL76UNOt5bOoVTOz5rNWZxgGt+Gsw=="], diff --git a/docs/configuration.md b/docs/configuration.md index a6587e8..473d6e6 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -20,8 +20,10 @@ Written by `/provider`, editable by hand. Every field is optional. "agent": "default", "thinking": "medium", "maxRetries": 3, + "maxSpendUsd": 5, + "subagentModel": "gpt-5-nano", "plugins": ["guard", "time"], - "toolSets": ["edit-plus", "git"], + "toolSets": ["edit-plus", "extra", "git"], "permission": { "bash": { "*": "ask", "git *": "allow" } }, @@ -42,12 +44,31 @@ Written by `/provider`, editable by hand. Every field is optional. | `agent` | default variant: `default`, `quick`, `deep`, `plan`, `review` | | `thinking` | default level: `off`, `low`, `medium`, `high`, `max` | | `maxRetries` | retries per model call for transient failures. Default 3 | -| `plugins` | which builtin plugins to enable. Omit for `["guard", "secrets", "protect", "time"]` | -| `toolSets` | optional tool sets beyond `core`: `edit-plus`, `git`, and `net`. Omit for the defaults; `net` is opt-in. See [tools](tools.md) | +| `maxSpendUsd` | session spend ceiling: warn at 80%, refuse the next turn at 100%. Headless exits non-zero naming the ceiling. Only enforced on priced models | +| `subagentModel` | model id for `explore` subagents, which search rather than reason. Omit to share the parent's model. `/cost` reports subagent spend separately | +| `plugins` | which builtin plugins to enable. Omit for `["guard", "secrets", "protect", "time", "no-force-push", "no-net-pipe", "no-root", "no-env-write"]` | +| `toolSets` | optional tool sets beyond `core`: `edit-plus`, `nav`, `extra`, `git`, and `net`. Omit for the defaults; `net` is opt-in. See [tools](tools.md) | | `permission` | which calls run, ask, or are refused, matched per command or path. See [permissions](permissions.md) | | `registryUrl` | index for `/registry`. Omit for the default. See [registry](registry.md) | | `mcpServers` | see [MCP](mcp.md) | +## Directories + +Beyond the config file, these locations are read on every start: + +| Path | Holds | +|---|---| +| `~/.shiro-neko/config.json` | the config above | +| `~/.shiro-neko/skills/*.md` `.shiro/skills/*.md` | auto-loaded skills — [extensions](extensions.md) | +| `~/.shiro-neko/tools/*.json` `.shiro/tools/*.json` | auto-loaded tool manifests | +| `~/.shiro-neko/plugins/*.json` `.shiro/plugins/*.json` | auto-loaded plugin manifests | +| `~/.shiro-neko/commands/*.md` `.shiro/commands/*.md` | custom slash commands — [custom commands](custom-commands.md) | +| `~/.shiro-neko/registry/{skills,plugins}/` | entries installed with `/registry` | +| `~/.shiro-neko/sessions/` | saved sessions, for `-c` / `-r` | + +`SHIRO_HOME` overrides the home directory for all of these, which is also how the test suite +isolates itself. + ## Provider presets `/provider` offers these. Each sets `baseURL` and the wire protocol for you. diff --git a/docs/custom-commands.md b/docs/custom-commands.md new file mode 100644 index 0000000..3d5e170 --- /dev/null +++ b/docs/custom-commands.md @@ -0,0 +1,76 @@ +# Custom slash commands + +A Markdown file becomes a slash command. Write the prompt once, run it with `/name` any time. + +Two directories are scanned, the project shadowing the user by name: + +| Origin | Directory | +|---|---| +| user | `~/.shiro-neko/commands/*.md` | +| project | `.shiro/commands/*.md` | + +The filename is the command: `.shiro/commands/review-diff.md` becomes `/review-diff`. Names are +letters, digits, dashes, and underscores; anything else is skipped. A custom command can never +shadow a built-in — `/cost` always runs the built-in `/cost`. + +## Format + +```markdown +--- +description: Review the staged diff for defects +agent: review +--- + +Review the staged changes. For each finding give file, line, what breaks, and the fix. +``` + +Frontmatter is optional but useful: + +- **`description`** — the one line shown in the `/` menu. Without it the first body line is used. +- **`agent`** — run this command under a specific agent variant (`default`, `quick`, `deep`, + `plan`, `review`). The variant is restored afterwards, so one command does not leak its agent + into the rest of the session. + +Everything after the frontmatter fence is the prompt. A file with an empty body is skipped, as +is one that fails to parse. + +## Arguments + +The body is a template, expanded against whatever you type after the command: + +- `$ARGUMENTS` — the whole argument string. +- `$1`, `$2`, … — positional arguments. A missing positional expands to nothing. + +```markdown +Compare $1 against $2 and report the differences. Context: $ARGUMENTS +``` + +`/compare src/a.ts src/b.ts` sends `Compare src/a.ts against src/b.ts … Context: src/a.ts src/b.ts`. + +## Shell substitution + +A `` !`command` `` inline runs the shell command and inlines its output before the prompt is +sent: + +```markdown +Review this diff: + +!`git diff --staged` +``` + +Every substitution runs through the **guard** before executing, exactly as a direct `bash` call +is — so a custom command cannot smuggle a destructive command past you. A substitution that +exits non-zero, or one the guard refuses, fails the command with the reason named. + +## When to write one + +- A prompt you find yourself retyping: a review shape, a release checklist, a project-specific + "how we test". +- A prompt that should pin an agent: a read-only review command that always runs under `review`. +- Project conventions the whole team should share: commit `.shiro/commands/` so everyone gets + the same commands. + +For behaviour that must survive across sessions rather than be invoked on demand, use +[memory](memory.md). For instructions the agent loads by task rather than by name, use a +[skill](skills.md). For extensions that add tools or refusal rules rather than prompts, see +[extensions](extensions.md). diff --git a/docs/extensions.md b/docs/extensions.md new file mode 100644 index 0000000..3ca986e --- /dev/null +++ b/docs/extensions.md @@ -0,0 +1,101 @@ +# Extensions: auto-loaded skills, tools, and plugins + +External extensions load automatically from two directories on every start, the project +shadowing the user by name: + +| Origin | Directories | +|---|---| +| user | `~/.shiro-neko/skills` `~/.shiro-neko/tools` `~/.shiro-neko/plugins` | +| project | `.shiro/skills` `.shiro/tools` `.shiro/plugins` | + +Drop a file in and it is live on the next start. No registry, no install command, no restart +of anything but the CLI itself. + +**Everything here is data, never code.** That is the same rule the [registry](registry.md) +enforces, and it is the whole security model. An external extension can add instructions, a +bounded tool, or a refusal rule — it cannot run arbitrary code, so it cannot read every file +the agent can read or lie about what it blocks. A malformed file is reported on the welcome +dashboard and skipped, never fatal. + +## Skills + +A skill is a Markdown file with frontmatter, exactly like a bundled one: + +```markdown +--- +name: deploy +description: Ship a release. Use when asked to deploy or cut a release. +--- + +# Deploy + +1. Confirm the tests pass. Do not deploy on a red suite. +2. Tag with the version from src/version.ts, not by hand. +``` + +Skills merge by name with the precedence `builtin < registry < user < project`, so your own +`debug.md` overrides the bundled `debug`. See [skills](skills.md) for the full format. + +## Tools + +A tool is a JSON manifest describing one bounded operation. Three kinds, each with a ceiling +on what it can do: + +```json +{ + "name": "recent-changes", + "description": "List the ten most recently changed files", + "kind": "shell", + "command": "git diff --name-only HEAD~10", + "autoApprove": true +} +``` + +| Field | Meaning | +|---|---| +| `name` | The tool name the model calls. Letters, digits, dashes, underscores. | +| `description` | What the model reads to decide when to use it. | +| `kind` | `shell`, `http`, or `read`. | +| `command` | For `shell`: the template to run, with an optional `{arg}` placeholder. | +| `url` | For `http`: the URL to fetch, with an optional `{arg}` placeholder. HTTPS only. | +| `path` | For `read`: the workspace file to return, with an optional `{arg}` placeholder. | +| `autoApprove` | `false` to require approval before running. Default `true`. | + +The model passes a single optional `arg` string, substituted into `{arg}`. + +**The limits are the point.** A `shell` tool runs a fixed template through the **guard** and +the platform shell — the same chain a built-in `bash` call goes through, so an installed tool +cannot do what the agent itself may not. An `http` tool fetches one HTTPS URL. A `read` tool +returns one workspace file, jailed to the workspace. None of them executes code from the +manifest. + +## Plugins + +A plugin is a refusal manifest — the same shape the registry installs — a name, an optional +prompt appendix, and deny rules matched against tool input: + +```json +{ + "name": "no-prod-config", + "description": "refuses to edit production config", + "appendix": "Production config is changed by hand, never by the agent.", + "deny": [ + { "tools": ["write_file", "edit_file"], "pathPattern": "config/production", "reason": "production config is hand-edited" }, + { "tools": ["bash"], "commandPattern": "kubectl\\s+apply", "reason": "deploys to the cluster" } + ] +} +``` + +A rule names the tools it covers and either a `pathPattern` (matched against the path a file +tool carries) or a `commandPattern` (matched against a `bash` command), both as case-insensitive +regexes, plus the `reason` handed to the model when it blocks. Patterns are validated on load; +an invalid regex is a reported error, not a crash. + +Refusal plugins compose with the built-in [plugins](plugins.md) — the first block wins. + +## Relationship to the registry + +The [registry](registry.md) fetches the same kinds of files over HTTPS with a confirmation +step. Auto-load is for your own and your project's files, which need no confirmation because +you wrote them. The two mechanisms share the loaders and the safety model; they differ only in +where the file comes from. diff --git a/docs/plugins.md b/docs/plugins.md index d49d263..1f5b960 100644 --- a/docs/plugins.md +++ b/docs/plugins.md @@ -19,7 +19,7 @@ the agent can read. That is a sandbox problem, not a loader problem — see ## Enabling ```json -{ "plugins": ["guard", "secrets", "protect", "time"] } +{ "plugins": ["guard", "secrets", "protect", "time", "no-force-push", "no-net-pipe", "no-root", "no-env-write"] } ``` That is also the default when the field is absent, and it lists **builtin** plugins only. @@ -128,6 +128,49 @@ write normally. Adds `current_time`, returning ISO 8601 plus the local string. Auto-approved; it reads nothing. Useful because models are confidently wrong about the date. +### The narrow safety refusals (default on) + +Six small plugins, each blocking one irreversible class of mistake. They are on by default for +the same reason the guard is: a safety check you have to opt into is not one. Each is data — a +name, a pattern list, and a refusal message — matched against the `bash` command string (or, for +`confirm-delete`, the path a `delete_file` carries). + +| Plugin | Refuses | Why | +|---|---|---| +| `no-force-push` | `git push --force`, `--force-with-lease`, `-f`, `+` | rewrites remote history | +| `no-net-pipe` | `curl … \| sh`, `wget … \| node`, `iex (iwr …)` | executes a download unseen | +| `no-root` | `sudo …`, elevated `runas` / `Start-Process -Verb RunAs` | nothing the agent does should need root | +| `no-env-write` | `export …KEY/TOKEN/SECRET/PASSWORD=…` | writes a credential into the environment | +| `no-main-commit` | `git commit`/`git merge` naming `main`/`master` | touches the default branch directly | +| `no-git-config` | `git config --global`, identity/runner keys | changes how git identifies or runs | + +Normal commands pass: `git push origin feature`, `npm test`, `export NODE_ENV=production`. The +patterns target the irreversible act, not the command family. + +### The advisory plugins (opt in) + +Three plugins carry only a prompt appendix — no blocking hook — so they shape behaviour without +ever refusing a call. Enable them in config when you want the nudge: + +| Plugin | Advises | +|---|---| +| `conventional-commit` | commit subjects as `type(scope): summary`, e.g. `fix(auth): reject expired tokens` | +| `tests-first` | for a bug, pin it with a failing test before fixing; watch it fail, then pass | +| `small-diffs` | one change does one thing; split a diff that is really two | + +```json +{ "plugins": ["guard", "secrets", "protect", "time", "conventional-commit", "small-diffs"] } +``` + +### `confirm-delete` (opt in) + +Refuses `delete_file` calls whose path is broad or ambiguous — a wildcard, a trailing slash, or +an empty path — so a delete is always one explicit file: + +``` +refusing to delete "src/*" (ambiguous or broad). Delete one explicit file. +``` + ### `bell` (opt in) Writes `\u0007` to stderr when a turn ends. Off by default — a bell after every turn is diff --git a/docs/skills.md b/docs/skills.md index 434a0cc..75bb6c8 100644 --- a/docs/skills.md +++ b/docs/skills.md @@ -3,9 +3,9 @@ A skill is a markdown file with instructions for one kind of task. Only its name and description sit in the system prompt; the body is loaded on demand. -That split matters. The nine bundled skills are roughly 14,000 characters of body against about -1,500 characters of catalogue — paid on every request. Putting every body in the prompt -would cost that on every turn, for instructions relevant to one turn in twenty. +That split matters. The twenty-nine bundled skills are tens of thousands of characters of body +against a small catalogue of names and descriptions — paid on every request. Putting every body +in the prompt would cost that on every turn, for instructions relevant to one turn in twenty. ## Format @@ -88,9 +88,18 @@ thing at a time, and stop at a target stated up front. Report the baseline along CI, Dockerfiles, and docs), apply one shape of change rather than improving as you pass, and never hand-merge a lockfile. -They are string constants in `src/skills-builtin.ts` rather than files, because -`bun build --compile` only embeds modules reachable through imports. A directory of `.md` -files would be missing from the shipped binary. +**`plan`** — break a non-trivial task into an ordered, verifiable sequence before writing code: +order by dependency rather than by file, one step one verifiable outcome, keep it small, and +replan when the ground moves. + +**`docs`** — write documentation grounded in the source: verify every claim against the code, +answer the reader's actual question, show a working example before describing one, and match +the house style. + +They are Markdown files in `src/skills-md/`, one per skill, loaded by `src/skills-builtin.ts` +as Bun raw-text imports. The `.md` file is the single source of truth — frontmatter and body +in proper Markdown — and Bun inlines every text import into the compiled binary, so the folder +ships with `bun build --compile` rather than being left behind on disk. ## How the agent uses one diff --git a/docs/tools.md b/docs/tools.md index 5ba53e3..eb20a15 100644 --- a/docs/tools.md +++ b/docs/tools.md @@ -46,7 +46,7 @@ Three more things sit around the rules: ## Tool sets Each tool costs its name, its description, and its JSON schema on **every request**. The current -registry has nineteen built-ins. `/tools` shows the live set; disabling an optional set removes +registry has forty-one built-ins. `/tools` shows the live set; disabling an optional set removes its schemas from both the request and the system prompt. | Tool | Bytes | Tool | Bytes | @@ -68,6 +68,8 @@ Sets let you switch off what a project does not need: |---|---|---| | `core` | `read_file` `write_file` `edit_file` `glob` `grep` `bash` | ~2,993 B | | `edit-plus` | `multi_edit` `list_dir` `read_many_files` `apply_patch` `move_file` `delete_file` | patch and file ops | +| `nav` | `find_symbol` `json_query` | navigation and structured reads | +| `extra` | 20 tools: line edits, fs inspect, git extensions, code/env reads | on by default | | `git` | `git_status` `git_diff` `git_log` `git_show` `git_blame` `git_branch` `git_commit_message` | ~2,180 B + message | | `net` | `web_fetch` | opt in | @@ -107,6 +109,59 @@ Both extra sets earn their place in most projects, but not all: - **Reading a lot, editing rarely?** Keep `edit-plus` for `list_dir` and `read_many_files` alone; they pay for themselves in round trips saved. +## The `extra` set + +Twenty tools across four families, on by default. Each follows the same rules as the core +tools: writes are jailed to the workspace, reads honour `.gitignore`, and every git call spawns +the binary with a fixed argument array, never a shell string. + +### Line edits + +Precise edits by line number, for changes that need no full-file rewrite and no exact-string +match. All refuse a path outside the workspace. + +| Tool | Does | +|---|---| +| `insert_lines` | Insert a block before a 1-based line, pushing the rest down. One past the end appends. | +| `delete_lines` | Delete an inclusive line range. Refuses the whole file — that is `delete_file`'s job. | +| `replace_lines` | Replace an inclusive line range with new text in one write. | +| `append_file` | Add text to the end of a file. | +| `prepend_file` | Add text to the top of a file, e.g. a header or import block. | +| `count_lines` | Line count for one file, or per file across a glob. A size read before opening something large. | + +### Filesystem + +| Tool | Does | +|---|---| +| `tree` | Indented directory tree, ignore-aware, directories first. A broad shape faster to scan than `list_dir`. | +| `file_info` | Size, line count, modified time, text-or-binary for one file. | +| `find_files` | Files whose *name* contains a substring (not a glob), e.g. `auth`. | +| `recent_files` | Files modified most recently, newest first. Find what a tool just touched. | +| `changed_files` | The working-tree delta git reports (modified, staged, untracked). | + +### Git extensions (read-only) + +Spawned with a fixed argv, so they are auto-approved like the core git tools. + +| Tool | Does | +|---|---| +| `git_log_file` | Commits that touched one file, newest first, with hash, date, subject. | +| `git_diff_commits` | Diff between two refs, optionally limited to one path. | +| `git_show_file` | A file's contents at a ref, e.g. `auth.ts` at `HEAD~3`. | +| `git_current_branch` | The current branch with its upstream and ahead/behind count. | +| `git_changed_in_ref` | Files changed between a ref and the working tree, names only. | + +### Code and environment + +| Tool | Does | +|---|---| +| `find_symbol` | Where a function, class, or type is *defined* across JS/TS, Python, Go, Rust. Matches declarations, not uses. | +| `json_query` | One value from a JSON file by dotted path (`scripts.build`), instead of reading it whole. | +| `outline` | Top-level declarations of a source file as a structural map. Read before opening a large file. | +| `read_symbol` | The full body of one top-level definition by name. | +| `env_info` | Platform, shell, and which runtimes and package managers are installed, before writing a command. | +| `count_tokens` | Estimate the token cost of a file or string (~4 chars per token) before sending it to the model. | + ## File tools ### `read_file` diff --git a/package.json b/package.json index f5d7ab5..efa22f8 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "shiro-neko", - "version": "0.1.0-beta.5", + "version": "1.0.0", "type": "module", "private": true, "bin": { @@ -16,6 +16,7 @@ }, "devDependencies": { "@types/bun": "^1.4.0", + "@types/pngjs": "^6.0.5", "@types/react": "19.2.18", "ink-testing-library": "4.0.0", "react-devtools-core": "^7.0.1" @@ -33,6 +34,7 @@ "ink-select-input": "6.2.0", "ink-spinner": "^5.0.0", "ink-text-input": "6.0.0", + "pngjs": "^7.0.0", "react": "19.2.8", "zod": "4.5.4" } diff --git a/src/autoload.ts b/src/autoload.ts new file mode 100644 index 0000000..17fd59f --- /dev/null +++ b/src/autoload.ts @@ -0,0 +1,186 @@ +import { tool, type ToolSet } from 'ai'; +import { homedir } from 'node:os'; +import { join } from 'node:path'; +import { z } from 'zod'; +import { jail } from './ignore'; +import { manifestToPlugin, parseManifest, type PluginManifest } from './registry'; +import type { Plugin } from './plugins'; + +/** + * Auto-registration and auto-loading of external skills, tools, and plugins. + * + * Everything here is *data*, never code — the same rule the registry enforces. + * An external tool is a bounded manifest (a shell template through the guard, an + * HTTP fetch, or a file read), an external plugin a refusal manifest, an external + * skill a markdown body. Loading arbitrary code from disk would let an entry read + * every file the agent can read and lie about what it blocks, so it is not offered. + * + * Directories, later shadowing earlier by name: + * ~/.shiro-neko/{tools,plugins,skills} (user) + * .shiro/{tools,plugins,skills} (project) + * Skills already load through skills.ts; this module adds tools and plugins and + * the one place cli turns them all on. + */ + +const home = () => process.env['SHIRO_HOME'] ?? homedir(); + +export type LoadError = { name: string; message: string }; + +function dirs(kind: 'tools' | 'plugins' | 'skills', cwd: string): string[] { + return [join(home(), '.shiro-neko', kind), join(cwd, '.shiro', kind)]; +} + +async function scan(dir: string, ext: string): Promise { + const files: string[] = []; + try { + for await (const f of new Bun.Glob(`*.${ext}`).scan({ cwd: dir, onlyFiles: true })) files.push(f); + } catch { + return []; + } + return files.sort(); +} + +const MAX_PATTERN = 200; +const nameSchema = z.string().min(1).max(40).regex(/^[a-z0-9][a-z0-9-_]*$/i); + +// --------------------------------------------------------------------------- +// External tools, as bounded manifests. +// --------------------------------------------------------------------------- + +/** + * Three kinds of tool, each with a ceiling on what it can do. None runs arbitrary + * code: `shell` interpolates a fixed template and runs it through the guard and + * the platform shell, `http` fetches a fixed URL, `read` returns a fixed file's + * contents (jailed to the workspace). The input is a single optional `arg` string + * substituted into a `{arg}` placeholder, so a manifest cannot take structure it + * was not declared for. + */ +const toolManifestSchema = z.object({ + name: nameSchema, + description: z.string().min(1).max(300), + kind: z.enum(['shell', 'http', 'read']), + /** The template with an optional `{arg}` placeholder. */ + command: z.string().max(500).optional(), + url: z.string().max(500).optional(), + path: z.string().max(300).optional(), + /** Set false to require approval before running. Default true (auto-approved). */ + autoApprove: z.boolean().optional(), +}); + +export type ToolManifest = z.infer; + +export function parseToolManifest(source: string): ToolManifest { + let raw: unknown; + try { + raw = JSON.parse(source); + } catch { + throw new Error('the tool manifest is not valid JSON'); + } + const parsed = toolManifestSchema.safeParse(raw); + if (!parsed.success) { + throw new Error(`the tool manifest is malformed: ${parsed.error.issues[0]?.message ?? 'unknown reason'}`); + } + const m = parsed.data; + if (m.kind === 'shell' && !m.command) throw new Error(`shell tool "${m.name}" needs a command template`); + if (m.kind === 'http' && !m.url) throw new Error(`http tool "${m.name}" needs a url`); + if (m.kind === 'read' && !m.path) throw new Error(`read tool "${m.name}" needs a path`); + return m; +} + +const MAX_TOOL_OUTPUT = 30_000; +const cap = (s: string) => (s.length <= MAX_TOOL_OUTPUT ? s : `${s.slice(0, MAX_TOOL_OUTPUT)}\n... [truncated]`); + +/** The guard an external shell tool runs through, supplied by cli so it shares the real chain. */ +export type ShellGuard = (command: string) => Promise; + +/** + * A manifest as a live tool. The guard is applied to every `shell` invocation, so + * an external tool cannot smuggle a destructive command past the user any more + * than a built-in bash call can. + */ +export function manifestToTool(manifest: ToolManifest, guard: ShellGuard) { + const inputSchema = z.object({ arg: z.string().optional().describe('optional argument substituted into {arg}') }); + const substitute = (template: string, arg: string) => template.replaceAll('{arg}', arg); + + return tool({ + description: `${manifest.description} (external ${manifest.kind} tool)`, + inputSchema, + execute: async ({ arg = '' }) => { + if (manifest.kind === 'read') { + const abs = jail(substitute(manifest.path!, arg)); + const file = Bun.file(abs); + if (!(await file.exists())) throw new Error(`no such file: ${manifest.path}`); + return cap(await file.text()); + } + + if (manifest.kind === 'http') { + const url = substitute(manifest.url!, arg); + if (!/^https:\/\//i.test(url)) throw new Error(`http tools may only fetch https URLs, got: ${url}`); + const res = await fetch(url, { redirect: 'follow', signal: AbortSignal.timeout(20_000) }); + if (!res.ok) throw new Error(`${url} returned ${res.status}`); + return cap(await res.text()); + } + + const command = substitute(manifest.command!, arg); + const blocked = await guard(command); + if (blocked) throw new Error(`refused: ${blocked}`); + const shell = process.platform === 'win32' ? ['cmd', '/c', command] : ['bash', '-lc', command]; + const proc = Bun.spawn(shell, { stdout: 'pipe', stderr: 'pipe' }); + const [out, err, code] = await Promise.all([ + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + proc.exited, + ]); + if (code !== 0) throw new Error(`exited ${code}: ${err.trim().slice(0, 300)}`); + return cap(out.trim() || '(no output)'); + }, + }); +} + +export type ExternalTools = { tools: ToolSet; autoApprove: string[]; errors: LoadError[] }; + +/** Loads every external tool manifest, project shadowing user by name. Bad files are reported and skipped. */ +export async function loadExternalTools(cwd: string, guard: ShellGuard): Promise { + const tools: ToolSet = {}; + const autoApprove: string[] = []; + const errors: LoadError[] = []; + + for (const dir of dirs('tools', cwd)) { + for (const file of await scan(dir, 'json')) { + const fallback = file.replace(/\.json$/i, ''); + try { + const manifest = parseToolManifest(await Bun.file(join(dir, file)).text()); + tools[manifest.name] = manifestToTool(manifest, guard); + if (manifest.autoApprove !== false) autoApprove.push(manifest.name); + } catch (e) { + errors.push({ name: fallback, message: e instanceof Error ? e.message : String(e) }); + } + } + } + return { tools, autoApprove, errors }; +} + +// --------------------------------------------------------------------------- +// External plugins, as refusal manifests (same shape the registry installs). +// --------------------------------------------------------------------------- + +export type ExternalPlugins = { plugins: Plugin[]; errors: LoadError[] }; + +/** Loads refusal-manifest plugins from disk, merging with any already installed via the registry. */ +export async function loadExternalPlugins(cwd: string): Promise { + const byName = new Map(); + const errors: LoadError[] = []; + + for (const dir of dirs('plugins', cwd)) { + for (const file of await scan(dir, 'json')) { + const fallback = file.replace(/\.json$/i, ''); + try { + const manifest: PluginManifest = parseManifest(await Bun.file(join(dir, file)).text()); + byName.set(manifest.name, manifestToPlugin(manifest)); + } catch (e) { + errors.push({ name: fallback, message: e instanceof Error ? e.message : String(e) }); + } + } + } + return { plugins: [...byName.values()], errors }; +} diff --git a/src/cli.tsx b/src/cli.tsx index 7ec69ca..6a0bef1 100644 --- a/src/cli.tsx +++ b/src/cli.tsx @@ -3,6 +3,7 @@ import { render } from 'ink'; import React from 'react'; import type { LanguageModel, ModelMessage } from 'ai'; import { resolveAgent, VARIANTS, isThinkingLevel, type AgentVariant } from './agents'; +import { loadExternalPlugins, loadExternalTools } from './autoload'; import { configPath, loadConfig, missingKeyMessage, resolveModel, writeConfigFile, type Config } from './config'; import type { FallbackEvent } from './fallback'; import { farewell } from './farewell'; @@ -18,12 +19,14 @@ import { createHost } from './plugins'; import { fetchModels, presetById } from './providers'; import * as registry from './registry'; import { Session } from './session'; +import { loadCustomCommands } from './custom-commands'; import { loadSkills } from './skills'; import * as store from './store'; import { createTaskTool, type SubagentApproval } from './subagent'; import { VERSION, versionLine } from './version'; import { createAskBridge } from './ui/Ask'; import { App, createApprovalBridge, createNoticeBus, createSubagentBus, type AppHooks } from './ui/App'; +import { Header, type HeaderFact } from './ui/Header'; import type { RegistryRow as AppRegistryRow } from './ui/Panels'; // SDK warnings go straight to stderr, which tears up the Ink render. @@ -154,6 +157,7 @@ if (resumeArg) { const mcp = has('--no-mcp') || !cfg.mcpServers ? undefined : await connectMcp(cfg.mcpServers); const instructions = has('--no-instructions') ? [] : await loadInstructions(); const skills = has('--no-skills') ? [] : await loadSkills(); +const customCommands = await loadCustomCommands(); const promptHistory = await store.loadHistory(); const installedPlugins = has('--no-plugins') ? { plugins: [], errors: [] } : await registry.loadInstalledPlugins(); @@ -202,9 +206,29 @@ const enabledPlugins = has('--no-plugins') ? [] : (cfg.plugins ?? DEFAULT_ENABLE const pluginErrors = enabledPlugins .filter((name) => !BUILTIN_PLUGINS.some((p) => p.name === name)) .map((name) => ({ plugin: name, message: 'no such plugin' })); + +// External skills, tools, and plugins auto-load from ~/.shiro-neko/ and +// .shiro/. All are data, never code; a bad file is reported, not fatal. +const externalPlugins = has('--no-plugins') ? { plugins: [], errors: [] } : await loadExternalPlugins(process.cwd()); + const plugins = createHost( - [...BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)), ...installedPlugins.plugins], - [...pluginErrors, ...installedPlugins.errors], + [ + ...BUILTIN_PLUGINS.filter((p) => enabledPlugins.includes(p.name)), + ...installedPlugins.plugins, + ...externalPlugins.plugins, + ], + [ + ...pluginErrors, + ...installedPlugins.errors, + ...externalPlugins.errors.map((e) => ({ plugin: e.name, message: e.message })), + ], +); + +// External shell tools run through the same guard chain as a built-in bash call, +// so an installed tool cannot do what the agent itself may not. Late-bound because +// the host above is what runs the chain. +const externalTools = await loadExternalTools(process.cwd(), async (command) => + plugins.guard({ toolName: 'bash', input: { command }, cwd: process.cwd() }), ); const memory = has('--no-memory') ? undefined : new Memory(process.cwd(), languageModel); @@ -261,8 +285,23 @@ const subagentGate: SubagentApproval = (req) => { return approveSubagent(req); }; +// A subagent doing search rather than reasoning can run on a cheaper model. +// It resolves against the same provider and key, so a configured `subagentModel` +// never needs a second credential. +const subagentModel = + cfg.subagentModel && cfg.subagentModel !== cfg.model && cfg.apiKey + ? resolveModel({ ...cfg, model: cfg.subagentModel }, reportFallback) + : (languageModel ?? unconfiguredModel); + +// Late-bound like `approveSubagent`: the task tool is built into `extraTools` +// before the Session that owns the spend ledger exists, so the usage callback is +// wired after construction. +let recordSubagent: (usage: { inputTokens: number; outputTokens: number }) => void = () => {}; + const session = new Session({ model: languageModel ?? unconfiguredModel, + modelId: cfg.model, + ...(cfg.subagentModel ? { subagentModelId: cfg.subagentModel } : {}), askApproval: bridge.ask, yolo, instructions, @@ -276,8 +315,10 @@ const session = new Session({ ...(memory ? { memory } : {}), ...(record.notebook ? { notebook: record.notebook } : {}), ...(cfg.maxRetries !== undefined ? { maxRetries: cfg.maxRetries } : {}), + ...(cfg.maxSpendUsd !== undefined ? { maxSpendUsd: cfg.maxSpendUsd } : {}), extraTools: { ...(mcp?.tools ?? {}), + ...externalTools.tools, git_commit_message: createCommitMessageTool({ model: languageModel ?? unconfiguredModel, ...(headless ? {} : { cwd: process.cwd() }), @@ -287,6 +328,9 @@ const session = new Session({ : { task: createTaskTool({ model: languageModel ?? unconfiguredModel, + subagentModel, + subagentModelId: cfg.subagentModel, + onUsage: (u) => recordSubagent(u), ...(headless ? {} : { report: subagents.emit }), // A worker's writes go through the parent's rules and the parent's // prompt. Headless has nobody to answer, so `worker` is withheld there @@ -295,7 +339,7 @@ const session = new Session({ }), }), }, - autoApprove: ['task', 'git_commit_message'], + autoApprove: ['task', 'git_commit_message', ...externalTools.autoApprove], messages: [...record.messages], onChange: (messages) => { // Debounced so a long tool loop does not hit the disk on every step. @@ -305,6 +349,7 @@ const session = new Session({ }); approveSubagent = session.approveForSubagent(); +recordSubagent = (u) => session.recordSubagentUsage(u); async function shutdown(code: number): Promise { clearTimeout(saveTimer); @@ -344,6 +389,7 @@ const hooks: AppHooks = { for await (const rel of walk({ limit: 5000 })) found.push(rel); return found; }, + customCommands: () => customCommands, registry: { list: async () => { const entries = await registry.fetchIndex(cfg.registryUrl); @@ -548,35 +594,46 @@ const hooks: AppHooks = { }, }; -const header = [ - needsProvider - ? `shiro-neko ${VERSION} no provider configured` - : `shiro-neko ${VERSION} ${cfg.provider}/${record.model} session ${record.id.slice(0, 8)}`, - `agent: ${agentVariant.name} thinking: ${agentVariant.thinking}`, - `cwd: ${process.cwd()}`, - restored ? `resumed ${record.messages.length} messages` : undefined, +// The welcome dashboard's environment facts, in scan order. Anything that should +// stop the user — a failed plugin, `--yolo`, a missing key — is given a tone so it +// lifts out of the quiet metadata rather than blending into it. +const facts: HeaderFact[] = [ + { label: 'agent', value: `${agentVariant.name} thinking ${agentVariant.thinking}` }, + restored ? { label: 'resumed', value: `${record.messages.length} messages` } : undefined, instructions.length > 0 - ? `instructions: ${instructions.map((i) => i.path.split(/[\\/]/).at(-1)).join(', ')}` - : 'no AGENTS.md found - /init writes one', - skills.length > 0 ? `skills: ${skills.map((s) => s.name).join(', ')}` : undefined, - plugins.plugins.length > 0 ? `plugins: ${plugins.plugins.map((p) => p.name).join(', ')}` : undefined, - ...plugins.errors.map((e) => `plugin ${e.plugin}: ${e.message}`), - memory && memory.all().length > 0 ? `memory: ${memory.all().length} notes about this project` : undefined, - mcp && Object.keys(mcp.tools).length > 0 ? `mcp: ${Object.keys(mcp.tools).length} tools` : undefined, - !mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0 - ? `mcp: ${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)` + ? { label: 'instructions', value: instructions.map((i) => i.path.split(/[\\/]/).at(-1)!).join(', ') } + : { label: 'instructions', value: 'none - /init writes an AGENTS.md', tone: 'info' }, + skills.length > 0 ? { label: 'skills', value: skills.map((s) => s.name).join(', ') } : undefined, + plugins.plugins.length > 0 + ? { label: 'plugins', value: plugins.plugins.map((p) => p.name).join(', ') } : undefined, - ...(mcp?.errors ?? []).map((e) => `mcp ${e.server} failed: ${e.message}`), + ...plugins.errors.map((e) => ({ label: 'plugin error', value: `${e.plugin}: ${e.message}`, tone: 'err' as const })), + memory && memory.all().length > 0 + ? { label: 'memory', value: `${memory.all().length} notes about this project` } + : undefined, + mcp && Object.keys(mcp.tools).length > 0 ? { label: 'mcp', value: `${Object.keys(mcp.tools).length} tools` } : undefined, + !mcp && cfg.mcpServers && Object.keys(cfg.mcpServers).length > 0 + ? { label: 'mcp', value: `${Object.keys(cfg.mcpServers).length} configured, not connected (--no-mcp)`, tone: 'warn' as const } + : undefined, + ...(mcp?.errors ?? []).map((e) => ({ label: 'mcp error', value: `${e.server}: ${e.message}`, tone: 'err' as const })), yolo - ? 'approvals: OFF (--yolo), but deny rules and the guard still apply' + ? { label: 'approvals', value: 'OFF (--yolo) - deny rules and the guard still apply', tone: 'warn' as const } : cfg.permission - ? `approvals: rules for ${Object.keys(cfg.permission).join(', ')}, defaults elsewhere` - : 'approvals: ask for write_file, edit_file, multi_edit, apply_patch, move_file, delete_file, bash, web_fetch, mcp__*', - cfg.toolSets ? `tool sets: core, ${cfg.toolSets.join(', ')}` : undefined, - '/help for commands', -] - .filter(Boolean) - .join('\n'); + ? { label: 'approvals', value: `rules for ${Object.keys(cfg.permission).join(', ')}, defaults elsewhere` } + : { label: 'approvals', value: 'ask for writes, bash, web_fetch, mcp' }, + cfg.toolSets ? { label: 'tool sets', value: `core, ${cfg.toolSets.join(', ')}` } : undefined, +].filter((f): f is HeaderFact => f !== undefined); + +const headerNode = ( +
+); // ctrl-c has to reach the App: with a command running it kills that command and // keeps the turn. Ink's own handler would exit the process before we saw the key. @@ -584,7 +641,9 @@ const app = render(