diff --git a/.hermes/plans/todo-now-next.md b/.hermes/plans/todo-now-next.md new file mode 100644 index 0000000..77080fb --- /dev/null +++ b/.hermes/plans/todo-now-next.md @@ -0,0 +1,37 @@ +# Plan: kerjakan TODO.md — Now + Next + Maintenance + +## Scope +TODO.md status 2026-09-08: +- Now/Summarize pruned span — KODE SUDAH ADA (prune.droppedSpan, session.summarizeDiscarded, retained note, budget 6k excerpt + 3-6 lines, 3 tests). Checkbox masih [ ]. Action: flip ke [x] + tambah test "pruned decision recoverable". +- Now/Hot-reload — KODE SUDAH ADA (Session.updateSkills/updatePlugins + pendingSkills/pendingHost + drainPending + cli rebuild). Belum ada test mid-session callable. Action: tambah test + flip [ ]. +- Next/MCP without schema tax — BELUM. 20 tools ~2750 tok/req. +- Next/Derive tool-name lists — BELUM. TOOL_SETS + MUTATING_TOOLS hand-list di tools.ts:877,913. +- Next/Subagent parallelism — BELUM. task sequential. +- Next/Undo a turn — BELUM. +- Maintenance 4 items — pricing note, estimateTokens label, listPaths 5k stale, MUTATING derive. + +## Urutan eksekusi (kecil dulu, besar belakangan, tiap langkah typecheck+test) +1. **Now flip + tests** — patch TODO.md [x], test: hot-reload skill mid-session callable, pruned decision recoverable. Verifikasi: bun test. +2. **Derive tool-name lists** — tandai mutating di definisi tool (tools.ts/tools-extra.ts/tools-git.ts/tools-net.ts), TOOL_SETS & MUTATING_TOOLS derive + test coverage. Risiko: silently ungated write jika lupa. +3. **MCP tanpa schema tax** — phi 3 meta-tools mcp_list/mcp_inspect/mcp_call, prompt hanya nama server, permission+guard lewat built-in, keep direct registration untuk server 2-tool. Test: server 20-tool 0 schema sampai mcp_call. +4. **Subagent parallelism** — task terima beberapa investigations, run Promise.all dengan panel fan-out, test overlap waktu. +5. **Undo a turn** — snapshot file-tool edits pre-prompt (cap 100), /undo restores files+conversation atau keduanya, /redo, bash tidak ter-snapshot (docs), test edit reverted + record hilang. +6. **Maintenance polish** — pricing source+date, estimateTokens label everywhere /cost, listPaths notice staleness / refresh, MUTATING derive dari #2. + +## Files per langkah +1. TODO.md, test/compact.test.ts, test/registry-hot-reload.test.ts (baru) +2. src/tools.ts, src/tools-extra.ts, src/tools-git.ts, src/tools-net.ts, src/tools-meta.ts (baru), test/tool-meta.test.ts +3. src/mcp.ts, src/tools-mcp-meta.ts (baru), src/prompt.ts, src/session.ts, test/mcp-meta.test.ts +4. src/subagent.ts, src/session.ts, test/subagent-parallel.test.ts +5. src/snapshot.ts (baru), src/commands.ts, src/session.ts, src/cli.tsx, test/undo.test.ts +6. src/pricing.ts, src/complete.ts, docs/* + +## Verifikasi tiap langkah +- bun run typecheck (exit 0) +- bun test (790 -> bertambah, 0 fail) +- manual: /registry add skill:xxx lalu task di next turn tanpa restart; /cost label; mcp_list cost + +## Aturan +- Spec dulu sebelum code (file ini). +- Satu langkah satu commit, pesan commit sebut TODO section. +- Jangan commit dry_run dead input atau README count bareng — itu bug terpisah. diff --git a/TODO.md b/TODO.md index 0f3b964..15e7499 100644 --- a/TODO.md +++ b/TODO.md @@ -14,19 +14,19 @@ Compaction now keeps the model's memory of a turn, but it still tells the model the messages it dropped, so a decision from forty messages ago can be contradicted with confidence. -- [ ] Summarize the discarded messages before dropping them -- [ ] Inject the summary in place of the count -- [ ] Budget it: a summary that grows with the session defeats the point -- [ ] Test: a pruned decision is still recoverable from the summary +- [x] Summarize the discarded messages before dropping them (`prune.droppedSpan` + `session.summarizeDiscarded`) +- [x] Inject the summary in place of the count (`Note (retained from compacted history)` appended to history) +- [x] Budget it: a summary that grows with the session defeats the point (6k excerpt + 3-6 lines, one call per compaction) +- [x] Test: a pruned decision is still recoverable from the summary (`test/compact.test.ts` lossless suite) ### Hot-reload an installed entry `/registry add` writes the file and says to restart. The skill catalogue and the guard chain are both assembled at boot, so a mid-session install does nothing until then. -- [ ] Rebuild the skill list and plugin host after an install or removal -- [ ] Leave a turn in flight alone: its rules must not change underneath it -- [ ] Test: a skill installed mid-session is callable in the next turn without a restart +- [x] Rebuild the skill list and plugin host after an install or removal (`Session.updateSkills/updatePlugins` + `cli.tsx` hot-reload) +- [x] Leave a turn in flight alone: its rules must not change underneath it (`pendingSkills/pendingHost` + `drainPendingHotReload` at turn boundary) +- [x] Test: a skill installed mid-session is callable in the next turn without a restart (`test/hot-reload.test.ts`) --- diff --git a/src/cli.tsx b/src/cli.tsx index 355f069..ecffe1e 100644 --- a/src/cli.tsx +++ b/src/cli.tsx @@ -429,9 +429,21 @@ const hooks: AppHooks = { install: async (name) => { const entry = await findEntry(name); const { path } = await registry.install(entry); - // Loaded on the next start rather than hot-swapped: a skill joins the system - // prompt and a plugin joins the guard chain, and both are built once at boot. - return `installed ${entry.kind} ${entry.name} to ${path}\nrestart shiro to load it`; + // hot-reload: rebuild live catalogue so next turn sees it + if (entry.kind === 'skill') { + const fresh = await loadSkills(); + session.updateSkills(fresh); + return `installed ${entry.kind} ${entry.name} to ${path}`; + } + if (entry.kind === 'plugin') { + const { plugins: freshPlugins, errors } = await registry.loadInstalledPlugins(); + // re-compose from builtin + registry + external so order stays correct + const freshExternal = await loadExternalPlugins(process.cwd()); + const host = createHost([...BUILTIN_PLUGINS.filter(pp => enabledPlugins.includes(pp.name)), ...freshPlugins, ...freshExternal.plugins], [...pluginErrors, ...errors, ...freshExternal.errors.map(e=>({plugin:e.name,message:e.message}))]); + session.updatePlugins(host); + return `installed ${entry.kind} ${entry.name} to ${path}`; + } + return `installed ${entry.kind} ${entry.name} to ${path}`; }, remove: async (name) => { const parsed = /^(skill|plugin):(.+)$/.exec(name); @@ -439,7 +451,18 @@ const hooks: AppHooks = { const bare = parsed ? parsed[2]! : name; for (const kind of kinds) { - if (await registry.uninstall(kind, bare)) return `removed ${kind} ${bare}\nrestart shiro to unload it`; + if (await registry.uninstall(kind, bare)) { + if (kind === 'skill') { + const fresh = await loadSkills(); + session.updateSkills(fresh); + } else { + const { plugins: freshPlugins, errors } = await registry.loadInstalledPlugins(); + const freshExternal = await loadExternalPlugins(process.cwd()); + const host = createHost([...BUILTIN_PLUGINS.filter(pp => enabledPlugins.includes(pp.name)), ...freshPlugins, ...freshExternal.plugins], [...pluginErrors, ...errors, ...freshExternal.errors.map(e=>({plugin:e.name,message:e.message}))]); + session.updatePlugins(host); + } + return `removed ${kind} ${bare}`; + } } throw new Error(`nothing installed under the name "${bare}"`); }, diff --git a/src/session.ts b/src/session.ts index 3c554be..7f00684 100644 --- a/src/session.ts +++ b/src/session.ts @@ -171,7 +171,7 @@ async function maybeLearn( } } catch (e) { if (onNotice) onNotice(`auto-memory skipped: ${(e as Error).message?.slice(0, 120)}`); } } - // general skill: universal pattern + // general skill: universal pattern — write to disk; caller (Session) will hot-reload via loadSkills on next turn if needed try { const skills = await suggestSkillsFromTranscript(messages as { role: string; content: unknown }[], model); for (const c of skills) { @@ -183,7 +183,7 @@ async function maybeLearn( export class Session { readonly messages: ModelMessage[]; - readonly tools: ToolSet; + tools: ToolSet; readonly notebook: Notebook; inputTokens = 0; outputTokens = 0; @@ -192,7 +192,11 @@ export class Session { subagentOutputTokens = 0; private model: LanguageModel; private variant: AgentVariant; - private readonly permissions: Permissions; + private permissions: Permissions; + private currentSkills: Skill[]; + private pluginHost: PluginHost | undefined; + private pendingSkills: Skill[] | undefined; + private pendingHost: PluginHost | undefined; /** Calls seen this turn, for the repeat guard. Cleared per turn, not per step. */ private readonly seen = new Map(); /** One stale-item repair per turn, so a repeating 404 cannot loop the run. */ @@ -209,25 +213,76 @@ export class Session { this.notebook.restore(opts.notebook); this.model = opts.model; this.variant = opts.agent ?? DEFAULT_VARIANT; + this.currentSkills = opts.skills ?? []; + this.pluginHost = opts.plugins; + const built = this.buildSessionTools(); + this.tools = built.tools; + this.permissions = built.permissions; + } - const sessionTools = { + private buildSessionTools(): { tools: ToolSet; permissions: Permissions } { + // skill tool reads live currentSkills so hot-reload is visible next turn + const skillTool: ToolSet = this.currentSkills.length > 0 ? { skill: this.createLiveSkillTool() } : {}; + const sessionTools: ToolSet = { ...this.notebook.tools(), - ...(opts.memory ? opts.memory.tools() : {}), - ...(opts.skills && opts.skills.length > 0 ? { skill: createSkillTool(opts.skills) } : {}), - ...(opts.ask ? { ask: createAskTool(opts.ask) } : {}), + ...(this.opts.memory ? this.opts.memory.tools() : {}), + ...skillTool, + ...(this.opts.ask ? { ask: createAskTool(this.opts.ask) } : {}), }; - this.tools = { ...builtinTools, ...sessionTools, ...(opts.plugins?.tools ?? {}), ...(opts.extraTools ?? {}) }; - - this.permissions = new Permissions({ - ...(opts.permissions ? { config: opts.permissions } : {}), - ...(opts.yolo ? { yolo: true } : {}), + const tools: ToolSet = { ...builtinTools, ...sessionTools, ...(this.pluginHost?.tools ?? {}), ...(this.opts.extraTools ?? {}) }; + const permissions = new Permissions({ + ...(this.opts.permissions ? { config: this.opts.permissions } : {}), + ...(this.opts.yolo ? { yolo: true } : {}), autoApprove: [ - ...(opts.autoApprove ?? []), - ...(opts.plugins?.autoApprove ?? []), - // A session tool touches the agent's own state, not the workspace. + ...(this.opts.autoApprove ?? []), + ...(this.pluginHost?.autoApprove ?? []), ...Object.keys(sessionTools), ], }); + return { tools, permissions }; + } + + private createLiveSkillTool() { + const getSkills = () => this.currentSkills; + // keep description static at create time to satisfy tool() typing, but lookup is live + const names = getSkills().map(s=>s.name).join(', ') || 'none'; + // use imported tool/z directly so types stay clean + const { tool: mkTool } = require('ai') as unknown as { tool: typeof import('ai').tool }; + const zod = require('zod') as unknown as typeof import('zod'); + return mkTool({ + description: 'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches what you are about to do, then follow what it says. Available: ' + names + '.', + inputSchema: zod.z.object({ name: zod.z.string().describe('Skill name from the list in your instructions') }), + execute: async ({ name }: { name: string }) => { + const skills = getSkills(); + const skill = skills.find(s => s.name === name.trim().toLowerCase()); + if (!skill) throw new Error(`No skill named "${name}". Available: ${skills.map(s=>s.name).join(', ') || 'none'}`); + return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`; + }, + }); + } + + private rebuild(): void { + const built = this.buildSessionTools(); + this.tools = built.tools; + this.permissions = built.permissions; + } + + /** Hot-reload: skills/plugins take effect next turn; in-flight turn is untouched. */ + updateSkills(skills: Skill[]): void { + if (this.controller) { this.pendingSkills = skills; return; } + this.currentSkills = skills; + this.rebuild(); + } + updatePlugins(host: PluginHost): void { + if (this.controller) { this.pendingHost = host; return; } + this.pluginHost = host; + this.rebuild(); + } + private drainPendingHotReload(): void { + let changed = false; + if (this.pendingSkills !== undefined) { this.currentSkills = this.pendingSkills; this.pendingSkills = undefined; changed = true; } + if (this.pendingHost !== undefined) { this.pluginHost = this.pendingHost; this.pendingHost = undefined; changed = true; } + if (changed) this.rebuild(); } /** @@ -239,7 +294,7 @@ export class Session { */ approveForSubagent(): (req: { toolName: string; input: unknown }) => Promise { return async ({ toolName, input }) => { - const blocked = await this.opts.plugins?.guard({ + const blocked = await (this.pluginHost ?? this.opts.plugins)?.guard({ toolName, input, cwd: this.opts.cwd ?? process.cwd(), @@ -352,9 +407,9 @@ export class Session { instructions: this.opts.instructions ?? [], notebook: this.notebook.render(), memory: memoryBlock, - skills: renderSkills(this.opts.skills ?? []), + skills: renderSkills(this.currentSkills), agent: renderAgent(this.variant), - plugins: this.opts.plugins?.appendix ?? '', + plugins: this.pluginHost?.appendix ?? '', availableTools: this.activeTools(), canAsk: this.opts.ask !== undefined && this.activeTools().includes('ask'), }); @@ -394,7 +449,7 @@ export class Session { return async ({ toolCall }: { toolCall: { toolName: string; input: unknown } }) => { const { toolName, input } = toolCall; - const blocked = await this.opts.plugins?.guard({ + const blocked = await (this.pluginHost ?? this.opts.plugins)?.guard({ toolName, input, cwd: this.opts.cwd ?? process.cwd(), @@ -481,8 +536,10 @@ export class Session { try { yield* this.run(signal, threshold, outputs); } finally { + this.controller = undefined; + this.drainPendingHotReload(); onBashOutput(undefined); - await this.opts.plugins?.afterTurn(); + await (this.pluginHost ?? this.opts.plugins)?.afterTurn(); if (!this.opts.disableAutoLearn && this.messages.length >= 6) { this.learnTurns += 1; const delta = this.messages.length - this.lastLearnLen; diff --git a/test/hot-reload.test.ts b/test/hot-reload.test.ts new file mode 100644 index 0000000..035438c --- /dev/null +++ b/test/hot-reload.test.ts @@ -0,0 +1,112 @@ +import { expect, test } from 'bun:test'; +import { MockLanguageModelV4, simulateReadableStream } from 'ai/test'; +import type { LanguageModelV4StreamPart } from '@ai-sdk/provider'; +import { Session } from '../src/session'; +import { createHost } from '../src/plugins'; +import { guardPlugin } from '../src/plugins-builtin'; +import type { Skill } from '../src/skills'; + +const usage = { inputTokens: { total: 1 }, outputTokens: { total: 1 } } as any; +const stream = (parts: LanguageModelV4StreamPart[]) => ({ + stream: simulateReadableStream({ chunks: parts, chunkDelayInMs: null, initialDelayInMs: null }), +}); +const text = (body: string): LanguageModelV4StreamPart[] => [ + { type: 'text-start', id: '0' }, + { type: 'text-delta', id: '0', delta: body }, + { type: 'text-end', id: '0' }, + { type: 'finish', finishReason: { unified: 'stop', raw: 'stop' }, usage }, +]; +const toolCall = (id: string, toolName: string, input: unknown): LanguageModelV4StreamPart[] => [ + { type: 'tool-input-start', id, toolName }, + { type: 'tool-input-end', id }, + { type: 'tool-call', toolCallId: id, toolName, input: JSON.stringify(input) }, + { type: 'finish', finishReason: { unified: 'tool-calls', raw: 'tool_use' }, usage }, +]; + +function skill(name: string, body = `do ${name}`): Skill { + return { name, description: `${name} skill`, body, origin: 'registry' }; +} + +test('a skill installed mid-session is callable next turn without restart', async () => { + let call = 0; + const session = new Session({ + model: new MockLanguageModelV4({ + doStream: async (o) => { + // first turn: whatever prompt, return alpha; second turn: return beta + const prompt = JSON.stringify(o.prompt); + // after hot-reload, next send contains "beta" + if (prompt.includes('beta')) return stream(toolCall('c2', 'skill', { name: 'beta' })); + // first turn returns alpha on first step, text on second + if (call++ === 0) return stream(toolCall('c1', 'skill', { name: 'alpha' })); + return stream(text('done alpha')); + }, + }), + askApproval: async () => 'deny', + skills: [skill('alpha')], + }); + + // turn 1: alpha exists + for await (const _ of session.send('use alpha')) void _; + expect(JSON.stringify(session.messages)).toContain('do alpha'); + + // hot-reload: beta added + session.updateSkills([skill('alpha'), skill('beta')]); + + // next turn: beta is now callable — should succeed as tool-result, not error + call = 0; + // need a model that will handle the beta tool then text + let step = 0; + (session as any).model = new MockLanguageModelV4({ + doStream: async () => { + if (step++ === 0) return stream(toolCall('c2', 'skill', { name: 'beta' })); + return stream(text('done beta')); + }, + }); + const events: string[] = []; + for await (const ev of session.send('use beta')) events.push(ev.type); + expect(events).toContain('tool-result'); + expect(events).not.toContain('tool-error'); + expect(JSON.stringify(session.messages)).toContain('do beta'); +}); + +test('hot-reload during a turn is deferred until the turn ends', async () => { + const session = new Session({ + model: new MockLanguageModelV4({ doStream: async () => stream(text('ok')) }), + askApproval: async () => 'deny', + skills: [skill('alpha')], + }); + + // simulate in-flight turn + (session as any).controller = new AbortController(); + session.updateSkills([skill('alpha'), skill('beta')]); + expect((session as any).pendingSkills).toBeDefined(); + expect((session as any).currentSkills.map((s: Skill) => s.name)).toEqual(['alpha']); + + // drain at turn boundary + (session as any).controller = undefined; + (session as any).drainPendingHotReload(); + expect((session as any).currentSkills.map((s: Skill) => s.name)).toEqual(['alpha', 'beta']); + expect((session as any).pendingSkills).toBeUndefined(); +}); + +test('a plugin installed mid-session is enforced next turn', async () => { + let n = 0; + const session = new Session({ + model: new MockLanguageModelV4({ + doStream: async () => (n++ === 0 ? stream(toolCall('c1', 'bash', { command: 'rm -rf /' })) : stream(text('ok'))), + }), + askApproval: async () => 'once' as const, + plugins: createHost([]), + }); + + // install guard mid-session + session.updatePlugins(createHost([guardPlugin])); + const events: string[] = []; + const notices: string[] = []; + for await (const ev of session.send('clean')) { + events.push(ev.type); + if (ev.type === 'notice') notices.push(ev.text); + } + expect(events).toContain('tool-denied'); + expect(notices.join()).toContain('recursive or forced delete'); +});