The compaction bug, which is the important one:
beta.2 taught the pruner to drop any assistant part whose reasoning item it had
removed. That was right about the 400 and wrong about everything else. On a
reasoning model every tool call carries a provider itemId, so past the threshold
the model could no longer see what it had already run, and re-ran the same tools
until maxSteps ended the turn. Reproduced at 12 model calls for a job needing 4,
with nothing but the user message reaching the wire.
The dependency is not the part, it is the itemId. A part carrying one is
serialised as `{ type: 'item_reference', id }`, a pointer to an item stored
provider-side that depends on its reasoning item. Without the itemId the same
content goes out inline and carries no dependency at all. Verified against the
provider's own serialiser: `text` with an itemId becomes item_reference, the
identical part without one becomes output_text.
So `dropOrphanedItems` becomes `detachOrphanedItems`: strip the itemId, keep the
content. Compaction may shorten the history; it must not blank it. The new test
asserts behaviour rather than shape — the loop must end because the model chose
to, and every call after the first must still carry the earlier exchange. A shape
assertion passed the whole time the model was losing its memory.
Registry, via `/registry [list|search|add|remove|installed]`:
Skills and plugins are treated differently on purpose. A skill is prompt text, so
installing one puts a stranger's words into the system prompt of every future
session in this project; the install shows the body first and the origin is
recorded, so /skills always says where an instruction came from. A plugin is a
JSON manifest of deny rules, evaluated by compiled code identical for every
install. Loading TypeScript from a URL is declined outright: a plugin that can
block tool calls could otherwise lie about blocking them.
Validated before anything is written: https only (file: and data: rejected), name
matched against ^[a-z0-9][a-z0-9-]*$ so it cannot escape its directory, size
caps on index and body, every regex compiled, pattern length capped since it runs
on every tool call, and the body's own name checked against the index. Installed
skills rank below your own, so an install can never shadow a skill you wrote.
Interface:
- Context is a percentage of the compaction threshold, amber from two thirds and
red at 90. A turn about to lose history now says so beforehand.
- Aligned command menu and registry tables; /skills and /plugins name origins.
538 tests, up from 488. The registry is tested against a real local HTTP server,
and the guard is proven to refuse a .env write end to end rather than assumed to.
121 lines
4.1 KiB
TypeScript
121 lines
4.1 KiB
TypeScript
import { tool } from 'ai';
|
|
import { homedir } from 'node:os';
|
|
import { join } from 'node:path';
|
|
import { z } from 'zod';
|
|
import { BUILTIN_SKILLS } from './skills-builtin';
|
|
|
|
export type SkillOrigin = 'builtin' | 'registry' | 'user' | 'project';
|
|
|
|
export type Skill = {
|
|
name: string;
|
|
description: string;
|
|
origin: SkillOrigin;
|
|
path?: string;
|
|
body: string;
|
|
};
|
|
|
|
const MAX_BODY = 20_000;
|
|
|
|
/**
|
|
* Minimal YAML frontmatter reader: `name` and `description` only.
|
|
* A real YAML parser would be a dependency for two string fields.
|
|
*/
|
|
export function parseSkill(source: string, origin: SkillOrigin, path?: string): Skill | undefined {
|
|
const match = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/.exec(source.trimStart());
|
|
if (!match) return undefined;
|
|
|
|
const meta: Record<string, string> = {};
|
|
for (const line of match[1]!.split(/\r?\n/)) {
|
|
const kv = /^([A-Za-z_-]+)\s*:\s*(.*)$/.exec(line.trim());
|
|
if (kv) meta[kv[1]!.toLowerCase()] = kv[2]!.replace(/^["']|["']$/g, '').trim();
|
|
}
|
|
|
|
const name = meta['name'];
|
|
const description = meta['description'];
|
|
if (!name || !description) return undefined;
|
|
|
|
return { name, description, origin, ...(path ? { path } : {}), body: match[2]!.trim().slice(0, MAX_BODY) };
|
|
}
|
|
|
|
/**
|
|
* Precedence, low to high. Installed skills sit below your own on purpose: a skill
|
|
* you wrote must never be shadowed by one fetched from a registry.
|
|
*/
|
|
const skillDirs = (cwd: string) => {
|
|
const home = join(process.env['SHIRO_HOME'] ?? homedir(), '.shiro-neko');
|
|
return [
|
|
{ dir: join(home, 'registry', 'skills'), origin: 'registry' as const },
|
|
{ dir: join(home, 'skills'), origin: 'user' as const },
|
|
{ dir: join(cwd, '.shiro', 'skills'), origin: 'project' as const },
|
|
];
|
|
};
|
|
|
|
/**
|
|
* Builtin, then installed, then user, then project. Later wins, so a project can
|
|
* override a bundled or installed skill by using the same name — and a skill you
|
|
* wrote yourself always beats one fetched from a registry.
|
|
*/
|
|
export async function loadSkills(cwd = process.cwd()): Promise<Skill[]> {
|
|
const byName = new Map<string, Skill>();
|
|
|
|
for (const { name, source } of BUILTIN_SKILLS) {
|
|
const skill = parseSkill(source, 'builtin');
|
|
if (skill) byName.set(skill.name, skill);
|
|
else byName.delete(name);
|
|
}
|
|
|
|
for (const { dir, origin } of skillDirs(cwd)) {
|
|
let files: string[] = [];
|
|
try {
|
|
for await (const f of new Bun.Glob('*.md').scan({ cwd: dir, onlyFiles: true })) files.push(f);
|
|
} catch {
|
|
continue;
|
|
}
|
|
for (const file of files.sort()) {
|
|
const path = join(dir, file);
|
|
try {
|
|
const skill = parseSkill(await Bun.file(path).text(), origin, path);
|
|
if (skill) byName.set(skill.name, skill);
|
|
} catch {
|
|
continue;
|
|
}
|
|
}
|
|
}
|
|
|
|
return [...byName.values()].sort((a, b) => a.name.localeCompare(b.name));
|
|
}
|
|
|
|
/**
|
|
* Catalogue for the system prompt: names and one-line descriptions only.
|
|
* Bodies stay out of context until the model asks, which is the point.
|
|
*/
|
|
export function renderSkills(skills: Skill[]): string {
|
|
if (skills.length === 0) return '';
|
|
const lines = skills.map((s) => `- ${s.name}: ${s.description}`);
|
|
return [
|
|
'',
|
|
'Skills available through the skill tool. Load one when its description matches the task,',
|
|
'before you start working, and follow it as if the user had written it:',
|
|
...lines,
|
|
].join('\n');
|
|
}
|
|
|
|
export function createSkillTool(skills: Skill[]) {
|
|
const names = skills.map((s) => s.name);
|
|
return tool({
|
|
description:
|
|
'Load a skill: detailed instructions for one kind of task. Call it as soon as a skill description matches ' +
|
|
`what you are about to do, then follow what it says. Available: ${names.join(', ') || 'none'}.`,
|
|
inputSchema: z.object({
|
|
name: z.string().describe('Skill name from the list in your instructions'),
|
|
}),
|
|
execute: async ({ name }) => {
|
|
const skill = skills.find((s) => s.name === name.trim().toLowerCase());
|
|
if (!skill) throw new Error(`No skill named "${name}". Available: ${names.join(', ') || 'none'}`);
|
|
return `Skill "${skill.name}" (${skill.origin}). Follow these instructions for this task.\n\n${skill.body}`;
|
|
},
|
|
});
|
|
}
|
|
|
|
export { skillDirs };
|