feat(tui): add new flow helpers for bash exit code handling and verification prompts

This commit is contained in:
asepharyana
2026-09-03 16:13:29 +07:00
parent 93e5464370
commit f98030f28a
5 changed files with 596 additions and 47 deletions
@@ -4,8 +4,22 @@ import {
conversationChars, conversationChars,
ErrorTracker, ErrorTracker,
truncateToolOutput, truncateToolOutput,
bashExitCode,
isToolFailure,
deriveVerifyCommand,
AgentTurnServiceImpl,
} from "./turn_service.ts"; } from "./turn_service.ts";
import { type ChatMessage, newConversation, systemMessage, userMessage, toolResultMessage } from "@zesdex/domain"; import {
type ChatMessage,
newConversation,
systemMessage,
userMessage,
assistantMessage,
toolResultMessage,
} from "@zesdex/domain";
import type { AgentTurnParams } from "@zesdex/agent";
import type { ToolExecutor } from "./index.ts";
import type { ProviderService } from "./ports.ts";
describe("truncateToolOutput", () => { describe("truncateToolOutput", () => {
it("short output is unchanged", () => { it("short output is unchanged", () => {
@@ -57,3 +71,196 @@ describe("conversationChars", () => {
expect(conversationChars(conv.messages)).toBe(3 + 11 + 6); expect(conversationChars(conv.messages)).toBe(3 + 11 + 6);
}); });
}); });
/* ── New flow helpers ─────────────────────────────────────────────── */
describe("bashExitCode / isToolFailure", () => {
it("extracts a non-zero exit code", () => {
expect(bashExitCode("boom\n\nExit code: 1 (1s)")).toBe(1);
expect(bashExitCode("done\n\nExit code: 0 (0.5s)")).toBe(0);
});
it("returns null when no exit-code line exists", () => {
expect(bashExitCode("plain output")).toBeNull();
});
it("isToolFailure flags bash non-zero exits as failures", () => {
expect(isToolFailure("bash", "nope\n\nExit code: 2 (1s)")).toBe(true);
expect(isToolFailure("bash", "ok\n\nExit code: 0 (1s)")).toBe(false);
});
it("isToolFailure still flags Error: prefixes for other tools", () => {
expect(isToolFailure("read", "Error: no such file")).toBe(true);
});
});
describe("deriveVerifyCommand", () => {
it("defaults to bun check+test for a Bun repo", () => {
// No real manifests in a control dir we guarantee to not exist.
expect(deriveVerifyCommand("/nonexistent-zesdex-dir")).toContain("bun");
});
});
/* ── runTurn flow tests (fakes, no network) ───────────────────────── */
interface ScriptedStep {
content?: string | null;
tools?: Array<{ name: string; args?: string; id?: string }>;
}
/** A ProviderService that replays scripted chatStream responses. */
function makeScriptedProvider(script: ScriptedStep[]): {
provider: ProviderService;
seen: Array<ChatMessage[]>;
} {
const seen: Array<ChatMessage[]> = [];
let step = 0;
const chatResponses: string[] = [];
const provider: ProviderService = {
async chat(_messages, _tools, _maxTokens, _temperature) {
// Used by compaction and review. For review tests we script below.
const text = chatResponses.shift() ?? "";
return { message: { role: "assistant", content: text || null }, usage: [10, 5] };
},
async chatStream(messages, _tools, _max, _temp, onEvent) {
// Snapshot a copy: runTurn freely mutates the live array (splice/compact).
seen.push([...messages]);
const s = script[Math.min(step, script.length - 1)] ?? { content: "done" };
step += 1;
if (s.content !== undefined && s.content !== null) onEvent({ kind: "token", content: s.content });
if (s.tools && s.tools.length > 0) {
const msg = assistantMessage(null) as ChatMessage;
msg.tool_calls = s.tools.map((t, i) => ({
id: t.id ?? `call-${i}`,
type: "function",
function: { name: t.name, arguments: t.args ?? "{}" },
}));
return { message: msg, usage: [10, 2] };
}
return {
message: { role: "assistant", content: s.content ?? null },
usage: [10, 2],
};
},
// expose a way for tests to script chat replies
} as ProviderService;
(provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => {
chatResponses.push(text);
};
return { provider, seen };
}
/** A ToolExecutor that echoes fixed outputs per tool. */
function fixedExecutor(outs: Record<string, string>): ToolExecutor {
return {
async execute(name) {
return outs[name] ?? "ok";
},
isParallelSafe() {
return false;
},
};
}
function buildParams(userText: string): {
params: AgentTurnParams;
events: Array<any>;
} {
const events: Array<any> = [];
const params: AgentTurnParams = {
messages: [userMessage(userText)],
session_dir: "/tmp/zesdex-flow-test",
workspace_roots: ["/nonexistent-zesdex-dir"], // no AGENTS.md/package.json side effects
turn_events: { push: (e) => events.push(e), drain: () => [] },
in_flight: { value: false },
abort: new AbortController(),
api_key: "k",
model: "m",
api_base: "https://x",
};
return { params, events };
}
/** Extract the `[Flow]` steering system message from a captured batch. */
function flowDirective(batch: ChatMessage[]): string {
const m = batch.find((x) => (x.content ?? "").includes("[Flow]"));
return m?.content ?? "";
}
describe("runTurn complexity steering", () => {
it("injects a complex-request directive for a complex prompt", async () => {
const { provider, seen } = makeScriptedProvider([
{ content: "final answer" },
]);
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
const { params } = buildParams("Please refactor the architecture across multiple files");
await svc.runTurn(params);
// The complexity directive system message must be present before the LLM.
const flow = flowDirective(seen[0] ?? []);
expect(flow).toContain("flagged as complex");
});
it("injects a keep-it-simple directive instead", async () => {
const { provider, seen } = makeScriptedProvider([{ content: "hi" }]);
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
const { params } = buildParams("what is 2+2");
await svc.runTurn(params);
const flow = flowDirective(seen[0] ?? []);
expect(flow).toContain("looks simple");
expect(flow).not.toContain("complex");
});
});
describe("runTurn verify-after-edit", () => {
it("injects a verify nudge after a write, cleared after a successful bash", async () => {
const { provider, seen } = makeScriptedProvider([
{ tools: [{ name: "write" }] },
{ tools: [{ name: "bash" }] },
{ content: "done" },
]);
const executor = fixedExecutor({ write: "wrote it", bash: "ok\n\nExit code: 0 (1s)" });
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
const { params } = buildParams("add a comment");
await svc.runTurn(params);
// The LLM call after the write should carry the verify nudge.
const second = seen[1] ?? [];
expect(second.some((m) => (m.content ?? "").includes("Run the verify command"))).toBe(true);
// The successful bash exits 0, which clears pendingVerify — so the nudge is
// injected exactly once, not repeated on every later call.
const finalBatch = seen[seen.length - 1] ?? [];
const nudgeCount = finalBatch.filter((m) => (m.content ?? "").includes("Run the verify command")).length;
expect(nudgeCount).toBe(1);
});
});
describe("runTurn convergence guard", () => {
it("stops after repeated identical read calls", async () => {
const script: ScriptedStep[] = [];
for (let i = 0; i < 6; i++) script.push({ tools: [{ name: "read" }] });
const { provider, seen } = makeScriptedProvider(script);
const executor = fixedExecutor({ read: "same content" });
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
const { params, events } = buildParams("read something");
await svc.runTurn(params);
const warned = events.some((e) => e.kind === "system_note" && /without any progress|repeated/.test(e.message ?? ""));
expect(warned).toBe(true);
// Not every call got issued — the guard broke early.
expect(seen.length).toBeLessThan(script.length);
});
});
describe("runTurn self-review", () => {
it("feeds a reviewer critique back as a system message after a mutator", async () => {
const { provider, seen } = makeScriptedProvider([
{ tools: [{ name: "edit" }] },
{ content: "fixed" },
]);
const withChat = provider as unknown as { setChatReply(s: string): void };
withChat.setChatReply("- [PRIORITY: high] handle empty input in parse()");
const executor = fixedExecutor({ edit: "edited" });
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
const { params, events } = buildParams("fix parse()");
await svc.runTurn(params);
// The reviewer critique must have reached the next LLM call's history.
const second = seen[1] ?? [];
expect(second.some((m) => (m.content ?? "").includes("[Reviewer]"))).toBe(true);
// review_usage emitted.
expect(events.some((e) => e.kind === "review_usage")).toBe(true);
});
});
+179 -5
View File
@@ -20,7 +20,9 @@ import {
mainAgentPromptWithProjectContext, mainAgentPromptWithProjectContext,
compactionPrompt, compactionPrompt,
errorRecoveryNote, errorRecoveryNote,
reviewerPrompt,
} from "@zesdex/agent"; } from "@zesdex/agent";
import { isComplexRequest } from "@zesdex/workflow";
import type { ProviderService } from "./ports.ts"; import type { ProviderService } from "./ports.ts";
import type { ToolExecutor } from "./index.ts"; import type { ToolExecutor } from "./index.ts";
@@ -38,11 +40,80 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000;
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"]; const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
const COMPACT_KEEP_TAIL = 6; const COMPACT_KEEP_TAIL = 6;
/** Tools that mutate the filesystem — after these, a verify run is expected. */
const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]);
/** Consecutive identical, non-progressing tool iterations before the loop stops. */
const MAX_NO_PROGRESS_STREAK = 4;
/** Self-review is bounded to this many passes per turn. */
const MAX_REVIEW_PASSES = 1;
/** Whether the output string denotes a tool error. */ /** Whether the output string denotes a tool error. */
function isErrorOutput(output: string): boolean { function isErrorOutput(output: string): boolean {
return output.startsWith("Error:"); return output.startsWith("Error:");
} }
/** Extract the numeric exit code from a `bash` tool result, or null if not a failure (0). */
export function bashExitCode(output: string): number | null {
const match = output.match(/Exit code:\s*(\d+)/);
if (!match) return null;
const code = Number(match[1]);
return Number.isInteger(code) && code >= 0 ? code : null;
}
/**
* Whether a tool result is a real failure. `Error:` prefixes cover most tools;
* a `bash` command that exits non-zero returns an `Exit code: N` line instead.
*/
export function isToolFailure(toolName: string, output: string): boolean {
if (isErrorOutput(output)) return true;
if (toolName === "bash" || toolName === "bash_output") {
const code = bashExitCode(output);
if (code === null) return false; // no exit-code line → no signal
return code !== 0;
}
return false;
}
/**
* Best-effort derivation of the repository's verify command from convention
* manifests. Defaults to a safe lint+test invocation for Bun.
*/
export function deriveVerifyCommand(root: string): string {
const fs = require("node:fs");
// Bun project: prefer an explicit `check` (typecheck) script then test.
try {
const pkg = JSON.parse(fs.readFileSync(`${root}/package.json`, "utf8")) as {
scripts?: Record<string, string>;
};
const s = pkg?.scripts ?? {};
const parts: string[] = [];
if (s["check"] && typeof s["check"] === "string") parts.push(`bun run check`);
else if (s["typecheck"] && typeof s["typecheck"] === "string") parts.push(`bun run typecheck`);
if (s["lint"] && typeof s["lint"] === "string") parts.push(`bun run lint`);
if (s["test"] && typeof s["test"] === "string") parts.push(`bun run test`);
if (parts.length > 0) return parts.join(" && ");
} catch {
/* no package.json — fall through */
}
// Rust project.
try {
if (fs.existsSync(`${root}/Cargo.toml`)) return "cargo check && cargo test";
} catch {
/* ignore */
}
// Java/Maven.
try {
if (fs.existsSync(`${root}/pom.xml`) || fs.existsSync(`${root}/build.gradle`)) {
return "mvn test";
}
} catch {
/* ignore */
}
return "bun run check && bun test";
}
/** Truncate a long tool output, preserving the head + truncation marker. */ /** Truncate a long tool output, preserving the head + truncation marker. */
export function truncateToolOutput(output: string): string { export function truncateToolOutput(output: string): string {
if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output; if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output;
@@ -136,7 +207,7 @@ async function executeToolCall(
output = `Error: ${(e as Error).message}`; output = `Error: ${(e as Error).message}`;
} }
const isError = isErrorOutput(output); const isError = isToolFailure(name, output);
const truncated = truncateToolOutput(output); const truncated = truncateToolOutput(output);
sink.push({ sink.push({
@@ -268,18 +339,43 @@ export class AgentTurnServiceImpl {
const abort = params.abort; const abort = params.abort;
const { in_flight } = params; const { in_flight } = params;
// Insert system prompt at index 0 with repo conventions loaded. // Insert system prompt at index 0 with repo conventions + verify command.
const projectContext = buildProjectContext(params.workspace_roots[0] ?? "."); const root = params.workspace_roots[0] ?? ".";
const systemPrompt = mainAgentPromptWithProjectContext(projectContext); const projectContext = buildProjectContext(root);
const verifyCommand = deriveVerifyCommand(root);
const systemPrompt = mainAgentPromptWithProjectContext(projectContext, verifyCommand);
params.messages.unshift(systemMessage(systemPrompt)); params.messages.unshift(systemMessage(systemPrompt));
const originalCount = params.messages.length; const originalCount = params.messages.length;
// Estimate request complexity from the last user message. // Estimate request complexity from the last user message.
const last = params.messages[params.messages.length - 1]; const last = params.messages[params.messages.length - 1];
const requestLen = last?.content?.length ?? 0; const requestLen = last?.content?.length ?? 0;
const userText = last?.content ?? "";
// Complexity steering: a one-shot directive so the model picks the right
// depth instead of relying on prose memory. Pure and cheap.
const complex = isComplexRequest(userText);
this.push(sink, {
kind: "system_note",
systemKind: "info",
message: complex ? "Complex request detected — plan before executing." : "Simple request — keep tool use minimal.",
});
params.messages.push(
systemMessage(
complex
? "[Flow] This request is flagged as complex. Enter a short plan with `plan_enter` and track steps with `todowrite` before starting edits."
: "[Flow] This request looks simple. If you can answer directly without tools, do so — do not spawn agents or workflows for it.",
),
);
const errors = new ErrorTracker(); const errors = new ErrorTracker();
let sawToolCalls = false; let sawToolCalls = false;
let sawMutator = false;
let pendingVerify = false;
let verifyPrompted = false;
let noProgressStreak = 0;
let lastSignature: string | null = null;
let reviewPasses = 0;
for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) { for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) {
// Check abort flag. // Check abort flag.
@@ -293,6 +389,26 @@ export class AgentTurnServiceImpl {
break; break;
} }
// Convergence guard: bail out of a loop stuck re-issuing the same call.
if (noProgressStreak >= MAX_NO_PROGRESS_STREAK) {
this.push(sink, {
kind: "system_note",
systemKind: "warn",
message: "Stopping: the same tool call is being repeated without any progress.",
});
break;
}
// Verify-after-edit: nudge the model to run the check before concluding.
if (pendingVerify && !verifyPrompted) {
params.messages.push(
systemMessage(
`[Flow] You just modified files. Run the verify command via \`bash\` now (${verifyCommand}) and resolve any failures before concluding your turn.`,
),
);
verifyPrompted = true;
}
// Auto-compact oversized history before the LLM call. // Auto-compact oversized history before the LLM call.
await this.autoCompactIfNeeded(params.messages); await this.autoCompactIfNeeded(params.messages);
@@ -343,12 +459,44 @@ export class AgentTurnServiceImpl {
return seq; return seq;
})(); })();
let mutated = false;
let verified = false;
let batchSignature = "";
for (let i = 0; i < toolCalls.length; i++) { for (let i = 0; i < toolCalls.length; i++) {
const tc = toolCalls[i]!; const tc = toolCalls[i]!;
const output = outputs[i]!; const output = outputs[i]!;
if (isErrorOutput(output)) errors.record(tc.function.name, output, params.messages); if (isToolFailure(tc.function.name, output)) errors.record(tc.function.name, output, params.messages);
if (MUTATOR_TOOLS.has(tc.function.name)) {
mutated = true;
sawMutator = true;
}
if ((tc.function.name === "bash" || tc.function.name === "bash_output") && bashExitCode(output) === 0) {
verified = true;
}
// Batch signature: concat of tool+output for no-progress detection.
batchSignature += `${tc.function.name}${output.length}`;
params.messages.push(toolResultMessage(tc.id, output)); params.messages.push(toolResultMessage(tc.id, output));
} }
// No-progress: byte-identical batch (same tools, same output lengths) → streak.
if (toolCalls.length === 1 && batchSignature === lastSignature) noProgressStreak += 1;
else noProgressStreak = 0;
lastSignature = batchSignature;
if (mutated) {
pendingVerify = true;
verifyPrompted = false; // allow a fresh nudge after the next edit batch
}
if (verified) {
pendingVerify = false; // a successful bash run satisfies the verify nudge
verifyPrompted = false;
}
// Bounded self-review pass after file mutations (at most one per turn).
if (sawMutator && reviewPasses < MAX_REVIEW_PASSES && !abort.signal.aborted) {
reviewPasses += 1;
await this.reviewWork(params.messages, abort, sink);
}
} }
// Remove the synthetic sys_msg before emitting to the transcript. // Remove the synthetic sys_msg before emitting to the transcript.
@@ -357,4 +505,30 @@ export class AgentTurnServiceImpl {
this.push(sink, { kind: "done" }); this.push(sink, { kind: "done" });
in_flight.value = false; in_flight.value = false;
} }
/** Run one bounded self-review of the work so far and feed critique back. */
private async reviewWork(
messages: ChatMessage[],
abort: AbortController,
sink: TurnEventSink,
): Promise<void> {
if (abort.signal.aborted) return;
try {
const result = await this.provider.chat(
[systemMessage(reviewerPrompt())],
undefined,
1024,
0.3,
);
const critique = result.message.content ?? "";
if (result.usage) {
this.push(sink, { kind: "review_usage", tokens_in: result.usage[0], tokens_out: result.usage[1] });
}
if (critique.trim() === "" || critique.trim().toUpperCase() === "NO ISSUES FOUND") return;
// Feed the critique back as a system message so the next iteration resolves it.
messages.push(systemMessage(`[Reviewer]\n${critique.trim()}`));
} catch {
/* review is best-effort — non-fatal */
}
}
} }
+105
View File
@@ -0,0 +1,105 @@
/**
* Tests pinning the rewritten prompts to be consistent with the REAL tool
* inventory: they must never reference the phantom `explore_codebase`, must
* list real tool names, and must carry the concrete verify command.
*/
import { describe, expect, test } from "bun:test";
import {
mainAgentPrompt,
mainAgentPromptWithProjectContext,
subagentDirective,
compactionPrompt,
reviewerPrompt,
consensusSynthesizerPrompt,
errorRecoveryNote,
} from "./prompt.ts";
/** All real tool names (mirrors registry.allTools()). */
const REAL_TOOLS = new Set([
"read", "write", "edit", "delete",
"grep", "glob", "semantic_search", "rebuild_index", "list_symbols",
"bash", "bash_output", "bash_kill", "git_operator", "git_worktree", "git_cred",
"cd", "dir_list", "dir_cache_update",
"plan_enter", "plan_ready", "sequential_think", "todowrite", "todofinish",
"workflow_run", "hive_mind", "note_finding", "read_findings",
"spawn_agents", "spawn_pipeline", "parallel_delegate",
"remember", "forget", "recall",
"web_search", "best_practice", "commit_convention", "pong",
]);
describe("prompt tool-consistency", () => {
test("main prompt never references non-existent explore_codebase", () => {
expect(mainAgentPrompt()).not.toContain("explore_codebase");
});
test("every backticked tool token in the main prompt is a real tool", () => {
const s = mainAgentPrompt();
const backticks = s.match(/`([a-z_]+)`/g) ?? [];
const tokens = backticks.map((t) => t.slice(1, -1));
// `plan` and `verify` appear in backticks as prose (a command/word), not tools.
const proseWords = new Set(["plan", "verify", "check"]);
const ghosts = tokens.filter((t) => !REAL_TOOLS.has(t) && !proseWords.has(t));
expect(ghosts).toEqual([]);
});
test("main prompt anchors the key real tools", () => {
const s = mainAgentPrompt();
for (const tool of ["plan_enter", "todowrite", "workflow_run", "spawn_agents", "hive_mind", "recall", "remember"]) {
expect(s).toContain(tool);
}
});
test("verify command is injected as a concrete line when provided", () => {
const s = mainAgentPrompt("bun run check && bun test");
expect(s).toContain("VERIFY COMMAND");
expect(s).toContain("bun run check && bun test");
});
test("without a verify command, no VERIFY COMMAND block is emitted", () => {
expect(mainAgentPrompt()).not.toContain("VERIFY COMMAND");
});
test("project-context variant appends the context block", () => {
const s = mainAgentPromptWithProjectContext("## Rules\nno threads", "bun test");
expect(s).toContain("## PROJECT CONTEXT");
expect(s).toContain("no threads");
expect(s).toContain("bun test");
});
});
describe("subagentDirective", () => {
test("includes the directive, cwd, root, and access tier", () => {
const s = subagentDirective("make the button green", "/p", "/p/src", "read");
expect(s).toContain("make the button green");
expect(s).toContain("/p");
expect(s).toContain("READ-ONLY");
});
test("no-tier call defaults to full access", () => {
const s = subagentDirective("do it", "/p", "/p");
expect(s).toContain("full tool set");
});
});
describe("other prompts", () => {
test("compaction asks for structured sections", () => {
expect(compactionPrompt()).toContain("## Requests");
expect(compactionPrompt()).toContain("## Work done");
});
test("reviewer emits NO ISSUES FOUND sentinel contract", () => {
expect(reviewerPrompt()).toContain("NO ISSUES FOUND");
});
test("consensus synthesizer covers the four sections", () => {
const s = consensusSynthesizerPrompt();
expect(s).toContain("AGREEMENTS");
expect(s).toContain("CONFLICTS");
expect(s).toContain("KEY FINDINGS");
expect(s).toContain("RECOMMENDATION");
});
test("error recovery note is present", () => {
expect(errorRecoveryNote("read", "nope")).toContain("[System note]");
});
});
+100 -29
View File
@@ -1,32 +1,54 @@
/** System prompts and directive templates. Mirrors `agent/prompt.rs`. */ /**
* System prompts and directive templates. Mirrors `agent/prompt.rs`.
*
* Every string here is written against the REAL built-in tool inventory
* (see `infrastructure/tools/registry.ts`). There is intentionally no
* reference to a hypothetical `explore_codebase`; orientation is done through
* the actual `read` / `grep` / `glob` / `semantic_search` tools.
*/
const MUTATOR_TOOLS = "`write`, `edit`, `delete`";
/** Shared operating rules referenced by main and sub-agent prompts. */
const SHARED_DISCIPLINE = `TOOL DISCIPLINE:
- Tool failures arrive prefixed with \`Error:\`. A \`bash\` command that exits non-zero returns an \`Exit code: N\` line (NOT an \`Error:\` prefix) — treat a non-zero exit the same as a failure.
- Read before you write: never edit a file you have not read first.
- Make the smallest correct change. Do not rewrite large sections unless the task explicitly calls for it.
- Do not re-read files whose contents are already in your context. Prefer targeted \`grep\`/\`read\` over broad scans (\`glob\`, \`semantic_search\`) once you know where the code lives.
- Follow the repository conventions from the PROJECT CONTEXT block when present.`;
/** Build the main-agent system prompt. */ /** Build the main-agent system prompt. */
export function mainAgentPrompt(): string { export function mainAgentPrompt(verifyCommand?: string): string {
return `You are Zesdex, an AI coding assistant. You have access to various tools via native function calling to help the user. return `You are Zesdex, an autonomous AI coding agent. You help by doing real work against the user's repository using native function-calling tools.
TOKEN BUDGET — BE EFFICIENT: HOW TO WORK (choose the lightest sufficient path):
- For simple/factual questions, answer directly. Do NOT call tools. 1. Simple or factual: answer directly. Do NOT call tools.
- For complex or unfamiliar code tasks, call \`explore_codebase\` ONCE at the start to locate relevant code, then work from that context. 2. Needs code context: orient ONCE with \`read\`/\`grep\`/\`glob\` (or \`semantic_search\` for fuzzy questions), then work from that context.
- Keep tool usage minimal: prefer \`grep\`/\`glob\`/\`read\` for targeted lookups; avoid re-reading files you already have in context. 3. Multi-step or complex: enter a plan with \`plan_enter\`, track steps with \`todowrite\`, then execute.
- Keep responses concise; do not repeat tool output verbatim. 4. Genuinely parallel or multi-angle exploration: prefer \`workflow_run\`, \`spawn_agents\`, \`parallel_delegate\`, or \`hive_mind\`. Do NOT spawn agents for work you can do directly in a few steps.
CRITICAL DIRECTIVES & PRIORITY HIERARCHY: ${SHARED_DISCIPLINE}
1. WORKFLOW FIRST: For any multi-step, complex, or non-trivial task, you MUST prioritise using \`workflow_run\` (to construct and execute a multi-phase YAML workflow) or \`hive_mind\` (to orchestrate parallel autonomous agents). Workflows are your primary strategy.
2. PLANNING & TODOs: Use \`plan_enter\` to establish high-level architectural plans and \`todowrite\` to maintain granular task checklists.
3. REASONING: Use \`seq_think\` for deep step-by-step analysis.
4. TOOL EXECUTION: Execute individual tools (file edits, terminal commands) within or guided by your workflows. If an error occurs, analyse and fix it.
VERIFY AFTER EDIT (CLAUDE-CODE STYLE): VERIFICATION CONTRACT:
- After modifying code (edit/write), run the repo's check command via \`bash\` before ending the turn: \`cargo check\` / \`cargo clippy\` / \`cargo test\` for Rust, or the equivalent lint/test (\`bun run lint && bun run test\`, \`npm test\`, etc.) for other stacks. Pick the project's actual verify command (see PROJECT CONTEXT / AGENTS.md when present). - After any of ${MUTATOR_TOOLS}, run the repository's verify command below and resolve failures before concluding your turn.
- If the check fails, fix the errors you can see and re-run; only end the turn after the check passes or you cannot resolve a failure yourself (then report it explicitly). - Only claim code compiles or tests pass after running a real check.
- Do NOT claim code compiles or works without running a real check. ${verifyCommand ? `\nVERIFY COMMAND (run via \`bash\`):\n${verifyCommand}` : ""}
Respond conversationally, concisely, and helpfully.`; OUTPUT CONTRACT:
- End with a concise summary of what you did and an explicit next step. Do not dump raw tool output into your reply.
- If you cannot complete something, state exactly what is blocking you.
- Reply in the same language the user writes in.
MEMORY:
- At the start of a task, \`recall\` memory that may be relevant; use \`remember\` to store any hard-won insight or lesson worth keeping.`;
} }
/** Main-agent prompt with an injected `## PROJECT CONTEXT` block. Empty context → base prompt. */ /** Main-agent prompt with an optional project-context block and verify command. */
export function mainAgentPromptWithProjectContext(projectContext: string): string { export function mainAgentPromptWithProjectContext(
const base = mainAgentPrompt(); projectContext: string,
verifyCommand?: string,
): string {
const base = mainAgentPrompt(verifyCommand);
const context = projectContext.trim(); const context = projectContext.trim();
if (context === "") return base; if (context === "") return base;
return `${base} return `${base}
@@ -35,29 +57,51 @@ export function mainAgentPromptWithProjectContext(projectContext: string): strin
${context}`; ${context}`;
} }
/** Build a subagent directive prompt. */ /** Build a subagent directive prompt. Access tier tells the subagent its limits. */
export function subagentDirective(directive: string, cwd: string, wsRoot: string): string { export function subagentDirective(
return `You are a focused subagent. directive: string,
cwd: string,
wsRoot: string,
access: "read" | "write" | "full" = "full",
): string {
const tierNote =
access === "read"
? "You have READ-ONLY tools: you may inspect files and search, but must NOT modify anything."
: access === "write"
? "You have WRITE tools (file edits) but NOT shell or network execution."
: "You have the full tool set.";
return `You are a focused subagent working autonomously on a single directive.
Current directory (PWD): ${cwd} Current directory (PWD): ${cwd}
Workspace root: ${wsRoot} Workspace root: ${wsRoot}
Access tier: ${access} — ${tierNote}
Your directive: ${SHARED_DISCIPLINE}
Your directive (source of truth — follow it exactly):
${directive} ${directive}
Complete the directive autonomously using the tools available to you. Return your final answer when done.`; Complete the directive autonomously with the tools available to you. When done, return a concise final answer that reports what you changed (or found) and any open issues. Do not ask for permission; act within your access tier.`;
} }
/** Build a conversation-compaction prompt. */ /** Build a conversation-compaction prompt. */
export function compactionPrompt(): string { export function compactionPrompt(): string {
return `You are a helpful assistant summarising conversation history. Provide a concise summary of the key user requests, decisions, tools executed, and modified files. Format as a clear bulleted list.`; return `You are condensing a long conversation to preserve context continuity without losing important detail.
Produce a concise markdown recap with these sections (skip any section that has no content):
- ## Requests: what the user asked for.
- ## Decisions: important choices made and why.
- ## Work done: tools executed and files created or modified (with paths).
- ## Open threads: unresolved items, pending questions, or what is still to do.
Be precise about file paths and current state. Do not invent details not present in the conversation.`;
} }
/** Directive for a lightweight context-scout subagent. */ /** Directive for a lightweight context-scout subagent. */
export function exploreScoutDirective(): string { export function exploreScoutDirective(): string {
return `You are a codebase context scout. Given the workspace root, quickly locate the code that is most relevant to the user's request: return `You are a codebase context scout. Given the workspace root, quickly locate the code that is most relevant to the user's request:
1. Run semantic_search once with the user's key terms. 1. Search for the relevant symbols/terms with \`grep\` or \`glob\`; use \`semantic_search\` only when the query is fuzzy.
2. Read up to the 3 most relevant files (use grep for symbols if needed). 2. Read up to the 3 most relevant files.
3. Report a concise bullet list (max 15 bullets, under 1500 characters) of what you found and exactly where (file paths). 3. Report a concise bullet list (max 15 bullets, under 1500 characters) of what you found and exactly where (file paths).
Do NOT rebuild the index. Do NOT enumerate unrelated files. Be brief.`; Do NOT rebuild the index. Do NOT enumerate unrelated files. Be brief.`;
} }
@@ -66,3 +110,30 @@ Do NOT rebuild the index. Do NOT enumerate unrelated files. Be brief.`;
export function errorRecoveryNote(toolName: string, lastError: string): string { export function errorRecoveryNote(toolName: string, lastError: string): string {
return `[System note] The tool \`${toolName}\` failed repeatedly with: "${lastError}". Try an alternative approach (verify paths, correct arguments, use a different tool, or finish without this tool). Do NOT retry the same call.`; return `[System note] The tool \`${toolName}\` failed repeatedly with: "${lastError}". Try an alternative approach (verify paths, correct arguments, use a different tool, or finish without this tool). Do NOT retry the same call.`;
} }
/** System prompt for the optional single-pass self-review after edits. */
export function reviewerPrompt(): string {
return `You are a code reviewer for an autonomous agent's work-in-progress. You are shown the conversation and the changes made so far.
Critique the work for:
- CORRECTNESS: would the changes work as intended? any obvious bugs, edge cases, or missing pieces?
- SPEC COMPLIANCE: did the agent actually do what was asked, or did it drift?
- QUALITY: are the changes minimal, conventional, and consistent with the repository's style?
Return a tight list of the most important ACTIONABLE issues only, each as:
- [PRIORITY: high|medium] <one concrete fix>, e.g. "high: handle empty input in parse()" or "medium: run the tests; they reference a removed export".
Do NOT praise the work. Do NOT restate the plan. If the work is already correct and complete, return exactly the single line "NO ISSUES FOUND". Be concrete enough that a next iteration can act without re-deriving context.`;
}
/** System prompt for reconciling multi-node hive-mind outputs into one consensus. */
export function consensusSynthesizerPrompt(): string {
return `You are a consensus synthesizer for a multi-agent hive mind. Several independent nodes analysed a problem and produced the outputs below. Distill them into ONE coherent consensus report with these sections:
- AGREEMENTS: points multiple nodes converge on.
- CONFLICTS: contradictory conclusions, with which node(s) support each side.
- KEY FINDINGS: the most important, actionable takeaways.
- RECOMMENDATION: a single recommended next action, or 'no clear consensus' if the outputs are too divergent.
Be concise and factual. If a node errored, note it and ignore its content.
Do not invent facts not present in the node outputs.`;
}
@@ -8,6 +8,7 @@
*/ */
import type { NodeOutput } from "@zesdex/workflow"; import type { NodeOutput } from "@zesdex/workflow";
import type { ToolCtx } from "../../agent/infrastructure/tools/mod.ts"; import type { ToolCtx } from "../../agent/infrastructure/tools/mod.ts";
import { consensusSynthesizerPrompt } from "@zesdex/agent";
import { MAX_NODE_OUTPUT_CHARS } from "./engine.ts"; import { MAX_NODE_OUTPUT_CHARS } from "./engine.ts";
/** /**
@@ -33,16 +34,7 @@ export async function synthesizeConsensus(
const systemMsg = { const systemMsg = {
role: "system" as const, role: "system" as const,
content: "You are a consensus synthesizer for a multi-agent hive mind. " + content: consensusSynthesizerPrompt(),
"Several independent nodes analysed a problem and produced the outputs " +
"below. Distill them into ONE coherent consensus report with these sections:\n" +
"- AGREEMENTS: points multiple nodes converge on.\n" +
"- CONFLICTS: contradictory conclusions, with which node(s) support each side.\n" +
"- KEY FINDINGS: the most important, actionable takeaways.\n" +
"- RECOMMENDATION: a single recommended next action, or 'no clear consensus' " +
"if the outputs are too divergent.\n" +
"Be concise and factual. If a node errored, note it and ignore its content.\n" +
"Do not invent facts not present in the node outputs.",
}; };
const userMsg = { const userMsg = {