feat(tui): add new flow helpers for bash exit code handling and verification prompts
This commit is contained in:
@@ -4,8 +4,22 @@ import {
|
||||
conversationChars,
|
||||
ErrorTracker,
|
||||
truncateToolOutput,
|
||||
bashExitCode,
|
||||
isToolFailure,
|
||||
deriveVerifyCommand,
|
||||
AgentTurnServiceImpl,
|
||||
} from "./turn_service.ts";
|
||||
import { type ChatMessage, newConversation, systemMessage, userMessage, toolResultMessage } from "@zesdex/domain";
|
||||
import {
|
||||
type ChatMessage,
|
||||
newConversation,
|
||||
systemMessage,
|
||||
userMessage,
|
||||
assistantMessage,
|
||||
toolResultMessage,
|
||||
} from "@zesdex/domain";
|
||||
import type { AgentTurnParams } from "@zesdex/agent";
|
||||
import type { ToolExecutor } from "./index.ts";
|
||||
import type { ProviderService } from "./ports.ts";
|
||||
|
||||
describe("truncateToolOutput", () => {
|
||||
it("short output is unchanged", () => {
|
||||
@@ -56,4 +70,197 @@ describe("conversationChars", () => {
|
||||
conv.messages.push(systemMessage("sys"), userMessage("hello world"), toolResultMessage("id", "output"));
|
||||
expect(conversationChars(conv.messages)).toBe(3 + 11 + 6);
|
||||
});
|
||||
});
|
||||
});
|
||||
/* ── New flow helpers ─────────────────────────────────────────────── */
|
||||
|
||||
describe("bashExitCode / isToolFailure", () => {
|
||||
it("extracts a non-zero exit code", () => {
|
||||
expect(bashExitCode("boom\n\nExit code: 1 (1s)")).toBe(1);
|
||||
expect(bashExitCode("done\n\nExit code: 0 (0.5s)")).toBe(0);
|
||||
});
|
||||
it("returns null when no exit-code line exists", () => {
|
||||
expect(bashExitCode("plain output")).toBeNull();
|
||||
});
|
||||
it("isToolFailure flags bash non-zero exits as failures", () => {
|
||||
expect(isToolFailure("bash", "nope\n\nExit code: 2 (1s)")).toBe(true);
|
||||
expect(isToolFailure("bash", "ok\n\nExit code: 0 (1s)")).toBe(false);
|
||||
});
|
||||
it("isToolFailure still flags Error: prefixes for other tools", () => {
|
||||
expect(isToolFailure("read", "Error: no such file")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("deriveVerifyCommand", () => {
|
||||
it("defaults to bun check+test for a Bun repo", () => {
|
||||
// No real manifests in a control dir we guarantee to not exist.
|
||||
expect(deriveVerifyCommand("/nonexistent-zesdex-dir")).toContain("bun");
|
||||
});
|
||||
});
|
||||
|
||||
/* ── runTurn flow tests (fakes, no network) ───────────────────────── */
|
||||
|
||||
interface ScriptedStep {
|
||||
content?: string | null;
|
||||
tools?: Array<{ name: string; args?: string; id?: string }>;
|
||||
}
|
||||
|
||||
/** A ProviderService that replays scripted chatStream responses. */
|
||||
function makeScriptedProvider(script: ScriptedStep[]): {
|
||||
provider: ProviderService;
|
||||
seen: Array<ChatMessage[]>;
|
||||
} {
|
||||
const seen: Array<ChatMessage[]> = [];
|
||||
let step = 0;
|
||||
const chatResponses: string[] = [];
|
||||
const provider: ProviderService = {
|
||||
async chat(_messages, _tools, _maxTokens, _temperature) {
|
||||
// Used by compaction and review. For review tests we script below.
|
||||
const text = chatResponses.shift() ?? "";
|
||||
return { message: { role: "assistant", content: text || null }, usage: [10, 5] };
|
||||
},
|
||||
async chatStream(messages, _tools, _max, _temp, onEvent) {
|
||||
// Snapshot a copy: runTurn freely mutates the live array (splice/compact).
|
||||
seen.push([...messages]);
|
||||
const s = script[Math.min(step, script.length - 1)] ?? { content: "done" };
|
||||
step += 1;
|
||||
if (s.content !== undefined && s.content !== null) onEvent({ kind: "token", content: s.content });
|
||||
if (s.tools && s.tools.length > 0) {
|
||||
const msg = assistantMessage(null) as ChatMessage;
|
||||
msg.tool_calls = s.tools.map((t, i) => ({
|
||||
id: t.id ?? `call-${i}`,
|
||||
type: "function",
|
||||
function: { name: t.name, arguments: t.args ?? "{}" },
|
||||
}));
|
||||
return { message: msg, usage: [10, 2] };
|
||||
}
|
||||
return {
|
||||
message: { role: "assistant", content: s.content ?? null },
|
||||
usage: [10, 2],
|
||||
};
|
||||
},
|
||||
// expose a way for tests to script chat replies
|
||||
} as ProviderService;
|
||||
(provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => {
|
||||
chatResponses.push(text);
|
||||
};
|
||||
return { provider, seen };
|
||||
}
|
||||
|
||||
/** A ToolExecutor that echoes fixed outputs per tool. */
|
||||
function fixedExecutor(outs: Record<string, string>): ToolExecutor {
|
||||
return {
|
||||
async execute(name) {
|
||||
return outs[name] ?? "ok";
|
||||
},
|
||||
isParallelSafe() {
|
||||
return false;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function buildParams(userText: string): {
|
||||
params: AgentTurnParams;
|
||||
events: Array<any>;
|
||||
} {
|
||||
const events: Array<any> = [];
|
||||
const params: AgentTurnParams = {
|
||||
messages: [userMessage(userText)],
|
||||
session_dir: "/tmp/zesdex-flow-test",
|
||||
workspace_roots: ["/nonexistent-zesdex-dir"], // no AGENTS.md/package.json side effects
|
||||
turn_events: { push: (e) => events.push(e), drain: () => [] },
|
||||
in_flight: { value: false },
|
||||
abort: new AbortController(),
|
||||
api_key: "k",
|
||||
model: "m",
|
||||
api_base: "https://x",
|
||||
};
|
||||
return { params, events };
|
||||
}
|
||||
|
||||
/** Extract the `[Flow]` steering system message from a captured batch. */
|
||||
function flowDirective(batch: ChatMessage[]): string {
|
||||
const m = batch.find((x) => (x.content ?? "").includes("[Flow]"));
|
||||
return m?.content ?? "";
|
||||
}
|
||||
|
||||
describe("runTurn complexity steering", () => {
|
||||
it("injects a complex-request directive for a complex prompt", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ content: "final answer" },
|
||||
]);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||
const { params } = buildParams("Please refactor the architecture across multiple files");
|
||||
await svc.runTurn(params);
|
||||
// The complexity directive system message must be present before the LLM.
|
||||
const flow = flowDirective(seen[0] ?? []);
|
||||
expect(flow).toContain("flagged as complex");
|
||||
});
|
||||
|
||||
it("injects a keep-it-simple directive instead", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([{ content: "hi" }]);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||
const { params } = buildParams("what is 2+2");
|
||||
await svc.runTurn(params);
|
||||
const flow = flowDirective(seen[0] ?? []);
|
||||
expect(flow).toContain("looks simple");
|
||||
expect(flow).not.toContain("complex");
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn verify-after-edit", () => {
|
||||
it("injects a verify nudge after a write, cleared after a successful bash", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ tools: [{ name: "write" }] },
|
||||
{ tools: [{ name: "bash" }] },
|
||||
{ content: "done" },
|
||||
]);
|
||||
const executor = fixedExecutor({ write: "wrote it", bash: "ok\n\nExit code: 0 (1s)" });
|
||||
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||
const { params } = buildParams("add a comment");
|
||||
await svc.runTurn(params);
|
||||
// The LLM call after the write should carry the verify nudge.
|
||||
const second = seen[1] ?? [];
|
||||
expect(second.some((m) => (m.content ?? "").includes("Run the verify command"))).toBe(true);
|
||||
// The successful bash exits 0, which clears pendingVerify — so the nudge is
|
||||
// injected exactly once, not repeated on every later call.
|
||||
const finalBatch = seen[seen.length - 1] ?? [];
|
||||
const nudgeCount = finalBatch.filter((m) => (m.content ?? "").includes("Run the verify command")).length;
|
||||
expect(nudgeCount).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn convergence guard", () => {
|
||||
it("stops after repeated identical read calls", async () => {
|
||||
const script: ScriptedStep[] = [];
|
||||
for (let i = 0; i < 6; i++) script.push({ tools: [{ name: "read" }] });
|
||||
const { provider, seen } = makeScriptedProvider(script);
|
||||
const executor = fixedExecutor({ read: "same content" });
|
||||
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||
const { params, events } = buildParams("read something");
|
||||
await svc.runTurn(params);
|
||||
const warned = events.some((e) => e.kind === "system_note" && /without any progress|repeated/.test(e.message ?? ""));
|
||||
expect(warned).toBe(true);
|
||||
// Not every call got issued — the guard broke early.
|
||||
expect(seen.length).toBeLessThan(script.length);
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn self-review", () => {
|
||||
it("feeds a reviewer critique back as a system message after a mutator", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ tools: [{ name: "edit" }] },
|
||||
{ content: "fixed" },
|
||||
]);
|
||||
const withChat = provider as unknown as { setChatReply(s: string): void };
|
||||
withChat.setChatReply("- [PRIORITY: high] handle empty input in parse()");
|
||||
const executor = fixedExecutor({ edit: "edited" });
|
||||
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||
const { params, events } = buildParams("fix parse()");
|
||||
await svc.runTurn(params);
|
||||
// The reviewer critique must have reached the next LLM call's history.
|
||||
const second = seen[1] ?? [];
|
||||
expect(second.some((m) => (m.content ?? "").includes("[Reviewer]"))).toBe(true);
|
||||
// review_usage emitted.
|
||||
expect(events.some((e) => e.kind === "review_usage")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -20,7 +20,9 @@ import {
|
||||
mainAgentPromptWithProjectContext,
|
||||
compactionPrompt,
|
||||
errorRecoveryNote,
|
||||
reviewerPrompt,
|
||||
} from "@zesdex/agent";
|
||||
import { isComplexRequest } from "@zesdex/workflow";
|
||||
import type { ProviderService } from "./ports.ts";
|
||||
import type { ToolExecutor } from "./index.ts";
|
||||
|
||||
@@ -38,11 +40,80 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000;
|
||||
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
|
||||
const COMPACT_KEEP_TAIL = 6;
|
||||
|
||||
/** Tools that mutate the filesystem — after these, a verify run is expected. */
|
||||
const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]);
|
||||
|
||||
/** Consecutive identical, non-progressing tool iterations before the loop stops. */
|
||||
const MAX_NO_PROGRESS_STREAK = 4;
|
||||
|
||||
/** Self-review is bounded to this many passes per turn. */
|
||||
const MAX_REVIEW_PASSES = 1;
|
||||
|
||||
/** Whether the output string denotes a tool error. */
|
||||
function isErrorOutput(output: string): boolean {
|
||||
return output.startsWith("Error:");
|
||||
}
|
||||
|
||||
/** Extract the numeric exit code from a `bash` tool result, or null if not a failure (0). */
|
||||
export function bashExitCode(output: string): number | null {
|
||||
const match = output.match(/Exit code:\s*(\d+)/);
|
||||
if (!match) return null;
|
||||
const code = Number(match[1]);
|
||||
return Number.isInteger(code) && code >= 0 ? code : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a tool result is a real failure. `Error:` prefixes cover most tools;
|
||||
* a `bash` command that exits non-zero returns an `Exit code: N` line instead.
|
||||
*/
|
||||
export function isToolFailure(toolName: string, output: string): boolean {
|
||||
if (isErrorOutput(output)) return true;
|
||||
if (toolName === "bash" || toolName === "bash_output") {
|
||||
const code = bashExitCode(output);
|
||||
if (code === null) return false; // no exit-code line → no signal
|
||||
return code !== 0;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort derivation of the repository's verify command from convention
|
||||
* manifests. Defaults to a safe lint+test invocation for Bun.
|
||||
*/
|
||||
export function deriveVerifyCommand(root: string): string {
|
||||
const fs = require("node:fs");
|
||||
// Bun project: prefer an explicit `check` (typecheck) script then test.
|
||||
try {
|
||||
const pkg = JSON.parse(fs.readFileSync(`${root}/package.json`, "utf8")) as {
|
||||
scripts?: Record<string, string>;
|
||||
};
|
||||
const s = pkg?.scripts ?? {};
|
||||
const parts: string[] = [];
|
||||
if (s["check"] && typeof s["check"] === "string") parts.push(`bun run check`);
|
||||
else if (s["typecheck"] && typeof s["typecheck"] === "string") parts.push(`bun run typecheck`);
|
||||
if (s["lint"] && typeof s["lint"] === "string") parts.push(`bun run lint`);
|
||||
if (s["test"] && typeof s["test"] === "string") parts.push(`bun run test`);
|
||||
if (parts.length > 0) return parts.join(" && ");
|
||||
} catch {
|
||||
/* no package.json — fall through */
|
||||
}
|
||||
// Rust project.
|
||||
try {
|
||||
if (fs.existsSync(`${root}/Cargo.toml`)) return "cargo check && cargo test";
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
// Java/Maven.
|
||||
try {
|
||||
if (fs.existsSync(`${root}/pom.xml`) || fs.existsSync(`${root}/build.gradle`)) {
|
||||
return "mvn test";
|
||||
}
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
return "bun run check && bun test";
|
||||
}
|
||||
|
||||
/** Truncate a long tool output, preserving the head + truncation marker. */
|
||||
export function truncateToolOutput(output: string): string {
|
||||
if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output;
|
||||
@@ -136,7 +207,7 @@ async function executeToolCall(
|
||||
output = `Error: ${(e as Error).message}`;
|
||||
}
|
||||
|
||||
const isError = isErrorOutput(output);
|
||||
const isError = isToolFailure(name, output);
|
||||
const truncated = truncateToolOutput(output);
|
||||
|
||||
sink.push({
|
||||
@@ -268,18 +339,43 @@ export class AgentTurnServiceImpl {
|
||||
const abort = params.abort;
|
||||
const { in_flight } = params;
|
||||
|
||||
// Insert system prompt at index 0 with repo conventions loaded.
|
||||
const projectContext = buildProjectContext(params.workspace_roots[0] ?? ".");
|
||||
const systemPrompt = mainAgentPromptWithProjectContext(projectContext);
|
||||
// Insert system prompt at index 0 with repo conventions + verify command.
|
||||
const root = params.workspace_roots[0] ?? ".";
|
||||
const projectContext = buildProjectContext(root);
|
||||
const verifyCommand = deriveVerifyCommand(root);
|
||||
const systemPrompt = mainAgentPromptWithProjectContext(projectContext, verifyCommand);
|
||||
params.messages.unshift(systemMessage(systemPrompt));
|
||||
const originalCount = params.messages.length;
|
||||
|
||||
// Estimate request complexity from the last user message.
|
||||
const last = params.messages[params.messages.length - 1];
|
||||
const requestLen = last?.content?.length ?? 0;
|
||||
const userText = last?.content ?? "";
|
||||
|
||||
// Complexity steering: a one-shot directive so the model picks the right
|
||||
// depth instead of relying on prose memory. Pure and cheap.
|
||||
const complex = isComplexRequest(userText);
|
||||
this.push(sink, {
|
||||
kind: "system_note",
|
||||
systemKind: "info",
|
||||
message: complex ? "Complex request detected — plan before executing." : "Simple request — keep tool use minimal.",
|
||||
});
|
||||
params.messages.push(
|
||||
systemMessage(
|
||||
complex
|
||||
? "[Flow] This request is flagged as complex. Enter a short plan with `plan_enter` and track steps with `todowrite` before starting edits."
|
||||
: "[Flow] This request looks simple. If you can answer directly without tools, do so — do not spawn agents or workflows for it.",
|
||||
),
|
||||
);
|
||||
|
||||
const errors = new ErrorTracker();
|
||||
let sawToolCalls = false;
|
||||
let sawMutator = false;
|
||||
let pendingVerify = false;
|
||||
let verifyPrompted = false;
|
||||
let noProgressStreak = 0;
|
||||
let lastSignature: string | null = null;
|
||||
let reviewPasses = 0;
|
||||
|
||||
for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) {
|
||||
// Check abort flag.
|
||||
@@ -293,6 +389,26 @@ export class AgentTurnServiceImpl {
|
||||
break;
|
||||
}
|
||||
|
||||
// Convergence guard: bail out of a loop stuck re-issuing the same call.
|
||||
if (noProgressStreak >= MAX_NO_PROGRESS_STREAK) {
|
||||
this.push(sink, {
|
||||
kind: "system_note",
|
||||
systemKind: "warn",
|
||||
message: "Stopping: the same tool call is being repeated without any progress.",
|
||||
});
|
||||
break;
|
||||
}
|
||||
|
||||
// Verify-after-edit: nudge the model to run the check before concluding.
|
||||
if (pendingVerify && !verifyPrompted) {
|
||||
params.messages.push(
|
||||
systemMessage(
|
||||
`[Flow] You just modified files. Run the verify command via \`bash\` now (${verifyCommand}) and resolve any failures before concluding your turn.`,
|
||||
),
|
||||
);
|
||||
verifyPrompted = true;
|
||||
}
|
||||
|
||||
// Auto-compact oversized history before the LLM call.
|
||||
await this.autoCompactIfNeeded(params.messages);
|
||||
|
||||
@@ -343,12 +459,44 @@ export class AgentTurnServiceImpl {
|
||||
return seq;
|
||||
})();
|
||||
|
||||
let mutated = false;
|
||||
let verified = false;
|
||||
let batchSignature = "";
|
||||
for (let i = 0; i < toolCalls.length; i++) {
|
||||
const tc = toolCalls[i]!;
|
||||
const output = outputs[i]!;
|
||||
if (isErrorOutput(output)) errors.record(tc.function.name, output, params.messages);
|
||||
if (isToolFailure(tc.function.name, output)) errors.record(tc.function.name, output, params.messages);
|
||||
if (MUTATOR_TOOLS.has(tc.function.name)) {
|
||||
mutated = true;
|
||||
sawMutator = true;
|
||||
}
|
||||
if ((tc.function.name === "bash" || tc.function.name === "bash_output") && bashExitCode(output) === 0) {
|
||||
verified = true;
|
||||
}
|
||||
// Batch signature: concat of tool+output for no-progress detection.
|
||||
batchSignature += `${tc.function.name} | ||||