Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
29585cc630 | ||
|
|
5c65f8f6ab | ||
|
|
7e2eda3bb4 | ||
|
|
e6718fa118 | ||
|
|
f98030f28a | ||
|
|
93e5464370 | ||
|
|
f5b04214cf | ||
|
|
008f737e1f | ||
|
|
2100e7ab54 | ||
|
|
c53fdfa7c9 | ||
|
|
5912a35bbb | ||
|
|
d86d6aeef2 |
@@ -1,3 +1,30 @@
|
|||||||
|
# [1.25.0](https://github.com/asepharyana/zesdex/compare/v1.24.0...v1.25.0) (2026-09-03)
|
||||||
|
|
||||||
|
|
||||||
|
### Features
|
||||||
|
|
||||||
|
* **tui:** enhance LLM interaction with max_tokens and temperature overrides; add tests for parallel delegation and workflow script parsing ([5c65f8f](https://github.com/asepharyana/zesdex/commit/5c65f8f6ab093dbfad682939a9fcae79a8a68661))
|
||||||
|
|
||||||
|
# [1.24.0](https://github.com/asepharyana/zesdex/compare/v1.23.2...v1.24.0) (2026-09-03)
|
||||||
|
|
||||||
|
|
||||||
|
### Features
|
||||||
|
|
||||||
|
* **tui:** add new flow helpers for bash exit code handling and verification prompts ([f98030f](https://github.com/asepharyana/zesdex/commit/f98030f28a20e3d9ba807bb98ddbf142511a3964))
|
||||||
|
* **tui:** enhance autocomplete functionality and UI interactions ([008f737](https://github.com/asepharyana/zesdex/commit/008f737e1f5d0eef6a07ebe0fff2c9730bd44eb4))
|
||||||
|
* **tui:** enhance command execution on autocomplete and improve UI responsiveness ([f5b0421](https://github.com/asepharyana/zesdex/commit/f5b04214cf385ba2e4a02f6ce202436ffab23c35))
|
||||||
|
* **tui:** enhance model selection and improve UI components ([c53fdfa](https://github.com/asepharyana/zesdex/commit/c53fdfa7c99887e8f3dfe68e7f1b2e766de039b1))
|
||||||
|
* **tui:** implement interactive menu system for model and config selection ([2100e7a](https://github.com/asepharyana/zesdex/commit/2100e7ab54460f28b67c3bb07855eef8f04eb9ec))
|
||||||
|
* **tui:** implement live model fetching and enhance model selection menus ([93e5464](https://github.com/asepharyana/zesdex/commit/93e54643703ce7f18286bd6a988e0fab18cc9e66))
|
||||||
|
* **tui:** implement message compaction and summarization logic for efficient conversation handling ([e6718fa](https://github.com/asepharyana/zesdex/commit/e6718fa11825cccc74a9d100755226b7dfefc106))
|
||||||
|
|
||||||
|
## [1.23.2](https://github.com/asepharyana/zesdex/compare/v1.23.1...v1.23.2) (2026-09-03)
|
||||||
|
|
||||||
|
|
||||||
|
### Bug Fixes
|
||||||
|
|
||||||
|
* **tui:** sliding-window stream reconciler + UI polish ([d86d6ae](https://github.com/asepharyana/zesdex/commit/d86d6aeef2dcd4924ced14f21502a84f7e916a26))
|
||||||
|
|
||||||
## [1.23.1](https://github.com/asepharyana/zesdex/compare/v1.23.0...v1.23.1) (2026-09-03)
|
## [1.23.1](https://github.com/asepharyana/zesdex/compare/v1.23.0...v1.23.1) (2026-09-03)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "zesdex",
|
"name": "zesdex",
|
||||||
"version": "1.23.1",
|
"version": "1.25.0",
|
||||||
"private": true,
|
"private": true,
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"packageManager": "bun@1.3.14",
|
"packageManager": "bun@1.3.14",
|
||||||
|
|||||||
@@ -4,8 +4,25 @@ import {
|
|||||||
conversationChars,
|
conversationChars,
|
||||||
ErrorTracker,
|
ErrorTracker,
|
||||||
truncateToolOutput,
|
truncateToolOutput,
|
||||||
|
bashExitCode,
|
||||||
|
isToolFailure,
|
||||||
|
deriveVerifyCommand,
|
||||||
|
AgentTurnServiceImpl,
|
||||||
|
takeRecentTail,
|
||||||
|
reduceMessagesToDigest,
|
||||||
|
compactMessagesWithAi,
|
||||||
} from "./turn_service.ts";
|
} from "./turn_service.ts";
|
||||||
import { type ChatMessage, newConversation, systemMessage, userMessage, toolResultMessage } from "@zesdex/domain";
|
import {
|
||||||
|
type ChatMessage,
|
||||||
|
newConversation,
|
||||||
|
systemMessage,
|
||||||
|
userMessage,
|
||||||
|
assistantMessage,
|
||||||
|
toolResultMessage,
|
||||||
|
} from "@zesdex/domain";
|
||||||
|
import type { AgentTurnParams } from "@zesdex/agent";
|
||||||
|
import type { ToolExecutor } from "./index.ts";
|
||||||
|
import type { ProviderService } from "./ports.ts";
|
||||||
|
|
||||||
describe("truncateToolOutput", () => {
|
describe("truncateToolOutput", () => {
|
||||||
it("short output is unchanged", () => {
|
it("short output is unchanged", () => {
|
||||||
@@ -56,4 +73,314 @@ describe("conversationChars", () => {
|
|||||||
conv.messages.push(systemMessage("sys"), userMessage("hello world"), toolResultMessage("id", "output"));
|
conv.messages.push(systemMessage("sys"), userMessage("hello world"), toolResultMessage("id", "output"));
|
||||||
expect(conversationChars(conv.messages)).toBe(3 + 11 + 6);
|
expect(conversationChars(conv.messages)).toBe(3 + 11 + 6);
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
/* ── New flow helpers ─────────────────────────────────────────────── */
|
||||||
|
|
||||||
|
describe("bashExitCode / isToolFailure", () => {
|
||||||
|
it("extracts a non-zero exit code", () => {
|
||||||
|
expect(bashExitCode("boom\n\nExit code: 1 (1s)")).toBe(1);
|
||||||
|
expect(bashExitCode("done\n\nExit code: 0 (0.5s)")).toBe(0);
|
||||||
|
});
|
||||||
|
it("returns null when no exit-code line exists", () => {
|
||||||
|
expect(bashExitCode("plain output")).toBeNull();
|
||||||
|
});
|
||||||
|
it("isToolFailure flags bash non-zero exits as failures", () => {
|
||||||
|
expect(isToolFailure("bash", "nope\n\nExit code: 2 (1s)")).toBe(true);
|
||||||
|
expect(isToolFailure("bash", "ok\n\nExit code: 0 (1s)")).toBe(false);
|
||||||
|
});
|
||||||
|
it("isToolFailure still flags Error: prefixes for other tools", () => {
|
||||||
|
expect(isToolFailure("read", "Error: no such file")).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("deriveVerifyCommand", () => {
|
||||||
|
it("defaults to bun check+test for a Bun repo", () => {
|
||||||
|
// No real manifests in a control dir we guarantee to not exist.
|
||||||
|
expect(deriveVerifyCommand("/nonexistent-zesdex-dir")).toContain("bun");
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
/* ── runTurn flow tests (fakes, no network) ───────────────────────── */
|
||||||
|
|
||||||
|
interface ScriptedStep {
|
||||||
|
content?: string | null;
|
||||||
|
tools?: Array<{ name: string; args?: string; id?: string }>;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** A ProviderService that replays scripted chatStream responses. */
|
||||||
|
function makeScriptedProvider(script: ScriptedStep[]): {
|
||||||
|
provider: ProviderService;
|
||||||
|
seen: Array<ChatMessage[]>;
|
||||||
|
chatCalls: Array<ChatMessage[]>;
|
||||||
|
} {
|
||||||
|
const seen: Array<ChatMessage[]> = [];
|
||||||
|
const chatCalls: Array<ChatMessage[]> = [];
|
||||||
|
let step = 0;
|
||||||
|
const chatResponses: string[] = [];
|
||||||
|
const provider: ProviderService = {
|
||||||
|
async chat(messages, _tools, _maxTokens, _temperature) {
|
||||||
|
// Used by compaction and review. For review tests we script below.
|
||||||
|
chatCalls.push([...messages]);
|
||||||
|
const text = chatResponses.shift() ?? "";
|
||||||
|
return { message: { role: "assistant", content: text || null }, usage: [10, 5] };
|
||||||
|
},
|
||||||
|
async chatStream(messages, _tools, _max, _temp, onEvent) {
|
||||||
|
// Snapshot a copy: runTurn freely mutates the live array (splice/compact).
|
||||||
|
seen.push([...messages]);
|
||||||
|
const s = script[Math.min(step, script.length - 1)] ?? { content: "done" };
|
||||||
|
step += 1;
|
||||||
|
if (s.content !== undefined && s.content !== null) onEvent({ kind: "token", content: s.content });
|
||||||
|
if (s.tools && s.tools.length > 0) {
|
||||||
|
const msg = assistantMessage(null) as ChatMessage;
|
||||||
|
msg.tool_calls = s.tools.map((t, i) => ({
|
||||||
|
id: t.id ?? `call-${i}`,
|
||||||
|
type: "function",
|
||||||
|
function: { name: t.name, arguments: t.args ?? "{}" },
|
||||||
|
}));
|
||||||
|
return { message: msg, usage: [10, 2] };
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
message: { role: "assistant", content: s.content ?? null },
|
||||||
|
usage: [10, 2],
|
||||||
|
};
|
||||||
|
},
|
||||||
|
// expose a way for tests to script chat replies
|
||||||
|
} as ProviderService;
|
||||||
|
(provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => {
|
||||||
|
chatResponses.push(text);
|
||||||
|
};
|
||||||
|
return { provider, seen, chatCalls };
|
||||||
|
}
|
||||||
|
|
||||||
|
/** A ToolExecutor that echoes fixed outputs per tool. */
|
||||||
|
function fixedExecutor(outs: Record<string, string>): ToolExecutor {
|
||||||
|
return {
|
||||||
|
async execute(name) {
|
||||||
|
return outs[name] ?? "ok";
|
||||||
|
},
|
||||||
|
isParallelSafe() {
|
||||||
|
return false;
|
||||||
|
},
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function buildParams(userText: string): {
|
||||||
|
params: AgentTurnParams;
|
||||||
|
events: Array<any>;
|
||||||
|
} {
|
||||||
|
const events: Array<any> = [];
|
||||||
|
const params: AgentTurnParams = {
|
||||||
|
messages: [userMessage(userText)],
|
||||||
|
session_dir: "/tmp/zesdex-flow-test",
|
||||||
|
workspace_roots: ["/nonexistent-zesdex-dir"], // no AGENTS.md/package.json side effects
|
||||||
|
turn_events: { push: (e) => events.push(e), drain: () => [] },
|
||||||
|
in_flight: { value: false },
|
||||||
|
abort: new AbortController(),
|
||||||
|
api_key: "k",
|
||||||
|
model: "m",
|
||||||
|
api_base: "https://x",
|
||||||
|
};
|
||||||
|
return { params, events };
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("runTurn", () => {
|
||||||
|
it("runs a simple one-shot prompt to completion", async () => {
|
||||||
|
const { provider } = makeScriptedProvider([{ content: "here is the answer" }]);
|
||||||
|
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||||
|
const { params } = buildParams("hello");
|
||||||
|
await svc.runTurn(params);
|
||||||
|
expect(params.in_flight.value).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("runTurn verify-after-edit", () => {
|
||||||
|
it("injects a verify nudge after a write, cleared after a successful bash", async () => {
|
||||||
|
const { provider, seen } = makeScriptedProvider([
|
||||||
|
{ tools: [{ name: "write" }] },
|
||||||
|
{ tools: [{ name: "bash" }] },
|
||||||
|
{ content: "done" },
|
||||||
|
]);
|
||||||
|
const executor = fixedExecutor({ write: "wrote it", bash: "ok\n\nExit code: 0 (1s)" });
|
||||||
|
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||||
|
const { params } = buildParams("add a comment");
|
||||||
|
await svc.runTurn(params);
|
||||||
|
// The LLM call after the write should carry the verify nudge.
|
||||||
|
const second = seen[1] ?? [];
|
||||||
|
expect(second.some((m) => (m.content ?? "").includes("Run the verify command"))).toBe(true);
|
||||||
|
// The successful bash exits 0, which clears pendingVerify — so the nudge is
|
||||||
|
// injected exactly once, not repeated on every later call.
|
||||||
|
const finalBatch = seen[seen.length - 1] ?? [];
|
||||||
|
const nudgeCount = finalBatch.filter((m) => (m.content ?? "").includes("Run the verify command")).length;
|
||||||
|
expect(nudgeCount).toBe(1);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("runTurn convergence guard", () => {
|
||||||
|
it("stops after repeated identical read calls", async () => {
|
||||||
|
const script: ScriptedStep[] = [];
|
||||||
|
for (let i = 0; i < 6; i++) script.push({ tools: [{ name: "read" }] });
|
||||||
|
const { provider, seen } = makeScriptedProvider(script);
|
||||||
|
const executor = fixedExecutor({ read: "same content" });
|
||||||
|
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||||
|
const { params, events } = buildParams("read something");
|
||||||
|
await svc.runTurn(params);
|
||||||
|
const warned = events.some((e) => e.kind === "system_note" && /without any progress|repeated/.test(e.message ?? ""));
|
||||||
|
expect(warned).toBe(true);
|
||||||
|
// Not every call got issued — the guard broke early.
|
||||||
|
expect(seen.length).toBeLessThan(script.length);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("runTurn self-review", () => {
|
||||||
|
it("feeds a reviewer critique back as a system message after a mutator", async () => {
|
||||||
|
const { provider, seen } = makeScriptedProvider([
|
||||||
|
{ tools: [{ name: "edit" }] },
|
||||||
|
{ content: "fixed" },
|
||||||
|
]);
|
||||||
|
const withChat = provider as unknown as { setChatReply(s: string): void };
|
||||||
|
withChat.setChatReply("- [PRIORITY: high] handle empty input in parse()");
|
||||||
|
const executor = fixedExecutor({ edit: "edited" });
|
||||||
|
const svc = new AgentTurnServiceImpl(provider as never, executor, [] as never);
|
||||||
|
const { params, events } = buildParams("fix parse()");
|
||||||
|
await svc.runTurn(params);
|
||||||
|
// The reviewer critique must have reached the next LLM call's history.
|
||||||
|
const second = seen[1] ?? [];
|
||||||
|
expect(second.some((m) => (m.content ?? "").includes("[Reviewer]"))).toBe(true);
|
||||||
|
// review_usage emitted.
|
||||||
|
expect(events.some((e) => e.kind === "review_usage")).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("passes the work-in-progress to the reviewer (not an empty slate)", async () => {
|
||||||
|
const { provider, chatCalls } = makeScriptedProvider([
|
||||||
|
{ tools: [{ name: "edit" }] },
|
||||||
|
{ content: "fixed" },
|
||||||
|
]);
|
||||||
|
(provider as unknown as { setChatReply(s: string): void }).setChatReply(
|
||||||
|
"- [PRIORITY: medium] add a null check",
|
||||||
|
);
|
||||||
|
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({ edit: "changed" }), [] as never);
|
||||||
|
const { params } = buildParams("harden the parser");
|
||||||
|
await svc.runTurn(params);
|
||||||
|
// The REVIEWER chat call carries the digest in its user message.
|
||||||
|
const reviewerCall = chatCalls.find((c) =>
|
||||||
|
c.some((m) => (m.content ?? "").includes("Review the work-in-progress below")),
|
||||||
|
);
|
||||||
|
expect(reviewerCall).toBeTruthy();
|
||||||
|
expect(reviewerCall!.some((m) => (m.content ?? "").includes("harden the parser"))).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
/* ── Compaction: efficient + accurate ─────────────────────────────── */
|
||||||
|
|
||||||
|
describe("takeRecentTail", () => {
|
||||||
|
it("keeps the most recent messages and evicts an oversized older blob", () => {
|
||||||
|
const messages = [
|
||||||
|
toolResultMessage("t0", "y".repeat(5000)), // huge old tool output
|
||||||
|
userMessage("old request"),
|
||||||
|
assistantMessage("recent reply A"),
|
||||||
|
userMessage("recent reply B"),
|
||||||
|
];
|
||||||
|
const { tail, evicted } = takeRecentTail(messages, 25, 2);
|
||||||
|
expect(evicted.length).toBe(2); // the blob + the old request
|
||||||
|
expect(tail.map((m) => m.content)).toEqual(["recent reply A", "recent reply B"]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("gives the whole history as tail when it is small enough", () => {
|
||||||
|
const messages = [userMessage("a"), assistantMessage("b")];
|
||||||
|
const { tail, evicted } = takeRecentTail(messages, 1000, 6);
|
||||||
|
expect(evicted).toEqual([]);
|
||||||
|
expect(tail.length).toBe(2);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("does not let a huge tool output blow the tail past the min count", () => {
|
||||||
|
// minTail 2: newest two kept regardless; the huge tool body stays evicted.
|
||||||
|
const messages = [
|
||||||
|
toolResultMessage("t0", "x".repeat(50000)),
|
||||||
|
userMessage("keep1"),
|
||||||
|
assistantMessage("keep2"),
|
||||||
|
];
|
||||||
|
const { tail, evicted } = takeRecentTail(messages, 8000, 2);
|
||||||
|
expect(tail.map((m) => m.content)).toEqual(["keep1", "keep2"]);
|
||||||
|
expect(evicted.length).toBe(1);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("reduceMessagesToDigest", () => {
|
||||||
|
it("stubs tool bodies entirely and previews text", () => {
|
||||||
|
const digest = reduceMessagesToDigest([
|
||||||
|
toolResultMessage("t0", "y".repeat(5000)),
|
||||||
|
userMessage("short request"),
|
||||||
|
]);
|
||||||
|
expect(digest).toContain("- tool");
|
||||||
|
expect(digest).not.toContain("yyyy");
|
||||||
|
expect(digest).toContain("short request");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("keeps user previews well above the shorter assistant cap", () => {
|
||||||
|
const long = "w".repeat(2000);
|
||||||
|
// User budget is 1500 chars — larger than the 400-char assistant cap, so
|
||||||
|
// a long user request survives far more of its body than a long assistant
|
||||||
|
// reply would (requests matter most for summary accuracy).
|
||||||
|
const digestUser = reduceMessagesToDigest([userMessage(long)]);
|
||||||
|
const digestAssistant = reduceMessagesToDigest([assistantMessage(long)]);
|
||||||
|
expect(digestUser.length).toBeGreaterThan(800);
|
||||||
|
expect(digestAssistant.length).toBeLessThan(500);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("respects the hard input cap", () => {
|
||||||
|
const many = [];
|
||||||
|
for (let i = 0; i < 200; i++) many.push(userMessage("r".repeat(300)));
|
||||||
|
const digest = reduceMessagesToDigest(many);
|
||||||
|
expect(digest.length).toBeLessThanOrEqual(20_000 + 40);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("compactMessagesWithAi", () => {
|
||||||
|
it("keeps the recent tail and prepends an AI summary", async () => {
|
||||||
|
// Long enough to bypass the min-size guard, with an oversized OLD tool blob
|
||||||
|
// (should be evicted + stubbed, not kept verbatim).
|
||||||
|
const messages: ChatMessage[] = [
|
||||||
|
userMessage("old request one"),
|
||||||
|
assistantMessage("old response"),
|
||||||
|
toolResultMessage("t0", "big".repeat(4000)),
|
||||||
|
];
|
||||||
|
for (let i = 0; i < 7; i++) messages.push(userMessage(`mid ${i}`));
|
||||||
|
messages.push(toolResultMessage("t1", "tail-result"));
|
||||||
|
messages.push(userMessage("newest request"));
|
||||||
|
messages.push(assistantMessage("newest response"));
|
||||||
|
|
||||||
|
const provider: ProviderService = {
|
||||||
|
async chat() {
|
||||||
|
return { message: { role: "assistant", content: "## Requests\n- old request one" }, usage: null };
|
||||||
|
},
|
||||||
|
async chatStream() {
|
||||||
|
return { message: { role: "assistant", content: null }, usage: null };
|
||||||
|
},
|
||||||
|
};
|
||||||
|
await compactMessagesWithAi(messages, provider);
|
||||||
|
// Summary first, recent tail preserved verbatim, tool blobs gone.
|
||||||
|
expect(messages[0]?.content).toContain("[AI Summary of Previous Conversation]");
|
||||||
|
expect(messages.some((m) => m.content === "newest request")).toBe(true);
|
||||||
|
expect(messages.some((m) => m.content === "newest response")).toBe(true);
|
||||||
|
expect(messages.some((m) => (m.content ?? "").includes("big".repeat(10)))).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("degrades gracefully on summarizer failure, preserving the tail", async () => {
|
||||||
|
const messages = [
|
||||||
|
userMessage("old"),
|
||||||
|
toolResultMessage("t0", "x".repeat(100)),
|
||||||
|
userMessage("recent request"),
|
||||||
|
];
|
||||||
|
const failing: ProviderService = {
|
||||||
|
async chat() {
|
||||||
|
throw new Error("network down");
|
||||||
|
},
|
||||||
|
async chatStream() {
|
||||||
|
return { message: { role: "assistant", content: null }, usage: null };
|
||||||
|
},
|
||||||
|
};
|
||||||
|
await compactMessagesWithAi(messages, failing);
|
||||||
|
// The recent message survives even when the LLM call fails.
|
||||||
|
expect(messages.some((m) => m.content === "recent request")).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ import {
|
|||||||
type ToolCall,
|
type ToolCall,
|
||||||
type ToolDef,
|
type ToolDef,
|
||||||
type JsonValue,
|
type JsonValue,
|
||||||
|
Roles,
|
||||||
systemMessage,
|
systemMessage,
|
||||||
userMessage,
|
userMessage,
|
||||||
assistantMessage,
|
assistantMessage,
|
||||||
@@ -20,6 +21,7 @@ import {
|
|||||||
mainAgentPromptWithProjectContext,
|
mainAgentPromptWithProjectContext,
|
||||||
compactionPrompt,
|
compactionPrompt,
|
||||||
errorRecoveryNote,
|
errorRecoveryNote,
|
||||||
|
reviewerPrompt,
|
||||||
} from "@zesdex/agent";
|
} from "@zesdex/agent";
|
||||||
import type { ProviderService } from "./ports.ts";
|
import type { ProviderService } from "./ports.ts";
|
||||||
import type { ToolExecutor } from "./index.ts";
|
import type { ToolExecutor } from "./index.ts";
|
||||||
@@ -38,11 +40,92 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000;
|
|||||||
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
|
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
|
||||||
const COMPACT_KEEP_TAIL = 6;
|
const COMPACT_KEEP_TAIL = 6;
|
||||||
|
|
||||||
|
/** Recent-tail budget (chars): the most recent history kept verbatim on compact. */
|
||||||
|
const COMPACT_TAIL_CHARS = 8_000;
|
||||||
|
/** Caps what the summarizer actually receives, keeping input small and focused. */
|
||||||
|
const COMPACT_MAX_INPUT_CHARS = 20_000;
|
||||||
|
/** Preview budget for a user request inside the digest (requests matter most). */
|
||||||
|
const COMPACT_USER_PREVIEW_CHARS = 1_500;
|
||||||
|
/** Preview budget for assistant text inside the digest. */
|
||||||
|
const COMPACT_TEXT_PREVIEW_CHARS = 400;
|
||||||
|
|
||||||
|
/** Tools that mutate the filesystem — after these, a verify run is expected. */
|
||||||
|
const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]);
|
||||||
|
|
||||||
|
/** Consecutive identical, non-progressing tool iterations before the loop stops. */
|
||||||
|
const MAX_NO_PROGRESS_STREAK = 4;
|
||||||
|
|
||||||
|
/** Self-review is bounded to this many passes per turn. */
|
||||||
|
const MAX_REVIEW_PASSES = 1;
|
||||||
|
|
||||||
|
/** Per-tool-call wall-clock budget before the tool is considered hung. */
|
||||||
|
const TOOL_TIMEOUT_MS = 90_000;
|
||||||
|
|
||||||
/** Whether the output string denotes a tool error. */
|
/** Whether the output string denotes a tool error. */
|
||||||
function isErrorOutput(output: string): boolean {
|
function isErrorOutput(output: string): boolean {
|
||||||
return output.startsWith("Error:");
|
return output.startsWith("Error:");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Extract the numeric exit code from a `bash` tool result, or null if not a failure (0). */
|
||||||
|
export function bashExitCode(output: string): number | null {
|
||||||
|
const match = output.match(/Exit code:\s*(\d+)/);
|
||||||
|
if (!match) return null;
|
||||||
|
const code = Number(match[1]);
|
||||||
|
return Number.isInteger(code) && code >= 0 ? code : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether a tool result is a real failure. `Error:` prefixes cover most tools;
|
||||||
|
* a `bash` command that exits non-zero returns an `Exit code: N` line instead.
|
||||||
|
*/
|
||||||
|
export function isToolFailure(toolName: string, output: string): boolean {
|
||||||
|
if (isErrorOutput(output)) return true;
|
||||||
|
if (toolName === "bash" || toolName === "bash_output") {
|
||||||
|
const code = bashExitCode(output);
|
||||||
|
if (code === null) return false; // no exit-code line → no signal
|
||||||
|
return code !== 0;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Best-effort derivation of the repository's verify command from convention
|
||||||
|
* manifests. Defaults to a safe lint+test invocation for Bun.
|
||||||
|
*/
|
||||||
|
export function deriveVerifyCommand(root: string): string {
|
||||||
|
const fs = require("node:fs");
|
||||||
|
// Bun project: prefer an explicit `check` (typecheck) script then test.
|
||||||
|
try {
|
||||||
|
const pkg = JSON.parse(fs.readFileSync(`${root}/package.json`, "utf8")) as {
|
||||||
|
scripts?: Record<string, string>;
|
||||||
|
};
|
||||||
|
const s = pkg?.scripts ?? {};
|
||||||
|
const parts: string[] = [];
|
||||||
|
if (s["check"] && typeof s["check"] === "string") parts.push(`bun run check`);
|
||||||
|
else if (s["typecheck"] && typeof s["typecheck"] === "string") parts.push(`bun run typecheck`);
|
||||||
|
if (s["lint"] && typeof s["lint"] === "string") parts.push(`bun run lint`);
|
||||||
|
if (s["test"] && typeof s["test"] === "string") parts.push(`bun run test`);
|
||||||
|
if (parts.length > 0) return parts.join(" && ");
|
||||||
|
} catch {
|
||||||
|
/* no package.json — fall through */
|
||||||
|
}
|
||||||
|
// Rust project.
|
||||||
|
try {
|
||||||
|
if (fs.existsSync(`${root}/Cargo.toml`)) return "cargo check && cargo test";
|
||||||
|
} catch {
|
||||||
|
/* ignore */
|
||||||
|
}
|
||||||
|
// Java/Maven.
|
||||||
|
try {
|
||||||
|
if (fs.existsSync(`${root}/pom.xml`) || fs.existsSync(`${root}/build.gradle`)) {
|
||||||
|
return "mvn test";
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
/* ignore */
|
||||||
|
}
|
||||||
|
return "bun run check && bun test";
|
||||||
|
}
|
||||||
|
|
||||||
/** Truncate a long tool output, preserving the head + truncation marker. */
|
/** Truncate a long tool output, preserving the head + truncation marker. */
|
||||||
export function truncateToolOutput(output: string): string {
|
export function truncateToolOutput(output: string): string {
|
||||||
if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output;
|
if (output.length <= TOOL_OUTPUT_MAX_CHARS) return output;
|
||||||
@@ -125,18 +208,32 @@ async function executeToolCall(
|
|||||||
executor: ToolExecutor,
|
executor: ToolExecutor,
|
||||||
sink: TurnEventSink,
|
sink: TurnEventSink,
|
||||||
tc: ToolCall,
|
tc: ToolCall,
|
||||||
|
signal?: AbortSignal,
|
||||||
): Promise<string> {
|
): Promise<string> {
|
||||||
const name = tc.function.name;
|
const name = tc.function.name;
|
||||||
const args = sanitizeToolArguments(tc.function.arguments);
|
const args = sanitizeToolArguments(tc.function.arguments);
|
||||||
|
|
||||||
let output: string;
|
const run = executor.execute(name, args as JsonValue).catch((e) => `Error: ${(e as Error).message}`);
|
||||||
try {
|
|
||||||
output = await executor.execute(name, args as JsonValue);
|
|
||||||
} catch (e) {
|
|
||||||
output = `Error: ${(e as Error).message}`;
|
|
||||||
}
|
|
||||||
|
|
||||||
const isError = isErrorOutput(output);
|
// Correct async cancellation: enforce a per-tool time budget and stop
|
||||||
|
// immediately if the turn is aborted while the tool is still running.
|
||||||
|
let timer: ReturnType<typeof setTimeout> | undefined;
|
||||||
|
const output = await Promise.race([
|
||||||
|
run,
|
||||||
|
new Promise<string>((resolve) => {
|
||||||
|
const onAbort = () => resolve("Error: Turn aborted by user");
|
||||||
|
timer = setTimeout(() => resolve(`Error: Tool timed out after ${TOOL_TIMEOUT_MS / 1000}s`), TOOL_TIMEOUT_MS);
|
||||||
|
signal?.addEventListener("abort", onAbort, { once: true });
|
||||||
|
// Release the abort listener + timer once the tool settles either way.
|
||||||
|
void run.finally(() => {
|
||||||
|
clearTimeout(timer);
|
||||||
|
signal?.removeEventListener("abort", onAbort);
|
||||||
|
});
|
||||||
|
}),
|
||||||
|
]);
|
||||||
|
if (timer) clearTimeout(timer);
|
||||||
|
|
||||||
|
const isError = isToolFailure(name, output);
|
||||||
const truncated = truncateToolOutput(output);
|
const truncated = truncateToolOutput(output);
|
||||||
|
|
||||||
sink.push({
|
sink.push({
|
||||||
@@ -156,6 +253,7 @@ async function executeToolCallsInParallel(
|
|||||||
executor: ToolExecutor,
|
executor: ToolExecutor,
|
||||||
sink: TurnEventSink,
|
sink: TurnEventSink,
|
||||||
toolCalls: ToolCall[],
|
toolCalls: ToolCall[],
|
||||||
|
signal?: AbortSignal,
|
||||||
): Promise<string[]> {
|
): Promise<string[]> {
|
||||||
// Simple bounded concurrency preserving input order.
|
// Simple bounded concurrency preserving input order.
|
||||||
const results: string[] = new Array(toolCalls.length);
|
const results: string[] = new Array(toolCalls.length);
|
||||||
@@ -165,7 +263,7 @@ async function executeToolCallsInParallel(
|
|||||||
while (true) {
|
while (true) {
|
||||||
const idx = next++;
|
const idx = next++;
|
||||||
if (idx >= toolCalls.length) return;
|
if (idx >= toolCalls.length) return;
|
||||||
results[idx] = await executeToolCall(executor, sink, toolCalls[idx]!);
|
results[idx] = await executeToolCall(executor, sink, toolCalls[idx]!, signal);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -179,8 +277,75 @@ async function executeToolCallsInParallel(
|
|||||||
/* -------------------------------------------------------------------------- */
|
/* -------------------------------------------------------------------------- */
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Compact oversized conversation history using AI summarisation. At most once
|
* Split history into a recent tail (kept verbatim, bound by CHAR budget so
|
||||||
* per turn. Keeps the last COMPACT_KEEP_TAIL messages.
|
* huge tool outputs don't monopolise it) and the evicted prefix to summarize.
|
||||||
|
*/
|
||||||
|
export function takeRecentTail(
|
||||||
|
messages: ChatMessage[],
|
||||||
|
tailChars = COMPACT_TAIL_CHARS,
|
||||||
|
minTail = COMPACT_KEEP_TAIL,
|
||||||
|
): { tail: ChatMessage[]; evicted: ChatMessage[] } {
|
||||||
|
if (messages.length <= minTail) return { tail: [...messages], evicted: [] };
|
||||||
|
let used = 0;
|
||||||
|
let keep = 0;
|
||||||
|
// Walk from the newest message backward. Always keep at least minTail; then
|
||||||
|
// stop once the aggregated char budget is exceeded.
|
||||||
|
for (let i = messages.length - 1; i >= 0; i--) {
|
||||||
|
const len = messages[i]!.content?.length ?? 0;
|
||||||
|
if (keep >= minTail && used + len > tailChars) break;
|
||||||
|
keep += 1;
|
||||||
|
used += len;
|
||||||
|
}
|
||||||
|
const tail = messages.slice(messages.length - keep);
|
||||||
|
const evicted = messages.slice(0, messages.length - keep);
|
||||||
|
return { tail, evicted };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reduce older history to a compact digest for the summarizer. Tool-result
|
||||||
|
* bodies are dropped entirely (they are noise for a summary); user requests
|
||||||
|
* get a generous preview; assistant text gets a shorter one. Caps the result
|
||||||
|
* at COMPACT_MAX_INPUT_CHARS so the summarizer sees a small, focused input.
|
||||||
|
*/
|
||||||
|
export function reduceMessagesToDigest(messages: ChatMessage[]): string {
|
||||||
|
let out = "";
|
||||||
|
let budget = COMPACT_MAX_INPUT_CHARS;
|
||||||
|
for (const m of messages) {
|
||||||
|
if (budget <= 0) break;
|
||||||
|
if (m.role === Roles.Tool) {
|
||||||
|
const name = m.name ?? "tool";
|
||||||
|
const line = `- tool ${name} executed\n`;
|
||||||
|
out += line;
|
||||||
|
budget -= line.length;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const text = m.content?.trim() ?? "";
|
||||||
|
if (text === "") continue; // assistant messages that only carried tool calls
|
||||||
|
const cap = m.role === Roles.User ? COMPACT_USER_PREVIEW_CHARS : COMPACT_TEXT_PREVIEW_CHARS;
|
||||||
|
let preview = text.replace(/\s*\n+\s*/g, " ");
|
||||||
|
if (preview.length > cap) {
|
||||||
|
preview = `${preview.slice(0, cap)}…[+${text.length - cap} ch]`;
|
||||||
|
}
|
||||||
|
const line = `- ${m.role}: ${preview}\n`;
|
||||||
|
out += line;
|
||||||
|
budget -= line.length;
|
||||||
|
}
|
||||||
|
// Hard guarantee: never exceed the summarizer input cap.
|
||||||
|
if (out.length > COMPACT_MAX_INPUT_CHARS) return out.slice(0, COMPACT_MAX_INPUT_CHARS);
|
||||||
|
return out.trim();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Compact oversized conversation history efficiently yet accurately.
|
||||||
|
*
|
||||||
|
* - Only the OLDER history is summarized; the recent tail is kept verbatim
|
||||||
|
* (bounded by char budget, not message count, so tool bloat can't win the
|
||||||
|
* budget).
|
||||||
|
* - The evicted history is REDUCED to a small digest before the summarizer
|
||||||
|
* sees it, so the LLM works on a focused input (cheaper + more accurate)
|
||||||
|
* instead of a ~60k-char dump.
|
||||||
|
* - On failure it degrades gracefully: keeps the recent tail and a marker,
|
||||||
|
* rather than wiping context to a useless sentinel.
|
||||||
*/
|
*/
|
||||||
export async function compactMessagesWithAi(
|
export async function compactMessagesWithAi(
|
||||||
messages: ChatMessage[],
|
messages: ChatMessage[],
|
||||||
@@ -188,17 +353,35 @@ export async function compactMessagesWithAi(
|
|||||||
): Promise<void> {
|
): Promise<void> {
|
||||||
if (messages.length <= COMPACT_KEEP_TAIL + 2) return;
|
if (messages.length <= COMPACT_KEEP_TAIL + 2) return;
|
||||||
|
|
||||||
const splitIdx = messages.length - COMPACT_KEEP_TAIL;
|
const { tail, evicted } = takeRecentTail(messages);
|
||||||
const evicted = messages.splice(0, splitIdx);
|
const digest = reduceMessagesToDigest(evicted);
|
||||||
|
if (digest.trim() === "") {
|
||||||
|
// Nothing worth summarizing (e.g. only stubbed tool bodies) — keep it all.
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
const summaryPrompt: ChatMessage[] = [systemMessage(compactionPrompt()), ...evicted, userMessage("Please summarise our previous conversation above for context continuity.")];
|
const summaryPrompt: ChatMessage[] = [
|
||||||
|
systemMessage(compactionPrompt()),
|
||||||
|
userMessage(
|
||||||
|
"The conversation below is the OLDER part of a session. Recent messages are kept separately, so do not preserve them. Compress the older part into the requested summary format.\n\n---\n" + digest,
|
||||||
|
),
|
||||||
|
];
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3);
|
const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3);
|
||||||
const summaryText = message.content ?? "Previous context summarised.";
|
const summaryText = message.content?.trim() ?? "";
|
||||||
messages.unshift(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText.trim()}`));
|
const rebuilt: ChatMessage[] = [];
|
||||||
|
if (summaryText !== "") {
|
||||||
|
rebuilt.push(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText}`));
|
||||||
|
} else {
|
||||||
|
rebuilt.push(systemMessage("[Earlier conversation summarized]"));
|
||||||
|
}
|
||||||
|
rebuilt.push(...tail);
|
||||||
|
messages.splice(0, messages.length, ...rebuilt);
|
||||||
} catch {
|
} catch {
|
||||||
messages.unshift(systemMessage("[Earlier conversation messages compacted to save context window]"));
|
// Graceful degradation: never wipe recent state on a summarizer failure.
|
||||||
|
const rebuilt: ChatMessage[] = [systemMessage("[Earlier context compressed; recent messages follow.]"), ...tail];
|
||||||
|
messages.splice(0, messages.length, ...rebuilt);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -268,9 +451,11 @@ export class AgentTurnServiceImpl {
|
|||||||
const abort = params.abort;
|
const abort = params.abort;
|
||||||
const { in_flight } = params;
|
const { in_flight } = params;
|
||||||
|
|
||||||
// Insert system prompt at index 0 with repo conventions loaded.
|
// Insert system prompt at index 0 with repo conventions + verify command.
|
||||||
const projectContext = buildProjectContext(params.workspace_roots[0] ?? ".");
|
const root = params.workspace_roots[0] ?? ".";
|
||||||
const systemPrompt = mainAgentPromptWithProjectContext(projectContext);
|
const projectContext = buildProjectContext(root);
|
||||||
|
const verifyCommand = deriveVerifyCommand(root);
|
||||||
|
const systemPrompt = mainAgentPromptWithProjectContext(projectContext, verifyCommand);
|
||||||
params.messages.unshift(systemMessage(systemPrompt));
|
params.messages.unshift(systemMessage(systemPrompt));
|
||||||
const originalCount = params.messages.length;
|
const originalCount = params.messages.length;
|
||||||
|
|
||||||
@@ -280,6 +465,12 @@ export class AgentTurnServiceImpl {
|
|||||||
|
|
||||||
const errors = new ErrorTracker();
|
const errors = new ErrorTracker();
|
||||||
let sawToolCalls = false;
|
let sawToolCalls = false;
|
||||||
|
let sawMutator = false;
|
||||||
|
let pendingVerify = false;
|
||||||
|
let verifyPrompted = false;
|
||||||
|
let noProgressStreak = 0;
|
||||||
|
let lastSignature: string | null = null;
|
||||||
|
let reviewPasses = 0;
|
||||||
|
|
||||||
for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) {
|
for (let iteration = 0; iteration < MAX_TURN_ITERATIONS; iteration++) {
|
||||||
// Check abort flag.
|
// Check abort flag.
|
||||||
@@ -293,12 +484,33 @@ export class AgentTurnServiceImpl {
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Convergence guard: bail out of a loop stuck re-issuing the same call.
|
||||||
|
if (noProgressStreak >= MAX_NO_PROGRESS_STREAK) {
|
||||||
|
this.push(sink, {
|
||||||
|
kind: "system_note",
|
||||||
|
systemKind: "warn",
|
||||||
|
message: "Stopping: the same tool call is being repeated without any progress.",
|
||||||
|
});
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify-after-edit: nudge the model to run the check before concluding.
|
||||||
|
if (pendingVerify && !verifyPrompted) {
|
||||||
|
params.messages.push(
|
||||||
|
systemMessage(
|
||||||
|
`[Flow] You just modified files. Run the verify command via \`bash\` now (${verifyCommand}) and resolve any failures before concluding your turn.`,
|
||||||
|
),
|
||||||
|
);
|
||||||
|
verifyPrompted = true;
|
||||||
|
}
|
||||||
|
|
||||||
// Auto-compact oversized history before the LLM call.
|
// Auto-compact oversized history before the LLM call.
|
||||||
await this.autoCompactIfNeeded(params.messages);
|
await this.autoCompactIfNeeded(params.messages);
|
||||||
|
|
||||||
// Adaptive generation parameters.
|
// Generation parameters: a config override wins, else fall back to the
|
||||||
const maxTokens = adaptiveMaxTokens(requestLen);
|
// adaptive defaults (lower temperature once the model is doing tool work).
|
||||||
const temperature = sawToolCalls ? 0.2 : 0.7;
|
const maxTokens = params.max_tokens ?? adaptiveMaxTokens(requestLen);
|
||||||
|
const temperature = params.temperature ?? (sawToolCalls ? 0.2 : 0.7);
|
||||||
|
|
||||||
this.push(sink, { kind: "stream_start" });
|
this.push(sink, { kind: "stream_start" });
|
||||||
|
|
||||||
@@ -336,19 +548,52 @@ export class AgentTurnServiceImpl {
|
|||||||
const parallel = toolCalls.length > 1 && toolCalls.every(isParallelSafe);
|
const parallel = toolCalls.length > 1 && toolCalls.every(isParallelSafe);
|
||||||
|
|
||||||
const outputs = parallel
|
const outputs = parallel
|
||||||
? await executeToolCallsInParallel(this.toolExecutor, sink, toolCalls)
|
? await executeToolCallsInParallel(this.toolExecutor, sink, toolCalls, abort.signal)
|
||||||
: await (async () => {
|
: await (async () => {
|
||||||
const seq: string[] = [];
|
const seq: string[] = [];
|
||||||
for (const tc of toolCalls) seq.push(await executeToolCall(this.toolExecutor, sink, tc));
|
for (const tc of toolCalls) seq.push(await executeToolCall(this.toolExecutor, sink, tc, abort.signal));
|
||||||
return seq;
|
return seq;
|
||||||
})();
|
})();
|
||||||
|
|
||||||
|
let mutated = false;
|
||||||
|
let verified = false;
|
||||||
|
let batchSignature = "";
|
||||||
for (let i = 0; i < toolCalls.length; i++) {
|
for (let i = 0; i < toolCalls.length; i++) {
|
||||||
const tc = toolCalls[i]!;
|
const tc = toolCalls[i]!;
|
||||||
const output = outputs[i]!;
|
const output = outputs[i]!;
|
||||||
if (isErrorOutput(output)) errors.record(tc.function.name, output, params.messages);
|
if (isToolFailure(tc.function.name, output)) errors.record(tc.function.name, output, params.messages);
|
||||||
|
if (MUTATOR_TOOLS.has(tc.function.name)) {
|
||||||
|
mutated = true;
|
||||||
|
sawMutator = true;
|
||||||
|
}
|
||||||
|
if ((tc.function.name === "bash" || tc.function.name === "bash_output") && bashExitCode(output) === 0) {
|
||||||
|
verified = true;
|
||||||
|
}
|
||||||
|
// Batch signature: concat of tool+output for no-progress detection.
|
||||||
|
batchSignature += `${tc.function.name} | ||||||