feat(tui): implement message compaction and summarization logic for efficient conversation handling
This commit is contained in:
@@ -8,6 +8,9 @@ import {
|
||||
isToolFailure,
|
||||
deriveVerifyCommand,
|
||||
AgentTurnServiceImpl,
|
||||
takeRecentTail,
|
||||
reduceMessagesToDigest,
|
||||
compactMessagesWithAi,
|
||||
} from "./turn_service.ts";
|
||||
import {
|
||||
type ChatMessage,
|
||||
@@ -264,3 +267,118 @@ describe("runTurn self-review", () => {
|
||||
expect(events.some((e) => e.kind === "review_usage")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
/* ── Compaction: efficient + accurate ─────────────────────────────── */
|
||||
|
||||
describe("takeRecentTail", () => {
|
||||
it("keeps the most recent messages and evicts an oversized older blob", () => {
|
||||
const messages = [
|
||||
toolResultMessage("t0", "y".repeat(5000)), // huge old tool output
|
||||
userMessage("old request"),
|
||||
assistantMessage("recent reply A"),
|
||||
userMessage("recent reply B"),
|
||||
];
|
||||
const { tail, evicted } = takeRecentTail(messages, 25, 2);
|
||||
expect(evicted.length).toBe(2); // the blob + the old request
|
||||
expect(tail.map((m) => m.content)).toEqual(["recent reply A", "recent reply B"]);
|
||||
});
|
||||
|
||||
it("gives the whole history as tail when it is small enough", () => {
|
||||
const messages = [userMessage("a"), assistantMessage("b")];
|
||||
const { tail, evicted } = takeRecentTail(messages, 1000, 6);
|
||||
expect(evicted).toEqual([]);
|
||||
expect(tail.length).toBe(2);
|
||||
});
|
||||
|
||||
it("does not let a huge tool output blow the tail past the min count", () => {
|
||||
// minTail 2: newest two kept regardless; the huge tool body stays evicted.
|
||||
const messages = [
|
||||
toolResultMessage("t0", "x".repeat(50000)),
|
||||
userMessage("keep1"),
|
||||
assistantMessage("keep2"),
|
||||
];
|
||||
const { tail, evicted } = takeRecentTail(messages, 8000, 2);
|
||||
expect(tail.map((m) => m.content)).toEqual(["keep1", "keep2"]);
|
||||
expect(evicted.length).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("reduceMessagesToDigest", () => {
|
||||
it("stubs tool bodies entirely and previews text", () => {
|
||||
const digest = reduceMessagesToDigest([
|
||||
toolResultMessage("t0", "y".repeat(5000)),
|
||||
userMessage("short request"),
|
||||
]);
|
||||
expect(digest).toContain("- tool");
|
||||
expect(digest).not.toContain("yyyy");
|
||||
expect(digest).toContain("short request");
|
||||
});
|
||||
|
||||
it("keeps user previews well above the shorter assistant cap", () => {
|
||||
const long = "w".repeat(2000);
|
||||
// User budget is 1500 chars — larger than the 400-char assistant cap, so
|
||||
// a long user request survives far more of its body than a long assistant
|
||||
// reply would (requests matter most for summary accuracy).
|
||||
const digestUser = reduceMessagesToDigest([userMessage(long)]);
|
||||
const digestAssistant = reduceMessagesToDigest([assistantMessage(long)]);
|
||||
expect(digestUser.length).toBeGreaterThan(800);
|
||||
expect(digestAssistant.length).toBeLessThan(500);
|
||||
});
|
||||
|
||||
it("respects the hard input cap", () => {
|
||||
const many = [];
|
||||
for (let i = 0; i < 200; i++) many.push(userMessage("r".repeat(300)));
|
||||
const digest = reduceMessagesToDigest(many);
|
||||
expect(digest.length).toBeLessThanOrEqual(20_000 + 40);
|
||||
});
|
||||
});
|
||||
|
||||
describe("compactMessagesWithAi", () => {
|
||||
it("keeps the recent tail and prepends an AI summary", async () => {
|
||||
// Long enough to bypass the min-size guard, with an oversized OLD tool blob
|
||||
// (should be evicted + stubbed, not kept verbatim).
|
||||
const messages: ChatMessage[] = [
|
||||
userMessage("old request one"),
|
||||
assistantMessage("old response"),
|
||||
toolResultMessage("t0", "big".repeat(4000)),
|
||||
];
|
||||
for (let i = 0; i < 7; i++) messages.push(userMessage(`mid ${i}`));
|
||||
messages.push(toolResultMessage("t1", "tail-result"));
|
||||
messages.push(userMessage("newest request"));
|
||||
messages.push(assistantMessage("newest response"));
|
||||
|
||||
const provider: ProviderService = {
|
||||
async chat() {
|
||||
return { message: { role: "assistant", content: "## Requests\n- old request one" }, usage: null };
|
||||
},
|
||||
async chatStream() {
|
||||
return { message: { role: "assistant", content: null }, usage: null };
|
||||
},
|
||||
};
|
||||
await compactMessagesWithAi(messages, provider);
|
||||
// Summary first, recent tail preserved verbatim, tool blobs gone.
|
||||
expect(messages[0]?.content).toContain("[AI Summary of Previous Conversation]");
|
||||
expect(messages.some((m) => m.content === "newest request")).toBe(true);
|
||||
expect(messages.some((m) => m.content === "newest response")).toBe(true);
|
||||
expect(messages.some((m) => (m.content ?? "").includes("big".repeat(10)))).toBe(false);
|
||||
});
|
||||
|
||||
it("degrades gracefully on summarizer failure, preserving the tail", async () => {
|
||||
const messages = [
|
||||
userMessage("old"),
|
||||
toolResultMessage("t0", "x".repeat(100)),
|
||||
userMessage("recent request"),
|
||||
];
|
||||
const failing: ProviderService = {
|
||||
async chat() {
|
||||
throw new Error("network down");
|
||||
},
|
||||
async chatStream() {
|
||||
return { message: { role: "assistant", content: null }, usage: null };
|
||||
},
|
||||
};
|
||||
await compactMessagesWithAi(messages, failing);
|
||||
// The recent message survives even when the LLM call fails.
|
||||
expect(messages.some((m) => m.content === "recent request")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
type ToolCall,
|
||||
type ToolDef,
|
||||
type JsonValue,
|
||||
Roles,
|
||||
systemMessage,
|
||||
userMessage,
|
||||
assistantMessage,
|
||||
@@ -40,6 +41,15 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000;
|
||||
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
|
||||
const COMPACT_KEEP_TAIL = 6;
|
||||
|
||||
/** Recent-tail budget (chars): the most recent history kept verbatim on compact. */
|
||||
const COMPACT_TAIL_CHARS = 8_000;
|
||||
/** Caps what the summarizer actually receives, keeping input small and focused. */
|
||||
const COMPACT_MAX_INPUT_CHARS = 20_000;
|
||||
/** Preview budget for a user request inside the digest (requests matter most). */
|
||||
const COMPACT_USER_PREVIEW_CHARS = 1_500;
|
||||
/** Preview budget for assistant text inside the digest. */
|
||||
const COMPACT_TEXT_PREVIEW_CHARS = 400;
|
||||
|
||||
/** Tools that mutate the filesystem — after these, a verify run is expected. */
|
||||
const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]);
|
||||
|
||||
@@ -250,8 +260,75 @@ async function executeToolCallsInParallel(
|
||||
/* -------------------------------------------------------------------------- */
|
||||
|
||||
/**
|
||||
* Compact oversized conversation history using AI summarisation. At most once
|
||||
* per turn. Keeps the last COMPACT_KEEP_TAIL messages.
|
||||
* Split history into a recent tail (kept verbatim, bound by CHAR budget so
|
||||
* huge tool outputs don't monopolise it) and the evicted prefix to summarize.
|
||||
*/
|
||||
export function takeRecentTail(
|
||||
messages: ChatMessage[],
|
||||
tailChars = COMPACT_TAIL_CHARS,
|
||||
minTail = COMPACT_KEEP_TAIL,
|
||||
): { tail: ChatMessage[]; evicted: ChatMessage[] } {
|
||||
if (messages.length <= minTail) return { tail: [...messages], evicted: [] };
|
||||
let used = 0;
|
||||
let keep = 0;
|
||||
// Walk from the newest message backward. Always keep at least minTail; then
|
||||
// stop once the aggregated char budget is exceeded.
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const len = messages[i]!.content?.length ?? 0;
|
||||
if (keep >= minTail && used + len > tailChars) break;
|
||||
keep += 1;
|
||||
used += len;
|
||||
}
|
||||
const tail = messages.slice(messages.length - keep);
|
||||
const evicted = messages.slice(0, messages.length - keep);
|
||||
return { tail, evicted };
|
||||
}
|
||||
|
||||
/**
|
||||
* Reduce older history to a compact digest for the summarizer. Tool-result
|
||||
* bodies are dropped entirely (they are noise for a summary); user requests
|
||||
* get a generous preview; assistant text gets a shorter one. Caps the result
|
||||
* at COMPACT_MAX_INPUT_CHARS so the summarizer sees a small, focused input.
|
||||
*/
|
||||
export function reduceMessagesToDigest(messages: ChatMessage[]): string {
|
||||
let out = "";
|
||||
let budget = COMPACT_MAX_INPUT_CHARS;
|
||||
for (const m of messages) {
|
||||
if (budget <= 0) break;
|
||||
if (m.role === Roles.Tool) {
|
||||
const name = m.name ?? "tool";
|
||||
const line = `- tool ${name} executed\n`;
|
||||
out += line;
|
||||
budget -= line.length;
|
||||
continue;
|
||||
}
|
||||
const text = m.content?.trim() ?? "";
|
||||
if (text === "") continue; // assistant messages that only carried tool calls
|
||||
const cap = m.role === Roles.User ? COMPACT_USER_PREVIEW_CHARS : COMPACT_TEXT_PREVIEW_CHARS;
|
||||
let preview = text.replace(/\s*\n+\s*/g, " ");
|
||||
if (preview.length > cap) {
|
||||
preview = `${preview.slice(0, cap)}…[+${text.length - cap} ch]`;
|
||||
}
|
||||
const line = `- ${m.role}: ${preview}\n`;
|
||||
out += line;
|
||||
budget -= line.length;
|
||||
}
|
||||
// Hard guarantee: never exceed the summarizer input cap.
|
||||
if (out.length > COMPACT_MAX_INPUT_CHARS) return out.slice(0, COMPACT_MAX_INPUT_CHARS);
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Compact oversized conversation history efficiently yet accurately.
|
||||
*
|
||||
* - Only the OLDER history is summarized; the recent tail is kept verbatim
|
||||
* (bounded by char budget, not message count, so tool bloat can't win the
|
||||
* budget).
|
||||
* - The evicted history is REDUCED to a small digest before the summarizer
|
||||
* sees it, so the LLM works on a focused input (cheaper + more accurate)
|
||||
* instead of a ~60k-char dump.
|
||||
* - On failure it degrades gracefully: keeps the recent tail and a marker,
|
||||
* rather than wiping context to a useless sentinel.
|
||||
*/
|
||||
export async function compactMessagesWithAi(
|
||||
messages: ChatMessage[],
|
||||
@@ -259,17 +336,35 @@ export async function compactMessagesWithAi(
|
||||
): Promise<void> {
|
||||
if (messages.length <= COMPACT_KEEP_TAIL + 2) return;
|
||||
|
||||
const splitIdx = messages.length - COMPACT_KEEP_TAIL;
|
||||
const evicted = messages.splice(0, splitIdx);
|
||||
const { tail, evicted } = takeRecentTail(messages);
|
||||
const digest = reduceMessagesToDigest(evicted);
|
||||
if (digest.trim() === "") {
|
||||
// Nothing worth summarizing (e.g. only stubbed tool bodies) — keep it all.
|
||||
return;
|
||||
}
|
||||
|
||||
const summaryPrompt: ChatMessage[] = [systemMessage(compactionPrompt()), ...evicted, userMessage("Please summarise our previous conversation above for context continuity.")];
|
||||
const summaryPrompt: ChatMessage[] = [
|
||||
systemMessage(compactionPrompt()),
|
||||
userMessage(
|
||||
"The conversation below is the OLDER part of a session. Recent messages are kept separately, so do not preserve them. Compress the older part into the requested summary format.\n\n---\n" + digest,
|
||||
),
|
||||
];
|
||||
|
||||
try {
|
||||
const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3);
|
||||
const summaryText = message.content ?? "Previous context summarised.";
|
||||
messages.unshift(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText.trim()}`));
|
||||
const summaryText = message.content?.trim() ?? "";
|
||||
const rebuilt: ChatMessage[] = [];
|
||||
if (summaryText !== "") {
|
||||
rebuilt.push(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText}`));
|
||||
} else {
|
||||
rebuilt.push(systemMessage("[Earlier conversation summarized]"));
|
||||
}
|
||||
rebuilt.push(...tail);
|
||||
messages.splice(0, messages.length, ...rebuilt);
|
||||
} catch {
|
||||
messages.unshift(systemMessage("[Earlier conversation messages compacted to save context window]"));
|
||||
// Graceful degradation: never wipe recent state on a summarizer failure.
|
||||
const rebuilt: ChatMessage[] = [systemMessage("[Earlier context compressed; recent messages follow.]"), ...tail];
|
||||
messages.splice(0, messages.length, ...rebuilt);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user