feat(tui): implement message compaction and summarization logic for efficient conversation handling

This commit is contained in:
asepharyana
2026-09-03 16:13:29 +07:00
parent f98030f28a
commit e6718fa118
2 changed files with 221 additions and 8 deletions
@@ -8,6 +8,9 @@ import {
isToolFailure, isToolFailure,
deriveVerifyCommand, deriveVerifyCommand,
AgentTurnServiceImpl, AgentTurnServiceImpl,
takeRecentTail,
reduceMessagesToDigest,
compactMessagesWithAi,
} from "./turn_service.ts"; } from "./turn_service.ts";
import { import {
type ChatMessage, type ChatMessage,
@@ -264,3 +267,118 @@ describe("runTurn self-review", () => {
expect(events.some((e) => e.kind === "review_usage")).toBe(true); expect(events.some((e) => e.kind === "review_usage")).toBe(true);
}); });
}); });
/* ── Compaction: efficient + accurate ─────────────────────────────── */
describe("takeRecentTail", () => {
it("keeps the most recent messages and evicts an oversized older blob", () => {
const messages = [
toolResultMessage("t0", "y".repeat(5000)), // huge old tool output
userMessage("old request"),
assistantMessage("recent reply A"),
userMessage("recent reply B"),
];
const { tail, evicted } = takeRecentTail(messages, 25, 2);
expect(evicted.length).toBe(2); // the blob + the old request
expect(tail.map((m) => m.content)).toEqual(["recent reply A", "recent reply B"]);
});
it("gives the whole history as tail when it is small enough", () => {
const messages = [userMessage("a"), assistantMessage("b")];
const { tail, evicted } = takeRecentTail(messages, 1000, 6);
expect(evicted).toEqual([]);
expect(tail.length).toBe(2);
});
it("does not let a huge tool output blow the tail past the min count", () => {
// minTail 2: newest two kept regardless; the huge tool body stays evicted.
const messages = [
toolResultMessage("t0", "x".repeat(50000)),
userMessage("keep1"),
assistantMessage("keep2"),
];
const { tail, evicted } = takeRecentTail(messages, 8000, 2);
expect(tail.map((m) => m.content)).toEqual(["keep1", "keep2"]);
expect(evicted.length).toBe(1);
});
});
describe("reduceMessagesToDigest", () => {
it("stubs tool bodies entirely and previews text", () => {
const digest = reduceMessagesToDigest([
toolResultMessage("t0", "y".repeat(5000)),
userMessage("short request"),
]);
expect(digest).toContain("- tool");
expect(digest).not.toContain("yyyy");
expect(digest).toContain("short request");
});
it("keeps user previews well above the shorter assistant cap", () => {
const long = "w".repeat(2000);
// User budget is 1500 chars — larger than the 400-char assistant cap, so
// a long user request survives far more of its body than a long assistant
// reply would (requests matter most for summary accuracy).
const digestUser = reduceMessagesToDigest([userMessage(long)]);
const digestAssistant = reduceMessagesToDigest([assistantMessage(long)]);
expect(digestUser.length).toBeGreaterThan(800);
expect(digestAssistant.length).toBeLessThan(500);
});
it("respects the hard input cap", () => {
const many = [];
for (let i = 0; i < 200; i++) many.push(userMessage("r".repeat(300)));
const digest = reduceMessagesToDigest(many);
expect(digest.length).toBeLessThanOrEqual(20_000 + 40);
});
});
describe("compactMessagesWithAi", () => {
it("keeps the recent tail and prepends an AI summary", async () => {
// Long enough to bypass the min-size guard, with an oversized OLD tool blob
// (should be evicted + stubbed, not kept verbatim).
const messages: ChatMessage[] = [
userMessage("old request one"),
assistantMessage("old response"),
toolResultMessage("t0", "big".repeat(4000)),
];
for (let i = 0; i < 7; i++) messages.push(userMessage(`mid ${i}`));
messages.push(toolResultMessage("t1", "tail-result"));
messages.push(userMessage("newest request"));
messages.push(assistantMessage("newest response"));
const provider: ProviderService = {
async chat() {
return { message: { role: "assistant", content: "## Requests\n- old request one" }, usage: null };
},
async chatStream() {
return { message: { role: "assistant", content: null }, usage: null };
},
};
await compactMessagesWithAi(messages, provider);
// Summary first, recent tail preserved verbatim, tool blobs gone.
expect(messages[0]?.content).toContain("[AI Summary of Previous Conversation]");
expect(messages.some((m) => m.content === "newest request")).toBe(true);
expect(messages.some((m) => m.content === "newest response")).toBe(true);
expect(messages.some((m) => (m.content ?? "").includes("big".repeat(10)))).toBe(false);
});
it("degrades gracefully on summarizer failure, preserving the tail", async () => {
const messages = [
userMessage("old"),
toolResultMessage("t0", "x".repeat(100)),
userMessage("recent request"),
];
const failing: ProviderService = {
async chat() {
throw new Error("network down");
},
async chatStream() {
return { message: { role: "assistant", content: null }, usage: null };
},
};
await compactMessagesWithAi(messages, failing);
// The recent message survives even when the LLM call fails.
expect(messages.some((m) => m.content === "recent request")).toBe(true);
});
});
+103 -8
View File
@@ -8,6 +8,7 @@ import {
type ToolCall, type ToolCall,
type ToolDef, type ToolDef,
type JsonValue, type JsonValue,
Roles,
systemMessage, systemMessage,
userMessage, userMessage,
assistantMessage, assistantMessage,
@@ -40,6 +41,15 @@ const PROJECT_CONTEXT_MAX_CHARS = 12_000;
const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"]; const RULE_FILENAMES = ["AGENTS.md", "agent.md", "CLAUDE.md", "claude.md", ".cursorrules", ".zesdexrules"];
const COMPACT_KEEP_TAIL = 6; const COMPACT_KEEP_TAIL = 6;
/** Recent-tail budget (chars): the most recent history kept verbatim on compact. */
const COMPACT_TAIL_CHARS = 8_000;
/** Caps what the summarizer actually receives, keeping input small and focused. */
const COMPACT_MAX_INPUT_CHARS = 20_000;
/** Preview budget for a user request inside the digest (requests matter most). */
const COMPACT_USER_PREVIEW_CHARS = 1_500;
/** Preview budget for assistant text inside the digest. */
const COMPACT_TEXT_PREVIEW_CHARS = 400;
/** Tools that mutate the filesystem — after these, a verify run is expected. */ /** Tools that mutate the filesystem — after these, a verify run is expected. */
const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]); const MUTATOR_TOOLS = new Set(["write", "edit", "delete"]);
@@ -250,8 +260,75 @@ async function executeToolCallsInParallel(
/* -------------------------------------------------------------------------- */ /* -------------------------------------------------------------------------- */
/** /**
* Compact oversized conversation history using AI summarisation. At most once * Split history into a recent tail (kept verbatim, bound by CHAR budget so
* per turn. Keeps the last COMPACT_KEEP_TAIL messages. * huge tool outputs don't monopolise it) and the evicted prefix to summarize.
*/
export function takeRecentTail(
messages: ChatMessage[],
tailChars = COMPACT_TAIL_CHARS,
minTail = COMPACT_KEEP_TAIL,
): { tail: ChatMessage[]; evicted: ChatMessage[] } {
if (messages.length <= minTail) return { tail: [...messages], evicted: [] };
let used = 0;
let keep = 0;
// Walk from the newest message backward. Always keep at least minTail; then
// stop once the aggregated char budget is exceeded.
for (let i = messages.length - 1; i >= 0; i--) {
const len = messages[i]!.content?.length ?? 0;
if (keep >= minTail && used + len > tailChars) break;
keep += 1;
used += len;
}
const tail = messages.slice(messages.length - keep);
const evicted = messages.slice(0, messages.length - keep);
return { tail, evicted };
}
/**
* Reduce older history to a compact digest for the summarizer. Tool-result
* bodies are dropped entirely (they are noise for a summary); user requests
* get a generous preview; assistant text gets a shorter one. Caps the result
* at COMPACT_MAX_INPUT_CHARS so the summarizer sees a small, focused input.
*/
export function reduceMessagesToDigest(messages: ChatMessage[]): string {
let out = "";
let budget = COMPACT_MAX_INPUT_CHARS;
for (const m of messages) {
if (budget <= 0) break;
if (m.role === Roles.Tool) {
const name = m.name ?? "tool";
const line = `- tool ${name} executed\n`;
out += line;
budget -= line.length;
continue;
}
const text = m.content?.trim() ?? "";
if (text === "") continue; // assistant messages that only carried tool calls
const cap = m.role === Roles.User ? COMPACT_USER_PREVIEW_CHARS : COMPACT_TEXT_PREVIEW_CHARS;
let preview = text.replace(/\s*\n+\s*/g, " ");
if (preview.length > cap) {
preview = `${preview.slice(0, cap)}…[+${text.length - cap} ch]`;
}
const line = `- ${m.role}: ${preview}\n`;
out += line;
budget -= line.length;
}
// Hard guarantee: never exceed the summarizer input cap.
if (out.length > COMPACT_MAX_INPUT_CHARS) return out.slice(0, COMPACT_MAX_INPUT_CHARS);
return out.trim();
}
/**
* Compact oversized conversation history efficiently yet accurately.
*
* - Only the OLDER history is summarized; the recent tail is kept verbatim
* (bounded by char budget, not message count, so tool bloat can't win the
* budget).
* - The evicted history is REDUCED to a small digest before the summarizer
* sees it, so the LLM works on a focused input (cheaper + more accurate)
* instead of a ~60k-char dump.
* - On failure it degrades gracefully: keeps the recent tail and a marker,
* rather than wiping context to a useless sentinel.
*/ */
export async function compactMessagesWithAi( export async function compactMessagesWithAi(
messages: ChatMessage[], messages: ChatMessage[],
@@ -259,17 +336,35 @@ export async function compactMessagesWithAi(
): Promise<void> { ): Promise<void> {
if (messages.length <= COMPACT_KEEP_TAIL + 2) return; if (messages.length <= COMPACT_KEEP_TAIL + 2) return;
const splitIdx = messages.length - COMPACT_KEEP_TAIL; const { tail, evicted } = takeRecentTail(messages);
const evicted = messages.splice(0, splitIdx); const digest = reduceMessagesToDigest(evicted);
if (digest.trim() === "") {
// Nothing worth summarizing (e.g. only stubbed tool bodies) — keep it all.
return;
}
const summaryPrompt: ChatMessage[] = [systemMessage(compactionPrompt()), ...evicted, userMessage("Please summarise our previous conversation above for context continuity.")]; const summaryPrompt: ChatMessage[] = [
systemMessage(compactionPrompt()),
userMessage(
"The conversation below is the OLDER part of a session. Recent messages are kept separately, so do not preserve them. Compress the older part into the requested summary format.\n\n---\n" + digest,
),
];
try { try {
const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3); const { message } = await provider.chat(summaryPrompt, undefined, 1024, 0.3);
const summaryText = message.content ?? "Previous context summarised."; const summaryText = message.content?.trim() ?? "";
messages.unshift(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText.trim()}`)); const rebuilt: ChatMessage[] = [];
if (summaryText !== "") {
rebuilt.push(systemMessage(`[AI Summary of Previous Conversation]\n${summaryText}`));
} else {
rebuilt.push(systemMessage("[Earlier conversation summarized]"));
}
rebuilt.push(...tail);
messages.splice(0, messages.length, ...rebuilt);
} catch { } catch {
messages.unshift(systemMessage("[Earlier conversation messages compacted to save context window]")); // Graceful degradation: never wipe recent state on a summarizer failure.
const rebuilt: ChatMessage[] = [systemMessage("[Earlier context compressed; recent messages follow.]"), ...tail];
messages.splice(0, messages.length, ...rebuilt);
} }
} }