Make compaction bounded and report it once per turn

The fixed three-message tool window collapsed long transcripts to a handful of messages: a 405-message run kept two of 202 tool calls, and the model re-ran what it could no longer see. Pruning now drops reasoning first and keeps the widest recent tool tail that fits a ladder, the SDK carries that view into later steps, and the turn emits one compaction event instead of one per step.

Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent)

Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
This commit is contained in:
Muhammad Zakir Ramadhan
2026-09-03 16:42:02 +07:00
co-authored by Sisyphus
parent 9e03512cb0
commit 83f4399e64
4 changed files with 227 additions and 9 deletions
+54
View File
@@ -141,3 +141,57 @@ export function prunePreservingItems(options: PruneOptions): ModelMessage[] {
const pruned = pruneMessages(options);
return dropOrphanedResults(detachOrphanedItems(options.messages, pruned));
}
/**
* How many trailing messages keep their tool content, widest first.
*
* One agent step is two messages — the assistant's tool call and the tool message
* answering it — so 64 is about 32 steps of memory.
*/
const KEEP_LADDER = [64, 32, 16, 8, 4] as const;
export type FitOptions = {
messages: ModelMessage[];
/** Estimated tokens the wire history must come in under. */
threshold: number;
estimate: (messages: ModelMessage[]) => number;
};
/**
* Prunes only as hard as the threshold requires.
*
* A fixed `before-last-3-messages` is catastrophic on an agent transcript, because
* nearly every assistant and tool message there consists of nothing but tool parts:
* stripping them empties the message, `emptyMessages: 'remove'` deletes it, and a
* 405-message history collapses to five. Measured on a synthetic run of 202 steps —
* two surviving tool calls out of 202.
*
* That is not a cost problem, it is a correctness one. The model loses its record of
* what it already ran, so it runs it again, the history grows, the threshold is
* crossed again, and the turn never converges. It looks like `git_status` and
* `list_dir` being called in a circle with a compaction notice between them.
*
* So: drop reasoning first, since it is never needed on the wire, and only reach for
* tool content if that was not enough — keeping as much of the recent tail as fits.
* The widest rung that comes in under the threshold wins; if even the narrowest does
* not, the narrowest is returned, because sending something is better than sending a
* request that will be rejected for size.
*/
export function pruneToFit({ messages, threshold, estimate }: FitOptions): ModelMessage[] {
const withoutReasoning = prunePreservingItems({ messages, reasoning: 'all', emptyMessages: 'remove' });
if (estimate(withoutReasoning) <= threshold) return withoutReasoning;
let narrowest = withoutReasoning;
for (const keep of KEEP_LADDER) {
narrowest = prunePreservingItems({
messages,
reasoning: 'all',
toolCalls: `before-last-${keep}-messages`,
emptyMessages: 'remove',
});
if (estimate(narrowest) <= threshold) return narrowest;
}
return narrowest;
}
export { KEEP_LADDER };
+42 -8
View File
@@ -15,7 +15,7 @@ import { Notebook, type NotebookState } from './notebook';
import { Permissions, type PermissionConfig } from './permission';
import type { PluginHost } from './plugins';
import { systemPrompt } from './prompt';
import { prunePreservingItems } from './prune';
import { pruneToFit } from './prune';
import { createSkillTool, renderSkills, type Skill } from './skills';
import { disabledToolNames, onBashOutput, tools as builtinTools, type ToolSetName } from './tools';
@@ -29,6 +29,8 @@ export type ApprovalRequest = {
suggestedPattern: string;
/** Set when the call is being asked about because it repeated, not because of a rule. */
repeated?: boolean;
/** Set when a `worker` subagent is asking, not the main agent. */
subagent?: boolean;
};
/** 'once' runs this call only; 'always' whitelists the suggested pattern for the session. */
@@ -136,6 +138,39 @@ export class Session {
});
}
/**
* The approval channel a `worker` subagent uses for its gated calls.
*
* Same rules, same prompt, same grants as a direct call: a subagent that could
* approve its own writes would be a way to launder a tool call past the user.
* Handed to `createTaskTool` from cli.tsx, which is where the two are wired.
*/
approveForSubagent(): (req: { toolName: string; input: unknown }) => Promise<boolean> {
return async ({ toolName, input }) => {
const blocked = await this.opts.plugins?.guard({
toolName,
input,
cwd: this.opts.cwd ?? process.cwd(),
});
if (blocked) return false;
const { decision, pattern } = this.permissions.check(toolName, input);
if (decision === 'deny') return false;
if (decision === 'allow') return true;
const answer = await this.opts.askApproval({
approvalId: `sub:${toolName}`,
toolName,
input,
...(pattern ? { matchedPattern: pattern } : {}),
suggestedPattern: this.permissions.suggest(toolName, input),
subagent: true,
});
if (answer === 'always') this.permissions.grant(toolName, this.permissions.suggest(toolName, input));
return answer !== 'deny';
};
}
setModel(model: LanguageModel): void {
this.model = model;
}
@@ -318,6 +353,7 @@ export class Session {
): AsyncGenerator<AgentEvent> {
// Each iteration is one model run. A run ends either finished, or suspended
// on tool approvals, in which case we collect decisions and run again.
let compactionReported = false;
while (true) {
const pending: ApprovalRequest[] = [];
const compactions: Extract<AgentEvent, { type: 'compacted' }>[] = [];
@@ -341,14 +377,12 @@ export class Session {
// visible to the steps that follow it, not only to the next turn.
const instructions = this.systemFor();
if (estimateTokens(messages) <= threshold) return { instructions };
const pruned = prunePreservingItems({
messages,
reasoning: 'all',
toolCalls: 'before-last-3-messages',
emptyMessages: 'remove',
});
const pruned = pruneToFit({ messages, threshold, estimate: estimateTokens });
// prepareStep cannot yield, so queue the notice and drain it in the loop.
compactions.push({ type: 'compacted', before: messages.length, after: pruned.length });
if (!compactionReported) {
compactions.push({ type: 'compacted', before: messages.length, after: pruned.length });
compactionReported = true;
}
return { instructions, messages: pruned };
},
});