Send a compacted history inline, with no provider item references

Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent)

Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
This commit is contained in:
Muhammad Zakir Ramadhan
2026-09-05 17:29:40 +07:00
co-authored by Sisyphus
parent a8a3ffcb5c
commit 4f6ed43fed
2 changed files with 51 additions and 7 deletions
+36 -7
View File
@@ -97,6 +97,31 @@ const ANSWER_PARTS = new Set(['tool-result', 'tool-error']);
const anyParts = (message: ModelMessage): Part[] => const anyParts = (message: ModelMessage): Part[] =>
Array.isArray(message.content) ? (message.content as Part[]) : []; Array.isArray(message.content) ? (message.content as Part[]) : [];
/**
* Every part again with its provider `itemId` gone, so the history goes out inline
* rather than as `item_reference` entries pointing at provider-side storage.
*
* A reference only resolves while the provider still holds that item; once it does
* not, the request is rejected with 404 "Item with id '...' not found" and no retry
* of the same history can succeed. The content is already in the local history, so
* inlining loses nothing.
*/
export function detachProviderItems(messages: ModelMessage[]): ModelMessage[] {
return messages.map((message) => {
const parts = anyParts(message);
if (parts.length === 0) return message;
let changed = false;
const next = parts.map((part) => {
if (itemId(part) === undefined) return part;
changed = true;
return withoutItemId(part);
});
return changed ? ({ ...message, content: next } as ModelMessage) : message;
});
}
/** /**
* Drops tool results whose tool call is gone. * Drops tool results whose tool call is gone.
* *
@@ -178,17 +203,21 @@ export type FitOptions = {
* request that will be rejected for size. * request that will be rejected for size.
*/ */
export function pruneToFit({ messages, threshold, estimate }: FitOptions): ModelMessage[] { export function pruneToFit({ messages, threshold, estimate }: FitOptions): ModelMessage[] {
const withoutReasoning = prunePreservingItems({ messages, reasoning: 'all', emptyMessages: 'remove' }); const withoutReasoning = detachProviderItems(
prunePreservingItems({ messages, reasoning: 'all', emptyMessages: 'remove' }),
);
if (estimate(withoutReasoning) <= threshold) return withoutReasoning; if (estimate(withoutReasoning) <= threshold) return withoutReasoning;
let narrowest = withoutReasoning; let narrowest = withoutReasoning;
for (const keep of KEEP_LADDER) { for (const keep of KEEP_LADDER) {
narrowest = prunePreservingItems({ narrowest = detachProviderItems(
messages, prunePreservingItems({
reasoning: 'all', messages,
toolCalls: `before-last-${keep}-messages`, reasoning: 'all',
emptyMessages: 'remove', toolCalls: `before-last-${keep}-messages`,
}); emptyMessages: 'remove',
}),
);
if (estimate(narrowest) <= threshold) return narrowest; if (estimate(narrowest) <= threshold) return narrowest;
} }
return narrowest; return narrowest;
+15
View File
@@ -423,6 +423,21 @@ test('reasoning is always dropped, whatever the threshold', () => {
expect(JSON.stringify(fitted)).not.toContain('deciding'); expect(JSON.stringify(fitted)).not.toContain('deciding');
}); });
test('compaction removes a plain assistant item reference without inline reasoning', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: `question ${'x'.repeat(2000)}` },
{
role: 'assistant',
content: [{ type: 'text', text: 'answer', providerOptions: { openai: { itemId: 'msg_plain' } } }],
},
];
const fitted = pruneToFit({ messages, threshold: 1, estimate });
expect(itemIds(fitted)).toEqual([]);
expect(JSON.stringify(fitted)).toContain('answer');
});
test('the user prompt survives even the narrowest rung', () => { test('the user prompt survives even the narrowest rung', () => {
const messages = transcript(200, 4000); const messages = transcript(200, 4000);
const fitted = pruneToFit({ messages, threshold: 100, estimate }); const fitted = pruneToFit({ messages, threshold: 100, estimate });