Enhance memory management and compaction features

- Implement global memory layer for cross-project patterns.
- Improve memory entry scoring with recency and tokenization.
- Adjust MAX_TEXT limit from 400 to 800 for better context retention.
- Add TTL pruning for stale memory entries.
- Capture and summarize dropped content during compaction.
- Update tests to reflect changes in memory behavior and compaction logic.
This commit is contained in:
asepharyana
2026-09-08 21:27:47 +07:00
parent b1ff00d840
commit 7db69f19da
8 changed files with 522 additions and 36 deletions
+60
View File
@@ -358,6 +358,66 @@ test('a compacted turn inlines a plain assistant item instead of referencing rem
* rejected with 404 "Item with id 'msg_...' not found", and every retry of the same
* history is rejected the same way, so resuming a session could never get going.
*/
test('lossless compaction appends a retained note when tool content was dropped', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3), ...bulkyExchange(4)];
let generateCalls = 0;
const session = new Session({
messages: [...messages],
compactThreshold: 1500,
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => {
generateCalls++;
return { content: [{ type: 'text', text: 'Retained: files f0..f2' }], finishReason: { unified: 'stop', raw: 'stop' }, usage, warnings: [] } as any;
},
}),
askApproval: async () => 'deny',
});
const events: AgentEvent[] = [];
for await (const ev of session.send('next')) events.push(ev);
expect(generateCalls).toBe(1);
expect(events.some((e) => e.type === 'compacted')).toBe(true);
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(true);
});
test('no retained note when history fits', async () => {
let generateCalls = 0;
const session = new Session({
messages: [...bulkyExchange(0)],
compactThreshold: 1_000_000,
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => {
generateCalls++;
return { content: [{ type: 'text', text: 'should not be called' }], finishReason: { unified: 'stop', raw: 'stop' }, usage, warnings: [] } as any;
},
}),
askApproval: async () => 'deny',
});
for await (const _ of session.send('next')) void _;
expect(generateCalls).toBe(0);
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(false);
});
test('a failing retained-note model does not break the turn', async () => {
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
const session = new Session({
messages: [...messages],
compactThreshold: 1000,
model: new MockLanguageModelV4({
doStream: async () => stream(text('ok')),
doGenerate: async () => { throw new Error('down'); },
}),
askApproval: async () => 'deny',
});
const events: AgentEvent[] = [];
for await (const ev of session.send('next')) events.push(ev);
expect(events.map((e) => e.type)).toContain('done');
expect(events.map((e) => e.type)).not.toContain('error');
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(false);
});
const staleItem = (id: string) =>
new APICallError({
message: `Item with id '${id}' not found.`,
+1 -1
View File
@@ -82,7 +82,7 @@ test('an empty note is refused', async () => {
test('long text is truncated', async () => {
const m = new Memory('/repo');
const entry = await m.add('fact', 'x'.repeat(2000));
expect(entry?.text.length).toBe(400);
expect(entry?.text.length).toBe(800);
});
test('search requires every term and records a hit', async () => {
+29 -1
View File
@@ -1,6 +1,6 @@
import { expect, test } from 'bun:test';
import type { ModelMessage } from 'ai';
import { detachOrphanedItems, dropOrphanedResults, pruneToFit, prunePreservingItems } from '../src/prune';
import { detachOrphanedItems, droppedSpan, dropOrphanedResults, estimateTokens, pruneToFit, prunePreservingItems } from '../src/prune';
const kinds = (messages: ModelMessage[]) =>
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
@@ -438,6 +438,34 @@ test('compaction removes a plain assistant item reference without inline reasoni
expect(JSON.stringify(fitted)).toContain('answer');
});
test('droppedSpan is empty when only reasoning was removed', () => {
const messages: ModelMessage[] = [
{ role: 'user', content: 'q' },
{
role: 'assistant',
content: [
{ type: 'reasoning', text: 't', providerOptions: { openai: { itemId: 'rs1' } } },
{ type: 'text', text: 'ans', providerOptions: { openai: { itemId: 'msg1' } } },
],
},
];
// pruneToFit with high threshold only strips reasoning, keeping ans
const fitted = pruneToFit({ messages, threshold: 20000, estimate: estimateTokens });
expect(droppedSpan(messages, fitted)).toEqual([]);
});
test('droppedSpan captures pruned tool content', () => {
const msgs: ModelMessage[] = [{ role: 'user', content: 'do thing' }];
for (let i = 0; i < 20; i++) {
msgs.push({ role: 'assistant', content: [{ type: 'tool-call', toolCallId: `t${i}`, toolName: 'grep', input: { pattern: 'x' } }] });
msgs.push({ role: 'tool', content: [{ type: 'tool-result', toolCallId: `t${i}`, toolName: 'grep', output: { type: 'text', value: 'x'.repeat(3000) } }] });
}
const fitted = pruneToFit({ messages: msgs, threshold: 6000, estimate: estimateTokens });
const span = droppedSpan(msgs, fitted);
expect(span.length).toBeGreaterThan(0);
expect(span.length + fitted.length).toBe(msgs.length);
});
test('the user prompt survives even the narrowest rung', () => {
const messages = transcript(200, 4000);
const fitted = pruneToFit({ messages, threshold: 100, estimate });