Enhance memory management and compaction features
- Implement global memory layer for cross-project patterns. - Improve memory entry scoring with recency and tokenization. - Adjust MAX_TEXT limit from 400 to 800 for better context retention. - Add TTL pruning for stale memory entries. - Capture and summarize dropped content during compaction. - Update tests to reflect changes in memory behavior and compaction logic.
This commit is contained in:
@@ -358,6 +358,66 @@ test('a compacted turn inlines a plain assistant item instead of referencing rem
|
||||
* rejected with 404 "Item with id 'msg_...' not found", and every retry of the same
|
||||
* history is rejected the same way, so resuming a session could never get going.
|
||||
*/
|
||||
|
||||
test('lossless compaction appends a retained note when tool content was dropped', async () => {
|
||||
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3), ...bulkyExchange(4)];
|
||||
let generateCalls = 0;
|
||||
const session = new Session({
|
||||
messages: [...messages],
|
||||
compactThreshold: 1500,
|
||||
model: new MockLanguageModelV4({
|
||||
doStream: async () => stream(text('ok')),
|
||||
doGenerate: async () => {
|
||||
generateCalls++;
|
||||
return { content: [{ type: 'text', text: 'Retained: files f0..f2' }], finishReason: { unified: 'stop', raw: 'stop' }, usage, warnings: [] } as any;
|
||||
},
|
||||
}),
|
||||
askApproval: async () => 'deny',
|
||||
});
|
||||
const events: AgentEvent[] = [];
|
||||
for await (const ev of session.send('next')) events.push(ev);
|
||||
expect(generateCalls).toBe(1);
|
||||
expect(events.some((e) => e.type === 'compacted')).toBe(true);
|
||||
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(true);
|
||||
});
|
||||
|
||||
test('no retained note when history fits', async () => {
|
||||
let generateCalls = 0;
|
||||
const session = new Session({
|
||||
messages: [...bulkyExchange(0)],
|
||||
compactThreshold: 1_000_000,
|
||||
model: new MockLanguageModelV4({
|
||||
doStream: async () => stream(text('ok')),
|
||||
doGenerate: async () => {
|
||||
generateCalls++;
|
||||
return { content: [{ type: 'text', text: 'should not be called' }], finishReason: { unified: 'stop', raw: 'stop' }, usage, warnings: [] } as any;
|
||||
},
|
||||
}),
|
||||
askApproval: async () => 'deny',
|
||||
});
|
||||
for await (const _ of session.send('next')) void _;
|
||||
expect(generateCalls).toBe(0);
|
||||
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(false);
|
||||
});
|
||||
|
||||
test('a failing retained-note model does not break the turn', async () => {
|
||||
const messages = [...bulkyExchange(0), ...bulkyExchange(1), ...bulkyExchange(2), ...bulkyExchange(3)];
|
||||
const session = new Session({
|
||||
messages: [...messages],
|
||||
compactThreshold: 1000,
|
||||
model: new MockLanguageModelV4({
|
||||
doStream: async () => stream(text('ok')),
|
||||
doGenerate: async () => { throw new Error('down'); },
|
||||
}),
|
||||
askApproval: async () => 'deny',
|
||||
});
|
||||
const events: AgentEvent[] = [];
|
||||
for await (const ev of session.send('next')) events.push(ev);
|
||||
expect(events.map((e) => e.type)).toContain('done');
|
||||
expect(events.map((e) => e.type)).not.toContain('error');
|
||||
expect(session.messages.some((m) => String(m.content).includes('retained from compacted'))).toBe(false);
|
||||
});
|
||||
|
||||
const staleItem = (id: string) =>
|
||||
new APICallError({
|
||||
message: `Item with id '${id}' not found.`,
|
||||
|
||||
+1
-1
@@ -82,7 +82,7 @@ test('an empty note is refused', async () => {
|
||||
test('long text is truncated', async () => {
|
||||
const m = new Memory('/repo');
|
||||
const entry = await m.add('fact', 'x'.repeat(2000));
|
||||
expect(entry?.text.length).toBe(400);
|
||||
expect(entry?.text.length).toBe(800);
|
||||
});
|
||||
|
||||
test('search requires every term and records a hit', async () => {
|
||||
|
||||
+29
-1
@@ -1,6 +1,6 @@
|
||||
import { expect, test } from 'bun:test';
|
||||
import type { ModelMessage } from 'ai';
|
||||
import { detachOrphanedItems, dropOrphanedResults, pruneToFit, prunePreservingItems } from '../src/prune';
|
||||
import { detachOrphanedItems, droppedSpan, dropOrphanedResults, estimateTokens, pruneToFit, prunePreservingItems } from '../src/prune';
|
||||
|
||||
const kinds = (messages: ModelMessage[]) =>
|
||||
messages.map((m) => (Array.isArray(m.content) ? `${m.role}:${m.content.map((p) => p.type).join('+')}` : m.role));
|
||||
@@ -438,6 +438,34 @@ test('compaction removes a plain assistant item reference without inline reasoni
|
||||
expect(JSON.stringify(fitted)).toContain('answer');
|
||||
});
|
||||
|
||||
test('droppedSpan is empty when only reasoning was removed', () => {
|
||||
const messages: ModelMessage[] = [
|
||||
{ role: 'user', content: 'q' },
|
||||
{
|
||||
role: 'assistant',
|
||||
content: [
|
||||
{ type: 'reasoning', text: 't', providerOptions: { openai: { itemId: 'rs1' } } },
|
||||
{ type: 'text', text: 'ans', providerOptions: { openai: { itemId: 'msg1' } } },
|
||||
],
|
||||
},
|
||||
];
|
||||
// pruneToFit with high threshold only strips reasoning, keeping ans
|
||||
const fitted = pruneToFit({ messages, threshold: 20000, estimate: estimateTokens });
|
||||
expect(droppedSpan(messages, fitted)).toEqual([]);
|
||||
});
|
||||
|
||||
test('droppedSpan captures pruned tool content', () => {
|
||||
const msgs: ModelMessage[] = [{ role: 'user', content: 'do thing' }];
|
||||
for (let i = 0; i < 20; i++) {
|
||||
msgs.push({ role: 'assistant', content: [{ type: 'tool-call', toolCallId: `t${i}`, toolName: 'grep', input: { pattern: 'x' } }] });
|
||||
msgs.push({ role: 'tool', content: [{ type: 'tool-result', toolCallId: `t${i}`, toolName: 'grep', output: { type: 'text', value: 'x'.repeat(3000) } }] });
|
||||
}
|
||||
const fitted = pruneToFit({ messages: msgs, threshold: 6000, estimate: estimateTokens });
|
||||
const span = droppedSpan(msgs, fitted);
|
||||
expect(span.length).toBeGreaterThan(0);
|
||||
expect(span.length + fitted.length).toBe(msgs.length);
|
||||
});
|
||||
|
||||
test('the user prompt survives even the narrowest rung', () => {
|
||||
const messages = transcript(200, 4000);
|
||||
const fitted = pruneToFit({ messages, threshold: 100, estimate });
|
||||
|
||||
Reference in New Issue
Block a user