diff --git a/src/prune.ts b/src/prune.ts index 7f23bef..1331afb 100644 --- a/src/prune.ts +++ b/src/prune.ts @@ -97,6 +97,31 @@ const ANSWER_PARTS = new Set(['tool-result', 'tool-error']); const anyParts = (message: ModelMessage): Part[] => Array.isArray(message.content) ? (message.content as Part[]) : []; +/** + * Every part again with its provider `itemId` gone, so the history goes out inline + * rather than as `item_reference` entries pointing at provider-side storage. + * + * A reference only resolves while the provider still holds that item; once it does + * not, the request is rejected with 404 "Item with id '...' not found" and no retry + * of the same history can succeed. The content is already in the local history, so + * inlining loses nothing. + */ +export function detachProviderItems(messages: ModelMessage[]): ModelMessage[] { + return messages.map((message) => { + const parts = anyParts(message); + if (parts.length === 0) return message; + + let changed = false; + const next = parts.map((part) => { + if (itemId(part) === undefined) return part; + changed = true; + return withoutItemId(part); + }); + + return changed ? ({ ...message, content: next } as ModelMessage) : message; + }); +} + /** * Drops tool results whose tool call is gone. * @@ -178,17 +203,21 @@ export type FitOptions = { * request that will be rejected for size. */ export function pruneToFit({ messages, threshold, estimate }: FitOptions): ModelMessage[] { - const withoutReasoning = prunePreservingItems({ messages, reasoning: 'all', emptyMessages: 'remove' }); + const withoutReasoning = detachProviderItems( + prunePreservingItems({ messages, reasoning: 'all', emptyMessages: 'remove' }), + ); if (estimate(withoutReasoning) <= threshold) return withoutReasoning; let narrowest = withoutReasoning; for (const keep of KEEP_LADDER) { - narrowest = prunePreservingItems({ - messages, - reasoning: 'all', - toolCalls: `before-last-${keep}-messages`, - emptyMessages: 'remove', - }); + narrowest = detachProviderItems( + prunePreservingItems({ + messages, + reasoning: 'all', + toolCalls: `before-last-${keep}-messages`, + emptyMessages: 'remove', + }), + ); if (estimate(narrowest) <= threshold) return narrowest; } return narrowest; diff --git a/test/prune.test.ts b/test/prune.test.ts index fce2ef2..f43e1d4 100644 --- a/test/prune.test.ts +++ b/test/prune.test.ts @@ -423,6 +423,21 @@ test('reasoning is always dropped, whatever the threshold', () => { expect(JSON.stringify(fitted)).not.toContain('deciding'); }); +test('compaction removes a plain assistant item reference without inline reasoning', () => { + const messages: ModelMessage[] = [ + { role: 'user', content: `question ${'x'.repeat(2000)}` }, + { + role: 'assistant', + content: [{ type: 'text', text: 'answer', providerOptions: { openai: { itemId: 'msg_plain' } } }], + }, + ]; + + const fitted = pruneToFit({ messages, threshold: 1, estimate }); + + expect(itemIds(fitted)).toEqual([]); + expect(JSON.stringify(fitted)).toContain('answer'); +}); + test('the user prompt survives even the narrowest rung', () => { const messages = transcript(200, 4000); const fitted = pruneToFit({ messages, threshold: 100, estimate });