diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index a2b7d7401..2141d3ff3 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -895,9 +895,19 @@ function createSnapcompactArchiveMigrationMessage(archiveText: string): Message */ const DEFAULT_SUMMARY_INPUT_WINDOW = 200_000; -/** Floor for one summarization window, so a tiny model still makes progress. */ +/** + * Floor for one summarization window, so a tiny model still makes progress. + * Scaled down (never below 1k) for models whose window cannot host the full + * floor next to the carried summary and output reserves. + */ const MIN_SUMMARY_INPUT_TOKENS = 16_384; +/** Smallest window worth planning for `model`; below this, overflow recovery gives up. */ +function minSummaryInputTokens(model: Model): number { + const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW; + return Math.min(MIN_SUMMARY_INPUT_TOKENS, Math.max(1_024, Math.floor(window / 8))); +} + /** * Usable conversation input for ONE summarization call: the summarizer's window * minus the summary it must emit, the previous summary it carries forward, and @@ -909,7 +919,7 @@ function summaryInputBudgetTokens(model: Model, maxTokens: number): number { // 0.8, not "window minus reserves": provider tokenizers disagree with the // local cl100k estimate by a few percent, and being wrong here is a hard // 400 on the one call that is supposed to rescue an oversized session. - return Math.max(MIN_SUMMARY_INPUT_TOKENS, Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS); + return Math.max(minSummaryInputTokens(model), Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS); } /** @@ -1012,7 +1022,7 @@ export async function generateSummary( // the rejection proves the plan was fiction, so converging on the real // cap must not spend a call per level of an imaginary ladder. const halved = Math.floor(Math.min(window.budgetTokens, windowTokens) / 2); - if (!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) || halved < MIN_SUMMARY_INPUT_TOKENS) { + if (!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) || halved < minSummaryInputTokens(model)) { throw error; } pending.splice( diff --git a/packages/agent/test/compaction-oversized-input.test.ts b/packages/agent/test/compaction-oversized-input.test.ts index d7141c756..d1ceb9afa 100644 --- a/packages/agent/test/compaction-oversized-input.test.ts +++ b/packages/agent/test/compaction-oversized-input.test.ts @@ -124,6 +124,31 @@ describe("summarization input budget", () => { } }); + test("keeps the window floor inside a small model's context", async () => { + // The absolute 16,384-token floor plus the carried summary and output + // reserves exceeds a 40k window outright, and overflow recovery would then + // bail at the very floor that caused the rejection. The floor scales with + // the window instead, so a small-context model still folds successfully. + const providerCapChars = 26_000; // what a 40k window can host next to the reserves + const spy = vi.spyOn(ai, "completeSimple").mockImplementation(async (_model, context) => { + const prompt = promptTextOf([_model, context]); + if (prompt.length > providerCapChars) { + throw new Error(`400 prompt is too long: ${prompt.length} tokens > ${providerCapChars} maximum`); + } + return createAssistantMessage("summary"); + }); + try { + const messages = Array.from({ length: 12 }, (_, i) => turn(i, 4_000)).flat(); + const summary = await generateSummary(messages, getModel(40_000), 16_384, "test-key"); + expect(summary).toBe("summary"); + for (const prompt of spy.mock.calls.map(promptTextOf)) { + expect(prompt.length).toBeLessThanOrEqual(providerCapChars); + } + } finally { + spy.mockRestore(); + } + }); + test("propagates a non-overflow failure instead of shrinking", async () => { let calls = 0; const spy = vi.spyOn(ai, "completeSimple").mockImplementation(async () => {