fix(compaction): scale summary window floor to the model's context

The absolute 16,384-token floor plus the carried summary and output
reserves exceeds windows below ~58k outright, and overflow recovery then
bailed at the very floor that caused the rejection, leaving compaction
unusable on small-context models. Scale the floor to window/8 (min 1k)
and use the same floor in overflow recovery.
This commit is contained in:
can1357
2026-08-19 01:06:05 +02:00
parent 17e47b3eb7
commit 75179e1dc0
2 changed files with 38 additions and 3 deletions
+13 -3
View File
@@ -895,9 +895,19 @@ function createSnapcompactArchiveMigrationMessage(archiveText: string): Message
*/
const DEFAULT_SUMMARY_INPUT_WINDOW = 200_000;
/** Floor for one summarization window, so a tiny model still makes progress. */
/**
* Floor for one summarization window, so a tiny model still makes progress.
* Scaled down (never below 1k) for models whose window cannot host the full
* floor next to the carried summary and output reserves.
*/
const MIN_SUMMARY_INPUT_TOKENS = 16_384;
/** Smallest window worth planning for `model`; below this, overflow recovery gives up. */
function minSummaryInputTokens(model: Model): number {
const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW;
return Math.min(MIN_SUMMARY_INPUT_TOKENS, Math.max(1_024, Math.floor(window / 8)));
}
/**
* Usable conversation input for ONE summarization call: the summarizer's window
* minus the summary it must emit, the previous summary it carries forward, and
@@ -909,7 +919,7 @@ function summaryInputBudgetTokens(model: Model, maxTokens: number): number {
// 0.8, not "window minus reserves": provider tokenizers disagree with the
// local cl100k estimate by a few percent, and being wrong here is a hard
// 400 on the one call that is supposed to rescue an oversized session.
return Math.max(MIN_SUMMARY_INPUT_TOKENS, Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS);
return Math.max(minSummaryInputTokens(model), Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS);
}
/**
@@ -1012,7 +1022,7 @@ export async function generateSummary(
// the rejection proves the plan was fiction, so converging on the real
// cap must not spend a call per level of an imaginary ladder.
const halved = Math.floor(Math.min(window.budgetTokens, windowTokens) / 2);
if (!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) || halved < MIN_SUMMARY_INPUT_TOKENS) {
if (!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) || halved < minSummaryInputTokens(model)) {
throw error;
}
pending.splice(
@@ -124,6 +124,31 @@ describe("summarization input budget", () => {
}
});
test("keeps the window floor inside a small model's context", async () => {
// The absolute 16,384-token floor plus the carried summary and output
// reserves exceeds a 40k window outright, and overflow recovery would then
// bail at the very floor that caused the rejection. The floor scales with
// the window instead, so a small-context model still folds successfully.
const providerCapChars = 26_000; // what a 40k window can host next to the reserves
const spy = vi.spyOn(ai, "completeSimple").mockImplementation(async (_model, context) => {
const prompt = promptTextOf([_model, context]);
if (prompt.length > providerCapChars) {
throw new Error(`400 prompt is too long: ${prompt.length} tokens > ${providerCapChars} maximum`);
}
return createAssistantMessage("summary");
});
try {
const messages = Array.from({ length: 12 }, (_, i) => turn(i, 4_000)).flat();
const summary = await generateSummary(messages, getModel(40_000), 16_384, "test-key");
expect(summary).toBe("summary");
for (const prompt of spy.mock.calls.map(promptTextOf)) {
expect(prompt.length).toBeLessThanOrEqual(providerCapChars);
}
} finally {
spy.mockRestore();
}
});
test("propagates a non-overflow failure instead of shrinking", async () => {
let calls = 0;
const spy = vi.spyOn(ai, "completeSimple").mockImplementation(async () => {