fix(compaction): scale summary window floor to the model's context
The absolute 16,384-token floor plus the carried summary and output reserves exceeds windows below ~58k outright, and overflow recovery then bailed at the very floor that caused the rejection, leaving compaction unusable on small-context models. Scale the floor to window/8 (min 1k) and use the same floor in overflow recovery.
This commit is contained in:
@@ -895,9 +895,19 @@ function createSnapcompactArchiveMigrationMessage(archiveText: string): Message
|
||||
*/
|
||||
const DEFAULT_SUMMARY_INPUT_WINDOW = 200_000;
|
||||
|
||||
/** Floor for one summarization window, so a tiny model still makes progress. */
|
||||
/**
|
||||
* Floor for one summarization window, so a tiny model still makes progress.
|
||||
* Scaled down (never below 1k) for models whose window cannot host the full
|
||||
* floor next to the carried summary and output reserves.
|
||||
*/
|
||||
const MIN_SUMMARY_INPUT_TOKENS = 16_384;
|
||||
|
||||
/** Smallest window worth planning for `model`; below this, overflow recovery gives up. */
|
||||
function minSummaryInputTokens(model: Model): number {
|
||||
const window = model.contextWindow && model.contextWindow > 0 ? model.contextWindow : DEFAULT_SUMMARY_INPUT_WINDOW;
|
||||
return Math.min(MIN_SUMMARY_INPUT_TOKENS, Math.max(1_024, Math.floor(window / 8)));
|
||||
}
|
||||
|
||||
/**
|
||||
* Usable conversation input for ONE summarization call: the summarizer's window
|
||||
* minus the summary it must emit, the previous summary it carries forward, and
|
||||
@@ -909,7 +919,7 @@ function summaryInputBudgetTokens(model: Model, maxTokens: number): number {
|
||||
// 0.8, not "window minus reserves": provider tokenizers disagree with the
|
||||
// local cl100k estimate by a few percent, and being wrong here is a hard
|
||||
// 400 on the one call that is supposed to rescue an oversized session.
|
||||
return Math.max(MIN_SUMMARY_INPUT_TOKENS, Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS);
|
||||
return Math.max(minSummaryInputTokens(model), Math.floor(window * 0.8) - maxTokens - MAX_SUMMARY_TOKENS);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1012,7 +1022,7 @@ export async function generateSummary(
|
||||
// the rejection proves the plan was fiction, so converging on the real
|
||||
// cap must not spend a call per level of an imaginary ladder.
|
||||
const halved = Math.floor(Math.min(window.budgetTokens, windowTokens) / 2);
|
||||
if (!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) || halved < MIN_SUMMARY_INPUT_TOKENS) {
|
||||
if (!AIError.is(AIError.classify(error), AIError.Flag.ContextOverflow) || halved < minSummaryInputTokens(model)) {
|
||||
throw error;
|
||||
}
|
||||
pending.splice(
|
||||
|
||||
@@ -124,6 +124,31 @@ describe("summarization input budget", () => {
|
||||
}
|
||||
});
|
||||
|
||||
test("keeps the window floor inside a small model's context", async () => {
|
||||
// The absolute 16,384-token floor plus the carried summary and output
|
||||
// reserves exceeds a 40k window outright, and overflow recovery would then
|
||||
// bail at the very floor that caused the rejection. The floor scales with
|
||||
// the window instead, so a small-context model still folds successfully.
|
||||
const providerCapChars = 26_000; // what a 40k window can host next to the reserves
|
||||
const spy = vi.spyOn(ai, "completeSimple").mockImplementation(async (_model, context) => {
|
||||
const prompt = promptTextOf([_model, context]);
|
||||
if (prompt.length > providerCapChars) {
|
||||
throw new Error(`400 prompt is too long: ${prompt.length} tokens > ${providerCapChars} maximum`);
|
||||
}
|
||||
return createAssistantMessage("summary");
|
||||
});
|
||||
try {
|
||||
const messages = Array.from({ length: 12 }, (_, i) => turn(i, 4_000)).flat();
|
||||
const summary = await generateSummary(messages, getModel(40_000), 16_384, "test-key");
|
||||
expect(summary).toBe("summary");
|
||||
for (const prompt of spy.mock.calls.map(promptTextOf)) {
|
||||
expect(prompt.length).toBeLessThanOrEqual(providerCapChars);
|
||||
}
|
||||
} finally {
|
||||
spy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
test("propagates a non-overflow failure instead of shrinking", async () => {
|
||||
let calls = 0;
|
||||
const spy = vi.spyOn(ai, "completeSimple").mockImplementation(async () => {
|
||||
|
||||
Reference in New Issue
Block a user