feat(coding-agent): changed context promotion to trigger on overflow errors instead of threshold
- Changed context promotion to trigger on context overflow errors instead of a configurable threshold percentage. - Removed the contextPromotion.thresholdPercent configuration setting. - Updated context promotion to retry immediately on the promoted model without requiring compaction. - Refactored context promotion logic to attempt promotion before compaction in the overflow handling flow. - Updated agent session to merge context promotion checks into the compaction method for unified overflow handling. - Updated tests to reflect overflow-based promotion triggering instead of threshold-based promotion.
This commit is contained in:
+5
-9
@@ -245,18 +245,15 @@ Related settings:
|
||||
|
||||
## Context promotion (model-level fallback chains)
|
||||
|
||||
Context promotion is a post-turn safety mechanism for small-context variants (for example `*-spark`) that should automatically promote to a larger-context sibling before hard overflow.
|
||||
Context promotion is an overflow recovery mechanism for small-context variants (for example `*-spark`) that automatically promotes to a larger-context sibling when the API rejects a request with a context length error.
|
||||
|
||||
### Trigger and order
|
||||
|
||||
On `agent_end` after a successful assistant turn (`stopReason` not `error`/`aborted`), `AgentSession` computes:
|
||||
When a turn fails with a context overflow error (e.g. `context_length_exceeded`), `AgentSession` attempts promotion **before** falling back to compaction:
|
||||
|
||||
- `contextTokens = calculateContextTokens(assistantMessage.usage)`
|
||||
- `contextPercent = contextTokens / currentModel.contextWindow * 100`
|
||||
|
||||
If `contextPercent >= contextPromotion.thresholdPercent` (default `90`) and `contextPromotion.enabled` is true, promotion is attempted.
|
||||
|
||||
Promotion check runs **before** threshold compaction check.
|
||||
1. If `contextPromotion.enabled` is true, resolve a promotion target (see below).
|
||||
2. If a target is found, switch to it and retry the request — no compaction needed.
|
||||
3. If no target is available, fall through to auto-compaction on the current model.
|
||||
|
||||
### Target selection
|
||||
|
||||
@@ -264,7 +261,6 @@ Selection is model-driven, not role-driven:
|
||||
|
||||
1. `currentModel.contextPromotionTarget` (if configured)
|
||||
2. smallest larger-context model on the same provider + API
|
||||
3. smallest larger-context model globally
|
||||
|
||||
Candidates are ignored unless credentials resolve (`ModelRegistry.getApiKey(...)`).
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added abort signal support to LSP file operations (`ensureFileOpen`, `refreshFile`) for cancellable file synchronization
|
||||
@@ -17,6 +18,8 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed context promotion to trigger on context overflow instead of a configurable threshold, promoting to a larger model before attempting compaction
|
||||
- Changed context promotion behavior to retry immediately on the promoted model without compacting, providing faster recovery from context limits
|
||||
- Changed default grep context lines from 1 before/3 after to 0 before/0 after for more focused search results
|
||||
- Changed escape key handling in custom editor to allow bypassing autocomplete dismissal when specified by parent controller
|
||||
- Changed workspace diagnostics to support abort signals for cancellable diagnostic runs
|
||||
@@ -30,6 +33,10 @@
|
||||
- Extended recency filter support to Brave provider alongside Perplexity
|
||||
- Changed GitHub issue comment fetching to use paginated API requests with 100 comments per page instead of single request with 50-comment limit
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed `contextPromotion.thresholdPercent` setting as context promotion now triggers only on overflow
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed LSP operations to properly respect abort signals and throw `ToolAbortError` when cancelled
|
||||
|
||||
@@ -820,6 +820,7 @@ export class ModelRegistry {
|
||||
maxTokens: modelDef.maxTokens,
|
||||
headers,
|
||||
compat: modelDef.compat,
|
||||
contextPromotionTarget: modelDef.contextPromotionTarget,
|
||||
} as Model<Api>);
|
||||
}
|
||||
|
||||
|
||||
@@ -305,10 +305,9 @@ export const SETTINGS_SCHEMA = {
|
||||
ui: {
|
||||
tab: "agent",
|
||||
label: "Auto-promote context",
|
||||
description: "Temporarily promote to a larger-context model role before hitting context limits",
|
||||
description: "Promote to a larger-context model on context overflow instead of compacting",
|
||||
},
|
||||
},
|
||||
"contextPromotion.thresholdPercent": { type: "number", default: 90 },
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────
|
||||
// Compaction settings
|
||||
@@ -1039,7 +1038,6 @@ export interface CompactionSettings {
|
||||
|
||||
export interface ContextPromotionSettings {
|
||||
enabled: boolean;
|
||||
thresholdPercent: number;
|
||||
}
|
||||
export interface RetrySettings {
|
||||
enabled: boolean;
|
||||
|
||||
@@ -599,7 +599,6 @@ export class AgentSession {
|
||||
if (didRetry) return; // Retry was initiated, don't proceed to compaction
|
||||
}
|
||||
|
||||
await this.#checkContextPromotion(msg);
|
||||
await this.#checkCompaction(msg);
|
||||
|
||||
// Check for incomplete todos (unless there was an error or abort)
|
||||
@@ -2690,44 +2689,34 @@ Be thorough - include exact file paths, function names, error messages, and tech
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if compaction is needed and run it.
|
||||
* Check if compaction or context promotion is needed and run it.
|
||||
* Called after agent_end and before prompt submission.
|
||||
*
|
||||
* Two cases:
|
||||
* 1. Overflow: LLM returned context overflow error, remove error message from agent state, compact, auto-retry
|
||||
* 2. Threshold: Context over threshold, compact, NO auto-retry (user continues manually)
|
||||
* Three cases (in order):
|
||||
* 1. Overflow + promotion: promote to larger model, retry without compacting
|
||||
* 2. Overflow + no promotion target: compact, auto-retry on same model
|
||||
* 3. Threshold: Context over threshold, compact, NO auto-retry (user continues manually)
|
||||
*
|
||||
* @param assistantMessage The assistant message to check
|
||||
* @param skipAbortedCheck If false, include aborted messages (for pre-prompt check). Default: true
|
||||
*/
|
||||
async #checkCompaction(assistantMessage: AssistantMessage, skipAbortedCheck = true): Promise<void> {
|
||||
const compactionSettings = this.settings.getGroup("compaction");
|
||||
if (!compactionSettings.enabled) return;
|
||||
|
||||
const pruneResult = await this.#pruneToolOutputs();
|
||||
|
||||
// Skip if message was aborted (user cancelled) - unless skipAbortedCheck is false
|
||||
if (skipAbortedCheck && assistantMessage.stopReason === "aborted") return;
|
||||
|
||||
const contextWindow = this.model?.contextWindow ?? 0;
|
||||
|
||||
// Skip overflow check if the message came from a different model.
|
||||
// This handles the case where user switched from a smaller-context model (e.g. opus)
|
||||
// to a larger-context model (e.g. codex) - the overflow error from the old model
|
||||
// shouldn't trigger compaction for the new model.
|
||||
const sameModel =
|
||||
this.model && assistantMessage.provider === this.model.provider && assistantMessage.model === this.model.id;
|
||||
|
||||
// Skip overflow check if the error is from before a compaction in the current path.
|
||||
// This handles the case where an error was kept after compaction (in the "kept" region).
|
||||
// The error shouldn't trigger another compaction since we already compacted.
|
||||
// Example: opus fails → switch to codex → compact → switch back to opus → opus error
|
||||
// Example: opus fails \u2192 switch to codex \u2192 compact \u2192 switch back to opus \u2192 opus error
|
||||
// is still in context but shouldn't trigger compaction again.
|
||||
const compactionEntry = getLatestCompactionEntry(this.sessionManager.getBranch());
|
||||
const errorIsFromBeforeCompaction =
|
||||
compactionEntry !== null && assistantMessage.timestamp < new Date(compactionEntry.timestamp).getTime();
|
||||
|
||||
// Case 1: Overflow - LLM returned context overflow error
|
||||
if (sameModel && !errorIsFromBeforeCompaction && isContextOverflow(assistantMessage, contextWindow)) {
|
||||
// Remove the error message from agent state (it IS saved to session for history,
|
||||
// but we don't want it in context for the retry)
|
||||
@@ -2735,23 +2724,43 @@ Be thorough - include exact file paths, function names, error messages, and tech
|
||||
if (messages.length > 0 && messages[messages.length - 1].role === "assistant") {
|
||||
this.agent.replaceMessages(messages.slice(0, -1));
|
||||
}
|
||||
await this.#runAutoCompaction("overflow", true);
|
||||
|
||||
// Try context promotion first \u2014 switch to a larger model and retry without compacting
|
||||
const promoted = await this.#tryContextPromotion(assistantMessage);
|
||||
if (promoted) {
|
||||
// Retry on the promoted (larger) model without compacting
|
||||
setTimeout(() => {
|
||||
this.agent.continue().catch(() => {});
|
||||
}, 100);
|
||||
return;
|
||||
}
|
||||
|
||||
// No promotion target available \u2014 fall through to compaction
|
||||
const compactionSettings = this.settings.getGroup("compaction");
|
||||
if (compactionSettings.enabled) {
|
||||
await this.#runAutoCompaction("overflow", true);
|
||||
}
|
||||
return;
|
||||
}
|
||||
const compactionSettings = this.settings.getGroup("compaction");
|
||||
if (!compactionSettings.enabled) return;
|
||||
|
||||
// Case 2: Threshold - turn succeeded but context is getting large
|
||||
// Skip if this was an error (non-overflow errors don't have usage data)
|
||||
if (assistantMessage.stopReason === "error") return;
|
||||
|
||||
const pruneResult = await this.#pruneToolOutputs();
|
||||
let contextTokens = calculateContextTokens(assistantMessage.usage);
|
||||
if (pruneResult) {
|
||||
contextTokens = Math.max(0, contextTokens - pruneResult.tokensSaved);
|
||||
}
|
||||
if (shouldCompact(contextTokens, contextWindow, compactionSettings)) {
|
||||
await this.#runAutoCompaction("threshold", false);
|
||||
// Try promotion first — if a larger model is available, switch instead of compacting
|
||||
const promoted = await this.#tryContextPromotion(assistantMessage);
|
||||
if (!promoted) {
|
||||
await this.#runAutoCompaction("threshold", false);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if agent stopped with incomplete todos and prompt to continue.
|
||||
*/
|
||||
@@ -2825,43 +2834,37 @@ Be thorough - include exact file paths, function names, error messages, and tech
|
||||
this.agent.continue().catch(() => {});
|
||||
}
|
||||
|
||||
async #checkContextPromotion(assistantMessage: AssistantMessage): Promise<void> {
|
||||
/**
|
||||
* Attempt context promotion to a larger model.
|
||||
* Returns true if promotion succeeded (caller should retry without compacting).
|
||||
*/
|
||||
async #tryContextPromotion(assistantMessage: AssistantMessage): Promise<boolean> {
|
||||
const promotionSettings = this.settings.getGroup("contextPromotion");
|
||||
if (!promotionSettings.enabled) return;
|
||||
if (assistantMessage.stopReason === "error" || assistantMessage.stopReason === "aborted") return;
|
||||
|
||||
if (!promotionSettings.enabled) return false;
|
||||
const currentModel = this.model;
|
||||
if (!currentModel) return;
|
||||
if (assistantMessage.provider !== currentModel.provider || assistantMessage.model !== currentModel.id) return;
|
||||
|
||||
if (!currentModel) return false;
|
||||
if (assistantMessage.provider !== currentModel.provider || assistantMessage.model !== currentModel.id)
|
||||
return false;
|
||||
const contextWindow = currentModel.contextWindow ?? 0;
|
||||
if (contextWindow <= 0) return;
|
||||
|
||||
const contextTokens = calculateContextTokens(assistantMessage.usage);
|
||||
if (contextTokens <= 0) return;
|
||||
|
||||
const thresholdPercent = Math.max(1, Math.min(100, promotionSettings.thresholdPercent));
|
||||
const contextPercent = (contextTokens / contextWindow) * 100;
|
||||
if (contextPercent < thresholdPercent) return;
|
||||
|
||||
if (contextWindow <= 0) return false;
|
||||
const targetModel = await this.#resolveContextPromotionTarget(currentModel, contextWindow);
|
||||
if (!targetModel) return;
|
||||
if (!targetModel) return false;
|
||||
|
||||
try {
|
||||
this.#closeProviderSessionsForModelSwitch(currentModel, targetModel);
|
||||
await this.setModelTemporary(targetModel);
|
||||
logger.debug("Context promotion switched model", {
|
||||
logger.debug("Context promotion switched model on overflow", {
|
||||
from: `${currentModel.provider}/${currentModel.id}`,
|
||||
to: `${targetModel.provider}/${targetModel.id}`,
|
||||
contextPercent,
|
||||
thresholdPercent,
|
||||
});
|
||||
return true;
|
||||
} catch (error) {
|
||||
logger.warn("Context promotion failed", {
|
||||
from: `${currentModel.provider}/${currentModel.id}`,
|
||||
to: `${targetModel.provider}/${targetModel.id}`,
|
||||
error: String(error),
|
||||
});
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2892,14 +2895,11 @@ Be thorough - include exact file paths, function names, error messages, and tech
|
||||
.filter(m => m.contextWindow > contextWindow)
|
||||
.sort((a, b) => a.contextWindow - b.contextWindow);
|
||||
addCandidate(anyLarger[0]);
|
||||
|
||||
for (const candidate of candidates) {
|
||||
if (modelsAreEqual(candidate, currentModel)) continue;
|
||||
if (candidate.contextWindow <= contextWindow) continue;
|
||||
|
||||
const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId);
|
||||
if (!apiKey) continue;
|
||||
|
||||
return candidate;
|
||||
}
|
||||
|
||||
@@ -2935,7 +2935,8 @@ Be thorough - include exact file paths, function names, error messages, and tech
|
||||
|
||||
const parsed = parseModelString(configuredTarget);
|
||||
if (parsed) {
|
||||
return availableModels.find(m => m.provider === parsed.provider && m.id === parsed.id);
|
||||
const explicitModel = availableModels.find(m => m.provider === parsed.provider && m.id === parsed.id);
|
||||
if (explicitModel) return explicitModel;
|
||||
}
|
||||
|
||||
return availableModels.find(m => m.provider === currentModel.provider && m.id === configuredTarget);
|
||||
|
||||
@@ -29,22 +29,23 @@ describe("AgentSession context promotion", () => {
|
||||
tempDir.removeSync();
|
||||
});
|
||||
|
||||
function createAssistantMessage(model: Model, contextTokens: number): AssistantMessage {
|
||||
function createOverflowMessage(model: Model): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text: "ok" }],
|
||||
content: [{ type: "text", text: "" }],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: contextTokens,
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: contextTokens,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
stopReason: "error",
|
||||
errorMessage: "context_length_exceeded: Your input exceeds the context window of this model.",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
@@ -58,7 +59,7 @@ describe("AgentSession context promotion", () => {
|
||||
throw new Error("Timed out waiting for condition");
|
||||
}
|
||||
|
||||
it("promotes to a larger-context model and clears codex websocket session state", async () => {
|
||||
it("promotes to a larger-context model on overflow and clears codex websocket session state", async () => {
|
||||
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
|
||||
const codexModel = modelRegistry.find("openai-codex", "gpt-5.3-codex");
|
||||
if (!sparkModel || !codexModel) {
|
||||
@@ -68,7 +69,6 @@ describe("AgentSession context promotion", () => {
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"contextPromotion.enabled": true,
|
||||
"contextPromotion.thresholdPercent": 90,
|
||||
});
|
||||
|
||||
const agent = new Agent({
|
||||
@@ -92,9 +92,9 @@ describe("AgentSession context promotion", () => {
|
||||
close: closeSpy,
|
||||
} satisfies ProviderSessionState);
|
||||
|
||||
const assistantMessage = createAssistantMessage(sparkModel, 120_000);
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
|
||||
const overflowMessage = createOverflowMessage(sparkModel);
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] });
|
||||
|
||||
await waitFor(() => session.model?.id === codexModel.id);
|
||||
|
||||
@@ -104,7 +104,7 @@ describe("AgentSession context promotion", () => {
|
||||
expect(session.providerSessionState.size).toBe(0);
|
||||
});
|
||||
|
||||
it("does not promote when context usage is below threshold", async () => {
|
||||
it("does not promote when promotion is disabled", async () => {
|
||||
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
|
||||
if (!sparkModel) {
|
||||
throw new Error("Expected codex spark model to exist");
|
||||
@@ -112,8 +112,7 @@ describe("AgentSession context promotion", () => {
|
||||
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"contextPromotion.enabled": true,
|
||||
"contextPromotion.thresholdPercent": 90,
|
||||
"contextPromotion.enabled": false,
|
||||
});
|
||||
|
||||
const agent = new Agent({
|
||||
@@ -137,9 +136,9 @@ describe("AgentSession context promotion", () => {
|
||||
close: closeSpy,
|
||||
} satisfies ProviderSessionState);
|
||||
|
||||
const assistantMessage = createAssistantMessage(sparkModel, 80_000);
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
|
||||
const overflowMessage = createOverflowMessage(sparkModel);
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] });
|
||||
|
||||
await Bun.sleep(30);
|
||||
|
||||
|
||||
Reference in New Issue
Block a user