feat(coding-agent): rebased pending context snapshot after compaction

- Update in-flight context snapshots after historical messages are modified by compaction or elision.
- Prevent stale run-start token counts from triggering false-positive dead-end "no progress" warnings during auto-continuation.
- Add regression test to verify that prompt headroom measurements correctly account for post-compaction context sizes.
This commit is contained in:
can1357
2026-07-13 01:27:30 +02:00
parent 46ed33f27b
commit bf5eb3769f
3 changed files with 101 additions and 1 deletions
+2
View File
@@ -13,6 +13,8 @@
### Fixed
- Fixed configured `retry.fallbackChains` never engaging on non-retryable provider errors (e.g. "Cloud Code Assist API returned an empty response"): a hard error on a model covered by a fallback chain now switches to the next candidate instead of failing the turn, while still never backoff-retrying the failing model itself
- Fixed the post-compaction transcript rebuild (auto-compaction and `/compact`) repainting the entire collapsed transcript — welcome banner included — below the stale pre-compaction scrollback: the rebuild collapses history behind the summary divider, shrinking the frame far below the committed row count, and the renderer's "duplication, never loss" resync re-showed everything without retracting native scrollback; both rebuild paths now request `clearScrollback` like auto-handoff already did
- Fixed mid-run auto-compaction spuriously warning "Compaction freed too little context to make progress" and pausing maintenance even when compaction genuinely shrank the context (observed: snapcompact took a 312k-token gpt-5.6 session to 86k real tokens and still dead-ended): the in-flight prompt's pending context snapshot — set once at run start and alive for the whole tool loop — was read as live residual context by the post-compaction headroom/retry-fit checks because the fresh compaction entry hides every earlier usage anchor; compaction (auto and `/compact`), the retry drop, and the dead-end shake rescue now rebase the snapshot onto the rewritten message set
- Fixed backgrounded Bash blocks continuing to repaint with live and final job output; they now freeze with a compact job notice while completion is delivered separately
### Removed
@@ -10115,6 +10115,7 @@ export class AgentSession {
const newEntries = this.sessionManager.getEntries();
const sessionContext = this.buildDisplaySessionContext();
this.agent.replaceMessages(sessionContext.messages);
this.#rebasePendingContextSnapshotAfterCompaction();
// Compaction discarded the conversation history that carried the approved
// plan reference. Clear the sent-flag so #buildPlanReferenceMessage re-reads
// the plan from disk and re-injects it on the next turn (issue #1246).
@@ -12534,7 +12535,11 @@ export class AgentSession {
if (signal.aborted) return undefined;
try {
const result = await this.shake("elide", { signal });
return result.toolResultsDropped + result.blocksDropped > 0 ? result : undefined;
if (result.toolResultsDropped + result.blocksDropped === 0) return undefined;
// The elide pass rewrote history; re-anchor the in-flight snapshot so
// the caller's headroom/retry-fit re-test measures the shaken context.
this.#rebasePendingContextSnapshotAfterCompaction();
return result;
} catch (error) {
logger.warn("Dead-end shake rescue failed", {
error: error instanceof Error ? error.message : String(error),
@@ -13059,6 +13064,7 @@ export class AgentSession {
const newEntries = this.sessionManager.getEntries();
const sessionContext = this.buildDisplaySessionContext();
this.agent.replaceMessages(sessionContext.messages);
this.#rebasePendingContextSnapshotAfterCompaction();
// Compaction discarded the conversation history that carried the approved
// plan reference. Clear the sent-flag so #buildPlanReferenceMessage re-reads
// the plan from disk and re-injects it on the next turn (issue #1246).
@@ -13124,6 +13130,7 @@ export class AgentSession {
(reason === "incomplete" && lastAssistant.stopReason === "length");
if (shouldDrop) {
this.agent.replaceMessages(messages.slice(0, -1));
this.#rebasePendingContextSnapshotAfterCompaction();
}
}
@@ -15941,6 +15948,28 @@ export class AgentSession {
this.#contextUsageRevision++;
}
/**
* Rebase the in-flight pending context snapshot onto the current message
* set after a compaction (or its dead-end rescue) rewrote history mid-run.
* The snapshot captures the prompt as submitted at run start and lives for
* the whole run; once a compaction entry lands, every earlier usage anchor
* is hidden from {@link getContextBreakdown}, so the stale run-start figure
* would be reported as live context until the next provider response. That
* inflated residual is what the post-compaction headroom/retry-fit checks
* measure — a run that started above the recovery band then trips the
* "freed too little context" dead-end even when compaction genuinely
* shrank the context. No-op while no prompt is in flight.
*/
#rebasePendingContextSnapshotAfterCompaction(): void {
if (!this.#pendingContextSnapshot) return;
const nonMessageTokens = computeNonMessageTokens(this);
this.#setPendingContextSnapshot({
promptTokens: nonMessageTokens + this.messages.reduce((sum, msg) => sum + estimateTokens(msg), 0),
nonMessageTokens,
cutoffCount: this.messages.length,
});
}
#ingestProviderUsageHeaders(response: ProviderResponseMetadata, model?: Model): void {
if (model?.provider !== "anthropic") return;
this.#modelRegistry.authStorage.ingestUsageHeaders("anthropic", response.headers, {
@@ -416,6 +416,75 @@ describe("AgentSession auto-compaction progress guard", () => {
const noProgress = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes(NO_PROGRESS_FRAGMENT));
expect(noProgress.length).toBe(0);
});
it("rebases the in-flight prompt snapshot so mid-run compaction is not misread as a dead-end", async () => {
// Regression: the pending context snapshot is set once per prompt and
// lives for the whole run. A fresh compaction entry hides every earlier
// usage anchor from getContextBreakdown, which then fell back to the
// stale run-start figure until the next provider response — a run
// submitted above the recovery band (0.8 × 170k = 136k here) tripped the
// "freed too little context" warning even though compaction had
// genuinely shrunk the context (observed live: 312k → 86k real tokens,
// warning still emitted).
const continueSpy = vi.spyOn(session.agent, "continue").mockResolvedValue();
// Hold the initial prompt in flight so the pending snapshot stays alive
// through the compaction, exactly like a live tool-loop run. The second
// agent.prompt call is the scheduled auto-continue — the "headroom was
// seen" signal the test awaits.
const gate = Promise.withResolvers<void>();
const firstPromptCall = Promise.withResolvers<void>();
const secondPromptCall = Promise.withResolvers<void>();
let promptCalls = 0;
const promptSpy = vi.spyOn(session.agent, "prompt").mockImplementation(() => {
promptCalls++;
if (promptCalls === 1) firstPromptCall.resolve();
if (promptCalls === 2) secondPromptCall.resolve();
return gate.promise as never;
});
const notices = collectNotices();
// The dead-end warning is the "no headroom was seen" signal: the headroom
// tail runs AFTER auto_compaction_end is emitted, so the test awaits one
// of the tail's two observable outcomes instead of the end event.
const noProgressSeen = Promise.withResolvers<void>();
session.subscribe(event => {
if (event.type === "notice" && event.message.includes(NO_PROGRESS_FRAGMENT)) noProgressSeen.resolve();
});
// ~150k-token prompt: above the recovery band, below the 170k threshold,
// so the pre-prompt maintenance pass stays quiet and the snapshot records
// the run-start size. agent.prompt is mocked, so the text never reaches
// the branch — it exists only in the in-flight snapshot.
const inFlightPrompt = session.prompt("x".repeat(600_000));
// The snapshot is written immediately before agent.prompt; awaiting the
// first (gated) call guarantees it is in place before the threshold turn
// lands — emitting earlier would race the submission pipeline and let
// compaction run against an unset snapshot.
await firstPromptCall.promise;
// Mid-run, the billed context crosses the threshold and compaction fires;
// the rewritten context (summary only) is tiny.
const assistantMsg = highUsageAssistant();
session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] });
// Wait for the headroom verdict while the prompt is still gated —
// releasing the gate earlier would clear the snapshot and mask the
// regression. Fixed behavior schedules the auto-continue (second prompt
// call); the regression emits the dead-end warning instead.
await Promise.race([secondPromptCall.promise, noProgressSeen.promise]);
gate.resolve();
await inFlightPrompt;
await session.waitForIdle();
// The stale 150k run-start snapshot must not be measured as residual
// context: no dead-end warning, and the auto-continue prompt ran
// (initial call + continuation).
expect(promptSpy).toHaveBeenCalledTimes(2);
expect(continueSpy).not.toHaveBeenCalled();
const noProgress = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes(NO_PROGRESS_FRAGMENT));
expect(noProgress.length).toBe(0);
});
/**
* Seed several large prior turns into the session branch so `prepareCompaction`
* returns a real preparation after the overflow recovery drops the failed