diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a17865705..24a3165de 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -117,6 +117,7 @@ - Fixed `omp install` of legacy pi extensions failing with `Cannot find module '/$bunfs/root/packages/coding-agent/src/extensibility/typebox.js'` on every released `omp--` binary. Commit `dc5c93462f` removed worker entrypoints from `scripts/ci-release-build-binaries.ts`; the inline comment then claimed the legacy-shim and package-barrel entrypoints (`typebox.ts`, `legacy-pi-{ai,coding-agent}-shim.ts`, `packages/{agent,natives,tui,utils}/...`) were "still" passed to `bun build --compile`, but they had never been re-added. The release binaries shipped without those files in bunfs, so `legacy-pi-compat.ts` redirected `typebox` imports to a bunfs path that didn't exist. `__resolveTypeBoxShimPath` now mirrors `__validateLegacyPiPackageRootOverrides` (#2168) by dropping the override when the shim file is missing, so a missing shim falls through to native `node_modules` resolution instead of emitting a dead bunfs URL ([#3414](https://github.com/can1357/oh-my-pi/issues/3414)). - Fixed every legacy `@(scope)/pi-*` and `@sinclair/typebox` import failing to load on the `omp-darwin-arm64` release binary (and any other `omp` built with Bun 1.3.14). `__validateLegacyPiPackageRootOverrides` and the `rewriteLegacyPiImports` emit path both depended on `--compile` extras being reachable as `/$bunfs/root/...` filesystem entries, but Bun 1.3.14 stopped exposing them through every API (`fs.existsSync`, `Bun.file().exists()`, `Bun.resolveSync`, `await import()` on the bunfs path or its `file://` URL all fail; only `/$bunfs/root/` itself answers). `legacy-pi-compat.ts` now keeps a JS-heap reference to every bundled pi-* surface in a lazy-loaded sibling `legacy-pi-bundled-registry.ts` and serves them through an `omp-legacy-pi-bundled:` virtual namespace whose `Bun.plugin().onLoad` returns synthetic re-exports — no bunfs path ever leaves the module in compiled mode, and dev / source-link / installed-package modes keep the historical `file://` rewrite. The matching `--compile` extras in `scripts/build-binary.ts`, the shared `scripts/binary-entrypoints.ts` list, and the dead `BUNFS_PACKAGE_ROOT` / `bunfsPath` / `__computeBunfsPackageRoot` / `__joinBunfsPath` helpers are gone. ([#3423](https://github.com/can1357/oh-my-pi/issues/3423)) +- Fixed auto-compaction thrashing on a session whose single most-recent kept turn already exceeds the compaction threshold. `prepareCompaction` keeps that turn verbatim (`findCutPoint` never cuts at tool results), so the rewritten context stays above threshold; the context-full / snapcompact success tail scheduled the agent-authored auto-continue (and the overflow/incomplete retry) unconditionally, so the next `agent_end` re-entered `#checkCompaction` over the same oversized tail and re-fired forever. This is the residual loop left after #3247 capped snapcompact's own frame projection — once the frame cap drops below one frame, snapcompact is skipped and the context-full summarizer path still made no headroom. `#runAutoCompaction` now gates the continuation/retry on a post-maintenance headroom check (`#compactionCreatedHeadroom`, sharing shake's `COMPACTION_RECOVERY_BAND` hysteresis from #2275); when a pass frees too little it pauses automatic maintenance and emits a single warning instead of looping. The post-turn threshold check also ignores an assistant's stale pre-compaction `usage` so the scheduled auto-continue cannot re-trip on the kept assistant's old high token count. ## [16.1.17] - 2026-06-24 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 5acab9e0e..58dc69f2c 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -420,14 +420,17 @@ export type AsyncJobSnapshotItem = Pick recoveryBand) return false; + if ( + typeof triggerContextTokens === "number" && + Number.isFinite(triggerContextTokens) && + triggerContextTokens > 0 + ) { + return residualTokens < triggerContextTokens; + } + return true; + } + /** * Internal: Run auto-compaction with events. * @@ -10601,11 +10652,17 @@ export class AgentSession { }; await this.#emitSessionEvent({ type: "auto_compaction_end", action, result, aborted: false, willRetry }); + // Post-maintenance progress guard. Snapcompact can project over budget and + // fall back to a context-full summary; the summarizer keeps `keepRecentTokens` + // of recent history verbatim and findCutPoint can only cut at turn + // boundaries (never tool results), so a single oversized recent turn (e.g. a + // huge tool result) leaves the rewritten context still above threshold. + // Scheduling the auto-continue regardless means the next agent_end re-enters + // #checkCompaction over the same oversized tail and re-fires forever + // (the snapcompact thrash). Mirror the shake recovery-band check: only + // auto-continue when compaction actually created headroom. let continuationScheduled = false; - if (!willRetry && reason !== "idle" && shouldAutoContinue) { - this.#scheduleAutoContinuePrompt(generation); - continuationScheduled = true; - } + const madeProgress = this.#compactionCreatedHeadroom(options.triggerContextTokens); if (willRetry) { const messages = this.agent.state.messages; @@ -10624,7 +10681,27 @@ export class AgentSession { } } - this.#scheduleAgentContinue({ delayMs: 100, generation }); + // Only retry when maintenance actually created headroom. An + // overflow/incomplete recovery whose kept recent turn alone still + // exceeds the window would retry straight back into the same + // overflow/length failure — the recovery-side twin of the threshold + // auto-continue thrash. Pause and surface the dead-end instead. + if (madeProgress) { + this.#scheduleAgentContinue({ delayMs: 100, generation }); + continuationScheduled = true; + } + } else if (reason !== "idle" && shouldAutoContinue && madeProgress) { + // Post-maintenance progress guard. Snapcompact can project over budget + // and fall back to a context-full summary; the summarizer keeps + // `keepRecentTokens` of recent history verbatim and findCutPoint can + // only cut at turn boundaries (never tool results), so a single + // oversized recent turn (e.g. a huge tool result) leaves the rewritten + // context still above threshold. Scheduling the auto-continue regardless + // means the next agent_end re-enters #checkCompaction over the same + // oversized tail and re-fires forever (the snapcompact thrash). Mirror + // the shake recovery-band check: only auto-continue when compaction + // actually created headroom. + this.#scheduleAutoContinuePrompt(generation); continuationScheduled = true; } else if (!suppressContinuation && this.agent.hasQueuedMessages()) { // Auto-compaction can complete while follow-up/steering/custom messages are waiting. @@ -10636,6 +10713,17 @@ export class AgentSession { }); continuationScheduled = true; } + + // A non-idle pass that wanted to continue (auto-continue or retry) but + // made no headroom is a dead-end: warn once so the user understands why + // maintenance paused instead of silently looping. + if (!madeProgress && reason !== "idle" && (willRetry || shouldAutoContinue)) { + this.emitNotice( + "warning", + "Compaction freed too little context to make progress — pausing automatic maintenance to avoid a compaction loop. The most recent turn alone is too large to reduce further; shrink it (e.g. clear large tool output) or switch to a larger-context model.", + "compaction", + ); + } return continuationScheduled ? COMPACTION_CHECK_CONTINUATION : COMPACTION_CHECK_NONE; } catch (error) { if (autoCompactionSignal.aborted) { @@ -10734,7 +10822,7 @@ export class AgentSession { if (typeof triggerContextTokens === "number" && Number.isFinite(triggerContextTokens)) { const correctedTokens = Math.max(0, triggerContextTokens - result.tokensFreed); const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); - const recoveryBand = Math.floor(thresholdTokens * SHAKE_RECOVERY_BAND); + const recoveryBand = Math.floor(thresholdTokens * COMPACTION_RECOVERY_BAND); stillOverThreshold = correctedTokens > recoveryBand; } else { const postShakeTokens = this.getContextUsage({ contextWindow })?.tokens ?? 0; diff --git a/packages/coding-agent/test/agent-session-auto-compaction-progress-guard.test.ts b/packages/coding-agent/test/agent-session-auto-compaction-progress-guard.test.ts new file mode 100644 index 000000000..92911cf17 --- /dev/null +++ b/packages/coding-agent/test/agent-session-auto-compaction-progress-guard.test.ts @@ -0,0 +1,229 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs"; +import * as path from "node:path"; +import { Agent } from "@oh-my-pi/pi-agent-core"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { loadExtensions } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/loader"; +import { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/runner"; +import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils"; + +/** + * Regression test for the auto-compaction thrash loop. + * + * When the most-recent kept turn alone exceeds the compaction threshold, + * `prepareCompaction` keeps it verbatim (findCutPoint never cuts at tool + * results), so a "successful" compaction leaves context still above threshold. + * The snapcompact strategy makes this visible: it projects over budget, falls + * back to a context-full summary ("could not bring the context under the + * limit"), and the success tail used to schedule the auto-continue regardless — + * the next agent_end re-entered #checkCompaction over the same oversized tail and + * re-fired forever. + * + * The fix gates the auto-continue (and the overflow/incomplete retry) on a + * post-maintenance headroom check; with no headroom it pauses and emits a single + * warning notice instead of looping. + */ +describe("AgentSession auto-compaction progress guard", () => { + let tempDir: TempDir; + let session: AgentSession; + let sessionManager: SessionManager; + let authStorage: AuthStorage; + let modelRegistry: ModelRegistry; + + const NOTICE_SOURCE = "compaction"; + const NO_PROGRESS_FRAGMENT = "Compaction freed too little context to make progress"; + + beforeEach(async () => { + tempDir = TempDir.createSync("@pi-auto-compaction-progress-"); + + // Short-circuit the actual summarization so the test makes no LLM call: the + // hook supplies the compaction result, then the production tail (events, + // progress guard, continuation scheduling) runs exactly as in a real pass. + const extensionsDir = path.join(getProjectAgentDir(tempDir.path()), "extensions"); + fs.mkdirSync(extensionsDir, { recursive: true }); + const extensionPath = path.join(extensionsDir, "compaction-short-circuit.ts"); + fs.writeFileSync( + extensionPath, + [ + "export default function(pi) {", + '\tpi.on("session_before_compact", async (event) => {', + "\t\treturn {", + "\t\t\tcompaction: {", + '\t\t\t\tsummary: "compacted",', + "\t\t\t\tshortSummary: undefined,", + "\t\t\t\tfirstKeptEntryId: event.preparation.firstKeptEntryId,", + "\t\t\t\ttokensBefore: event.preparation.tokensBefore,", + "\t\t\t\tdetails: {},", + "\t\t\t},", + "\t\t};", + "\t});", + "}", + ].join("\n"), + ); + + authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + modelRegistry = new ModelRegistry(authStorage); + sessionManager = SessionManager.create(tempDir.path(), tempDir.path()); + + const extensionsResult = await loadExtensions([extensionPath], tempDir.path()); + const extensionRunner = new ExtensionRunner( + extensionsResult.extensions, + extensionsResult.runtime, + tempDir.path(), + sessionManager, + modelRegistry, + ); + + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) { + throw new Error("Expected built-in anthropic model to exist"); + } + + const agent = new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }); + + // Seed a minimal branch so prepareCompaction() returns a preparation. + sessionManager.appendMessage({ + role: "user", + content: "hello", + timestamp: Date.now(), + }); + + session = new AgentSession({ + agent, + sessionManager, + settings: Settings.isolated({ + // Auto-continue ON so the guarded auto-continue path is exercised. + "compaction.autoContinue": true, + }), + modelRegistry, + extensionRunner, + }); + }); + + afterEach(async () => { + try { + await session?.dispose(); + } finally { + authStorage?.close(); + await tempDir?.remove(); + vi.restoreAllMocks(); + } + }); + + /** Build a threshold-tripping assistant turn (contextWindow 200k, ~80% threshold). */ + function highUsageAssistant() { + return { + role: "assistant" as const, + content: [{ type: "text" as const, text: "Done." }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + stopReason: "stop" as const, + usage: { + input: 190000, + output: 1000, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 191000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: Date.now(), + }; + } + + function collectNotices() { + const notices: { level: string; message: string; source?: string }[] = []; + session.subscribe(event => { + if (event.type === "notice") { + notices.push({ level: event.level, message: event.message, source: event.source }); + } + }); + return notices; + } + + function countCompactionStarts() { + let starts = 0; + session.subscribe(event => { + if (event.type === "auto_compaction_start") starts++; + }); + return () => starts; + } + + it("pauses (no continuation, single warning) when compaction creates no headroom", async () => { + const continueSpy = vi.spyOn(session.agent, "continue").mockResolvedValue(); + // Auto-continue runs through agent.prompt (#promptWithMessage), not + // agent.continue — spy both so "no continuation" is actually proven. + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined as never); + // Residual context stays above the recovery band after the rewrite: the most + // recent turn alone is too large to reduce. + vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: 190000, contextWindow: 200000, percent: 95 }); + + const notices = collectNotices(); + const startCount = countCompactionStarts(); + + const { promise: compactionDone, resolve: onCompactionDone } = Promise.withResolvers(); + session.subscribe(event => { + if (event.type === "auto_compaction_end") onCompactionDone(); + }); + + const assistantMsg = highUsageAssistant(); + session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] }); + + await compactionDone; + await session.waitForIdle(); + + // Compaction ran exactly once and did not schedule a continuation turn + // (neither the auto-continue prompt nor a queued-message continue). + expect(startCount()).toBe(1); + expect(promptSpy).not.toHaveBeenCalled(); + expect(continueSpy).not.toHaveBeenCalled(); + expect(session.isStreaming).toBe(false); + + const noProgress = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes(NO_PROGRESS_FRAGMENT)); + expect(noProgress.length).toBe(1); + expect(noProgress[0].level).toBe("warning"); + }); + + it("auto-continues (no warning) when compaction creates headroom", async () => { + // The auto-continue path runs #scheduleAutoContinuePrompt → #promptWithMessage + // → agent.prompt. Stub both prompt and continue so no real agent loop runs. + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined as never); + vi.spyOn(session.agent, "continue").mockResolvedValue(); + // Residual context drops well under the threshold: real reduction happened. + vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: 1000, contextWindow: 200000, percent: 0.5 }); + + const notices = collectNotices(); + + const { promise: compactionDone, resolve: onCompactionDone } = Promise.withResolvers(); + session.subscribe(event => { + if (event.type === "auto_compaction_end") onCompactionDone(); + }); + + const assistantMsg = highUsageAssistant(); + session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] }); + + await compactionDone; + await session.waitForIdle(); + + // Headroom was created, so the guard scheduled the agent-authored + // continuation prompt and stayed silent. + expect(promptSpy).toHaveBeenCalledTimes(1); + const noProgress = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes(NO_PROGRESS_FRAGMENT)); + expect(noProgress.length).toBe(0); + }); +});