diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 700ac4f98..368a72c4b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -10,6 +10,7 @@ - Fixed tool-argument repair applying lossy transformations (such as stringifying objects or stripping unrecognized keys) when validating union schemas (`anyOf`/`oneOf`), preventing corrupted tool call and subagent payloads - Fixed 400 errors when communicating with local OpenAI-compatible inference servers that reject `chat_template_kwargs.reasoning_effort` by improving reasoning effort parameter fallback and compatibility handling +- Fixed DeepSeek-family models on hosts like Fireworks losing reasoning whenever tools were offered: a redundant `tool_choice: "auto"` is now omitted so the provider keeps thinking enabled; forced and `"none"` selectors still take priority ([#1207](https://github.com/can1357/oh-my-pi/issues/1207)) ## [17.3.8] - 2026-08-19 diff --git a/packages/coding-agent/src/session/session-maintenance.ts b/packages/coding-agent/src/session/session-maintenance.ts index 57c57d050..1fe21f2a1 100644 --- a/packages/coding-agent/src/session/session-maintenance.ts +++ b/packages/coding-agent/src/session/session-maintenance.ts @@ -1144,6 +1144,11 @@ export class SessionMaintenance { if (!model) return; const method = resolveSpeculationMethod(model, settings); if (!method) return; + this.#startSpeculationRun(contextTokens, method); + } + + /** Install and launch one background speculation run for `method`. */ + #startSpeculationRun(contextTokens: number, method: "remote" | "handoff" | "soft"): void { const controller = new AbortController(); const run: SpeculationRun = { controller, promise: Promise.resolve(), contextTokensAtStart: contextTokens }; this.#speculation = run; @@ -1156,6 +1161,49 @@ export class SessionMaintenance { }); } + /** + * Grace band above the compaction threshold: when a single turn jumps past + * the threshold before the background speculation armed (or even started), + * the threshold pass keeps serving the user instead of blocking on a + * synchronous summarization — the speculation finishes in the background + * and the next maintenance boundary splices it in for free. Returns true + * while deferral is in effect (a run was live, or one was started here); + * the caller MUST skip its blocking compaction then. + * + * Deferral ends — and the blocking pass resumes — once context grows past + * `threshold + lead`, clamped to keep {@link SPECULATION_LEAD_MIN_TOKENS} + * of headroom below the window. A provider overflow inside the band is + * recovered by the existing overflow path (compact + retry). Never defers + * for local-first method orders (shake/snapcompact are instant), when + * async compaction is disabled, or when a `session_before_compact` + * extension must keep exact blocking semantics. + */ + deferThresholdCompactionToSpeculation(contextTokens: number, contextWindow: number): boolean { + if (contextWindow <= 0 || this.#host.isDisposed()) return false; + const settings = this.#host.settings.getGroup("compaction"); + if (!settings.enabled || settings.asyncEnabled === false || !hasConfiguredCompactionMethod(settings)) + return false; + if (this.isCompacting || this.#host.isGeneratingHandoff()) return false; + if (this.#host.extensionRunner?.hasHandlers("session_before_compact")) return false; + const model = this.#model; + if (!model) return false; + const method = resolveSpeculationMethod(model, settings); + if (!method) return false; + const thresholdTokens = resolveThresholdTokens(contextWindow, settings); + const graceCapTokens = Math.min( + thresholdTokens + resolveSpeculationLeadTokens(thresholdTokens), + contextWindow - SPECULATION_LEAD_MIN_TOKENS, + ); + if (contextTokens >= graceCapTokens) return false; + const run = this.#speculation; + if (run) { + if (run.armed) return false; // ready — the real pass splices it in now + return true; // still summarizing in the background + } + this.#startSpeculationRun(contextTokens, method); + return true; + } + /** Produce and arm one speculative compaction result off a branch snapshot. */ async #runSpeculation( run: SpeculationRun, @@ -1425,6 +1473,16 @@ export class SessionMaintenance { // pre-prompt retry useful; the new agent loop may warn for its own turn. return; } + // Grace band: a live (or just-started) background speculation absorbs the + // blocking summarization; the user's prompt goes out immediately and the + // armed result is spliced in at the next boundary. + if (this.deferThresholdCompactionToSpeculation(contextTokens, contextWindow)) { + logger.debug("Pre-prompt threshold deferred to speculative compaction", { + contextTokens, + contextWindow, + }); + return; + } // Auto-promote first: switching to a larger-context model avoids compacting // the history at all. The post-turn threshold path already promotes before @@ -1504,6 +1562,16 @@ export class SessionMaintenance { this.maybeStartSpeculativeCompaction(contextTokens, contextWindow); return; } + // Grace band: keep the tool loop moving while a background speculation + // (live or started here) produces the summary; checked before the + // persistence barrier so deferred boundaries never await the journal. + if (this.deferThresholdCompactionToSpeculation(contextTokens, contextWindow)) { + logger.debug("Mid-run threshold deferred to speculative compaction", { + contextTokens, + contextWindow, + }); + return; + } if (!(await this.#host.persistTurnMessagesForMidRunCompaction(context))) return; if (this.#midTurnCompactionDeadEnds.has(activeMessages)) { @@ -1794,6 +1862,17 @@ export class SessionMaintenance { contextPromotionEnabled: this.#host.settings.get("contextPromotion.enabled") === true, }); if (shouldThresholdCompact) { + // Grace band: a live (or just-started) background speculation absorbs + // the blocking summarization; the session stays responsive and the + // armed result lands at the next boundary. Deferral delays promotion + // by at most the band — the eventual real pass still promotes first. + if (this.deferThresholdCompactionToSpeculation(postMaintenanceContextTokens, contextWindow)) { + logger.debug("Post-turn threshold deferred to speculative compaction", { + postMaintenanceContextTokens, + contextWindow, + }); + return COMPACTION_CHECK_NONE; + } // Try promotion first — if a larger model is available, switch instead of compacting const promoted = await this.#tryContextPromotion(assistantMessage); if (!promoted) { diff --git a/packages/coding-agent/test/compaction-speculation.test.ts b/packages/coding-agent/test/compaction-speculation.test.ts index d90eb25c0..6378e8516 100644 --- a/packages/coding-agent/test/compaction-speculation.test.ts +++ b/packages/coding-agent/test/compaction-speculation.test.ts @@ -287,4 +287,53 @@ describe("async speculative compaction", () => { expect(maintenance.speculationState).toBe("idle"); }); + + it("defers a threshold pass that jumped past the band, then commits the armed result for free", async () => { + const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async preparation => ({ + summary: "grace summary", + firstKeptEntryId: preparation.firstKeptEntryId, + tokensBefore: preparation.tokensBefore, + details: {}, + })); + + // One large turn skipped the pre-threshold band entirely: deferral must + // start the speculation itself and keep the pass non-blocking. + expect(maintenance.deferThresholdCompactionToSpeculation(THRESHOLD + 1_000, CONTEXT_WINDOW)).toBe(true); + expect(maintenance.speculationState).toBe("running"); + // While the run is in flight, later boundaries inside the band keep deferring. + expect(maintenance.deferThresholdCompactionToSpeculation(THRESHOLD + 1_500, CONTEXT_WINDOW)).toBe(true); + await waitForState("armed"); + + // Armed: deferral ends so the real pass splices the result in immediately. + expect(maintenance.deferThresholdCompactionToSpeculation(THRESHOLD + 2_000, CONTEXT_WINDOW)).toBe(false); + await maintenance.runAutoCompaction("threshold", false, false, false, { + triggerContextTokens: THRESHOLD + 2_000, + }); + + const entry = sessionManager.getEntries().findLast(item => item.type === "compaction"); + expect(entry?.type === "compaction" ? entry.summary : undefined).toBe("grace summary"); + expect(compactSpy).toHaveBeenCalledTimes(1); + }); + + it("stops deferring at the grace cap so the blocking pass reclaims context", () => { + const compactSpy = vi.spyOn(compactionModule, "compact"); + + // Lead floor (8192) bounds the band for a 50K threshold: at the cap the + // blocking pass must own the recovery again. + const graceCap = THRESHOLD + 8_192; + expect(maintenance.deferThresholdCompactionToSpeculation(graceCap, CONTEXT_WINDOW)).toBe(false); + expect(maintenance.speculationState).toBe("idle"); + expect(compactSpy).not.toHaveBeenCalled(); + }); + + it("never defers when async compaction is disabled or a local method leads", () => { + maintenance = createMaintenance({ asyncEnabled: false }); + expect(maintenance.deferThresholdCompactionToSpeculation(THRESHOLD + 1, CONTEXT_WINDOW)).toBe(false); + expect(maintenance.speculationState).toBe("idle"); + + // Snapcompact is local and effectively instant — blocking on it is fine. + maintenance = createMaintenance({ methodOrder: ["snapcompact", "soft"] }); + expect(maintenance.deferThresholdCompactionToSpeculation(THRESHOLD + 1, CONTEXT_WINDOW)).toBe(false); + expect(maintenance.speculationState).toBe("idle"); + }); });