diff --git a/packages/coding-agent/src/auto-thinking/classifier.ts b/packages/coding-agent/src/auto-thinking/classifier.ts index 4755ad074..62e922e47 100644 --- a/packages/coding-agent/src/auto-thinking/classifier.ts +++ b/packages/coding-agent/src/auto-thinking/classifier.ts @@ -15,7 +15,6 @@ * the caller falls back to a concrete level and continues the turn. */ import { type AssistantMessage, completeSimple, Effort, type Model } from "@oh-my-pi/pi-ai"; -import { THINKING_EFFORTS } from "@oh-my-pi/pi-catalog/effort"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { prompt } from "@oh-my-pi/pi-utils"; @@ -96,10 +95,9 @@ export async function classifyDifficulty( backend === ONLINE_AUTO_THINKING_MODEL_KEY ? await classifyOnline(input, deps, ceiling) : await classifyLocal(input, backend, deps); - // Policy clamp before the model clamp: a hallucinated `max` must not cross a - // ceiling the user did not opt into, even on a model that supports the tier. - const capped = THINKING_EFFORTS.indexOf(effort) > THINKING_EFFORTS.indexOf(ceiling) ? ceiling : effort; - return clampAutoThinkingEffort(deps.model, capped); + // The ceiling goes into the clamp itself: capping the request alone is not + // enough, because a sparse ladder snaps an excluded request back up. + return clampAutoThinkingEffort(deps.model, effort, ceiling); } async function classifyOnline(input: string, deps: ClassifyDifficultyDeps, ceiling: Effort): Promise { diff --git a/packages/coding-agent/src/thinking.ts b/packages/coding-agent/src/thinking.ts index 85739955d..9ef31c328 100644 --- a/packages/coding-agent/src/thinking.ts +++ b/packages/coding-agent/src/thinking.ts @@ -197,6 +197,11 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu * above Low (falling back to the full supported set only when the model maxes * out below Low). Within that pool the request snaps to the highest level not * exceeding it, or the pool minimum when the request is below the pool. + * `ceiling` bounds the pool from above, so a policy ceiling survives the model + * clamp: a sparse ladder such as `["max"]` must not snap an `xhigh` request up + * to `max`. When no supported tier sits at or below the ceiling there is + * nothing legal to pick and the result is `undefined` — auto leaves the current + * level alone rather than billing a tier the caller excluded. * * Returns `undefined` for reasoning-capable models without a controllable * effort surface (`thinking.efforts` empty — e.g. devin-agent models, where @@ -205,12 +210,19 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu * forward a concrete effort that would then trip {@link requireSupportedEffort} * downstream. */ -export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort): Effort | undefined { +export function clampAutoThinkingEffort( + model: Model | undefined, + effort: Effort, + ceiling: Effort = Effort.Max, +): Effort | undefined { const supported = model ? getSupportedEfforts(model) : THINKING_EFFORTS; if (supported.length === 0) return undefined; const lowIndex = THINKING_EFFORTS.indexOf(Effort.Low); - const eligible = supported.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex); - const pool = eligible.length > 0 ? eligible : supported; + const ceilingIndex = THINKING_EFFORTS.indexOf(ceiling); + const withinCeiling = supported.filter(level => THINKING_EFFORTS.indexOf(level) <= ceilingIndex); + if (withinCeiling.length === 0) return undefined; + const eligible = withinCeiling.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex); + const pool = eligible.length > 0 ? eligible : withinCeiling; const requestedIndex = THINKING_EFFORTS.indexOf(effort); let chosen = pool[0]; for (const candidate of pool) { @@ -253,12 +265,16 @@ export function resolveTaskEffortLevel(model: Model | undefined, effort: TaskEff * the model's `defaultLevel`, otherwise High, clamped into the auto range. * * Deliberately stays below {@link Effort.Max}: the placeholder must not bill the - * top tier for a turn nobody classified, so a `defaultLevel` of `max` is capped - * at XHigh before clamping. Classification itself may still resolve Max on - * models that expose the tier. Returns `undefined` for non-reasoning models. + * top tier for a turn nobody classified, so XHigh is passed as a hard ceiling + * rather than only capping the preferred level — otherwise a sparse `["max"]` + * ladder would snap straight back up. A model whose ladder offers nothing at or + * below XHigh therefore has no provisional level, and `auto` leaves the current + * one in place. Classification itself may still resolve Max on models that + * expose the tier when the user opts in. Returns `undefined` for non-reasoning + * models. */ export function resolveProvisionalAutoLevel(model: Model | undefined): Effort | undefined { if (!model?.reasoning) return undefined; const preferred = model.thinking?.defaultLevel ?? Effort.High; - return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred); + return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred, Effort.XHigh); } diff --git a/packages/coding-agent/test/auto-thinking-classifier.test.ts b/packages/coding-agent/test/auto-thinking-classifier.test.ts index 7636bedab..71cae11ec 100644 --- a/packages/coding-agent/test/auto-thinking-classifier.test.ts +++ b/packages/coding-agent/test/auto-thinking-classifier.test.ts @@ -325,6 +325,33 @@ describe("auto thinking classifier helpers", () => { expect(await classifyDifficulty("untangle this cross-service race", fixture.deps)).toBe(Effort.XHigh); }); + it("never bills a max-only ladder without opt-in", async () => { + // `["max"]` has nothing at or below the default ceiling, so the model clamp + // must not snap the request back up — auto yields nothing and the session + // keeps its current level. + const defaulted = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "xhigh"); + expect(await classifyDifficulty("cut over the storage layer", defaulted.deps)).toBeUndefined(); + + vi.restoreAllMocks(); + + const optedIn = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "max", "max"); + expect(await classifyDifficulty("cut over the storage layer", optedIn.deps)).toBe(Effort.Max); + }); + + it("has no provisional level on a max-only ladder", () => { + expect(resolveProvisionalAutoLevel(buildLadderModel("mock-max-only", [Effort.Max]))).toBeUndefined(); + }); + + it("keeps the ceiling out of the pool for sparse ladders", () => { + const maxOnly = buildLadderModel("mock-max-only", [Effort.Max]); + + expect(clampAutoThinkingEffort(maxOnly, Effort.XHigh, Effort.XHigh)).toBeUndefined(); + expect(clampAutoThinkingEffort(maxOnly, Effort.Max)).toBe(Effort.Max); + expect( + clampAutoThinkingEffort(buildLadderModel("mock-hm", [Effort.High, Effort.Max]), Effort.Max, Effort.XHigh), + ).toBe(Effort.High); + }); + it("keeps the provisional auto level below max even when the model defaults to it", () => { const maxDefaultModel = buildModel({ id: "mock-max-default",