fix(coding-agent): keep the auto ceiling out of the model clamp

Capping the classifier result before `clampAutoThinkingEffort` was not
enough. The clamp seeds `chosen` with `pool[0]`, so a sparse ladder whose
tiers all sit above the request snaps upward instead of down: on
`thinking.efforts: ["max"]` an `xhigh` request returned `max`, letting the
default setting bill the top tier with no opt-in. The same upward snap made
`resolveProvisionalAutoLevel` hand back `max`, breaking the invariant its
doc comment had just claimed.

`clampAutoThinkingEffort` now takes the ceiling and intersects it with the
model's supported tiers, returning `undefined` when nothing is eligible so
auto leaves the current level alone instead of billing an excluded tier.
The classifier passes its configured ceiling and the provisional level
passes XHigh.
This commit is contained in:
Éverton Toffanetto
2026-07-26 04:04:07 -03:00
parent e3e4fd3a31
commit 3933bbd0f4
3 changed files with 53 additions and 12 deletions
@@ -15,7 +15,6 @@
* the caller falls back to a concrete level and continues the turn.
*/
import { type AssistantMessage, completeSimple, Effort, type Model } from "@oh-my-pi/pi-ai";
import { THINKING_EFFORTS } from "@oh-my-pi/pi-catalog/effort";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { prompt } from "@oh-my-pi/pi-utils";
@@ -96,10 +95,9 @@ export async function classifyDifficulty(
backend === ONLINE_AUTO_THINKING_MODEL_KEY
? await classifyOnline(input, deps, ceiling)
: await classifyLocal(input, backend, deps);
// Policy clamp before the model clamp: a hallucinated `max` must not cross a
// ceiling the user did not opt into, even on a model that supports the tier.
const capped = THINKING_EFFORTS.indexOf(effort) > THINKING_EFFORTS.indexOf(ceiling) ? ceiling : effort;
return clampAutoThinkingEffort(deps.model, capped);
// The ceiling goes into the clamp itself: capping the request alone is not
// enough, because a sparse ladder snaps an excluded request back up.
return clampAutoThinkingEffort(deps.model, effort, ceiling);
}
async function classifyOnline(input: string, deps: ClassifyDifficultyDeps, ceiling: Effort): Promise<Effort> {
+23 -7
View File
@@ -197,6 +197,11 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu
* above Low (falling back to the full supported set only when the model maxes
* out below Low). Within that pool the request snaps to the highest level not
* exceeding it, or the pool minimum when the request is below the pool.
* `ceiling` bounds the pool from above, so a policy ceiling survives the model
* clamp: a sparse ladder such as `["max"]` must not snap an `xhigh` request up
* to `max`. When no supported tier sits at or below the ceiling there is
* nothing legal to pick and the result is `undefined` — auto leaves the current
* level alone rather than billing a tier the caller excluded.
*
* Returns `undefined` for reasoning-capable models without a controllable
* effort surface (`thinking.efforts` empty — e.g. devin-agent models, where
@@ -205,12 +210,19 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu
* forward a concrete effort that would then trip {@link requireSupportedEffort}
* downstream.
*/
export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort): Effort | undefined {
export function clampAutoThinkingEffort(
model: Model | undefined,
effort: Effort,
ceiling: Effort = Effort.Max,
): Effort | undefined {
const supported = model ? getSupportedEfforts(model) : THINKING_EFFORTS;
if (supported.length === 0) return undefined;
const lowIndex = THINKING_EFFORTS.indexOf(Effort.Low);
const eligible = supported.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex);
const pool = eligible.length > 0 ? eligible : supported;
const ceilingIndex = THINKING_EFFORTS.indexOf(ceiling);
const withinCeiling = supported.filter(level => THINKING_EFFORTS.indexOf(level) <= ceilingIndex);
if (withinCeiling.length === 0) return undefined;
const eligible = withinCeiling.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex);
const pool = eligible.length > 0 ? eligible : withinCeiling;
const requestedIndex = THINKING_EFFORTS.indexOf(effort);
let chosen = pool[0];
for (const candidate of pool) {
@@ -253,12 +265,16 @@ export function resolveTaskEffortLevel(model: Model | undefined, effort: TaskEff
* the model's `defaultLevel`, otherwise High, clamped into the auto range.
*
* Deliberately stays below {@link Effort.Max}: the placeholder must not bill the
* top tier for a turn nobody classified, so a `defaultLevel` of `max` is capped
* at XHigh before clamping. Classification itself may still resolve Max on
* models that expose the tier. Returns `undefined` for non-reasoning models.
* top tier for a turn nobody classified, so XHigh is passed as a hard ceiling
* rather than only capping the preferred level — otherwise a sparse `["max"]`
* ladder would snap straight back up. A model whose ladder offers nothing at or
* below XHigh therefore has no provisional level, and `auto` leaves the current
* one in place. Classification itself may still resolve Max on models that
* expose the tier when the user opts in. Returns `undefined` for non-reasoning
* models.
*/
export function resolveProvisionalAutoLevel(model: Model | undefined): Effort | undefined {
if (!model?.reasoning) return undefined;
const preferred = model.thinking?.defaultLevel ?? Effort.High;
return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred);
return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred, Effort.XHigh);
}
@@ -325,6 +325,33 @@ describe("auto thinking classifier helpers", () => {
expect(await classifyDifficulty("untangle this cross-service race", fixture.deps)).toBe(Effort.XHigh);
});
it("never bills a max-only ladder without opt-in", async () => {
// `["max"]` has nothing at or below the default ceiling, so the model clamp
// must not snap the request back up — auto yields nothing and the session
// keeps its current level.
const defaulted = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "xhigh");
expect(await classifyDifficulty("cut over the storage layer", defaulted.deps)).toBeUndefined();
vi.restoreAllMocks();
const optedIn = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "max", "max");
expect(await classifyDifficulty("cut over the storage layer", optedIn.deps)).toBe(Effort.Max);
});
it("has no provisional level on a max-only ladder", () => {
expect(resolveProvisionalAutoLevel(buildLadderModel("mock-max-only", [Effort.Max]))).toBeUndefined();
});
it("keeps the ceiling out of the pool for sparse ladders", () => {
const maxOnly = buildLadderModel("mock-max-only", [Effort.Max]);
expect(clampAutoThinkingEffort(maxOnly, Effort.XHigh, Effort.XHigh)).toBeUndefined();
expect(clampAutoThinkingEffort(maxOnly, Effort.Max)).toBe(Effort.Max);
expect(
clampAutoThinkingEffort(buildLadderModel("mock-hm", [Effort.High, Effort.Max]), Effort.Max, Effort.XHigh),
).toBe(Effort.High);
});
it("keeps the provisional auto level below max even when the model defaults to it", () => {
const maxDefaultModel = buildModel({
id: "mock-max-default",