Merge PR #6680: feat(coding-agent): add opt-in max ceiling for auto thinking (@everton-dgn)

This commit is contained in:
can1357
2026-07-30 01:48:51 +02:00
9 changed files with 282 additions and 26 deletions
@@ -78,11 +78,12 @@ describe("auto thinking classifier helpers", () => {
expect(parseCliThinkingLevel("bogus")).toBeUndefined();
});
it("maps online 4-way classifier labels to effort levels", () => {
it("maps online level labels to effort levels", () => {
expect(parseDifficultyLevel("x-high")).toBe(Effort.XHigh);
expect(parseDifficultyLevel("The answer is HIGH.")).toBe(Effort.High);
expect(parseDifficultyLevel("med")).toBe(Effort.Medium);
expect(parseDifficultyLevel("low")).toBe(Effort.Low);
expect(parseDifficultyLevel("max")).toBe(Effort.Max);
expect(parseDifficultyLevel("unknown")).toBeUndefined();
});
@@ -115,6 +116,34 @@ describe("auto thinking classifier helpers", () => {
}
});
it("keeps the local classifier capped at xhigh even when opted in to max", async () => {
// The local backend only ever emits trivial/moderate/hard, so a sparse
// ladder must not let the opt-in ceiling snap `hard` up to a tier the
// on-device model never selected. `max` is the model's only tier at or
// above Low, and the local ceiling hides it, so nothing is eligible —
// falling through to `minimal` would breach the Low floor.
const fixture = await createLocalClassifierFixture("qwen3-1.7b");
const sparse = buildLadderModel("mock-minimal-max", [Effort.Minimal, Effort.Max]);
vi.spyOn(tinyModelClient, "complete").mockResolvedValue("hard");
try {
const settings = Settings.isolated({
"providers.autoThinkingModel": "qwen3-1.7b",
"providers.autoThinkingMaxEffort": "max",
});
expect(
await classifyDifficulty("cut over the storage layer", {
settings,
registry: fixture.registry,
model: sparse,
}),
).toBeUndefined();
} finally {
fixture.cleanup();
}
});
it("uses a larger local non-reasoning classifier floor", async () => {
let maxTokens: number | undefined;
const fixture = await createLocalClassifierFixture("qwen2.5-1.5b");
@@ -202,6 +231,155 @@ describe("auto thinking classifier helpers", () => {
expect(options).toMatchObject({ disableReasoning: true, maxTokens: 1024 });
});
function createOnlineFixture(targetModel: Model, answer: string, maxEffort: "xhigh" | "max" = "xhigh") {
const classifierModel = getBundledModel("anthropic", "claude-sonnet-4-6");
if (!classifierModel) throw new Error("Expected bundled Claude Sonnet 4.6 model");
const settings = {
get(path: string) {
if (path === "providers.autoThinkingModel") return "online";
return path === "providers.autoThinkingMaxEffort" ? maxEffort : undefined;
},
getModelRole(role: string) {
return role === "smol" ? `${classifierModel.provider}/${classifierModel.id}` : undefined;
},
getStorage() {
return undefined;
},
} as never;
const registry = {
getAvailable: () => [classifierModel],
getApiKey: async () => "test-key",
resolver: () => async () => "test-key",
} as never;
const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({
stopReason: "stop",
content: [{ type: "text", text: answer }],
} as never);
return { deps: { settings, registry, model: targetModel }, completeSimpleMock };
}
function buildLadderModel(id: string, efforts: Effort[]): Model {
return buildModel({
id,
name: id,
api: "openai-completions",
provider: "mock",
baseUrl: "https://example.com",
reasoning: true,
thinking: { mode: "effort", efforts },
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 4096,
});
}
const MAX_LADDER = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max];
const XHIGH_LADDER = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
it("offers the max label only when opted in on a model that exposes the tier", async () => {
const optedIn = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "high", "max");
await classifyDifficulty("refactor the scheduler", optedIn.deps);
const optedInRequest = optedIn.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
// The label alone is inert: the criteria and the tie-break exception are
// what make the tier reachable, so all three must ship together.
expect(optedInRequest.systemPrompt[0]).toContain("`max`");
expect(optedInRequest.systemPrompt[0]).toContain("no reproduction to work from");
expect(optedInRequest.systemPrompt[0]).toContain("except between `xhigh` and `max`");
vi.restoreAllMocks();
const defaulted = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "high");
await classifyDifficulty("refactor the scheduler", defaulted.deps);
const defaultedRequest = defaulted.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
expect(defaultedRequest.systemPrompt[0]).not.toMatch(/\bmax\b/);
expect(defaultedRequest.systemPrompt[0]).toContain("`xhigh`");
// The tie-break exception is what makes the top tier reachable, so it must
// not leak into the prompt of a user who did not opt in.
expect(defaultedRequest.systemPrompt[0]).toContain("choose the lower one.");
expect(defaultedRequest.systemPrompt[0]).not.toContain("no reproduction to work from");
vi.restoreAllMocks();
const unsupported = createOnlineFixture(buildLadderModel("mock-xhigh", XHIGH_LADDER), "high", "max");
await classifyDifficulty("refactor the scheduler", unsupported.deps);
const unsupportedRequest = unsupported.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
expect(unsupportedRequest.systemPrompt[0]).not.toMatch(/\bmax\b/);
expect(unsupportedRequest.systemPrompt[0]).toContain("choose the lower one.");
});
it("resolves max only when opted in, and snaps it to the ceiling otherwise", async () => {
const optedIn = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "max", "max");
expect(await classifyDifficulty("untangle this cross-service race", optedIn.deps)).toBe(Effort.Max);
vi.restoreAllMocks();
// Hallucinated `max` on a max-capable model must not cross the default ceiling.
const defaulted = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "max");
expect(await classifyDifficulty("untangle this cross-service race", defaulted.deps)).toBe(Effort.XHigh);
});
it("resolves the sparse ladder's max tier when opted in", async () => {
const fixture = createOnlineFixture(buildLadderModel("mock-sparse", [Effort.High, Effort.Max]), "max", "max");
expect(await classifyDifficulty("cut over the storage layer", fixture.deps)).toBe(Effort.Max);
});
it("takes the first label when the classifier echoes several", async () => {
// `earliest()` is deliberately conservative: an echoed list resolves to the
// lowest-positioned label rather than the model's final word.
const fixture = createOnlineFixture(
buildLadderModel("mock-max", MAX_LADDER),
"low, medium, high, xhigh, max",
"max",
);
expect(await classifyDifficulty("rename a helper", fixture.deps)).toBe(Effort.Low);
});
it("snaps a hallucinated max back to the model's ceiling instead of failing the turn", async () => {
const fixture = createOnlineFixture(buildLadderModel("mock-xhigh", XHIGH_LADDER), "max");
expect(await classifyDifficulty("untangle this cross-service race", fixture.deps)).toBe(Effort.XHigh);
});
it("resolves no level on a max-only ladder without opt-in", async () => {
// `["max"]` has nothing at or below the default ceiling, so the model clamp
// must not snap the request back up — auto yields nothing and the session
// keeps its current level.
const defaulted = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "xhigh");
expect(await classifyDifficulty("cut over the storage layer", defaulted.deps)).toBeUndefined();
vi.restoreAllMocks();
const optedIn = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "max", "max");
expect(await classifyDifficulty("cut over the storage layer", optedIn.deps)).toBe(Effort.Max);
});
it("has no provisional level on a max-only ladder", () => {
expect(resolveProvisionalAutoLevel(buildLadderModel("mock-max-only", [Effort.Max]))).toBeUndefined();
});
it("stops at the highest tier under the ceiling on a sparse ladder", async () => {
const fixture = createOnlineFixture(buildLadderModel("mock-hm", [Effort.High, Effort.Max]), "max");
expect(await classifyDifficulty("cut over the storage layer", fixture.deps)).toBe(Effort.High);
});
it("keeps the provisional auto level below max even when the model defaults to it", () => {
const maxDefaultModel = buildModel({
id: "mock-max-default",
name: "mock-max-default",
api: "openai-completions",
provider: "mock",
baseUrl: "https://example.com",
reasoning: true,
thinking: { mode: "effort", efforts: MAX_LADDER, defaultLevel: Effort.Max },
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 4096,
});
expect(resolveProvisionalAutoLevel(maxDefaultModel)).toBe(Effort.XHigh);
});
it("clamps auto effort to model support while never resolving below low", () => {
const model = getBundledModel("anthropic", "claude-sonnet-4-6");
if (!model) throw new Error("Expected bundled Claude Sonnet 4.6 model");