fix(coding-agent): clamped auto thinking to undefined for models without controllable effort
Devin provider models (devin-agent) advertise reasoning: true but no thinking.efforts metadata — Cascade selects effort by routing to sibling model ids, not a wire param. getSupportedEfforts(model) therefore returns []. clampAutoThinkingEffort previously short-circuited that empty supported list by returning the requested effort as-is, so the auto-thinking classifier-resolved level (e.g. low) reached stream.ts:1163 where requireSupportedEffort threw 'Thinking effort low is not supported by devin/<id>. Supported efforts: '. In --print mode the user saw the error text; in the TUI it was silently swallowed, producing the reported 'working then empty response' symptom. Returns undefined when supported is empty so the result mirrors clampThinkingLevelForModel's behavior on the same shape (the explicit --thinking low / high paths already worked because of this). Updates classifyDifficulty's return type to Effort | undefined and threads through to the existing #applyAutoThinkingLevel undefined-effort early-return. #applyAutoThinkingLevel also short-circuits the classifier call up front for these models — there is no effort to pick. Fixes #3356
This commit is contained in:
@@ -8,6 +8,7 @@
|
||||
- Fixed llama.cpp discovery to prefer per-model `/v1/models` `meta.n_ctx`/`meta.n_ctx_train` values, refresh selected models after lazy load, and bypass fresh-cache reuse so server restarts update context windows. ([#3310](https://github.com/can1357/oh-my-pi/issues/3310))
|
||||
- Fixed `task.maxConcurrency: 0` serializing subagent spawns instead of running them unbounded. The settings UI labels `0` as "Unlimited", but the session-scoped spawn `Semaphore` clamped `max` via `Math.max(1, max)`, so the second subagent body in a batch always waited for the first to release the seat. The constructor now treats `max <= 0` (and any non-finite input) as unbounded via `Number.POSITIVE_INFINITY`, matching the eval `parallel()`/`pipeline()` worker-pool semantics ([#3305](https://github.com/can1357/oh-my-pi/issues/3305)).
|
||||
- Fixed MCP tool calls forwarding empty optional placeholder arguments (`""` and `{}`) to `tools/call`; optional placeholders are now omitted while required fields and meaningful falsy values are preserved. ([#3302](https://github.com/can1357/oh-my-pi/issues/3302))
|
||||
- Fixed Devin provider models silently producing empty responses under the default `defaultThinkingLevel: auto`. Devin models advertise `reasoning: true` but no `thinking.efforts` (Cascade selects effort by routing to sibling model ids, not a wire param), so `getSupportedEfforts(model)` was empty; `clampAutoThinkingEffort` returned the classifier-picked effort as-is, which then tripped `requireSupportedEffort` in `pi-ai/stream.ts` with `Thinking effort low is not supported by devin/<id>. Supported efforts: ` (silently swallowed by the TUI). `clampAutoThinkingEffort` now returns `undefined` when the model has no controllable effort surface, matching `clampThinkingLevelForModel`; the auto-thinking turn hook also short-circuits the classifier call for these models. ([#3356](https://github.com/can1357/oh-my-pi/issues/3356))
|
||||
|
||||
## [16.1.16] - 2026-06-23
|
||||
|
||||
|
||||
@@ -55,10 +55,12 @@ export interface ClassifyDifficultyDeps {
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify `promptText` and return a concrete effort clamped to `deps.model`.
|
||||
* Classify `promptText` and return a concrete effort clamped to `deps.model`,
|
||||
* or `undefined` when the model has no controllable effort surface (auto has
|
||||
* nothing to pick — the caller leaves the prior reasoning level in place).
|
||||
* @throws when the backend cannot produce a usable classification.
|
||||
*/
|
||||
export async function classifyDifficulty(promptText: string, deps: ClassifyDifficultyDeps): Promise<Effort> {
|
||||
export async function classifyDifficulty(promptText: string, deps: ClassifyDifficultyDeps): Promise<Effort | undefined> {
|
||||
const backend = deps.settings.get("providers.autoThinkingModel");
|
||||
const input = prepareClassifierInput(promptText);
|
||||
const effort =
|
||||
|
||||
@@ -7367,6 +7367,10 @@ export class AgentSession {
|
||||
async #applyAutoThinkingLevel(promptText: string, generation: number): Promise<void> {
|
||||
const model = this.model;
|
||||
if (!model?.reasoning) return;
|
||||
// Models with reasoning but no controllable effort surface (devin-agent
|
||||
// Cascade routes effort via sibling model ids, not a wire param) have
|
||||
// nothing to pick — skip classification rather than discard its result.
|
||||
if (getSupportedEfforts(model).length === 0) return;
|
||||
|
||||
let resolved: Effort | undefined;
|
||||
if (this.#magicKeywordEnabled("ultrathink") && containsUltrathink(promptText)) {
|
||||
|
||||
@@ -180,10 +180,17 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu
|
||||
* above Low (falling back to the full supported set only when the model maxes
|
||||
* out below Low). Within that pool the request snaps to the highest level not
|
||||
* exceeding it, or the pool minimum when the request is below the pool.
|
||||
*
|
||||
* Returns `undefined` for reasoning-capable models without a controllable
|
||||
* effort surface (`thinking.efforts` empty — e.g. devin-agent models, where
|
||||
* Cascade selects effort by routing to sibling model ids). Matches
|
||||
* {@link clampThinkingLevelForModel}: with no effort to pick, `auto` must not
|
||||
* forward a concrete effort that would then trip {@link requireSupportedEffort}
|
||||
* downstream.
|
||||
*/
|
||||
export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort): Effort {
|
||||
export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort): Effort | undefined {
|
||||
const supported = model ? getSupportedEfforts(model) : THINKING_EFFORTS;
|
||||
if (supported.length === 0) return effort;
|
||||
if (supported.length === 0) return undefined;
|
||||
const lowIndex = THINKING_EFFORTS.indexOf(Effort.Low);
|
||||
const eligible = supported.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex);
|
||||
const pool = eligible.length > 0 ? eligible : supported;
|
||||
|
||||
@@ -18,6 +18,7 @@ import {
|
||||
parseConfiguredThinkingLevel,
|
||||
parseEffort,
|
||||
parseThinkingLevel,
|
||||
resolveProvisionalAutoLevel,
|
||||
} from "@oh-my-pi/pi-coding-agent/thinking";
|
||||
import type { TinyMemoryLocalModelKey } from "@oh-my-pi/pi-coding-agent/tiny/models";
|
||||
import { tinyModelClient } from "@oh-my-pi/pi-coding-agent/tiny/title-client";
|
||||
@@ -141,6 +142,29 @@ describe("auto thinking classifier helpers", () => {
|
||||
expect(clampAutoThinkingEffort(model, Effort.Minimal)).toBe(Effort.Low);
|
||||
});
|
||||
|
||||
it("returns undefined for reasoning models without controllable efforts (devin-agent shape)", () => {
|
||||
// Repro for https://github.com/can1357/oh-my-pi/issues/3356 — Devin
|
||||
// models report `reasoning: true` but expose no `thinking.efforts` (Cascade
|
||||
// selects effort by routing to sibling model ids). `auto` must not invent
|
||||
// a concrete effort here, or `requireSupportedEffort` throws in stream.ts.
|
||||
const devinModel = {
|
||||
id: "glm-5-2",
|
||||
name: "GLM-5.2",
|
||||
api: "devin-agent",
|
||||
provider: "devin",
|
||||
baseUrl: "https://server.codeium.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 4096,
|
||||
} as Model;
|
||||
|
||||
expect(clampAutoThinkingEffort(devinModel, Effort.Low)).toBeUndefined();
|
||||
expect(clampAutoThinkingEffort(devinModel, Effort.XHigh)).toBeUndefined();
|
||||
expect(resolveProvisionalAutoLevel(devinModel)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("accepts max as the top configured thinking alias", () => {
|
||||
expect(parseEffort("max")).toBe(Effort.XHigh);
|
||||
expect(parseThinkingLevel("max")).toBeUndefined();
|
||||
|
||||
Reference in New Issue
Block a user