fix(ai): preserve xhigh thinking effort

getSupportedEfforts() intersected explicit model.thinking metadata
with heuristic capability inference from known model IDs. For custom
OpenAI-compatible proxy models that intentionally override thinking
levels in models.yml, that intersection dropped xhigh, so request
construction stopped treating xhigh as supported and the outgoing
payload lost the high-effort marker. Trust the explicit metadata when
it is present and fall back to inference only otherwise.

Fixes #969
This commit is contained in:
can1357
2026-05-09 03:11:18 +02:00
parent 0319aae382
commit ef410f0f7c
2 changed files with 87 additions and 12 deletions
+6 -12
View File
@@ -182,8 +182,11 @@ export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
}
/**
* Returns supported thinking efforts from canonical model rules constrained by
* explicit model metadata.
* Returns the supported thinking efforts declared on the model metadata.
*
* Catalog enrichment is responsible for normalizing bundled model metadata up front.
* Runtime callers must treat explicit `model.thinking` on custom models as authoritative
* so proxy-specific overrides from `models.yml` survive request construction.
*
* @throws Error when a reasoning-capable model is missing thinking metadata
*/
@@ -194,12 +197,7 @@ export function getSupportedEfforts<TApi extends Api>(model: ApiModel<TApi>): re
if (!model.thinking) {
throw new Error(`Model ${model.provider}/${model.id} is missing thinking metadata`);
}
const configuredEfforts = expandEffortRange(model.thinking);
const parsedModel = parseKnownModel(model.id);
if (parsedModel.family === "unknown") {
return configuredEfforts;
}
return intersectEfforts(configuredEfforts, inferSupportedEfforts(parsedModel, model));
return expandEffortRange(model.thinking);
}
/**
@@ -421,10 +419,6 @@ function expandEffortRange(thinking: ThinkingConfig): readonly Effort[] {
return THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
}
function intersectEfforts(left: readonly Effort[], right: readonly Effort[]): readonly Effort[] {
return left.filter(effort => right.includes(effort));
}
function inferSupportedEfforts<TApi extends Api>(parsedModel: ParsedModel, model: ApiModel<TApi>): readonly Effort[] {
switch (parsedModel.family) {
case "openai":
+81
View File
@@ -0,0 +1,81 @@
import { afterEach, describe, expect, it } from "bun:test";
import { Effort, getSupportedEfforts } from "../src/model-thinking";
import { streamOpenAICompletions } from "../src/providers/openai-completions";
import type { Context, Model } from "../src/types";
const originalFetch = global.fetch;
afterEach(() => {
global.fetch = originalFetch;
});
const testContext: Context = {
messages: [{ role: "user", content: "hello", timestamp: 0 }],
};
function createSseResponse(events: unknown[]): Response {
const payload = `${events.map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`).join("\n\n")}\n\n`;
return new Response(payload, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
function customOpenAICompatModel(): Model<"openai-completions"> {
return {
id: "gpt-5.1",
name: "GPT-5.1 proxy",
api: "openai-completions",
provider: "custom",
baseUrl: "https://proxy.example.com/v1",
reasoning: true,
thinking: {
mode: "effort",
minLevel: Effort.Low,
maxLevel: Effort.XHigh,
},
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128_000,
maxTokens: 16_384,
};
}
describe("issue #969 — custom thinking metadata must preserve explicit xhigh", () => {
it("uses the configured xhigh effort for custom OpenAI-compatible models", async () => {
const model = customOpenAICompatModel();
let payload: Record<string, unknown> | undefined;
global.fetch = Object.assign(
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
return createSseResponse([
{
id: "chatcmpl-969",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: { content: "ok" } }],
},
{
id: "chatcmpl-969",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
},
"[DONE]",
]);
},
{ preconnect: originalFetch.preconnect },
);
expect(getSupportedEfforts(model)).toContain(Effort.XHigh);
const result = await streamOpenAICompletions(model, testContext, {
apiKey: "test-key",
reasoning: "xhigh",
}).result();
expect(result.stopReason).toBe("stop");
expect(payload?.reasoning_effort).toBe("xhigh");
});
});