fix(ai): preserve xhigh thinking effort
getSupportedEfforts() intersected explicit model.thinking metadata with heuristic capability inference from known model IDs. For custom OpenAI-compatible proxy models that intentionally override thinking levels in models.yml, that intersection dropped xhigh, so request construction stopped treating xhigh as supported and the outgoing payload lost the high-effort marker. Trust the explicit metadata when it is present and fall back to inference only otherwise. Fixes #969
This commit is contained in:
@@ -182,8 +182,11 @@ export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns supported thinking efforts from canonical model rules constrained by
|
||||
* explicit model metadata.
|
||||
* Returns the supported thinking efforts declared on the model metadata.
|
||||
*
|
||||
* Catalog enrichment is responsible for normalizing bundled model metadata up front.
|
||||
* Runtime callers must treat explicit `model.thinking` on custom models as authoritative
|
||||
* so proxy-specific overrides from `models.yml` survive request construction.
|
||||
*
|
||||
* @throws Error when a reasoning-capable model is missing thinking metadata
|
||||
*/
|
||||
@@ -194,12 +197,7 @@ export function getSupportedEfforts<TApi extends Api>(model: ApiModel<TApi>): re
|
||||
if (!model.thinking) {
|
||||
throw new Error(`Model ${model.provider}/${model.id} is missing thinking metadata`);
|
||||
}
|
||||
const configuredEfforts = expandEffortRange(model.thinking);
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
if (parsedModel.family === "unknown") {
|
||||
return configuredEfforts;
|
||||
}
|
||||
return intersectEfforts(configuredEfforts, inferSupportedEfforts(parsedModel, model));
|
||||
return expandEffortRange(model.thinking);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -421,10 +419,6 @@ function expandEffortRange(thinking: ThinkingConfig): readonly Effort[] {
|
||||
return THINKING_EFFORTS.slice(minIndex, maxIndex + 1);
|
||||
}
|
||||
|
||||
function intersectEfforts(left: readonly Effort[], right: readonly Effort[]): readonly Effort[] {
|
||||
return left.filter(effort => right.includes(effort));
|
||||
}
|
||||
|
||||
function inferSupportedEfforts<TApi extends Api>(parsedModel: ParsedModel, model: ApiModel<TApi>): readonly Effort[] {
|
||||
switch (parsedModel.family) {
|
||||
case "openai":
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { Effort, getSupportedEfforts } from "../src/model-thinking";
|
||||
import { streamOpenAICompletions } from "../src/providers/openai-completions";
|
||||
import type { Context, Model } from "../src/types";
|
||||
|
||||
const originalFetch = global.fetch;
|
||||
|
||||
afterEach(() => {
|
||||
global.fetch = originalFetch;
|
||||
});
|
||||
|
||||
const testContext: Context = {
|
||||
messages: [{ role: "user", content: "hello", timestamp: 0 }],
|
||||
};
|
||||
|
||||
function createSseResponse(events: unknown[]): Response {
|
||||
const payload = `${events.map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`).join("\n\n")}\n\n`;
|
||||
return new Response(payload, {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
}
|
||||
|
||||
function customOpenAICompatModel(): Model<"openai-completions"> {
|
||||
return {
|
||||
id: "gpt-5.1",
|
||||
name: "GPT-5.1 proxy",
|
||||
api: "openai-completions",
|
||||
provider: "custom",
|
||||
baseUrl: "https://proxy.example.com/v1",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
minLevel: Effort.Low,
|
||||
maxLevel: Effort.XHigh,
|
||||
},
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 16_384,
|
||||
};
|
||||
}
|
||||
|
||||
describe("issue #969 — custom thinking metadata must preserve explicit xhigh", () => {
|
||||
it("uses the configured xhigh effort for custom OpenAI-compatible models", async () => {
|
||||
const model = customOpenAICompatModel();
|
||||
let payload: Record<string, unknown> | undefined;
|
||||
global.fetch = Object.assign(
|
||||
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record<string, unknown>;
|
||||
return createSseResponse([
|
||||
{
|
||||
id: "chatcmpl-969",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: model.id,
|
||||
choices: [{ index: 0, delta: { content: "ok" } }],
|
||||
},
|
||||
{
|
||||
id: "chatcmpl-969",
|
||||
object: "chat.completion.chunk",
|
||||
created: 0,
|
||||
model: model.id,
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
},
|
||||
"[DONE]",
|
||||
]);
|
||||
},
|
||||
{ preconnect: originalFetch.preconnect },
|
||||
);
|
||||
|
||||
expect(getSupportedEfforts(model)).toContain(Effort.XHigh);
|
||||
const result = await streamOpenAICompletions(model, testContext, {
|
||||
apiKey: "test-key",
|
||||
reasoning: "xhigh",
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(payload?.reasoning_effort).toBe("xhigh");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user