diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 4710b863d..d39bff11a 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -182,8 +182,11 @@ export function linkOpenAIPromotionTargets(models: ApiModel[]): void { } /** - * Returns supported thinking efforts from canonical model rules constrained by - * explicit model metadata. + * Returns the supported thinking efforts declared on the model metadata. + * + * Catalog enrichment is responsible for normalizing bundled model metadata up front. + * Runtime callers must treat explicit `model.thinking` on custom models as authoritative + * so proxy-specific overrides from `models.yml` survive request construction. * * @throws Error when a reasoning-capable model is missing thinking metadata */ @@ -194,12 +197,7 @@ export function getSupportedEfforts(model: ApiModel): re if (!model.thinking) { throw new Error(`Model ${model.provider}/${model.id} is missing thinking metadata`); } - const configuredEfforts = expandEffortRange(model.thinking); - const parsedModel = parseKnownModel(model.id); - if (parsedModel.family === "unknown") { - return configuredEfforts; - } - return intersectEfforts(configuredEfforts, inferSupportedEfforts(parsedModel, model)); + return expandEffortRange(model.thinking); } /** @@ -421,10 +419,6 @@ function expandEffortRange(thinking: ThinkingConfig): readonly Effort[] { return THINKING_EFFORTS.slice(minIndex, maxIndex + 1); } -function intersectEfforts(left: readonly Effort[], right: readonly Effort[]): readonly Effort[] { - return left.filter(effort => right.includes(effort)); -} - function inferSupportedEfforts(parsedModel: ParsedModel, model: ApiModel): readonly Effort[] { switch (parsedModel.family) { case "openai": diff --git a/packages/ai/test/issue-969-repro.test.ts b/packages/ai/test/issue-969-repro.test.ts new file mode 100644 index 000000000..1b347470e --- /dev/null +++ b/packages/ai/test/issue-969-repro.test.ts @@ -0,0 +1,81 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { Effort, getSupportedEfforts } from "../src/model-thinking"; +import { streamOpenAICompletions } from "../src/providers/openai-completions"; +import type { Context, Model } from "../src/types"; + +const originalFetch = global.fetch; + +afterEach(() => { + global.fetch = originalFetch; +}); + +const testContext: Context = { + messages: [{ role: "user", content: "hello", timestamp: 0 }], +}; + +function createSseResponse(events: unknown[]): Response { + const payload = `${events.map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`).join("\n\n")}\n\n`; + return new Response(payload, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +function customOpenAICompatModel(): Model<"openai-completions"> { + return { + id: "gpt-5.1", + name: "GPT-5.1 proxy", + api: "openai-completions", + provider: "custom", + baseUrl: "https://proxy.example.com/v1", + reasoning: true, + thinking: { + mode: "effort", + minLevel: Effort.Low, + maxLevel: Effort.XHigh, + }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 16_384, + }; +} + +describe("issue #969 — custom thinking metadata must preserve explicit xhigh", () => { + it("uses the configured xhigh effort for custom OpenAI-compatible models", async () => { + const model = customOpenAICompatModel(); + let payload: Record | undefined; + global.fetch = Object.assign( + async (_input: string | URL | Request, init?: RequestInit): Promise => { + payload = JSON.parse(typeof init?.body === "string" ? init.body : "{}") as Record; + return createSseResponse([ + { + id: "chatcmpl-969", + object: "chat.completion.chunk", + created: 0, + model: model.id, + choices: [{ index: 0, delta: { content: "ok" } }], + }, + { + id: "chatcmpl-969", + object: "chat.completion.chunk", + created: 0, + model: model.id, + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + }, + "[DONE]", + ]); + }, + { preconnect: originalFetch.preconnect }, + ); + + expect(getSupportedEfforts(model)).toContain(Effort.XHigh); + const result = await streamOpenAICompletions(model, testContext, { + apiKey: "test-key", + reasoning: "xhigh", + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(payload?.reasoning_effort).toBe("xhigh"); + }); +});