fix(catalog): corrected qwen3.8 max discovery metadata
Curated reasoning, multimodal input, context limits, and the provider-specific effort ladder for the discovered Alibaba Token Plan model. Fixes #8019
This commit is contained in:
@@ -9,6 +9,7 @@ import {
|
||||
import type { Model, ModelSpec, OpenAICompat } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
function chatModel(compat: OpenAICompat): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
@@ -160,4 +161,18 @@ describe("OpenAI compat policy", () => {
|
||||
expect(chatPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
|
||||
expect(responsesPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
|
||||
});
|
||||
|
||||
it("routes Token Plan qwen3.8-max effort selections onto the wire", () => {
|
||||
const model = getBundledModel<"openai-completions">("alibaba-token-plan", "qwen3.8-max");
|
||||
for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) {
|
||||
const params = chatParams();
|
||||
applyChatCompletionsCompatPolicy(
|
||||
params,
|
||||
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }),
|
||||
);
|
||||
expect(params.reasoning_effort).toBe(effort);
|
||||
expect(params.enable_thinking).toBeUndefined();
|
||||
expect(params.chat_template_kwargs).toBeUndefined();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed dynamically discovered `alibaba-token-plan/qwen3.8-max` metadata so thinking controls and image input are available ([#8019](https://github.com/can1357/oh-my-pi/issues/8019)).
|
||||
|
||||
## [17.2.11] - 2026-08-07
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -172,7 +172,13 @@ export function applyGeneratedModelPolicies(models: ModelSpec<Api>[]): void {
|
||||
*/
|
||||
export function rebakeModelThinking(model: ModelSpec<Api>): void {
|
||||
if (isVariantCollapsedSpec(model)) return;
|
||||
if (model.provider === "alibaba-token-plan" && model.id === "qwen3.8-max-preview" && model.thinking) return;
|
||||
if (
|
||||
model.provider === "alibaba-token-plan" &&
|
||||
(model.id === "qwen3.8-max-preview" || model.id === "qwen3.8-max") &&
|
||||
model.thinking
|
||||
) {
|
||||
return;
|
||||
}
|
||||
const requiresProviderAuthoredEffort =
|
||||
model.provider === "umans" && (model.thinking?.requiresEffort === true || model.id === "umans-kimi-k2.7");
|
||||
const thinking = resolveModelThinking({ ...model, thinking: undefined }, buildCompat(model));
|
||||
|
||||
+5293
-2080
File diff suppressed because it is too large
Load Diff
@@ -2689,6 +2689,13 @@ const ALIBABA_TOKEN_PLAN_REASONING: ThinkingConfig = {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
|
||||
};
|
||||
// Qwen3.8-Max uses the OpenAI `reasoning_effort` dialect; the generic Qwen
|
||||
// dialect emits only the legacy binary `enable_thinking` toggle.
|
||||
const ALIBABA_TOKEN_PLAN_QWEN_EFFORT_COMPAT: OpenAICompat = {
|
||||
...ALIBABA_TOKEN_PLAN_COMPAT,
|
||||
supportsReasoningEffort: true,
|
||||
thinkingFormat: "openai",
|
||||
};
|
||||
|
||||
export const ALIBABA_TOKEN_PLAN_STATIC_MODELS: readonly ModelSpec<"openai-completions">[] = [
|
||||
{
|
||||
@@ -2707,10 +2714,25 @@ export const ALIBABA_TOKEN_PLAN_STATIC_MODELS: readonly ModelSpec<"openai-comple
|
||||
efforts: [Effort.Low, Effort.High, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
},
|
||||
compat: {
|
||||
...ALIBABA_TOKEN_PLAN_COMPAT,
|
||||
supportsReasoningEffort: true,
|
||||
compat: ALIBABA_TOKEN_PLAN_QWEN_EFFORT_COMPAT,
|
||||
},
|
||||
{
|
||||
id: "qwen3.8-max",
|
||||
name: "Qwen3.8 Max",
|
||||
api: "openai-completions",
|
||||
provider: "alibaba-token-plan",
|
||||
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
cost: ALIBABA_TOKEN_PLAN_COST,
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
defaultLevel: Effort.XHigh,
|
||||
},
|
||||
compat: ALIBABA_TOKEN_PLAN_QWEN_EFFORT_COMPAT,
|
||||
},
|
||||
{
|
||||
id: "qwen3.7-max",
|
||||
|
||||
@@ -13,6 +13,7 @@ describe("QwenCloud Token Plan provider", () => {
|
||||
test("ships the documented Individual text-model allowlist", () => {
|
||||
expect(ALIBABA_TOKEN_PLAN_STATIC_MODELS.map(model => model.id)).toEqual([
|
||||
"qwen3.8-max-preview",
|
||||
"qwen3.8-max",
|
||||
"qwen3.7-max",
|
||||
"qwen3.7-plus",
|
||||
"qwen3.6-flash",
|
||||
@@ -143,6 +144,23 @@ describe("QwenCloud Token Plan provider", () => {
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 64_000,
|
||||
});
|
||||
expect(models?.find(model => model.id === "qwen3.8-max")).toMatchObject({
|
||||
id: "qwen3.8-max",
|
||||
provider: "alibaba-token-plan",
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 131_072,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
defaultLevel: Effort.XHigh,
|
||||
},
|
||||
compat: {
|
||||
supportsReasoningEffort: true,
|
||||
thinkingFormat: "openai",
|
||||
},
|
||||
});
|
||||
expect(options.dynamicModelsAuthoritative).toBe(true);
|
||||
});
|
||||
|
||||
|
||||
@@ -145,7 +145,7 @@ describe("generated model policies", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("preserves QwenCloud's mandatory qwen3.8 effort ladder", () => {
|
||||
it("preserves QwenCloud's provider-authored qwen3.8 effort ladders", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({
|
||||
id: "qwen3.8-max-preview",
|
||||
@@ -157,15 +157,32 @@ describe("generated model policies", () => {
|
||||
requiresEffort: true,
|
||||
},
|
||||
}),
|
||||
createSpec({
|
||||
id: "qwen3.8-max",
|
||||
api: "openai-completions",
|
||||
provider: "alibaba-token-plan",
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
defaultLevel: Effort.XHigh,
|
||||
},
|
||||
}),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.thinking).toEqual({
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.High, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
});
|
||||
expect(models.map(model => model.thinking)).toEqual([
|
||||
{
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.High, Effort.XHigh],
|
||||
requiresEffort: true,
|
||||
},
|
||||
{
|
||||
mode: "effort",
|
||||
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
|
||||
defaultLevel: Effort.XHigh,
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("pins zai glm-5.2 base id to 1M context", () => {
|
||||
|
||||
Reference in New Issue
Block a user