fix(catalog): corrected qwen3.8 max discovery metadata

Curated reasoning, multimodal input, context limits, and the provider-specific effort ladder for the discovered Alibaba Token Plan model.

Fixes #8019
This commit is contained in:
roboomp
2026-08-08 14:37:24 +00:00
parent 08819b279c
commit 155fdaedba
7 changed files with 5385 additions and 2090 deletions
@@ -9,6 +9,7 @@ import {
import type { Model, ModelSpec, OpenAICompat } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
function chatModel(compat: OpenAICompat): Model<"openai-completions"> {
return buildModel({
@@ -160,4 +161,18 @@ describe("OpenAI compat policy", () => {
expect(chatPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
expect(responsesPolicy.stream.reasoningDeltasMayBeCumulative).toBe(true);
});
it("routes Token Plan qwen3.8-max effort selections onto the wire", () => {
const model = getBundledModel<"openai-completions">("alibaba-token-plan", "qwen3.8-max");
for (const effort of [Effort.Low, Effort.Medium, Effort.XHigh]) {
const params = chatParams();
applyChatCompletionsCompatPolicy(
params,
resolveOpenAICompatPolicy(model, { endpoint: "chat-completions", reasoning: effort }),
);
expect(params.reasoning_effort).toBe(effort);
expect(params.enable_thinking).toBeUndefined();
expect(params.chat_template_kwargs).toBeUndefined();
}
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed dynamically discovered `alibaba-token-plan/qwen3.8-max` metadata so thinking controls and image input are available ([#8019](https://github.com/can1357/oh-my-pi/issues/8019)).
## [17.2.11] - 2026-08-07
### Fixed
@@ -172,7 +172,13 @@ export function applyGeneratedModelPolicies(models: ModelSpec<Api>[]): void {
*/
export function rebakeModelThinking(model: ModelSpec<Api>): void {
if (isVariantCollapsedSpec(model)) return;
if (model.provider === "alibaba-token-plan" && model.id === "qwen3.8-max-preview" && model.thinking) return;
if (
model.provider === "alibaba-token-plan" &&
(model.id === "qwen3.8-max-preview" || model.id === "qwen3.8-max") &&
model.thinking
) {
return;
}
const requiresProviderAuthoredEffort =
model.provider === "umans" && (model.thinking?.requiresEffort === true || model.id === "umans-kimi-k2.7");
const thinking = resolveModelThinking({ ...model, thinking: undefined }, buildCompat(model));
File diff suppressed because it is too large Load Diff
@@ -2689,6 +2689,13 @@ const ALIBABA_TOKEN_PLAN_REASONING: ThinkingConfig = {
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
};
// Qwen3.8-Max uses the OpenAI `reasoning_effort` dialect; the generic Qwen
// dialect emits only the legacy binary `enable_thinking` toggle.
const ALIBABA_TOKEN_PLAN_QWEN_EFFORT_COMPAT: OpenAICompat = {
...ALIBABA_TOKEN_PLAN_COMPAT,
supportsReasoningEffort: true,
thinkingFormat: "openai",
};
export const ALIBABA_TOKEN_PLAN_STATIC_MODELS: readonly ModelSpec<"openai-completions">[] = [
{
@@ -2707,10 +2714,25 @@ export const ALIBABA_TOKEN_PLAN_STATIC_MODELS: readonly ModelSpec<"openai-comple
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
},
compat: {
...ALIBABA_TOKEN_PLAN_COMPAT,
supportsReasoningEffort: true,
compat: ALIBABA_TOKEN_PLAN_QWEN_EFFORT_COMPAT,
},
{
id: "qwen3.8-max",
name: "Qwen3.8 Max",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 1_000_000,
maxTokens: 131_072,
thinking: {
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
defaultLevel: Effort.XHigh,
},
compat: ALIBABA_TOKEN_PLAN_QWEN_EFFORT_COMPAT,
},
{
id: "qwen3.7-max",
@@ -13,6 +13,7 @@ describe("QwenCloud Token Plan provider", () => {
test("ships the documented Individual text-model allowlist", () => {
expect(ALIBABA_TOKEN_PLAN_STATIC_MODELS.map(model => model.id)).toEqual([
"qwen3.8-max-preview",
"qwen3.8-max",
"qwen3.7-max",
"qwen3.7-plus",
"qwen3.6-flash",
@@ -143,6 +144,23 @@ describe("QwenCloud Token Plan provider", () => {
contextWindow: 1_000_000,
maxTokens: 64_000,
});
expect(models?.find(model => model.id === "qwen3.8-max")).toMatchObject({
id: "qwen3.8-max",
provider: "alibaba-token-plan",
reasoning: true,
input: ["text", "image"],
contextWindow: 1_000_000,
maxTokens: 131_072,
thinking: {
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
defaultLevel: Effort.XHigh,
},
compat: {
supportsReasoningEffort: true,
thinkingFormat: "openai",
},
});
expect(options.dynamicModelsAuthoritative).toBe(true);
});
@@ -145,7 +145,7 @@ describe("generated model policies", () => {
});
});
it("preserves QwenCloud's mandatory qwen3.8 effort ladder", () => {
it("preserves QwenCloud's provider-authored qwen3.8 effort ladders", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "qwen3.8-max-preview",
@@ -157,15 +157,32 @@ describe("generated model policies", () => {
requiresEffort: true,
},
}),
createSpec({
id: "qwen3.8-max",
api: "openai-completions",
provider: "alibaba-token-plan",
thinking: {
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
defaultLevel: Effort.XHigh,
},
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
});
expect(models.map(model => model.thinking)).toEqual([
{
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
},
{
mode: "effort",
efforts: [Effort.Low, Effort.Medium, Effort.XHigh],
defaultLevel: Effort.XHigh,
},
]);
});
it("pins zai glm-5.2 base id to 1M context", () => {