Merge PR #6447: feat: add native Alibaba Token Plan provider (@eggpeat)

This commit is contained in:
can1357
2026-07-24 02:25:25 +02:00
24 changed files with 1142 additions and 20 deletions
+2
View File
@@ -80,6 +80,8 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
| `AI_GATEWAY_API_KEY` | Vercel AI Gateway auth | Using `vercel-ai-gateway` provider | |
| `CLOUDFLARE_AI_GATEWAY_API_KEY` | Cloudflare AI Gateway auth | Using `cloudflare-ai-gateway` provider | Base URL must be configured as `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic` |
| `ALIBABA_CODING_PLAN_API_KEY` | Alibaba Coding Plan auth | Using `alibaba-coding-plan` provider | |
| `ALIBABA_TOKEN_PLAN_API_KEY` | QwenCloud Token Plan auth | Using `alibaba-token-plan` provider | Preferred provider-specific name |
| `BAILIAN_TOKEN_PLAN_API_KEY` | QwenCloud Token Plan auth | Using `alibaba-token-plan` provider | Compatible with Qwen Code's Token Plan preset |
| `DEEPSEEK_API_KEY` | DeepSeek auth | Using DeepSeek models | |
| `KILO_API_KEY` | Kilo auth | Using Kilo models | |
| `OLLAMA_CLOUD_API_KEY` | Ollama Cloud auth | Using `ollama-cloud` provider | |
+1
View File
@@ -12,6 +12,7 @@
- Added Vercel AI Gateway Responses cache anchors and cache lifetimes, emitted only with automatic caching.
- Added opt-in OpenAI GPT-5.6 explicit prompt-cache controls for Responses and Chat Completions. Existing requests remain implicit; the policy marks at most one existing stable-history block and is rejected locally on unsupported explicit routes.
- Forwarded `statefulResponses` through `streamSimple`, so diagnostic callers can explicitly disable OpenAI Responses `previous_response_id` chaining.
- Added native QwenCloud Token Plan API-key login, model discovery, and an optional interactive console-Cookie prompt for 5-hour and 7-day quota reporting ([#6151](https://github.com/can1357/oh-my-pi/issues/6151)).
### Fixed
+3
View File
@@ -73,6 +73,8 @@ Unified LLM API with automatic model discovery, provider configuration, token an
- **Xiaomi MiMo** (requires `XIAOMI_API_KEY`)
- **ZenMux** (requires `ZENMUX_API_KEY`)
- **Qwen Portal** (supports `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY`)
- **QwenCloud Token Plan** (supports `/login alibaba-token-plan`, `ALIBABA_TOKEN_PLAN_API_KEY`, or `BAILIAN_TOKEN_PLAN_API_KEY`; interactive login optionally stores a `home.qwencloud.com` Cookie request header for best-effort 5-hour and 7-day quota reporting)
To enable quota reporting, sign in to the Token Plan dashboard, copy the `Cookie` request-header value from a `home.qwencloud.com` request in browser developer tools, and paste it at the second login prompt. Press Enter to skip; the Cookie is sensitive and session-lived, so rerun login when it expires.
- **Cloudflare AI Gateway** (requires `CLOUDFLARE_AI_GATEWAY_API_KEY` and provider-specific gateway base URL)
- **Ollama** (local OpenAI-compatible runtime; optional `OLLAMA_API_KEY`)
- **Ollama Cloud** (hosted native Ollama API; requires `OLLAMA_CLOUD_API_KEY`)
@@ -953,6 +955,7 @@ In Node.js environments, you can set environment variables to avoid passing API
| Ollama | `OLLAMA_API_KEY` (optional for local deployments) |
| Ollama Cloud | `OLLAMA_CLOUD_API_KEY` |
| Qwen Portal | `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY` |
| QwenCloud Token Plan | `ALIBABA_TOKEN_PLAN_API_KEY` or `BAILIAN_TOKEN_PLAN_API_KEY` |
| zAI | `ZAI_API_KEY` |
| Umans AI Coding Plan | `UMANS_AI_CODING_PLAN_API_KEY` |
| MiniMax Code | `MINIMAX_CODE_API_KEY` (international) or `MINIMAX_CODE_CN_API_KEY` (China) |
+17 -6
View File
@@ -11,6 +11,7 @@ import { Database, type Statement } from "bun:sqlite";
import { createHash } from "node:crypto";
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import { $env, getAgentDbPath, logger } from "@oh-my-pi/pi-utils";
import type { ApiKeyResolver } from "./auth-retry";
import * as AIError from "./error";
@@ -42,6 +43,7 @@ import type {
UsageReport,
} from "./usage";
import { resolveUsedFraction } from "./usage";
import { alibabaTokenPlanRankingStrategy, alibabaTokenPlanUsageProvider } from "./usage/alibaba-token-plan";
import { claudeRankingStrategy, claudeUsageProvider } from "./usage/claude";
import { cursorUsageProvider } from "./usage/cursor";
import { googleGeminiCliUsageProvider } from "./usage/gemini";
@@ -587,6 +589,7 @@ async function defaultConfigValueResolver(config: string): Promise<string | unde
// ─────────────────────────────────────────────────────────────────────────────
const DEFAULT_USAGE_PROVIDERS: UsageProvider[] = [
alibabaTokenPlanUsageProvider,
openaiCodexUsageProvider,
kimiUsageProvider,
antigravityUsageProvider,
@@ -974,6 +977,7 @@ function resolveDefaultUsageProvider(provider: Provider): UsageProvider | undefi
}
const DEFAULT_RANKING_STRATEGIES = new Map<Provider, CredentialRankingStrategy>([
["alibaba-token-plan", alibabaTokenPlanRankingStrategy],
["openai-codex", codexRankingStrategy],
["anthropic", claudeRankingStrategy],
["google-antigravity", antigravityRankingStrategy],
@@ -3012,11 +3016,13 @@ export class AuthStorage {
return report;
}
// Failure: apply a short jittered cool-down so the credential doesn't
// re-hit the endpoint on every poll. Serve the last good value when we
// have one (keeps the credential in the report); otherwise cache null
// so a cold or throttled credential stops re-bursting until the window
// expires and the next poll retries.
const lastGood = this.#usageCache.getStale<UsageReport | null>(cacheKey)?.value ?? null;
// re-hit the endpoint on every poll. Most providers serve the last good
// value through transient failures. Session-cookie providers can opt out
// so an expired login does not display stale quota indefinitely.
const retainLastGood = this.#usageProviderResolver?.(request.provider)?.retainLastGoodOnFailure !== false;
const lastGood = retainLastGood
? (this.#usageCache.getStale<UsageReport | null>(cacheKey)?.value ?? null)
: null;
const backoffJitter = USAGE_FAILURE_BACKOFF_MS * (Math.random() * 0.5 - 0.25);
const coolDown = Date.now() + USAGE_FAILURE_BACKOFF_MS + backoffJitter;
this.#usageCache.set(cacheKey, { value: lastGood, expiresAt: coolDown });
@@ -5992,7 +5998,12 @@ function matchesReplacementCredential(
): boolean {
if (!existing || existing.type !== incoming.type) return false;
if (incoming.type === "api_key") {
return existing.type === "api_key" && existing.key === incoming.key;
if (existing.type !== "api_key") return false;
if (existing.key === incoming.key) return true;
if (provider !== "alibaba-token-plan") return false;
const existingToken = parseAlibabaTokenPlanCredential(existing.key)?.token;
const incomingToken = parseAlibabaTokenPlanCredential(incoming.key)?.token;
return existingToken !== undefined && existingToken === incomingToken;
}
const incomingIdentifiers = extractOAuthCredentialIdentifiers(incoming);
const incomingIdentityKey = resolveProviderCredentialIdentityKey(provider, incomingIdentifiers);
@@ -13,6 +13,7 @@ import type {
ResolvedOpenAISharedCompat,
VercelGatewayRouting,
} from "@oh-my-pi/pi-catalog/types";
import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import {
COREWEAVE_PROJECT_HEADER,
coreWeaveProjectHeaders,
@@ -258,6 +259,12 @@ export function resolveOpenAIRequestSetup(
baseUrl = resolveGitHubCopilotBaseUrl(model.baseUrl, rawApiKey) ?? model.baseUrl;
}
if (model.provider === "alibaba-token-plan") {
const credential = parseAlibabaTokenPlanCredential(rawApiKey);
if (!credential) throw new AIError.ConfigurationError("Invalid QwenCloud Token Plan credential");
apiKey = credential.token;
}
if (options.alibabaCodingPlanAuth && model.provider === "alibaba-coding-plan") {
try {
const parsed = JSON.parse(rawApiKey);
@@ -0,0 +1,43 @@
import { serializeAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import * as AIError from "../error";
import { createApiKeyLogin } from "./api-key-login";
import type { OAuthController, OAuthLoginCallbacks } from "./oauth/types";
import type { ProviderDefinition } from "./types";
const TOKEN_PLAN_BASE_URL = "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
const loginApiKey = createApiKeyLogin({
providerLabel: "QwenCloud Token Plan",
authUrl: "https://home.qwencloud.com/billing/subscription/token-plan-individual",
instructions: "Subscribe to Token Plan Individual and copy its dedicated API key",
promptMessage: "Paste your QwenCloud Token Plan API key",
placeholder: "sk-sp-...",
validation: {
kind: "models-endpoint",
provider: "QwenCloud Token Plan",
modelsUrl: `${TOKEN_PLAN_BASE_URL}/models`,
},
});
export async function loginAlibabaTokenPlan(options: OAuthController): Promise<string> {
if (!options.onPrompt) {
throw new AIError.OnPromptRequiredError("QwenCloud Token Plan");
}
const apiKey = await loginApiKey(options);
const cookie = await options.onPrompt({
message:
"Paste the Cookie request header from home.qwencloud.com for optional quota reporting, or press Enter to skip",
placeholder: "login_aliyunid_csrf=...; ...",
allowEmpty: true,
});
if (options.signal?.aborted) {
throw new AIError.LoginCancelledError();
}
return serializeAlibabaTokenPlanCredential(apiKey, cookie);
}
export const alibabaTokenPlanProvider = {
id: "alibaba-token-plan",
name: "QwenCloud Token Plan",
login: (cb: OAuthLoginCallbacks) => loginAlibabaTokenPlan(cb),
} as const satisfies ProviderDefinition;
+2
View File
@@ -1,6 +1,7 @@
import type { KnownProvider } from "@oh-my-pi/pi-catalog";
import { aimlApiProvider } from "./aimlapi";
import { alibabaCodingPlanProvider } from "./alibaba-coding-plan";
import { alibabaTokenPlanProvider } from "./alibaba-token-plan";
import { amazonBedrockProvider } from "./amazon-bedrock";
import { anthropicProvider } from "./anthropic";
import { azureProvider } from "./azure";
@@ -94,6 +95,7 @@ const ALL = [
gitlabDuoProvider,
gitLabDuoWorkflowProvider,
alibabaCodingPlanProvider,
alibabaTokenPlanProvider,
aimlApiProvider,
zhipuCodingPlanProvider,
umansProvider,
+2
View File
@@ -304,6 +304,8 @@ export interface UsageProvider {
supports?(params: UsageFetchParams): boolean;
/** True when fetchUsage contacts upstream and can authenticate the credential for health checks. */
validatesCredentials?: boolean;
/** Whether a failed refresh may serve the previous successful report. Defaults to true. */
retainLastGoodOnFailure?: boolean;
}
/** Request context used when ranking usage for a specific model. */
+202
View File
@@ -0,0 +1,202 @@
import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import type {
CredentialRankingStrategy,
UsageFetchContext,
UsageFetchParams,
UsageLimit,
UsageProvider,
UsageReport,
} from "../usage";
import { isRecord } from "../utils";
import { toNumber } from "./shared";
const PROVIDER = "alibaba-token-plan";
const CONSOLE_ORIGIN = "https://home.qwencloud.com";
const DASHBOARD_URL = `${CONSOLE_ORIGIN}/billing/subscription/token-plan-individual`;
const USER_INFO_URL = `${CONSOLE_ORIGIN}/tool/user/info.json`;
const USAGE_URL = `${CONSOLE_ORIGIN}/data/api.json?product=sfm_bailian&action=IntlBroadScopeAspnGateway`;
const GATEWAY_ACTION = "IntlBroadScopeAspnGateway";
const USAGE_API = "zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/usage";
const BROWSER_USER_AGENT =
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/143.0.0.0 Safari/537.36";
const FIVE_HOURS_MS = 5 * 60 * 60 * 1000;
const SEVEN_DAYS_MS = 7 * 24 * 60 * 60 * 1000;
function extractCookieValue(header: string, name: string): string | undefined {
for (const segment of header.split(";")) {
const separator = segment.indexOf("=");
if (separator < 0 || segment.slice(0, separator).trim() !== name) continue;
const value = segment.slice(separator + 1).trim();
return value || undefined;
}
return undefined;
}
function parseResetTime(value: unknown): number | undefined {
const parsed = toNumber(value);
if (parsed === undefined || parsed <= 0) return undefined;
return parsed < 1_000_000_000_000 ? parsed * 1000 : parsed;
}
function parseUsedFraction(value: unknown): number | undefined {
const parsed = toNumber(value);
if (parsed === undefined || parsed < 0) return undefined;
return Math.min(1, parsed > 1 ? parsed / 100 : parsed);
}
function usageStatus(usedFraction: number): UsageLimit["status"] {
if (usedFraction >= 1) return "exhausted";
if (usedFraction >= 0.8) return "warning";
return "ok";
}
function buildLimit(
id: "5h" | "7d",
label: string,
durationMs: number,
usedFraction: number | undefined,
resetsAt: number | undefined,
accountId: string | undefined,
): UsageLimit | undefined {
if (usedFraction === undefined) return undefined;
return {
id: `credits:${id}`,
label,
scope: { provider: PROVIDER, ...(accountId ? { accountId } : {}), windowId: id },
window: { id, label, durationMs, ...(resetsAt ? { resetsAt } : {}) },
amount: { used: usedFraction * 100, usedFraction, unit: "percent" },
status: usageStatus(usedFraction),
};
}
function accountIdFromUserData(value: Record<string, unknown>): string | undefined {
for (const key of ["accountId", "userId", "aliyunId", "loginId"]) {
const candidate = value[key];
if (typeof candidate === "string" && candidate.trim()) return candidate.trim();
if (typeof candidate === "number" && Number.isFinite(candidate)) return String(candidate);
}
return undefined;
}
async function fetchAlibabaTokenPlanUsage(
params: UsageFetchParams,
ctx: UsageFetchContext,
): Promise<UsageReport | null> {
if (params.provider !== PROVIDER || params.credential.type !== "api_key" || !params.credential.apiKey) return null;
const credential = parseAlibabaTokenPlanCredential(params.credential.apiKey);
if (!credential?.cookie) return null;
const cookie = credential.cookie;
try {
const userResponse = await ctx.fetch(USER_INFO_URL, {
headers: {
Accept: "application/json, text/plain, */*",
Cookie: cookie,
Referer: `${CONSOLE_ORIGIN}/`,
"User-Agent": BROWSER_USER_AGENT,
},
redirect: "manual",
signal: params.signal,
});
if (!userResponse.ok) {
ctx.logger?.warn("QwenCloud session lookup failed", { provider: PROVIDER, status: userResponse.status });
return null;
}
const userPayload: unknown = await userResponse.json();
if (!isRecord(userPayload) || !isRecord(userPayload.data) || typeof userPayload.data.secToken !== "string") {
ctx.logger?.warn("QwenCloud session response invalid", { provider: PROVIDER });
return null;
}
const secToken = userPayload.data.secToken;
const csrf = extractCookieValue(cookie, "login_aliyunid_csrf") ?? extractCookieValue(cookie, "csrf");
const headers: Record<string, string> = {
Accept: "application/json, text/plain, */*",
"Content-Type": "application/x-www-form-urlencoded",
Cookie: cookie,
Origin: CONSOLE_ORIGIN,
Referer: DASHBOARD_URL,
"User-Agent": BROWSER_USER_AGENT,
"X-Requested-With": "XMLHttpRequest",
};
if (csrf) {
headers["x-xsrf-token"] = csrf;
headers["x-csrf-token"] = csrf;
}
const body = new URLSearchParams({
product: "sfm_bailian",
action: GATEWAY_ACTION,
region: "ap-southeast-1",
sec_token: secToken,
params: JSON.stringify({ Api: USAGE_API, Data: {} }),
});
const usageResponse = await ctx.fetch(USAGE_URL, {
method: "POST",
headers,
body,
redirect: "manual",
signal: params.signal,
});
if (!usageResponse.ok) {
ctx.logger?.warn("QwenCloud usage fetch failed", { provider: PROVIDER, status: usageResponse.status });
return null;
}
const payload: unknown = await usageResponse.json();
if (!isRecord(payload) || payload.successResponse === false || !isRecord(payload.data)) {
ctx.logger?.warn("QwenCloud usage response invalid", { provider: PROVIDER });
return null;
}
const accountId = accountIdFromUserData(userPayload.data);
const limits = [
buildLimit(
"5h",
"5 Hour Credits",
FIVE_HOURS_MS,
parseUsedFraction(payload.data.per5HourPercentage),
parseResetTime(payload.data.per5HourResetTime),
accountId,
),
buildLimit(
"7d",
"7 Day Credits",
SEVEN_DAYS_MS,
parseUsedFraction(payload.data.per1WeekPercentage),
parseResetTime(payload.data.per1WeekResetTime),
accountId,
),
].filter((limit): limit is UsageLimit => limit !== undefined);
if (limits.length === 0) return null;
return {
provider: PROVIDER,
fetchedAt: Date.now(),
limits,
metadata: { source: "qwencloud-console", ...(accountId ? { accountId } : {}) },
};
} catch (error) {
ctx.logger?.warn("QwenCloud usage request failed", {
provider: PROVIDER,
error: error instanceof Error ? error.name : "unknown",
});
return null;
}
}
export const alibabaTokenPlanUsageProvider: UsageProvider = {
id: PROVIDER,
retainLastGoodOnFailure: false,
fetchUsage: fetchAlibabaTokenPlanUsage,
supports: params =>
params.provider === PROVIDER &&
params.credential.type === "api_key" &&
Boolean(params.credential.apiKey && parseAlibabaTokenPlanCredential(params.credential.apiKey)?.cookie),
};
export const alibabaTokenPlanRankingStrategy: CredentialRankingStrategy = {
findWindowLimits: report => ({
primary: report.limits.find(limit => limit.id === "credits:5h"),
secondary: report.limits.find(limit => limit.id === "credits:7d"),
}),
windowDefaults: {
primaryMs: FIVE_HOURS_MS,
secondaryMs: SEVEN_DAYS_MS,
},
};
@@ -0,0 +1,115 @@
import { describe, expect, test } from "bun:test";
import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
import type { UsageFetchParams } from "@oh-my-pi/pi-ai/usage";
import {
alibabaTokenPlanRankingStrategy,
alibabaTokenPlanUsageProvider,
} from "@oh-my-pi/pi-ai/usage/alibaba-token-plan";
import { serializeAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
function params(apiKey: string): UsageFetchParams {
return {
provider: "alibaba-token-plan",
credential: { type: "api_key", apiKey },
accountKey: "account-1",
};
}
describe("QwenCloud Token Plan opt-in usage", () => {
test("fetches quota windows with the Cookie stored during login", async () => {
const requests: { url: string; init?: RequestInit }[] = [];
const fetchMock: FetchImpl = (input, init) => {
requests.push({ url: String(input), init });
if (requests.length === 1) {
return Promise.resolve(
Response.json({ code: "200", data: { secToken: "sec-token", accountId: "account-1" } }),
);
}
return Promise.resolve(
Response.json({
code: "200",
successResponse: true,
data: {
per5HourPercentage: 0.25,
per5HourResetTime: 1_800_000_000_000,
per1WeekPercentage: 0.5,
per1WeekResetTime: 1_800_100_000_000,
},
}),
);
};
const cookie = "session_id=test; login_aliyunid_csrf=csrf-token; locale=en-US";
const credential = serializeAlibabaTokenPlanCredential("sk-sp-test", cookie);
const report = await alibabaTokenPlanUsageProvider.fetchUsage(params(credential), { fetch: fetchMock });
expect(requests).toHaveLength(2);
expect(requests[0]?.url).toBe("https://home.qwencloud.com/tool/user/info.json");
expect(new Headers(requests[0]?.init?.headers).get("Cookie")).toBe(cookie);
expect(requests[0]?.init?.redirect).toBe("manual");
expect(requests[1]?.url).toBe(
"https://home.qwencloud.com/data/api.json?product=sfm_bailian&action=IntlBroadScopeAspnGateway",
);
const usageHeaders = new Headers(requests[1]?.init?.headers);
expect(usageHeaders.get("Cookie")).toBe(cookie);
expect(usageHeaders.get("Origin")).toBe("https://home.qwencloud.com");
expect(usageHeaders.get("Referer")).toBe("https://home.qwencloud.com/billing/subscription/token-plan-individual");
expect(usageHeaders.get("X-Requested-With")).toBe("XMLHttpRequest");
expect(usageHeaders.get("x-xsrf-token")).toBe("csrf-token");
expect(usageHeaders.get("x-csrf-token")).toBe("csrf-token");
expect(requests[1]?.init?.redirect).toBe("manual");
const body = new URLSearchParams(String(requests[1]?.init?.body));
expect(body.get("params")).toBe(
JSON.stringify({ Api: "zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/usage", Data: {} }),
);
expect(report).toMatchObject({
provider: "alibaba-token-plan",
metadata: { source: "qwencloud-console", accountId: "account-1" },
limits: [
{
id: "credits:5h",
window: { id: "5h", durationMs: 18_000_000, resetsAt: 1_800_000_000_000 },
amount: { used: 25, usedFraction: 0.25, unit: "percent" },
},
{
id: "credits:7d",
window: { id: "7d", durationMs: 604_800_000, resetsAt: 1_800_100_000_000 },
amount: { used: 50, usedFraction: 0.5, unit: "percent" },
},
],
});
if (!report) throw new Error("expected QwenCloud usage report");
const windows = alibabaTokenPlanRankingStrategy.findWindowLimits(report, { modelId: "qwen3.7-plus" });
expect(windows.primary?.id).toBe("credits:5h");
expect(windows.secondary?.id).toBe("credits:7d");
});
test("does not claim quota support for API-key-only credentials", async () => {
let fetched = false;
const fetchMock: FetchImpl = () => {
fetched = true;
return Promise.resolve(Response.json({}));
};
const request = params("sk-sp-test");
expect(alibabaTokenPlanUsageProvider.supports?.(request)).toBe(false);
expect(await alibabaTokenPlanUsageProvider.fetchUsage(request, { fetch: fetchMock })).toBeNull();
expect(fetched).toBe(false);
});
test("fails closed when the stored console session has expired", async () => {
let requestCount = 0;
const fetchMock: FetchImpl = () => {
requestCount++;
return Promise.resolve(
requestCount === 1
? Response.json({ code: "200", data: { secToken: "sec-token" } })
: Response.json({ code: "ConsoleNeedLogin", message: "You need to log in.", successResponse: false }),
);
};
const credential = serializeAlibabaTokenPlanCredential("sk-sp-test", "session_id=expired");
expect(await alibabaTokenPlanUsageProvider.fetchUsage(params(credential), { fetch: fetchMock })).toBeNull();
expect(requestCount).toBe(2);
});
});
@@ -0,0 +1,77 @@
import { describe, expect, test } from "bun:test";
import { resolveOpenAIRequestSetup } from "@oh-my-pi/pi-ai/providers/openai-shared";
import { loginAlibabaTokenPlan } from "@oh-my-pi/pi-ai/registry/alibaba-token-plan";
import { getOAuthProviders } from "@oh-my-pi/pi-ai/registry/oauth";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
describe("QwenCloud Token Plan login", () => {
test("opens the Individual subscription page and validates without inference", async () => {
const authRequests: { url: string; instructions?: string }[] = [];
let requestedUrl = "";
let authorization = "";
const apiKey = await loginAlibabaTokenPlan({
onAuth: request => authRequests.push(request),
onPrompt: async prompt => (prompt.allowEmpty ? "" : " sk-sp-test "),
fetch: (input, init) => {
requestedUrl = String(input);
authorization = new Headers(init?.headers).get("Authorization") ?? "";
return Promise.resolve(Response.json({ data: [{ id: "qwen3.7-plus" }] }));
},
});
expect(apiKey).toBe("sk-sp-test");
expect(authRequests).toEqual([
{
url: "https://home.qwencloud.com/billing/subscription/token-plan-individual",
instructions: "Subscribe to Token Plan Individual and copy its dedicated API key",
},
]);
expect(requestedUrl).toBe("https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1/models");
expect(authorization).toBe("Bearer sk-sp-test");
});
test("stores an optional console Cookie while sending only the API key to inference", async () => {
const prompts = ["sk-sp-test", "session_id=test; login_aliyunid_csrf=csrf-token"];
const credential = await loginAlibabaTokenPlan({
onAuth: () => {},
onPrompt: async () => prompts.shift() ?? "",
fetch: () => Promise.resolve(Response.json({ data: [{ id: "qwen3.7-plus" }] })),
});
expect(JSON.parse(credential)).toEqual({
token: "sk-sp-test",
cookie: "session_id=test; login_aliyunid_csrf=csrf-token",
});
const model = getBundledModel<"openai-completions">("alibaba-token-plan", "qwen3.7-plus");
if (!model) throw new Error("expected bundled QwenCloud Token Plan model");
const setup = resolveOpenAIRequestSetup(model, {
apiKey: credential,
messages: [],
});
expect(setup.headers.Authorization).toBe("Bearer sk-sp-test");
expect(JSON.stringify(setup)).not.toContain("session_id=test");
});
test("rejects malformed compound credentials before inference setup", () => {
const model = getBundledModel<"openai-completions">("alibaba-token-plan", "qwen3.7-plus");
if (!model) throw new Error("expected bundled QwenCloud Token Plan model");
for (const apiKey of [
' {"token":"sk-sp-test","cookie":"session=secret"',
'"token":"sk-sp-test","cookie":"session=secret"}',
]) {
expect(() => resolveOpenAIRequestSetup(model, { apiKey, messages: [] })).toThrow(
"Invalid QwenCloud Token Plan credential",
);
}
});
test("registers Token Plan separately from the legacy Alibaba Coding Plan", () => {
const providers = getOAuthProviders();
expect(providers.find(provider => provider.id === "alibaba-token-plan")).toMatchObject({
name: "QwenCloud Token Plan",
available: true,
});
expect(providers.some(provider => provider.id === "alibaba-coding-plan")).toBe(true);
});
});
@@ -9,6 +9,7 @@ import * as deepseekModule from "@oh-my-pi/pi-ai/registry/deepseek";
import * as kagiModule from "@oh-my-pi/pi-ai/registry/kagi";
import * as ollamaCloudModule from "@oh-my-pi/pi-ai/registry/ollama-cloud";
import * as aiStream from "@oh-my-pi/pi-ai/stream";
import { serializeAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import { removeWithRetries } from "../../utils/src/temp";
function countCredentialRows(dbPath: string, provider: string): number {
@@ -126,6 +127,55 @@ describe("AuthStorage api-key login upsert", () => {
expect(rotatedKeys).toEqual(["first-kagi-key", "second-kagi-key"]);
});
it("replaces Token Plan Cookies by API-token identity without collapsing different tokens", () => {
if (!store) throw new Error("test setup failed");
const firstToken = "sk-sp-first";
const secondToken = "sk-sp-second";
store.upsertAuthCredentialForProvider("alibaba-token-plan", {
type: "api_key",
key: serializeAlibabaTokenPlanCredential(firstToken, "session=old"),
source: "login",
});
store.upsertAuthCredentialForProvider("alibaba-token-plan", {
type: "api_key",
key: serializeAlibabaTokenPlanCredential(firstToken, "session=fresh"),
source: "login",
});
store.upsertAuthCredentialForProvider("alibaba-token-plan", {
type: "api_key",
key: serializeAlibabaTokenPlanCredential(secondToken, "session=second"),
source: "login",
});
expect(store.listAuthCredentials("alibaba-token-plan").map(entry => entry.credential)).toEqual([
{
type: "api_key",
key: serializeAlibabaTokenPlanCredential(firstToken, "session=fresh"),
source: "login",
},
{
type: "api_key",
key: serializeAlibabaTokenPlanCredential(secondToken, "session=second"),
source: "login",
},
]);
store.upsertAuthCredentialForProvider("alibaba-token-plan", {
type: "api_key",
key: firstToken,
source: "login",
});
expect(store.listAuthCredentials("alibaba-token-plan").map(entry => entry.credential)).toEqual([
{ type: "api_key", key: firstToken, source: "login" },
{
type: "api_key",
key: serializeAlibabaTokenPlanCredential(secondToken, "session=second"),
source: "login",
},
]);
});
it("hard-deletes superseded api-key rows when a different key replaces them", () => {
if (!store || !dbPath) throw new Error("test setup failed");
@@ -19,7 +19,9 @@ import {
type StoredAuthCredential,
} from "@oh-my-pi/pi-ai/auth-storage";
import type { UsageLimit, UsageReport } from "@oh-my-pi/pi-ai/usage";
import { alibabaTokenPlanUsageProvider } from "@oh-my-pi/pi-ai/usage/alibaba-token-plan";
import * as claudeUsage from "@oh-my-pi/pi-ai/usage/claude";
import { serializeAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
function anthropicReports(reports: UsageReport[] | null): UsageReport[] {
return (reports ?? []).filter(r => r.provider === "anthropic");
@@ -325,6 +327,67 @@ describe("AuthStorage usage cache: last-good failure fallback", () => {
expect(calls).toBe(3);
});
});
describe("AuthStorage usage cache: provider failure policy", () => {
it("drops stale QwenCloud quota after the optional console session expires", async () => {
const store = makeStore([
{
id: 1,
provider: "alibaba-token-plan",
credential: {
type: "api_key",
key: serializeAlibabaTokenPlanCredential("sk-sp-test", "session_id=test"),
},
disabledCause: null,
},
]);
let usageCalls = 0;
const usageFetch = Object.assign(
(input: string | URL | Request) => {
if (String(input).endsWith("/tool/user/info.json")) {
return Promise.resolve(Response.json({ code: "200", data: { secToken: "sec-token" } }));
}
usageCalls++;
return Promise.resolve(
usageCalls === 1
? Response.json({
code: "200",
successResponse: true,
data: {
per5HourPercentage: 0.25,
per5HourResetTime: 1_800_000_000_000,
per1WeekPercentage: 0.5,
per1WeekResetTime: 1_800_100_000_000,
},
})
: Response.json({
code: "ConsoleNeedLogin",
message: "You need to log in.",
successResponse: false,
}),
);
},
{ preconnect: fetch.preconnect },
);
const storage = new AuthStorage(store, {
usageFetch,
usageProviderResolver: provider =>
provider === "alibaba-token-plan" ? alibabaTokenPlanUsageProvider : undefined,
});
await storage.reload();
try {
const first = (await storage.fetchUsageReports()) ?? [];
expect(first.filter(report => report.provider === "alibaba-token-plan")).toHaveLength(1);
expect(usageCalls).toBe(1);
expireCachePayloads(store);
const second = (await storage.fetchUsageReports()) ?? [];
expect(second.filter(report => report.provider === "alibaba-token-plan")).toHaveLength(0);
expect(usageCalls).toBe(2);
} finally {
storage.close();
}
});
});
describe("AuthStorage usage cache: jitter", () => {
it("writes per-credential cache TTLs with ±25% jitter so refreshes decorrelate", async () => {
+1
View File
@@ -10,6 +10,7 @@
- Added an opt-in Vercel AI Gateway automatic prompt-cache compatibility option alongside provider routing preferences.
- Added Vercel AI Gateway Responses cache-anchor and cache-lifetime compatibility controls.
- Added resolved OpenAI GPT-5.6 prompt-cache breakpoint capability metadata, keeping older models and compatible endpoints opt-in only.
- Added the native `alibaba-token-plan` provider with QwenCloud Token Plan Individual discovery and a curated chat-model fallback catalog ([#6151](https://github.com/can1357/oh-my-pi/issues/6151)).
## [17.0.9] - 2026-07-23
+19 -9
View File
@@ -29,6 +29,7 @@ import {
} from "../src/provider-models/descriptor-types";
import { PROVIDER_DESCRIPTORS } from "../src/provider-models/descriptors";
import {
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
ANTHROPIC_CURATED_FALLBACK_MODELS,
buildFireworksFastSeed,
buildXaiOAuthStaticSeed,
@@ -104,13 +105,16 @@ async function resolveProviderApiKey(providerId: string, catalog: CatalogDiscove
return undefined;
}
type CatalogProviderFetchResult = { models: ModelSpec[]; succeeded: boolean };
async function fetchProviderModelsFromCatalog(descriptor: CatalogProviderDescriptor): Promise<ModelSpec[]> {
async function fetchProviderModelsFromCatalog(
descriptor: CatalogProviderDescriptor,
): Promise<CatalogProviderFetchResult> {
const apiKey = await resolveProviderApiKey(descriptor.providerId, descriptor.catalogDiscovery);
if (!apiKey && !allowsUnauthenticatedCatalogDiscovery(descriptor)) {
console.log(`No ${descriptor.catalogDiscovery.label} credentials found (env or agent.db), using fallback models`);
return [];
return { models: [], succeeded: false };
}
try {
@@ -129,19 +133,19 @@ async function fetchProviderModelsFromCatalog(descriptor: CatalogProviderDescrip
console.warn(
`${descriptor.catalogDiscovery.label} dynamic fetch failed (stale cache merge), using fallback models`,
);
return [];
return { models: [], succeeded: false };
}
const models = result.models.filter(model => model.provider === descriptor.providerId);
if (models.length === 0) {
console.warn(`${descriptor.catalogDiscovery.label} discovery returned no models, using fallback models`);
return [];
console.warn(`${descriptor.catalogDiscovery.label} discovery returned no models`);
return { models: [], succeeded: true };
}
console.log(`Fetched ${models.length} models from ${descriptor.catalogDiscovery.label} model manager`);
// The manager returns built models; models.json stores specs (sparse compat).
return models.map(model => toModelSpec(model));
return { models: models.map(model => toModelSpec(model)), succeeded: true };
} catch (error) {
console.error(`Failed to fetch ${descriptor.catalogDiscovery.label} models:`, error);
return [];
return { models: [], succeeded: false };
}
}
@@ -485,12 +489,12 @@ async function generateModels() {
const catalogProviderModelBatches = await Promise.all(
catalogProviderDescriptors.map(async descriptor => ({
descriptor,
models: await fetchProviderModelsFromCatalog(descriptor),
...(await fetchProviderModelsFromCatalog(descriptor)),
})),
);
const authoritativeCatalogProviders = new Set(
catalogProviderModelBatches
.filter(batch => batch.descriptor.dynamicModelsAuthoritative === true && batch.models.length > 0)
.filter(batch => batch.descriptor.dynamicModelsAuthoritative === true && batch.succeeded)
.map(batch => batch.descriptor.providerId),
);
const catalogProviderModels = catalogProviderModelBatches.flatMap(batch => batch.models);
@@ -517,6 +521,12 @@ async function generateModels() {
// persisted `modelRoles.default = "xai-oauth/<id>"` is honored before the
// async refresh fires (interactive boot does not await refresh).
allModels.push(...buildXaiOAuthStaticSeed());
// Seed QwenCloud's documented Token Plan models when credentialed
// discovery is unavailable. A successful `/models` response is authoritative
// for the subscribed edition and must not be widened by the fallback.
if (!authoritativeCatalogProviders.has("alibaba-token-plan")) {
allModels.push(...ALIBABA_TOKEN_PLAN_STATIC_MODELS);
}
// Seed Anthropic models that are live on the first-party API or in limited
// release but that models.dev has not catalogued yet (e.g. Claude Fable 5 /
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
+11 -3
View File
@@ -18,7 +18,10 @@ import { isMimoModelIdOrName } from "../src/identity/family";
import { getLongestModelLikeIdSegment } from "../src/identity/id";
import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference";
import { resolveModelThinking } from "../src/model-thinking";
import { resolveWaferServerlessThinkingFormat } from "../src/provider-models/openai-compat";
import {
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
resolveWaferServerlessThinkingFormat,
} from "../src/provider-models/openai-compat";
import type { Api, Model, ModelSpec } from "../src/types";
import { isVariantCollapsedSpec } from "../src/variant-collapse";
import { buildCanonicalModelIndex, buildCanonicalReferenceData } from "./equivalence";
@@ -78,11 +81,12 @@ export function applyGeneratedModelPolicies(models: ModelSpec<Api>[]): void {
* Recompute `thinking` from the canonical deriver, replacing any baked value.
* Mirrors `buildModel`'s trust-or-derive resolution with trust disabled: the
* generator is the authority that produces the trusted values. Collapsed
* effort-tier variants are exempt — their collapse table authored the
* routing/off-suppression metadata and the deriver cannot reproduce it.
* effort-tier variants and provider-authored wire ladders are exempt because
* the generic deriver cannot reproduce that routing metadata.
*/
export function rebakeModelThinking(model: ModelSpec<Api>): void {
if (isVariantCollapsedSpec(model)) return;
if (model.provider === "alibaba-token-plan" && model.id === "qwen3.8-max-preview" && model.thinking) return;
const requiresProviderAuthoredEffort =
model.provider === "umans" && (model.thinking?.requiresEffort === true || model.id === "umans-kimi-k2.7");
const thinking = resolveModelThinking({ ...model, thinking: undefined }, buildCompat(model));
@@ -208,6 +212,10 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
model.contextWindow = copilotLimits.contextWindow;
model.maxTokens = copilotLimits.maxTokens;
}
if (model.provider === "alibaba-token-plan") {
const reference = ALIBABA_TOKEN_PLAN_STATIC_MODELS.find(candidate => candidate.id === model.id);
if (reference) model.name = reference.name;
}
if (model.provider === "ollama-cloud") {
model.omitMaxOutputTokens = true;
+4 -1
View File
@@ -41,7 +41,10 @@ export const KNOWN_HOSTS = {
zai: { providers: ["zai"], urlMarkers: ["api.z.ai"] },
zhipu: { providers: ["zhipu-coding-plan"], urlMarkers: ["open.bigmodel.cn"] },
kilo: { providers: ["kilo"], urlMarkers: ["api.kilo.ai"] },
alibabaDashscope: { providers: ["alibaba-coding-plan"], urlMarkers: ["dashscope"] },
alibabaDashscope: {
providers: ["alibaba-coding-plan", "alibaba-token-plan"],
urlMarkers: ["dashscope", "token-plan."],
},
umans: { providers: ["umans"], urlMarkers: ["api.code.umans.ai"] },
xiaomi: { providers: ["xiaomi"], providerPrefixes: ["xiaomi-token-plan-"], urlMarkers: ["xiaomimimo.com"] },
xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] },
@@ -10,6 +10,7 @@ const DEFAULT_MODEL_PROVIDER_ORDER = [
"kimi-code",
"moonshot",
"qwen-portal",
"alibaba-token-plan",
"zai",
"xai-oauth",
"xai",
+192 -1
View File
@@ -7565,6 +7565,197 @@
}
}
},
"alibaba-token-plan": {
"deepseek-v4-pro": {
"id": "deepseek-v4-pro",
"name": "DeepSeek V4 Pro",
"api": "openai-completions",
"provider": "alibaba-token-plan",
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 384000,
"thinking": {
"mode": "effort",
"efforts": [
"high",
"max"
]
},
"compat": {
"supportsDeveloperRole": false
}
},
"glm-5.2": {
"id": "glm-5.2",
"name": "GLM-5.2",
"api": "openai-completions",
"provider": "alibaba-token-plan",
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"max"
]
},
"compat": {
"supportsDeveloperRole": false
}
},
"qwen3.6-flash": {
"id": "qwen3.6-flash",
"name": "Qwen3.6 Flash",
"api": "openai-completions",
"provider": "alibaba-token-plan",
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
},
"compat": {
"supportsDeveloperRole": false
}
},
"qwen3.7-max": {
"id": "qwen3.7-max",
"name": "Qwen3.7 Max",
"api": "openai-completions",
"provider": "alibaba-token-plan",
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
},
"compat": {
"supportsDeveloperRole": false
}
},
"qwen3.7-plus": {
"id": "qwen3.7-plus",
"name": "Qwen3.7 Plus",
"api": "openai-completions",
"provider": "alibaba-token-plan",
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 64000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
},
"compat": {
"supportsDeveloperRole": false
}
},
"qwen3.8-max-preview": {
"id": "qwen3.8-max-preview",
"name": "Qwen3.8 Max Preview",
"api": "openai-completions",
"provider": "alibaba-token-plan",
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 983616,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"high",
"xhigh"
],
"requiresEffort": true
},
"compat": {
"supportsDeveloperRole": false,
"supportsReasoningEffort": true
}
}
},
"amazon-bedrock": {
"anthropic.claude-3-5-haiku-20241022-v1:0": {
"id": "anthropic.claude-3-5-haiku-20241022-v1:0",
@@ -94609,4 +94800,4 @@
}
}
}
}
}
@@ -11,6 +11,7 @@ import { ollamaCloudModelManagerOptions } from "./ollama";
import {
aimlApiModelManagerOptions,
alibabaCodingPlanModelManagerOptions,
alibabaTokenPlanModelManagerOptions,
anthropicModelManagerOptions,
basetenModelManagerOptions,
cerebrasModelManagerOptions,
@@ -76,6 +77,14 @@ export const CATALOG_PROVIDERS = [
createModelManagerOptions: (config: ModelManagerConfig) => alibabaCodingPlanModelManagerOptions(config),
catalogDiscovery: { label: "Alibaba Coding Plan" },
},
{
id: "alibaba-token-plan",
defaultModel: "qwen3.7-plus",
envVars: ["ALIBABA_TOKEN_PLAN_API_KEY", "BAILIAN_TOKEN_PLAN_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => alibabaTokenPlanModelManagerOptions(config),
dynamicModelsAuthoritative: true,
catalogDiscovery: { label: "QwenCloud Token Plan" },
},
{
id: "baseten",
defaultModel: "moonshotai/Kimi-K2.7-Code",
@@ -17,6 +17,7 @@ import type { ModelManagerOptions } from "../model-manager";
import { getBundledModels } from "../models";
import type { Api, FetchImpl, Model, ModelSpec, OpenAICompat, Provider, ThinkingConfig } from "../types";
import { discoveryFetch, isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
import { parseAlibabaTokenPlanCredential } from "../wire/alibaba-token-plan";
import { coreWeaveProjectHeaders } from "../wire/coreweave";
import {
COPILOT_API_HEADERS,
@@ -2429,6 +2430,164 @@ export function alibabaCodingPlanModelManagerOptions(
};
}
// ---------------------------------------------------------------------------
// Alibaba Token Plan
// ---------------------------------------------------------------------------
export const ALIBABA_TOKEN_PLAN_BASE_URL = "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
const ALIBABA_TOKEN_PLAN_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } as const;
const ALIBABA_TOKEN_PLAN_COMPAT: OpenAICompat = {
supportsDeveloperRole: false,
};
const ALIBABA_TOKEN_PLAN_REASONING: ThinkingConfig = {
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High],
};
export const ALIBABA_TOKEN_PLAN_STATIC_MODELS: readonly ModelSpec<"openai-completions">[] = [
{
id: "qwen3.8-max-preview",
name: "Qwen3.8 Max Preview",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 983_616,
maxTokens: 131_072,
thinking: {
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
},
compat: {
...ALIBABA_TOKEN_PLAN_COMPAT,
supportsReasoningEffort: true,
},
},
{
id: "qwen3.7-max",
name: "Qwen3.7 Max",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 1_000_000,
maxTokens: 65_536,
thinking: ALIBABA_TOKEN_PLAN_REASONING,
compat: ALIBABA_TOKEN_PLAN_COMPAT,
},
{
id: "qwen3.7-plus",
name: "Qwen3.7 Plus",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 1_000_000,
maxTokens: 64_000,
thinking: ALIBABA_TOKEN_PLAN_REASONING,
compat: ALIBABA_TOKEN_PLAN_COMPAT,
},
{
id: "qwen3.6-flash",
name: "Qwen3.6 Flash",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text", "image"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 1_000_000,
maxTokens: 65_536,
thinking: ALIBABA_TOKEN_PLAN_REASONING,
compat: ALIBABA_TOKEN_PLAN_COMPAT,
},
{
id: "glm-5.2",
name: "GLM-5.2",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 1_000_000,
maxTokens: 131_072,
thinking: {
mode: "effort",
efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.Max],
},
compat: ALIBABA_TOKEN_PLAN_COMPAT,
},
{
id: "deepseek-v4-pro",
name: "DeepSeek V4 Pro",
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
reasoning: true,
input: ["text"],
cost: ALIBABA_TOKEN_PLAN_COST,
contextWindow: 1_000_000,
maxTokens: 384_000,
thinking: {
mode: "effort",
efforts: [Effort.High, Effort.Max],
},
compat: ALIBABA_TOKEN_PLAN_COMPAT,
},
];
export interface AlibabaTokenPlanModelManagerConfig {
apiKey?: string;
baseUrl?: string;
fetch?: FetchImpl;
}
export function alibabaTokenPlanModelManagerOptions(
config?: AlibabaTokenPlanModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
const credential = config?.apiKey ? parseAlibabaTokenPlanCredential(config.apiKey) : undefined;
const apiKey = credential?.token;
const baseUrl = config?.baseUrl ?? ALIBABA_TOKEN_PLAN_BASE_URL;
return {
providerId: "alibaba-token-plan",
dynamicModelsAuthoritative: true,
staticModels: ALIBABA_TOKEN_PLAN_STATIC_MODELS,
...(apiKey && {
fetchDynamicModels: () =>
fetchOpenAICompatibleModels({
api: "openai-completions",
provider: "alibaba-token-plan",
baseUrl,
apiKey,
filterModel: (_entry, model) =>
ALIBABA_TOKEN_PLAN_STATIC_MODELS.some(reference => reference.id === model.id),
mapModel: (_entry, defaults) => {
const reference = ALIBABA_TOKEN_PLAN_STATIC_MODELS.find(model => model.id === defaults.id);
return reference
? {
...reference,
id: defaults.id,
api: defaults.api,
provider: defaults.provider,
baseUrl: defaults.baseUrl,
}
: defaults;
},
fetch: config?.fetch,
}),
}),
};
}
// ---------------------------------------------------------------------------
// 11. Vercel AI Gateway
// ---------------------------------------------------------------------------
@@ -0,0 +1,27 @@
export interface AlibabaTokenPlanCredential {
token: string;
cookie?: string;
}
const TOKEN_PATTERN = /^sk-[A-Za-z0-9._~+/-]+={0,2}$/;
export function parseAlibabaTokenPlanCredential(value: string): AlibabaTokenPlanCredential | null {
const trimmed = value.trim();
if (!trimmed) return null;
if (!trimmed.startsWith("{")) return TOKEN_PATTERN.test(trimmed) ? { token: trimmed } : null;
try {
const parsed = JSON.parse(trimmed) as { token?: unknown; cookie?: unknown };
if (typeof parsed.token !== "string" || !TOKEN_PATTERN.test(parsed.token.trim())) return null;
if (parsed.cookie !== undefined && typeof parsed.cookie !== "string") return null;
const token = parsed.token.trim();
const cookie = parsed.cookie?.trim();
return cookie ? { token, cookie } : { token };
} catch {
return null;
}
}
export function serializeAlibabaTokenPlanCredential(token: string, cookie: string): string {
const trimmedCookie = cookie.trim();
return trimmedCookie ? JSON.stringify({ token, cookie: trimmedCookie }) : token;
}
@@ -0,0 +1,112 @@
import { describe, expect, test } from "bun:test";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import { CATALOG_PROVIDERS } from "@oh-my-pi/pi-catalog/provider-models/descriptors";
import {
ALIBABA_TOKEN_PLAN_BASE_URL,
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
alibabaTokenPlanModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
import { serializeAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
describe("QwenCloud Token Plan provider", () => {
test("ships the documented Individual text-model allowlist", () => {
expect(ALIBABA_TOKEN_PLAN_STATIC_MODELS.map(model => model.id)).toEqual([
"qwen3.8-max-preview",
"qwen3.7-max",
"qwen3.7-plus",
"qwen3.6-flash",
"glm-5.2",
"deepseek-v4-pro",
]);
const preview = ALIBABA_TOKEN_PLAN_STATIC_MODELS[0];
expect(preview).toMatchObject({
provider: "alibaba-token-plan",
baseUrl: ALIBABA_TOKEN_PLAN_BASE_URL,
contextWindow: 983_616,
maxTokens: 131_072,
input: ["text", "image"],
thinking: {
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
},
compat: {
supportsDeveloperRole: false,
supportsReasoningEffort: true,
},
});
expect(ALIBABA_TOKEN_PLAN_STATIC_MODELS.find(model => model.id === "glm-5.2")?.thinking?.efforts).toEqual([
Effort.Minimal,
Effort.Low,
Effort.Medium,
Effort.High,
Effort.Max,
]);
});
test("discovers the subscribed allowlist from the native models endpoint", async () => {
let requestedUrl = "";
let authorization = "";
const fetchMock: FetchImpl = (input, init) => {
requestedUrl = String(input);
authorization = new Headers(init?.headers).get("Authorization") ?? "";
return Promise.resolve(
Response.json({
data: [
{
id: "qwen3.7-plus",
name: "server metadata must not replace curated metadata",
owned_by: "qwencloud",
context_length: 262_144,
max_completion_tokens: 16_384,
},
{ id: "wan2.7-image", owned_by: "qwencloud" },
],
}),
);
};
const apiKey = ` ${serializeAlibabaTokenPlanCredential("sk-sp-test", "session_id=test")} `;
const options = alibabaTokenPlanModelManagerOptions({ apiKey, fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
expect(requestedUrl).toBe(`${ALIBABA_TOKEN_PLAN_BASE_URL}/models`);
expect(authorization).toBe("Bearer sk-sp-test");
expect(models).toHaveLength(1);
expect(models?.[0]).toMatchObject({
id: "qwen3.7-plus",
provider: "alibaba-token-plan",
name: "Qwen3.7 Plus",
contextWindow: 1_000_000,
maxTokens: 64_000,
});
expect(options.dynamicModelsAuthoritative).toBe(true);
});
test("rejects malformed compound credentials before model discovery", () => {
let fetched = false;
const fetchMock: FetchImpl = () => {
fetched = true;
return Promise.resolve(Response.json({ data: [] }));
};
const options = alibabaTokenPlanModelManagerOptions({
apiKey: ' {"token":"sk-sp-test","cookie":"session=secret"',
fetch: fetchMock,
});
expect(options.fetchDynamicModels).toBeUndefined();
expect(fetched).toBe(false);
});
test("uses Token Plan-specific environment keys and authoritative discovery", () => {
const descriptor = CATALOG_PROVIDERS.find(provider => provider.id === "alibaba-token-plan");
expect(descriptor).toMatchObject({
defaultModel: "qwen3.7-plus",
envVars: ["ALIBABA_TOKEN_PLAN_API_KEY", "BAILIAN_TOKEN_PLAN_API_KEY"],
dynamicModelsAuthoritative: true,
catalogDiscovery: { label: "QwenCloud Token Plan" },
});
});
});
@@ -140,6 +140,29 @@ describe("generated model policies", () => {
});
});
it("preserves QwenCloud's mandatory qwen3.8 effort ladder", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "qwen3.8-max-preview",
api: "openai-completions",
provider: "alibaba-token-plan",
thinking: {
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
},
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Low, Effort.High, Effort.XHigh],
requiresEffort: true,
});
});
it("pins zai glm-5.2 base id to 1M context", () => {
const models = [
createSpec({