feat(ai): complete Wafer Serverless bundled catalog from live /v1/models
The earlier bundled catalog was missing three Serverless-only models and had two entries whose metadata didn't match what `/v1/models` returns. Cross-referenced against the live https://pass.wafer.ai/v1/models response (public, unauthenticated) and rebuilt the wafer-serverless block from it. Added: - `qwen3.7-max` — 256k ctx, reasoning, $5/$15/$0.50 per M. Canonical id is lowercase; round-trips verbatim on the wire. - `deepseek-v4-flash` — 1M ctx, reasoning, $0.14/$0.28/$0.01 per M. - `deepseek-v4-pro` — 1M ctx, reasoning, $1.74/$3.48/$0.02 per M. Fixed: - `Kimi-K2.6` cost was 0; live pricing is 88/384/9 cents per M → $0.88/$3.84/$0.09. - `Qwen3.6-35B-A3B` was bundled with 32k context and text-only; live `/v1/models` reports 256k and `vision: true`. All three new reasoning models carry the same zai-style thinking compat as the existing GLM/Kimi entries (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`) — Wafer normalizes reasoning output to `reasoning_content` regardless of upstream provider. Tests extended to cover the new entries plus the corrected Kimi pricing and Qwen3.6 context/vision (5 cases, 47 expect() calls, all passing).
This commit is contained in:
@@ -5,7 +5,7 @@
|
||||
### Added
|
||||
|
||||
- Added `CheckCredentialsOptions.completionProbe` (and `completionTimeoutMs`) so `AuthStorage.checkCredentials` can additionally exercise each credential against the provider's chat-completion endpoint after refresh-on-expiry. Result lands on `CredentialHealthResult.completion` ({ok, reason?, modelId?, latencyMs?}) without disturbing the usage `ok` field. Public types: `CompletionProbe`, `CompletionProbeInput`, `CompletionProbeCredential`, `CredentialCompletionResult`. The probe is invoked even when no `UsageProvider` is registered for the row, and is skipped when OAuth refresh fails (the stale bytes would only mask the upstream failure).
|
||||
- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Qwen3.5-397B-A17B, Kimi-K2.6, Qwen3.6-35B-A3B}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. GLM-family entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`).
|
||||
- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Kimi-K2.6, Qwen3.5-397B-A17B, Qwen3.6-35B-A3B, qwen3.7-max, deepseek-v4-flash, deepseek-v4-pro}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. Reasoning-capable entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`).
|
||||
|
||||
### Changed
|
||||
|
||||
|
||||
@@ -70711,6 +70711,64 @@
|
||||
}
|
||||
},
|
||||
"wafer-serverless": {
|
||||
"deepseek-v4-flash": {
|
||||
"id": "deepseek-v4-flash",
|
||||
"name": "DeepSeek V4 Flash",
|
||||
"api": "openai-completions",
|
||||
"provider": "wafer-serverless",
|
||||
"baseUrl": "https://pass.wafer.ai/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.14,
|
||||
"output": 0.28,
|
||||
"cacheRead": 0.01,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false,
|
||||
"thinkingFormat": "zai",
|
||||
"reasoningContentField": "reasoning_content"
|
||||
},
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"deepseek-v4-pro": {
|
||||
"id": "deepseek-v4-pro",
|
||||
"name": "DeepSeek V4 Pro",
|
||||
"api": "openai-completions",
|
||||
"provider": "wafer-serverless",
|
||||
"baseUrl": "https://pass.wafer.ai/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.74,
|
||||
"output": 3.48,
|
||||
"cacheRead": 0.02,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false,
|
||||
"thinkingFormat": "zai",
|
||||
"reasoningContentField": "reasoning_content"
|
||||
},
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"GLM-5.1": {
|
||||
"id": "GLM-5.1",
|
||||
"name": "GLM-5.1",
|
||||
@@ -70752,9 +70810,9 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"input": 0.88,
|
||||
"output": 3.84,
|
||||
"cacheRead": 0.09,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
@@ -70800,19 +70858,49 @@
|
||||
"provider": "wafer-serverless",
|
||||
"baseUrl": "https://pass.wafer.ai/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.15,
|
||||
"output": 1,
|
||||
"cacheRead": 0.02,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 65536,
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false
|
||||
}
|
||||
},
|
||||
"qwen3.7-max": {
|
||||
"id": "qwen3.7-max",
|
||||
"name": "Qwen3.7-Max",
|
||||
"api": "openai-completions",
|
||||
"provider": "wafer-serverless",
|
||||
"baseUrl": "https://pass.wafer.ai/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"input": 5,
|
||||
"output": 15,
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 32768,
|
||||
"maxTokens": 32768,
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 65536,
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false
|
||||
"supportsDeveloperRole": false,
|
||||
"thinkingFormat": "zai",
|
||||
"reasoningContentField": "reasoning_content"
|
||||
},
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -85,7 +85,7 @@ describe("Wafer Pass provider", () => {
|
||||
});
|
||||
|
||||
describe("Wafer Serverless provider", () => {
|
||||
it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Kimi-K2.6, Qwen3.6)", () => {
|
||||
it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Qwen3.6, Qwen3.7-Max, Kimi-K2.6, DeepSeek V4 Flash/Pro)", () => {
|
||||
const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1");
|
||||
expect(glm).toBeDefined();
|
||||
expect(glm.provider).toBe("wafer-serverless");
|
||||
@@ -99,15 +99,40 @@ describe("Wafer Serverless provider", () => {
|
||||
const kimi = getBundledModel<"openai-completions">("wafer-serverless", "Kimi-K2.6");
|
||||
expect(kimi).toBeDefined();
|
||||
expect(kimi.contextWindow).toBe(262144);
|
||||
// Kimi-K2.6 carries its real cents/M pricing (88 / 384 / 9 → 0.88 / 3.84 / 0.09).
|
||||
expect(kimi.cost).toEqual({ input: 0.88, output: 3.84, cacheRead: 0.09, cacheWrite: 0 });
|
||||
|
||||
const qwen36 = getBundledModel<"openai-completions">("wafer-serverless", "Qwen3.6-35B-A3B");
|
||||
expect(qwen36).toBeDefined();
|
||||
// Serverless-only Qwen3.6 — context capped at 32k per docs.
|
||||
expect(qwen36.contextWindow).toBe(32768);
|
||||
// Qwen3.6 advertises 256k context and vision per the live /v1/models response.
|
||||
expect(qwen36.contextWindow).toBe(256000);
|
||||
expect(qwen36.input).toEqual(["text", "image"]);
|
||||
|
||||
const qwen37max = getBundledModel<"openai-completions">("wafer-serverless", "qwen3.7-max");
|
||||
expect(qwen37max).toBeDefined();
|
||||
// Wafer's canonical id is lowercase `qwen3.7-max` — must round-trip verbatim.
|
||||
expect(qwen37max.id).toBe("qwen3.7-max");
|
||||
expect(qwen37max.name).toBe("Qwen3.7-Max");
|
||||
expect(qwen37max.reasoning).toBe(true);
|
||||
expect(qwen37max.compat?.thinkingFormat).toBe("zai");
|
||||
|
||||
const dsFlash = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-flash");
|
||||
expect(dsFlash).toBeDefined();
|
||||
expect(dsFlash.contextWindow).toBe(1000000);
|
||||
expect(dsFlash.reasoning).toBe(true);
|
||||
expect(dsFlash.compat?.reasoningContentField).toBe("reasoning_content");
|
||||
|
||||
const dsPro = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-pro");
|
||||
expect(dsPro).toBeDefined();
|
||||
expect(dsPro.contextWindow).toBe(1000000);
|
||||
expect(dsPro.reasoning).toBe(true);
|
||||
});
|
||||
|
||||
it("does not expose Serverless-only ids on the Wafer Pass catalog", () => {
|
||||
expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "qwen3.7-max")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "deepseek-v4-flash")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "deepseek-v4-pro")).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user