feat(ai): complete Wafer Serverless bundled catalog from live /v1/models

The earlier bundled catalog was missing three Serverless-only models and
had two entries whose metadata didn't match what `/v1/models` returns.
Cross-referenced against the live https://pass.wafer.ai/v1/models response
(public, unauthenticated) and rebuilt the wafer-serverless block from it.

Added:
- `qwen3.7-max` — 256k ctx, reasoning, $5/$15/$0.50 per M.
  Canonical id is lowercase; round-trips verbatim on the wire.
- `deepseek-v4-flash` — 1M ctx, reasoning, $0.14/$0.28/$0.01 per M.
- `deepseek-v4-pro` — 1M ctx, reasoning, $1.74/$3.48/$0.02 per M.

Fixed:
- `Kimi-K2.6` cost was 0; live pricing is 88/384/9 cents per M
  → $0.88/$3.84/$0.09.
- `Qwen3.6-35B-A3B` was bundled with 32k context and text-only;
  live `/v1/models` reports 256k and `vision: true`.

All three new reasoning models carry the same zai-style thinking compat
as the existing GLM/Kimi entries (`thinkingFormat: "zai"`,
`reasoningContentField: "reasoning_content"`) — Wafer normalizes reasoning
output to `reasoning_content` regardless of upstream provider.

Tests extended to cover the new entries plus the corrected Kimi pricing
and Qwen3.6 context/vision (5 cases, 47 expect() calls, all passing).
This commit is contained in:
bench-local
2026-05-27 20:53:34 -07:00
parent f6ca76728b
commit 1f38f8d99c
3 changed files with 126 additions and 13 deletions
+1 -1
View File
@@ -5,7 +5,7 @@
### Added
- Added `CheckCredentialsOptions.completionProbe` (and `completionTimeoutMs`) so `AuthStorage.checkCredentials` can additionally exercise each credential against the provider's chat-completion endpoint after refresh-on-expiry. Result lands on `CredentialHealthResult.completion` ({ok, reason?, modelId?, latencyMs?}) without disturbing the usage `ok` field. Public types: `CompletionProbe`, `CompletionProbeInput`, `CompletionProbeCredential`, `CredentialCompletionResult`. The probe is invoked even when no `UsageProvider` is registered for the row, and is skipped when OAuth refresh fails (the stale bytes would only mask the upstream failure).
- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Qwen3.5-397B-A17B, Kimi-K2.6, Qwen3.6-35B-A3B}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. GLM-family entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`).
- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Kimi-K2.6, Qwen3.5-397B-A17B, Qwen3.6-35B-A3B, qwen3.7-max, deepseek-v4-flash, deepseek-v4-pro}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. Reasoning-capable entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`).
### Changed
+97 -9
View File
@@ -70711,6 +70711,64 @@
}
},
"wafer-serverless": {
"deepseek-v4-flash": {
"id": "deepseek-v4-flash",
"name": "DeepSeek V4 Flash",
"api": "openai-completions",
"provider": "wafer-serverless",
"baseUrl": "https://pass.wafer.ai/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.14,
"output": 0.28,
"cacheRead": 0.01,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 65536,
"compat": {
"supportsDeveloperRole": false,
"thinkingFormat": "zai",
"reasoningContentField": "reasoning_content"
},
"thinking": {
"mode": "effort",
"minLevel": "minimal",
"maxLevel": "xhigh"
}
},
"deepseek-v4-pro": {
"id": "deepseek-v4-pro",
"name": "DeepSeek V4 Pro",
"api": "openai-completions",
"provider": "wafer-serverless",
"baseUrl": "https://pass.wafer.ai/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 1.74,
"output": 3.48,
"cacheRead": 0.02,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 65536,
"compat": {
"supportsDeveloperRole": false,
"thinkingFormat": "zai",
"reasoningContentField": "reasoning_content"
},
"thinking": {
"mode": "effort",
"minLevel": "minimal",
"maxLevel": "xhigh"
}
},
"GLM-5.1": {
"id": "GLM-5.1",
"name": "GLM-5.1",
@@ -70752,9 +70810,9 @@
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"input": 0.88,
"output": 3.84,
"cacheRead": 0.09,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -70800,19 +70858,49 @@
"provider": "wafer-serverless",
"baseUrl": "https://pass.wafer.ai/v1",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 0.15,
"output": 1,
"cacheRead": 0.02,
"cacheWrite": 0
},
"contextWindow": 256000,
"maxTokens": 65536,
"compat": {
"supportsDeveloperRole": false
}
},
"qwen3.7-max": {
"id": "qwen3.7-max",
"name": "Qwen3.7-Max",
"api": "openai-completions",
"provider": "wafer-serverless",
"baseUrl": "https://pass.wafer.ai/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"input": 5,
"output": 15,
"cacheRead": 0.5,
"cacheWrite": 0
},
"contextWindow": 32768,
"maxTokens": 32768,
"contextWindow": 256000,
"maxTokens": 65536,
"compat": {
"supportsDeveloperRole": false
"supportsDeveloperRole": false,
"thinkingFormat": "zai",
"reasoningContentField": "reasoning_content"
},
"thinking": {
"mode": "effort",
"minLevel": "minimal",
"maxLevel": "xhigh"
}
}
},
+28 -3
View File
@@ -85,7 +85,7 @@ describe("Wafer Pass provider", () => {
});
describe("Wafer Serverless provider", () => {
it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Kimi-K2.6, Qwen3.6)", () => {
it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Qwen3.6, Qwen3.7-Max, Kimi-K2.6, DeepSeek V4 Flash/Pro)", () => {
const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1");
expect(glm).toBeDefined();
expect(glm.provider).toBe("wafer-serverless");
@@ -99,15 +99,40 @@ describe("Wafer Serverless provider", () => {
const kimi = getBundledModel<"openai-completions">("wafer-serverless", "Kimi-K2.6");
expect(kimi).toBeDefined();
expect(kimi.contextWindow).toBe(262144);
// Kimi-K2.6 carries its real cents/M pricing (88 / 384 / 9 → 0.88 / 3.84 / 0.09).
expect(kimi.cost).toEqual({ input: 0.88, output: 3.84, cacheRead: 0.09, cacheWrite: 0 });
const qwen36 = getBundledModel<"openai-completions">("wafer-serverless", "Qwen3.6-35B-A3B");
expect(qwen36).toBeDefined();
// Serverless-only Qwen3.6 — context capped at 32k per docs.
expect(qwen36.contextWindow).toBe(32768);
// Qwen3.6 advertises 256k context and vision per the live /v1/models response.
expect(qwen36.contextWindow).toBe(256000);
expect(qwen36.input).toEqual(["text", "image"]);
const qwen37max = getBundledModel<"openai-completions">("wafer-serverless", "qwen3.7-max");
expect(qwen37max).toBeDefined();
// Wafer's canonical id is lowercase `qwen3.7-max` — must round-trip verbatim.
expect(qwen37max.id).toBe("qwen3.7-max");
expect(qwen37max.name).toBe("Qwen3.7-Max");
expect(qwen37max.reasoning).toBe(true);
expect(qwen37max.compat?.thinkingFormat).toBe("zai");
const dsFlash = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-flash");
expect(dsFlash).toBeDefined();
expect(dsFlash.contextWindow).toBe(1000000);
expect(dsFlash.reasoning).toBe(true);
expect(dsFlash.compat?.reasoningContentField).toBe("reasoning_content");
const dsPro = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-pro");
expect(dsPro).toBeDefined();
expect(dsPro.contextWindow).toBe(1000000);
expect(dsPro.reasoning).toBe(true);
});
it("does not expose Serverless-only ids on the Wafer Pass catalog", () => {
expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined();
expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined();
expect(getBundledModel("wafer-pass", "qwen3.7-max")).toBeUndefined();
expect(getBundledModel("wafer-pass", "deepseek-v4-flash")).toBeUndefined();
expect(getBundledModel("wafer-pass", "deepseek-v4-pro")).toBeUndefined();
});
});