diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index b1b641da7..cdf78c05b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,7 +5,7 @@ ### Added - Added `CheckCredentialsOptions.completionProbe` (and `completionTimeoutMs`) so `AuthStorage.checkCredentials` can additionally exercise each credential against the provider's chat-completion endpoint after refresh-on-expiry. Result lands on `CredentialHealthResult.completion` ({ok, reason?, modelId?, latencyMs?}) without disturbing the usage `ok` field. Public types: `CompletionProbe`, `CompletionProbeInput`, `CompletionProbeCredential`, `CredentialCompletionResult`. The probe is invoked even when no `UsageProvider` is registered for the row, and is skipped when OAuth refresh fails (the stale bytes would only mask the upstream failure). -- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Qwen3.5-397B-A17B, Kimi-K2.6, Qwen3.6-35B-A3B}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. GLM-family entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`). +- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Kimi-K2.6, Qwen3.5-397B-A17B, Qwen3.6-35B-A3B, qwen3.7-max, deepseek-v4-flash, deepseek-v4-pro}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. Reasoning-capable entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`). ### Changed diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index af8011a18..ff4585fe8 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -70711,6 +70711,64 @@ } }, "wafer-serverless": { + "deepseek-v4-flash": { + "id": "deepseek-v4-flash", + "name": "DeepSeek V4 Flash", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0.01, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false, + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content" + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "deepseek-v4-pro": { + "id": "deepseek-v4-pro", + "name": "DeepSeek V4 Pro", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.74, + "output": 3.48, + "cacheRead": 0.02, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false, + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content" + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "GLM-5.1": { "id": "GLM-5.1", "name": "GLM-5.1", @@ -70752,9 +70810,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.88, + "output": 3.84, + "cacheRead": 0.09, "cacheWrite": 0 }, "contextWindow": 262144, @@ -70800,19 +70858,49 @@ "provider": "wafer-serverless", "baseUrl": "https://pass.wafer.ai/v1", "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.15, + "output": 1, + "cacheRead": 0.02, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false + } + }, + "qwen3.7-max": { + "id": "qwen3.7-max", + "name": "Qwen3.7-Max", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 5, + "output": 15, + "cacheRead": 0.5, "cacheWrite": 0 }, - "contextWindow": 32768, - "maxTokens": 32768, + "contextWindow": 256000, + "maxTokens": 65536, "compat": { - "supportsDeveloperRole": false + "supportsDeveloperRole": false, + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content" + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" } } }, diff --git a/packages/ai/test/wafer.test.ts b/packages/ai/test/wafer.test.ts index c2b9ea2f8..2d82da5cc 100644 --- a/packages/ai/test/wafer.test.ts +++ b/packages/ai/test/wafer.test.ts @@ -85,7 +85,7 @@ describe("Wafer Pass provider", () => { }); describe("Wafer Serverless provider", () => { - it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Kimi-K2.6, Qwen3.6)", () => { + it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Qwen3.6, Qwen3.7-Max, Kimi-K2.6, DeepSeek V4 Flash/Pro)", () => { const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1"); expect(glm).toBeDefined(); expect(glm.provider).toBe("wafer-serverless"); @@ -99,15 +99,40 @@ describe("Wafer Serverless provider", () => { const kimi = getBundledModel<"openai-completions">("wafer-serverless", "Kimi-K2.6"); expect(kimi).toBeDefined(); expect(kimi.contextWindow).toBe(262144); + // Kimi-K2.6 carries its real cents/M pricing (88 / 384 / 9 → 0.88 / 3.84 / 0.09). + expect(kimi.cost).toEqual({ input: 0.88, output: 3.84, cacheRead: 0.09, cacheWrite: 0 }); const qwen36 = getBundledModel<"openai-completions">("wafer-serverless", "Qwen3.6-35B-A3B"); expect(qwen36).toBeDefined(); - // Serverless-only Qwen3.6 — context capped at 32k per docs. - expect(qwen36.contextWindow).toBe(32768); + // Qwen3.6 advertises 256k context and vision per the live /v1/models response. + expect(qwen36.contextWindow).toBe(256000); + expect(qwen36.input).toEqual(["text", "image"]); + + const qwen37max = getBundledModel<"openai-completions">("wafer-serverless", "qwen3.7-max"); + expect(qwen37max).toBeDefined(); + // Wafer's canonical id is lowercase `qwen3.7-max` — must round-trip verbatim. + expect(qwen37max.id).toBe("qwen3.7-max"); + expect(qwen37max.name).toBe("Qwen3.7-Max"); + expect(qwen37max.reasoning).toBe(true); + expect(qwen37max.compat?.thinkingFormat).toBe("zai"); + + const dsFlash = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-flash"); + expect(dsFlash).toBeDefined(); + expect(dsFlash.contextWindow).toBe(1000000); + expect(dsFlash.reasoning).toBe(true); + expect(dsFlash.compat?.reasoningContentField).toBe("reasoning_content"); + + const dsPro = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-pro"); + expect(dsPro).toBeDefined(); + expect(dsPro.contextWindow).toBe(1000000); + expect(dsPro.reasoning).toBe(true); }); it("does not expose Serverless-only ids on the Wafer Pass catalog", () => { expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined(); expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined(); + expect(getBundledModel("wafer-pass", "qwen3.7-max")).toBeUndefined(); + expect(getBundledModel("wafer-pass", "deepseek-v4-flash")).toBeUndefined(); + expect(getBundledModel("wafer-pass", "deepseek-v4-pro")).toBeUndefined(); }); });