From f6ca76728bbef4be709a633756ddf76fcccb1c82 Mon Sep 17 00:00:00 2001 From: bench-local Date: Wed, 27 May 2026 20:45:33 -0700 Subject: [PATCH] feat(ai): add Wafer Pass and Wafer Serverless providers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wafer (https://wafer.ai) exposes a single OpenAI-compatible endpoint (`https://pass.wafer.ai/v1`) for two SKUs whose entitlement differs server-side, so we model them as two parallel providers — mirroring the firepass/fireworks split so a user with both subscriptions can switch without re-pasting: - `wafer-pass` — flat-rate. `/v1/models` is filtered to entries whose `wafer.tier === "pass_included"`. - `wafer-serverless` — pay-as-you-go superset of Pass. Both issue `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate via `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` are wired through `getEnvApiKey`. Bundled catalog: - `wafer-pass`: GLM-5.1, Qwen3.5-397B-A17B. - `wafer-serverless`: GLM-5.1, Qwen3.5-397B-A17B, Kimi-K2.6, Qwen3.6-35B-A3B. Dynamic discovery via `/v1/models` overlays additional models at runtime and folds the `wafer` envelope (tier, capabilities, cents/M pricing) into the canonical `Model<"openai-completions">` shape. GLM-family entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`) so reasoning tokens land in the right field. Cents-per-million → dollars-per-million via /100. Tests (`packages/ai/test/wafer.test.ts`, 5 cases): bundled catalog contract for both providers and wire-id pass-through (case-sensitive, no rewrite — `GLM-5.1` must round-trip verbatim or upstream 404s). Optional `packages/ai/test/wafer.live.ts` exercises a real round-trip against `pass.wafer.ai` when `WAFER_PASS_API_KEY` is set. --- README.md | 4 +- docs/environment-variables.md | 2 + packages/ai/CHANGELOG.md | 1 + packages/ai/README.md | 2 + packages/ai/src/auth-storage.ts | 12 ++ packages/ai/src/models.json | 160 ++++++++++++++++++ .../ai/src/provider-models/descriptors.ts | 14 ++ .../ai/src/provider-models/openai-compat.ts | 130 ++++++++++++++ packages/ai/src/stream.ts | 2 + packages/ai/src/types.ts | 2 + packages/ai/src/utils/oauth/index.ts | 12 ++ packages/ai/src/utils/oauth/types.ts | 2 + packages/ai/src/utils/oauth/wafer.ts | 50 ++++++ packages/ai/test/wafer.live.ts | 93 ++++++++++ packages/ai/test/wafer.test.ts | 113 +++++++++++++ packages/coding-agent/src/cli/args.ts | 2 + 16 files changed, 599 insertions(+), 2 deletions(-) create mode 100644 packages/ai/src/utils/oauth/wafer.ts create mode 100644 packages/ai/test/wafer.live.ts create mode 100644 packages/ai/test/wafer.test.ts diff --git a/README.md b/README.md index 480b4715a..59a450a5f 100644 --- a/README.md +++ b/README.md @@ -251,13 +251,13 @@ Auth tags below: `oauth` signs in with your provider account, `plan` routes thro Direct APIs and gateways. Mix providers per role. -Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google Antigravity `oauth` · xAI · Mistral · Groq · Cerebras · Fireworks · Together · Hugging Face · NVIDIA · OpenRouter · Synthetic · Vercel AI Gateway · Cloudflare AI Gateway · Perplexity `oauth` +Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google Antigravity `oauth` · xAI · Mistral · Groq · Cerebras · Fireworks · Together · Hugging Face · NVIDIA · OpenRouter · Synthetic · Vercel AI Gateway · Cloudflare AI Gateway · Wafer Serverless · Perplexity `oauth` ### Coding plans Subscription-routed. `/login` attaches the session. -Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen +Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · Wafer Pass `plan` · OpenCode Go · OpenCode Zen ### Run it yourself diff --git a/docs/environment-variables.md b/docs/environment-variables.md index b25e7d81b..dd472006c 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -74,6 +74,8 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not | `DEEPSEEK_API_KEY` | DeepSeek auth | Using DeepSeek models | | | `KILO_API_KEY` | Kilo auth | Using Kilo models | | | `OLLAMA_CLOUD_API_KEY` | Ollama Cloud auth | Using `ollama-cloud` provider | | +| `WAFER_PASS_API_KEY` | Wafer Pass auth | Using `wafer-pass` provider | Flat-rate Wafer subscription; validated against `https://pass.wafer.ai/v1/models` | +| `WAFER_SERVERLESS_API_KEY` | Wafer Serverless auth | Using `wafer-serverless` provider | Pay-as-you-go Wafer SKU; validated against `https://pass.wafer.ai/v1/models` | | `GITLAB_TOKEN` | GitLab Duo auth | Using `gitlab-duo` provider | | ### GitHub/Copilot token chains diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f3301a257..b1b641da7 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,6 +5,7 @@ ### Added - Added `CheckCredentialsOptions.completionProbe` (and `completionTimeoutMs`) so `AuthStorage.checkCredentials` can additionally exercise each credential against the provider's chat-completion endpoint after refresh-on-expiry. Result lands on `CredentialHealthResult.completion` ({ok, reason?, modelId?, latencyMs?}) without disturbing the usage `ok` field. Public types: `CompletionProbe`, `CompletionProbeInput`, `CompletionProbeCredential`, `CredentialCompletionResult`. The probe is invoked even when no `UsageProvider` is registered for the row, and is skipped when OAuth refresh fails (the stale bytes would only mask the upstream failure). +- Added Wafer Pass and Wafer Serverless providers (`wafer-pass`, `wafer-serverless`). OpenAI-compatible (`https://pass.wafer.ai/v1`), bearer auth, `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate the key against `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` environment variables wired into `getEnvApiKey`. Bundled catalog seeds `wafer-pass/{GLM-5.1, Qwen3.5-397B-A17B}` and `wafer-serverless/{GLM-5.1, Qwen3.5-397B-A17B, Kimi-K2.6, Qwen3.6-35B-A3B}`; dynamic discovery via `/v1/models` overlays additional models at runtime. Pass-tier discovery filters `wafer.tier === "pass_included"`. GLM-family entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`). ### Changed diff --git a/packages/ai/README.md b/packages/ai/README.md index 99be71531..54eeb9876 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -62,6 +62,8 @@ Unified LLM API with automatic model discovery, provider configuration, token an - **Hugging Face Inference** - **xAI** - **Venice** (requires `VENICE_API_KEY`) +- **Wafer Pass** (requires `WAFER_PASS_API_KEY`; flat-rate subscription, includes GLM-5.1 and Qwen3.5-397B-A17B) +- **Wafer Serverless** (requires `WAFER_SERVERLESS_API_KEY`; pay-as-you-go) - **OpenRouter** - **Kilo Gateway** (supports OAuth `/login kilo` or `KILO_API_KEY`) - **LiteLLM** (requires `LITELLM_API_KEY`) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 7ae39f06d..92bdc4e0c 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -1619,6 +1619,18 @@ export class AuthStorage { await saveApiKeyCredential(apiKey); return; } + case "wafer-pass": { + const { loginWaferPass } = await import("./utils/oauth/wafer"); + const apiKey = await loginWaferPass(ctrl); + await saveApiKeyCredential(apiKey); + return; + } + case "wafer-serverless": { + const { loginWaferServerless } = await import("./utils/oauth/wafer"); + const apiKey = await loginWaferServerless(ctrl); + await saveApiKeyCredential(apiKey); + return; + } case "zai": { const { loginZai } = await import("./utils/oauth/zai"); const apiKey = await loginZai(ctrl); diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index 995a5ace8..af8011a18 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -70656,6 +70656,166 @@ } } }, + "wafer-pass": { + "GLM-5.1": { + "id": "GLM-5.1", + "name": "GLM-5.1", + "api": "openai-completions", + "provider": "wafer-pass", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.2, + "output": 3.6, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 202752, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false, + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content" + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "Qwen3.5-397B-A17B": { + "id": "Qwen3.5-397B-A17B", + "name": "Qwen3.5-397B-A17B", + "api": "openai-completions", + "provider": "wafer-pass", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.48, + "output": 2.88, + "cacheRead": 0.05, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false + } + } + }, + "wafer-serverless": { + "GLM-5.1": { + "id": "GLM-5.1", + "name": "GLM-5.1", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.2, + "output": 3.6, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 202752, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false, + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content" + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "Kimi-K2.6": { + "id": "Kimi-K2.6", + "name": "Kimi-K2.6", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false, + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content" + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "Qwen3.5-397B-A17B": { + "id": "Qwen3.5-397B-A17B", + "name": "Qwen3.5-397B-A17B", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.48, + "output": 2.88, + "cacheRead": 0.05, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "compat": { + "supportsDeveloperRole": false + } + }, + "Qwen3.6-35B-A3B": { + "id": "Qwen3.6-35B-A3B", + "name": "Qwen3.6-35B-A3B", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32768, + "maxTokens": 32768, + "compat": { + "supportsDeveloperRole": false + } + } + }, "xai": { "grok-2": { "id": "grok-2", diff --git a/packages/ai/src/provider-models/descriptors.ts b/packages/ai/src/provider-models/descriptors.ts index ceb972bbc..e6a80d648 100644 --- a/packages/ai/src/provider-models/descriptors.ts +++ b/packages/ai/src/provider-models/descriptors.ts @@ -39,6 +39,8 @@ import { veniceModelManagerOptions, vercelAiGatewayModelManagerOptions, vllmModelManagerOptions, + waferPassModelManagerOptions, + waferServerlessModelManagerOptions, xaiModelManagerOptions, xaiOAuthModelManagerOptions, xiaomiModelManagerOptions, @@ -169,6 +171,18 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [ // models.json would permanently drop the model from the catalog with no // automated mechanism to restore it. descriptor("firepass", "kimi-k2.6-turbo", config => firepassModelManagerOptions(config)), + catalogDescriptor( + "wafer-pass", + "GLM-5.1", + config => waferPassModelManagerOptions(config), + catalog("Wafer Pass", ["WAFER_PASS_API_KEY"], { oauthProvider: "wafer-pass" }), + ), + catalogDescriptor( + "wafer-serverless", + "GLM-5.1", + config => waferServerlessModelManagerOptions(config), + catalog("Wafer Serverless", ["WAFER_SERVERLESS_API_KEY"], { oauthProvider: "wafer-serverless" }), + ), descriptor("xai", "grok-4-fast-non-reasoning", config => xaiModelManagerOptions(config)), catalogDescriptor( "xai-oauth", diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index 1728225ee..aa61b76d6 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -1037,6 +1037,136 @@ export function firepassModelManagerOptions( }; } +// --------------------------------------------------------------------------- +// 7.7 Wafer (Pass + Serverless) +// --------------------------------------------------------------------------- + +export interface WaferModelManagerConfig { + apiKey?: string; + baseUrl?: string; +} + +const WAFER_DEFAULT_BASE_URL = "https://pass.wafer.ai/v1"; +const WAFER_MAX_TOKENS_CAP = 65536; + +/** + * Shared mapper for Wafer's `/v1/models` records. + * + * Wafer wraps each entry with a `wafer` envelope describing tier, capabilities, + * and cents-per-million pricing. The mapper folds that metadata into the + * canonical `Model<"openai-completions">` shape and applies zai-family thinking + * compat when the entry advertises reasoning support (GLM-family on the Pass + * SKU). Cents-per-million → dollars-per-million via /100. + */ +interface WaferRecord { + context_length?: unknown; + tier?: unknown; + capabilities?: { vision?: unknown; reasoning?: unknown; tools?: unknown }; + pricing?: { + input_cents_per_million?: unknown; + output_cents_per_million?: unknown; + cache_read_cents_per_million?: unknown; + }; + display_name?: unknown; +} + +function readWaferRecord(entry: OpenAICompatibleModelRecord): WaferRecord | undefined { + const raw = (entry as { wafer?: unknown }).wafer; + return raw && typeof raw === "object" ? (raw as WaferRecord) : undefined; +} + +function mapWaferModel( + providerId: "wafer-pass" | "wafer-serverless", + baseUrl: string, + entry: OpenAICompatibleModelRecord, + defaults: Model<"openai-completions">, +): Model<"openai-completions"> { + const wafer = readWaferRecord(entry); + const capabilities = wafer?.capabilities ?? {}; + const reasoning = capabilities.reasoning === true; + const vision = capabilities.vision === true; + const contextWindow = toPositiveNumber( + wafer?.context_length, + toPositiveNumber((entry as { max_model_len?: unknown }).max_model_len, defaults.contextWindow), + ); + const maxTokens = Math.min(contextWindow, WAFER_MAX_TOKENS_CAP); + const pricing = wafer?.pricing ?? {}; + // Wafer publishes cents-per-million; OMP catalog stores dollars-per-million. + const cost = { + input: toPositiveNumber(pricing.input_cents_per_million, 0) / 100, + output: toPositiveNumber(pricing.output_cents_per_million, 0) / 100, + cacheRead: toPositiveNumber(pricing.cache_read_cents_per_million, 0) / 100, + cacheWrite: 0, + }; + const name = toModelName(wafer?.display_name, defaults.name); + const base: Model<"openai-completions"> = { + ...defaults, + id: defaults.id, + name, + api: "openai-completions", + provider: providerId, + baseUrl, + reasoning, + input: vision ? (["text", "image"] as const) : ["text"], + cost, + contextWindow, + maxTokens, + }; + if (reasoning) { + return { + ...base, + compat: { + thinkingFormat: "zai", + reasoningContentField: "reasoning_content", + supportsDeveloperRole: false, + }, + }; + } + return { + ...base, + compat: { supportsDeveloperRole: false }, + }; +} + +function createWaferOptions( + providerId: "wafer-pass" | "wafer-serverless", + config: WaferModelManagerConfig | undefined, +): ModelManagerOptions<"openai-completions"> { + const apiKey = config?.apiKey; + const baseUrl = config?.baseUrl ?? WAFER_DEFAULT_BASE_URL; + const passOnly = providerId === "wafer-pass"; + return { + providerId, + ...(apiKey && { + fetchDynamicModels: () => + fetchOpenAICompatibleModels({ + api: "openai-completions", + provider: providerId, + baseUrl, + apiKey, + filterModel: entry => { + if (!passOnly) return true; + const wafer = readWaferRecord(entry); + return wafer?.tier === "pass_included"; + }, + mapModel: (entry, defaults) => mapWaferModel(providerId, baseUrl, entry, defaults), + }), + }), + }; +} + +export function waferPassModelManagerOptions( + config?: WaferModelManagerConfig, +): ModelManagerOptions<"openai-completions"> { + return createWaferOptions("wafer-pass", config); +} + +export function waferServerlessModelManagerOptions( + config?: WaferModelManagerConfig, +): ModelManagerOptions<"openai-completions"> { + return createWaferOptions("wafer-serverless", config); +} + // --------------------------------------------------------------------------- // 7. Mistral // --------------------------------------------------------------------------- diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 0530a1404..161f8bb70 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -183,6 +183,8 @@ const serviceProviderMap: Record = { "xai-oauth": () => $pickenv("XAI_OAUTH_TOKEN", "XAI_API_KEY"), fireworks: "FIREWORKS_API_KEY", firepass: "FIREPASS_API_KEY", + "wafer-pass": "WAFER_PASS_API_KEY", + "wafer-serverless": "WAFER_SERVERLESS_API_KEY", openrouter: "OPENROUTER_API_KEY", kilo: "KILO_API_KEY", "vercel-ai-gateway": "AI_GATEWAY_API_KEY", diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index b9b48f330..7cb2c2339 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -142,6 +142,8 @@ export type KnownProvider = | "venice" | "vllm" | "xiaomi" + | "wafer-pass" + | "wafer-serverless" | "zenmux" | "lm-studio"; export type Provider = KnownProvider | string; diff --git a/packages/ai/src/utils/oauth/index.ts b/packages/ai/src/utils/oauth/index.ts index ad79a65a7..0d274b12b 100644 --- a/packages/ai/src/utils/oauth/index.ts +++ b/packages/ai/src/utils/oauth/index.ts @@ -235,6 +235,16 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [ name: "xAI Grok OAuth (SuperGrok Subscription)", available: true, }, + { + id: "wafer-pass", + name: "Wafer Pass (flat-rate subscription)", + available: true, + }, + { + id: "wafer-serverless", + name: "Wafer Serverless (pay-as-you-go)", + available: true, + }, ]; const customOAuthProviders = new Map(); @@ -359,6 +369,8 @@ export async function refreshOAuthToken( case "cloudflare-ai-gateway": case "vercel-ai-gateway": case "qwen-portal": + case "wafer-pass": + case "wafer-serverless": case "zenmux": case "vllm": // API keys / static bearer tokens don't expire, return as-is diff --git a/packages/ai/src/utils/oauth/types.ts b/packages/ai/src/utils/oauth/types.ts index 0dc045ec3..70710e6bf 100644 --- a/packages/ai/src/utils/oauth/types.ts +++ b/packages/ai/src/utils/oauth/types.ts @@ -48,6 +48,8 @@ export type OAuthProvider = | "together" | "venice" | "vercel-ai-gateway" + | "wafer-pass" + | "wafer-serverless" | "vllm" | "xai-oauth" | "xiaomi" diff --git a/packages/ai/src/utils/oauth/wafer.ts b/packages/ai/src/utils/oauth/wafer.ts new file mode 100644 index 000000000..490ceada6 --- /dev/null +++ b/packages/ai/src/utils/oauth/wafer.ts @@ -0,0 +1,50 @@ +/** + * Wafer login flows. + * + * Wafer (https://wafer.ai) exposes a single OpenAI-compatible base URL + * (`https://pass.wafer.ai/v1`) for two SKUs: + * + * - **Wafer Pass** — flat-rate subscription. The key authorizes models whose + * catalog entries carry `wafer.tier = "pass_included"`. + * - **Wafer Serverless** — pay-as-you-go. Superset of Pass; the same `/v1/models` + * endpoint returns the full per-account model list. + * + * Both SKUs issue `wfr_…` keys. The key prefix alone does not distinguish + * tiers — the entitlement is per-account on the server side — so we expose + * two parallel logins / env vars (`WAFER_PASS_API_KEY`, `WAFER_SERVERLESS_API_KEY`) + * mirroring the firepass/fireworks split, letting users with both + * subscriptions switch between them without re-pasting. + * + * Validation uses the shared `/v1/models` endpoint, which works for both + * tiers and is cheap (no token spend). + */ +import { createApiKeyLogin } from "./api-key-login"; + +const WAFER_AUTH_URL = "https://wafer.ai/dashboard"; +const WAFER_MODELS_URL = "https://pass.wafer.ai/v1/models"; + +export const loginWaferPass = createApiKeyLogin({ + providerLabel: "Wafer Pass", + authUrl: WAFER_AUTH_URL, + instructions: "Create or copy your Wafer Pass API key from the Wafer dashboard", + promptMessage: "Paste your Wafer Pass API key", + placeholder: "wfr_...", + validation: { + kind: "models-endpoint", + provider: "Wafer Pass", + modelsUrl: WAFER_MODELS_URL, + }, +}); + +export const loginWaferServerless = createApiKeyLogin({ + providerLabel: "Wafer Serverless", + authUrl: WAFER_AUTH_URL, + instructions: "Create or copy your Wafer Serverless API key from the Wafer dashboard", + promptMessage: "Paste your Wafer Serverless API key", + placeholder: "wfr_...", + validation: { + kind: "models-endpoint", + provider: "Wafer Serverless", + modelsUrl: WAFER_MODELS_URL, + }, +}); diff --git a/packages/ai/test/wafer.live.ts b/packages/ai/test/wafer.live.ts new file mode 100644 index 000000000..f8dcd8caa --- /dev/null +++ b/packages/ai/test/wafer.live.ts @@ -0,0 +1,93 @@ +/** + * Live Wafer Pass smoke. NOT part of the bun test suite — run manually: + * WAFER_PASS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts + * + * Validates that the bundled `wafer-pass/GLM-5.1` entry round-trips a real + * streaming chat completion against `https://pass.wafer.ai/v1`, with the wire + * `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a non-empty + * assistant text returned. + */ +import { getBundledModel } from "../src/models"; +import { streamOpenAICompletions } from "../src/providers/openai-completions"; +import type { Context, Model } from "../src/types"; + +const apiKey = process.env.WAFER_PASS_API_KEY ?? process.env.WAFER_SERVERLESS_API_KEY; +if (!apiKey) { + console.error("WAFER_PASS_API_KEY (or WAFER_SERVERLESS_API_KEY) env var is required"); + process.exit(2); +} + +const providerId = process.env.WAFER_PASS_API_KEY ? "wafer-pass" : "wafer-serverless"; +const model = getBundledModel<"openai-completions">(providerId, "GLM-5.1"); +console.log(`Model: ${model.provider}/${model.id} -> ${model.baseUrl}`); +console.log(`compat.thinkingFormat: ${model.compat?.thinkingFormat ?? "(none)"}`); + +interface CapturedRequest { + url: string; + body: string | null; +} + +const originalFetch = global.fetch; +const captured: { value: CapturedRequest | null } = { value: null }; +type FetchInput = Parameters[0]; +global.fetch = (async (input: FetchInput, init?: RequestInit) => { + const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + captured.value = { url, body: typeof init?.body === "string" ? init.body : null }; + return originalFetch(input as Parameters[0], init); +}) as typeof global.fetch; + +const context: Context = { + systemPrompt: ["Reply with exactly two words."], + messages: [{ role: "user", content: "Say hi.", timestamp: Date.now() }], +}; + +const stream = streamOpenAICompletions(model as Model<"openai-completions">, context, { apiKey }); +let text = ""; +let stopReason: string | undefined; +let cost = 0; +let firstError: unknown; +let inputTokens = 0; +let outputTokens = 0; +for await (const ev of stream) { + if (ev.type === "text_delta") text += ev.delta; + else if (ev.type === "done") { + stopReason = ev.reason; + const usage = ev.message.usage; + cost = usage?.cost?.total ?? 0; + inputTokens = usage?.input ?? 0; + outputTokens = usage?.output ?? 0; + } else if (ev.type === "error") { + firstError = ev.error.errorMessage ?? ev.error; + stopReason = ev.reason; + } +} + +const snapshot = (captured as { value: CapturedRequest | null }).value; +const parsedBody = snapshot?.body ? (JSON.parse(snapshot.body) as { model?: unknown }) : null; +console.log("wire url:", snapshot?.url); +console.log("wire model:", parsedBody?.model); +console.log("text:", JSON.stringify(text.slice(0, 200))); +console.log("stopReason:", stopReason); +console.log("usage:", { input: inputTokens, output: outputTokens, costUSD: cost }); + +if (firstError) { + console.error("\nLIVE FAIL — Wafer rejected the request:", firstError); + process.exit(1); +} +if (snapshot?.url !== "https://pass.wafer.ai/v1/chat/completions") { + console.error("\nLIVE FAIL — wire url was not the documented endpoint"); + process.exit(1); +} +if (parsedBody?.model !== "GLM-5.1") { + console.error("\nLIVE FAIL — wire model id was not preserved verbatim:", parsedBody?.model); + process.exit(1); +} +if (text.trim().length === 0) { + console.error("\nLIVE FAIL — assistant returned empty text"); + process.exit(1); +} + +console.log( + `\nLIVE OK — Wafer ${providerId} round-trip: GLM-5.1 endpoint preserved, ` + + `${inputTokens}→${outputTokens} tokens, stopReason=${stopReason}.`, +); diff --git a/packages/ai/test/wafer.test.ts b/packages/ai/test/wafer.test.ts new file mode 100644 index 000000000..c2b9ea2f8 --- /dev/null +++ b/packages/ai/test/wafer.test.ts @@ -0,0 +1,113 @@ +/** + * Wafer Pass + Wafer Serverless provider wiring. + * + * Wafer exposes a single OpenAI-compatible base URL (`https://pass.wafer.ai/v1`) + * for two SKUs whose entitlement differs server-side: + * - `wafer-pass` (flat-rate) + * - `wafer-serverless` (pay-as-you-go) + * + * Both providers route through `openai-completions` and the catalog id matches + * the wire id (no rewrite). These tests defend the bundled catalog contract and + * the case-sensitive id pass-through against the wire. + */ +import { afterEach, describe, expect, it } from "bun:test"; +import { getBundledModel } from "../src/models"; +import { streamOpenAICompletions } from "../src/providers/openai-completions"; +import type { Context, Model } from "../src/types"; + +const originalFetch = global.fetch; + +afterEach(() => { + global.fetch = originalFetch; +}); + +function sseResponse(events: unknown[]): Response { + const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`; + return new Response(payload, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); +} + +describe("Wafer Pass provider", () => { + it("ships a bundled GLM-5.1 entry with zai-family thinking compat", () => { + const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1"); + expect(model).toBeDefined(); + expect(model.id).toBe("GLM-5.1"); + expect(model.provider).toBe("wafer-pass"); + expect(model.api).toBe("openai-completions"); + expect(model.baseUrl).toBe("https://pass.wafer.ai/v1"); + expect(model.reasoning).toBe(true); + expect(model.input).toEqual(["text"]); + expect(model.compat?.thinkingFormat).toBe("zai"); + expect(model.compat?.reasoningContentField).toBe("reasoning_content"); + expect(model.compat?.supportsDeveloperRole).toBe(false); + }); + + it("ships a bundled Qwen3.5-397B-A17B entry with vision input and no reasoning", () => { + const model = getBundledModel<"openai-completions">("wafer-pass", "Qwen3.5-397B-A17B"); + expect(model).toBeDefined(); + expect(model.id).toBe("Qwen3.5-397B-A17B"); + expect(model.provider).toBe("wafer-pass"); + expect(model.reasoning).toBe(false); + expect(model.input).toEqual(["text", "image"]); + }); + + it("preserves the catalog id verbatim on the wire (no rewrite, case-sensitive)", async () => { + const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1"); + const captured: { url: string | null; body: string | null } = { url: null, body: null }; + global.fetch = (async (input: unknown, init?: RequestInit) => { + captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : String(input); + captured.body = typeof init?.body === "string" ? init.body : null; + return sseResponse(["[DONE]"]); + }) as typeof global.fetch; + + const context: Context = { + systemPrompt: ["t"], + messages: [{ role: "user", content: "hi", timestamp: Date.now() }], + }; + const stream = streamOpenAICompletions(model as Model<"openai-completions">, context, { + apiKey: "wfr_test", + }); + for await (const _event of stream) { + /* drain */ + } + + expect(captured.url).toBe("https://pass.wafer.ai/v1/chat/completions"); + expect(captured.body).not.toBeNull(); + const parsed = JSON.parse(captured.body ?? "{}") as { model?: unknown }; + // Wafer's docs note model names are case-insensitive on input, but the + // canonical id has mixed case; we must round-trip it unchanged so users + // who pin `GLM-5.1` don't end up with usage rows under `glm-5.1` or + // hitting the upstream 404 path. + expect(parsed.model).toBe("GLM-5.1"); + }); +}); + +describe("Wafer Serverless provider", () => { + it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Kimi-K2.6, Qwen3.6)", () => { + const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1"); + expect(glm).toBeDefined(); + expect(glm.provider).toBe("wafer-serverless"); + expect(glm.baseUrl).toBe("https://pass.wafer.ai/v1"); + expect(glm.compat?.thinkingFormat).toBe("zai"); + + const qwen35 = getBundledModel<"openai-completions">("wafer-serverless", "Qwen3.5-397B-A17B"); + expect(qwen35).toBeDefined(); + expect(qwen35.provider).toBe("wafer-serverless"); + + const kimi = getBundledModel<"openai-completions">("wafer-serverless", "Kimi-K2.6"); + expect(kimi).toBeDefined(); + expect(kimi.contextWindow).toBe(262144); + + const qwen36 = getBundledModel<"openai-completions">("wafer-serverless", "Qwen3.6-35B-A3B"); + expect(qwen36).toBeDefined(); + // Serverless-only Qwen3.6 — context capped at 32k per docs. + expect(qwen36.contextWindow).toBe(32768); + }); + + it("does not expose Serverless-only ids on the Wafer Pass catalog", () => { + expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined(); + expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined(); + }); +}); diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 6aaf5941d..50454cea0 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -247,6 +247,8 @@ export function getExtraHelpText(): string { OPENCODE_API_KEY - OpenCode Zen/OpenCode Go models CURSOR_ACCESS_TOKEN - Cursor AI models AI_GATEWAY_API_KEY - Vercel AI Gateway + WAFER_PASS_API_KEY - Wafer Pass (flat-rate subscription; GLM-5.1, Qwen3.5) + WAFER_SERVERLESS_API_KEY - Wafer Serverless (pay-as-you-go) ${chalk.dim("# Cloud Providers")} AWS_PROFILE - AWS Bedrock (or AWS_ACCESS_KEY_ID + AWS_SECRET_ACCESS_KEY)