diff --git a/README.md b/README.md index f4fa65676..8c000fa56 100644 --- a/README.md +++ b/README.md @@ -292,7 +292,7 @@ Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google A Subscription-routed. `/login` attaches the session. -Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · Wafer Pass `plan` · OpenCode Go · OpenCode Zen +Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen ### Run it yourself diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 20ed8c00f..dd485c7d6 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -82,7 +82,6 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not | `DEEPSEEK_API_KEY` | DeepSeek auth | Using DeepSeek models | | | `KILO_API_KEY` | Kilo auth | Using Kilo models | | | `OLLAMA_CLOUD_API_KEY` | Ollama Cloud auth | Using `ollama-cloud` provider | | -| `WAFER_PASS_API_KEY` | Wafer Pass auth | Using `wafer-pass` provider | Flat-rate Wafer subscription; validated against `https://pass.wafer.ai/v1/models` | | `WAFER_SERVERLESS_API_KEY` | Wafer Serverless auth | Using `wafer-serverless` provider | Pay-as-you-go Wafer SKU; validated against `https://pass.wafer.ai/v1/models` | | `GITLAB_TOKEN` | GitLab Duo auth | Using `gitlab-duo` provider | | diff --git a/docs/providers.md b/docs/providers.md index abfea6ab4..c5f65e009 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -123,7 +123,6 @@ Each provider has one or more environment variables that supply a key when no st | `gitlab-duo` | `GITLAB_TOKEN` | | `opencode-zen`, `opencode-go` | `OPENCODE_API_KEY` | | `firepass` | `FIREPASS_API_KEY` | -| `wafer-pass` | `WAFER_PASS_API_KEY` | | `wafer-serverless` | `WAFER_SERVERLESS_API_KEY` | | `xiaomi` | `XIAOMI_API_KEY` | | `ollama-cloud` | `OLLAMA_CLOUD_API_KEY` | @@ -131,7 +130,7 @@ Each provider has one or more environment variables that supply a key when no st | `lm-studio` | `LM_STUDIO_API_KEY` (optional; keyless by default) | | `llama.cpp` | `LLAMA_CPP_API_KEY` (only when the server requires auth) | -OAuth-backed providers such as `anthropic`, `github-copilot`, `cursor`, `ollama-cloud`, `qwen-portal`, `kimi-code`, `xai-oauth`, `wafer-pass`, `wafer-serverless`, `google-gemini-cli`, and `google-antigravity` are normally reached through `/login` rather than an environment variable. See [Environment variables](./environment-variables.md) for search-tool and configuration variables not listed here. +OAuth-backed providers such as `anthropic`, `github-copilot`, `cursor`, `ollama-cloud`, `qwen-portal`, `kimi-code`, `xai-oauth`, `wafer-serverless`, `google-gemini-cli`, and `google-antigravity` are normally reached through `/login` rather than an environment variable. See [Environment variables](./environment-variables.md) for search-tool and configuration variables not listed here. ### `.env` discovery and precedence diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f78603bfe..30ea9f884 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -10,6 +10,10 @@ - Added `llama.cpp` to the interactive `/login` provider list, accepting an optional API key while defaulting to local no-auth mode. +### Removed + +- Removed Wafer Pass (`wafer-pass`) login support; Wafer Serverless remains available as `wafer-serverless`. + ## [16.1.8] - 2026-06-20 ### Changed diff --git a/packages/ai/README.md b/packages/ai/README.md index f485d9621..5621aefe1 100644 --- a/packages/ai/README.md +++ b/packages/ai/README.md @@ -62,7 +62,6 @@ Unified LLM API with automatic model discovery, provider configuration, token an - **Hugging Face Inference** - **xAI** - **Venice** (requires `VENICE_API_KEY`) -- **Wafer Pass** (requires `WAFER_PASS_API_KEY`; flat-rate subscription, includes GLM-5.1 and Qwen3.5-397B-A17B) - **Wafer Serverless** (requires `WAFER_SERVERLESS_API_KEY`; pay-as-you-go) - **OpenRouter** - **Kilo Gateway** (supports OAuth `/login kilo` or `KILO_API_KEY`) diff --git a/packages/ai/src/registry/oauth/wafer.ts b/packages/ai/src/registry/oauth/wafer.ts index d9d74a557..f1d33c9d9 100644 --- a/packages/ai/src/registry/oauth/wafer.ts +++ b/packages/ai/src/registry/oauth/wafer.ts @@ -1,41 +1,15 @@ /** - * Wafer login flows. + * Wafer Serverless login flow. * - * Wafer (https://wafer.ai) exposes a single OpenAI-compatible base URL - * (`https://pass.wafer.ai/v1`) for two SKUs: - * - * - **Wafer Pass** — flat-rate subscription. The key authorizes models whose - * catalog entries carry `wafer.tier = "pass_included"`. - * - **Wafer Serverless** — pay-as-you-go. Superset of Pass; the same `/v1/models` - * endpoint returns the full per-account model list. - * - * Both SKUs issue `wfr_…` keys. The key prefix alone does not distinguish - * tiers — the entitlement is per-account on the server side — so we expose - * two parallel logins / env vars (`WAFER_PASS_API_KEY`, `WAFER_SERVERLESS_API_KEY`) - * mirroring the firepass/fireworks split, letting users with both - * subscriptions switch between them without re-pasting. - * - * Validation uses the shared `/v1/models` endpoint, which works for both - * tiers and is cheap (no token spend). + * Wafer (https://wafer.ai) exposes a pay-as-you-go OpenAI-compatible SKU at + * `https://pass.wafer.ai/v1`. Keys use the `wfr_…` prefix and are validated + * against `/v1/models`, which is cheap (no token spend). */ import { createApiKeyLogin } from "../api-key-login"; const WAFER_AUTH_URL = "https://wafer.ai/dashboard"; const WAFER_MODELS_URL = "https://pass.wafer.ai/v1/models"; -export const loginWaferPass = createApiKeyLogin({ - providerLabel: "Wafer Pass", - authUrl: WAFER_AUTH_URL, - instructions: "Create or copy your Wafer Pass API key from the Wafer dashboard", - promptMessage: "Paste your Wafer Pass API key", - placeholder: "wfr_...", - validation: { - kind: "models-endpoint", - provider: "Wafer Pass", - modelsUrl: WAFER_MODELS_URL, - }, -}); - export const loginWaferServerless = createApiKeyLogin({ providerLabel: "Wafer Serverless", authUrl: WAFER_AUTH_URL, diff --git a/packages/ai/src/registry/registry.ts b/packages/ai/src/registry/registry.ts index 10d6e1d8e..078e2a551 100644 --- a/packages/ai/src/registry/registry.ts +++ b/packages/ai/src/registry/registry.ts @@ -51,7 +51,6 @@ import { umansProvider } from "./umans"; import { veniceProvider } from "./venice"; import { vercelAiGatewayProvider } from "./vercel-ai-gateway"; import { vllmProvider } from "./vllm"; -import { waferPassProvider } from "./wafer-pass"; import { waferServerlessProvider } from "./wafer-serverless"; import { xaiProvider } from "./xai"; import { xaiOauthProvider } from "./xai-oauth"; @@ -96,7 +95,6 @@ const ALL = [ xiaomiTokenPlanAmsProvider, xiaomiTokenPlanCnProvider, firepassProvider, - waferPassProvider, deepseekProvider, moonshotProvider, cerebrasProvider, diff --git a/packages/ai/src/registry/wafer-pass.ts b/packages/ai/src/registry/wafer-pass.ts deleted file mode 100644 index abbd0c3bb..000000000 --- a/packages/ai/src/registry/wafer-pass.ts +++ /dev/null @@ -1,12 +0,0 @@ -import type { OAuthLoginCallbacks } from "./oauth/types"; -import type { ProviderDefinition } from "./types"; - -export const waferPassProvider = { - id: "wafer-pass", - name: "Wafer Pass (flat-rate subscription)", - login: async (cb: OAuthLoginCallbacks) => { - // Lazy import: keep heavy OAuth flow modules out of the eager registry graph. - const { loginWaferPass } = await import("./oauth/wafer"); - return loginWaferPass(cb); - }, -} as const satisfies ProviderDefinition; diff --git a/packages/ai/test/wafer.live.ts b/packages/ai/test/wafer.live.ts index 949108903..dc83aba61 100644 --- a/packages/ai/test/wafer.live.ts +++ b/packages/ai/test/wafer.live.ts @@ -1,25 +1,24 @@ /** - * Live Wafer Pass smoke. NOT part of the bun test suite — run manually: - * WAFER_PASS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts + * Live Wafer Serverless smoke. NOT part of the bun test suite — run manually: + * WAFER_SERVERLESS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts * - * Validates that the bundled `wafer-pass/GLM-5.1` entry round-trips a real - * streaming chat completion against `https://pass.wafer.ai/v1`, with the wire - * `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a non-empty - * assistant text returned. + * Validates that the bundled `wafer-serverless/GLM-5.1` entry round-trips a + * real streaming chat completion against `https://pass.wafer.ai/v1`, with the + * wire `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a + * non-empty assistant text returned. */ import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; -const apiKey = process.env.WAFER_PASS_API_KEY ?? process.env.WAFER_SERVERLESS_API_KEY; +const apiKey = process.env.WAFER_SERVERLESS_API_KEY; if (!apiKey) { - console.error("WAFER_PASS_API_KEY (or WAFER_SERVERLESS_API_KEY) env var is required"); + console.error("WAFER_SERVERLESS_API_KEY env var is required"); process.exit(2); } -const providerId = process.env.WAFER_PASS_API_KEY ? "wafer-pass" : "wafer-serverless"; -const model = getBundledModel<"openai-completions">(providerId, "GLM-5.1"); +const model = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1"); console.log(`Model: ${model.provider}/${model.id} -> ${model.baseUrl}`); console.log(`compat.thinkingFormat: ${model.compat?.thinkingFormat ?? "(none)"}`); @@ -89,6 +88,6 @@ if (text.trim().length === 0) { } console.log( - `\nLIVE OK — Wafer ${providerId} round-trip: GLM-5.1 endpoint preserved, ` + + `\nLIVE OK — Wafer Serverless round-trip: GLM-5.1 endpoint preserved, ` + `${inputTokens}→${outputTokens} tokens, stopReason=${stopReason}.`, ); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index fc4cf28f6..f92b60a6e 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -6,6 +6,10 @@ - Fixed the `moonshot` provider with no path to the Kimi China API: model discovery now honors a `MOONSHOT_BASE_URL` override (redirecting to `api.moonshot.cn`), and `KIMI_API_KEY` resolves as a fallback for `MOONSHOT_API_KEY`. ([#2883](https://github.com/can1357/oh-my-pi/issues/2883)) +### Removed + +- Removed bundled Wafer Pass (`wafer-pass`) catalog entries and generation support; Wafer Serverless remains available as `wafer-serverless`. + ## [16.1.8] - 2026-06-20 ### Fixed diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 756e40772..5967c67d5 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -61,6 +61,7 @@ const packageRoot = path.join(import.meta.dir, ".."); * and never written to models.json. */ const DISCOVERY_ONLY_PROVIDERS = new Set(["ollama", "vllm", "lm-studio", "litellm"]); +const RETIRED_PROVIDERS = new Set(["wafer-pass"]); async function resolveProviderApiKey(providerId: string, catalog: CatalogDiscoveryConfig): Promise { for (const envVar of catalog.envVars ?? []) { @@ -532,6 +533,7 @@ async function generateModels() { if ( !fetchedKeys.has(`${model.provider}/${model.id}`) && !DISCOVERY_ONLY_PROVIDERS.has(model.provider) && + !RETIRED_PROVIDERS.has(model.provider) && !authoritativeCatalogProviders.has(model.provider) && !modelsDevSnapshotExcludedProviders.has(model.provider) ) { @@ -569,7 +571,7 @@ async function generateModels() { // Group by provider and sort each provider's models const providers: Record> = {}; for (const model of allModels) { - if (DISCOVERY_ONLY_PROVIDERS.has(model.provider)) continue; + if (DISCOVERY_ONLY_PROVIDERS.has(model.provider) || RETIRED_PROVIDERS.has(model.provider)) continue; if (!providers[model.provider]) { providers[model.provider] = {}; } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 99bc884f4..7c18a7ebb 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -77590,64 +77590,6 @@ } } }, - "wafer-pass": { - "GLM-5.1": { - "id": "GLM-5.1", - "name": "GLM-5.1", - "api": "openai-completions", - "provider": "wafer-pass", - "baseUrl": "https://pass.wafer.ai/v1", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 202752, - "maxTokens": 65536, - "compat": { - "supportsDeveloperRole": false, - "thinkingFormat": "zai", - "reasoningContentField": "reasoning_content" - }, - "thinking": { - "mode": "effort", - "efforts": [ - "minimal", - "low", - "medium", - "high" - ] - } - }, - "Qwen3.5-397B-A17B": { - "id": "Qwen3.5-397B-A17B", - "name": "Qwen3.5-397B-A17B", - "api": "openai-completions", - "provider": "wafer-pass", - "baseUrl": "https://pass.wafer.ai/v1", - "reasoning": false, - "input": [ - "text", - "image" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 262144, - "maxTokens": 65536, - "compat": { - "supportsDeveloperRole": false - } - } - }, "wafer-serverless": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index 9f1b644c1..f2601ede6 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -41,7 +41,6 @@ import { veniceModelManagerOptions, vercelAiGatewayModelManagerOptions, vllmModelManagerOptions, - waferPassModelManagerOptions, waferServerlessModelManagerOptions, xaiModelManagerOptions, xaiOAuthModelManagerOptions, @@ -349,13 +348,6 @@ export const CATALOG_PROVIDERS = [ createModelManagerOptions: (config: ModelManagerConfig) => vllmModelManagerOptions(config), catalogDiscovery: { label: "vLLM", allowUnauthenticated: true }, }, - { - id: "wafer-pass", - defaultModel: "GLM-5.1", - envVars: ["WAFER_PASS_API_KEY"], - createModelManagerOptions: (config: ModelManagerConfig) => waferPassModelManagerOptions(config), - catalogDiscovery: { label: "Wafer Pass", oauthProvider: "wafer-pass" }, - }, { id: "wafer-serverless", defaultModel: "GLM-5.1", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 91bbc92ad..02d632535 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1568,7 +1568,7 @@ export function firepassModelManagerOptions( } // --------------------------------------------------------------------------- -// 7.7 Wafer (Pass + Serverless) +// 7.7 Wafer Serverless // --------------------------------------------------------------------------- export interface WaferModelManagerConfig { @@ -1581,13 +1581,14 @@ const WAFER_DEFAULT_BASE_URL = "https://pass.wafer.ai/v1"; const WAFER_MAX_TOKENS_CAP = 65536; /** - * Shared mapper for Wafer's `/v1/models` records. + * Mapper for Wafer Serverless `/v1/models` records. * - * Wafer wraps each entry with a `wafer` envelope describing tier, capabilities, - * and cents-per-million pricing. The mapper folds that metadata into the - * canonical `ModelSpec<"openai-completions">` shape and applies zai-family thinking - * compat when the entry advertises reasoning support (GLM-family on the Pass - * SKU). Cents-per-million → dollars-per-million via /100. + * Wafer wraps each entry with a `wafer` envelope describing capabilities and + * pricing. The mapper folds that metadata into the canonical + * `ModelSpec<"openai-completions">` shape and applies upstream-specific thinking + * compat when the entry advertises reasoning support. Wafer pricing is exposed + * through internal wholesale units; the public Serverless rate equals + * `cents × 125 / 10000`. */ interface WaferRecord { context_length?: unknown; @@ -1608,7 +1609,7 @@ function readWaferRecord(entry: OpenAICompatibleModelRecord): WaferRecord | unde } function mapWaferModel( - providerId: "wafer-pass" | "wafer-serverless", + providerId: "wafer-serverless", baseUrl: string, entry: OpenAICompatibleModelRecord, defaults: ModelSpec<"openai-completions">, @@ -1624,25 +1625,12 @@ function mapWaferModel( ); const maxTokens = contextWindow !== null ? Math.min(contextWindow, WAFER_MAX_TOKENS_CAP) : null; const pricing = wafer?.pricing ?? {}; - // Wafer's `/v1/models` exposes pricing through `*_cents_per_million` fields, - // but the values are an internal wholesale unit, not literal cents — across - // every published Serverless model on wafer.ai the user-facing rate equals - // `cents × 125 / 10000` (i.e. wholesale × 1.25 / 100; GLM-5.1's `120` → - // $1.50/M, Kimi-K2.6's `88` → $1.10/M, etc.). The multiply-first form keeps - // the result a finite dyadic for every observed value. - // For the Pass SKU the per-token rate is bundled in the flat-rate - // subscription, so we follow the convention shared with - // `kimi-code`/`firepass`/`alibaba-coding-plan` and seed every Pass model with - // `cost: 0` regardless of what the upstream envelope says. - const isPassSku = providerId === "wafer-pass"; - const cost = isPassSku - ? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } - : { - input: (toPositiveNumber(pricing.input_cents_per_million, 0) * 125) / 10000, - output: (toPositiveNumber(pricing.output_cents_per_million, 0) * 125) / 10000, - cacheRead: (toPositiveNumber(pricing.cache_read_cents_per_million, 0) * 125) / 10000, - cacheWrite: 0, - }; + const cost = { + input: (toPositiveNumber(pricing.input_cents_per_million, 0) * 125) / 10000, + output: (toPositiveNumber(pricing.output_cents_per_million, 0) * 125) / 10000, + cacheRead: (toPositiveNumber(pricing.cache_read_cents_per_million, 0) * 125) / 10000, + cacheWrite: 0, + }; const name = toModelName(wafer?.display_name, defaults.name); const base: ModelSpec<"openai-completions"> = { ...defaults, @@ -1688,13 +1676,12 @@ function mapWaferModel( }; } -function createWaferOptions( - providerId: "wafer-pass" | "wafer-serverless", - config: WaferModelManagerConfig | undefined, +export function waferServerlessModelManagerOptions( + config?: WaferModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { const apiKey = config?.apiKey; const baseUrl = config?.baseUrl ?? WAFER_DEFAULT_BASE_URL; - const passOnly = providerId === "wafer-pass"; + const providerId = "wafer-serverless" as const; return { providerId, ...(apiKey && { @@ -1704,11 +1691,6 @@ function createWaferOptions( provider: providerId, baseUrl, apiKey, - filterModel: entry => { - if (!passOnly) return true; - const wafer = readWaferRecord(entry); - return wafer?.tier === "pass_included"; - }, mapModel: (entry, defaults) => mapWaferModel(providerId, baseUrl, entry, defaults), fetch: config?.fetch, }), @@ -1716,18 +1698,6 @@ function createWaferOptions( }; } -export function waferPassModelManagerOptions( - config?: WaferModelManagerConfig, -): ModelManagerOptions<"openai-completions"> { - return createWaferOptions("wafer-pass", config); -} - -export function waferServerlessModelManagerOptions( - config?: WaferModelManagerConfig, -): ModelManagerOptions<"openai-completions"> { - return createWaferOptions("wafer-serverless", config); -} - // --------------------------------------------------------------------------- // 7. Mistral // --------------------------------------------------------------------------- diff --git a/packages/catalog/test/wafer.test.ts b/packages/catalog/test/wafer.test.ts index b207a2eae..12a2fc8b2 100644 --- a/packages/catalog/test/wafer.test.ts +++ b/packages/catalog/test/wafer.test.ts @@ -1,90 +1,17 @@ /** - * Wafer Pass + Wafer Serverless provider wiring. + * Wafer Serverless provider wiring. * - * Wafer exposes a single OpenAI-compatible base URL (`https://pass.wafer.ai/v1`) - * for two SKUs whose entitlement differs server-side: - * - `wafer-pass` (flat-rate) - * - `wafer-serverless` (pay-as-you-go) - * - * Both providers route through `openai-completions` and the catalog id matches - * the wire id (no rewrite). These tests defend the bundled catalog contract and - * the case-sensitive id pass-through against the wire. + * Wafer Serverless exposes an OpenAI-compatible base URL + * (`https://pass.wafer.ai/v1`) and routes through `openai-completions`. The + * catalog id matches the wire id (no rewrite). These tests defend the bundled + * catalog contract and the case-sensitive id pass-through against the wire. */ import { describe, expect, it } from "bun:test"; -import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { Context } from "@oh-my-pi/pi-ai/types"; import { createModelManager } from "@oh-my-pi/pi-catalog/model-manager"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; -import { - waferPassModelManagerOptions, - waferServerlessModelManagerOptions, -} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import { waferServerlessModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types"; -function sseResponse(events: unknown[]): Response { - const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`; - return new Response(payload, { - status: 200, - headers: { "content-type": "text/event-stream" }, - }); -} - -describe("Wafer Pass provider", () => { - it("ships a bundled GLM-5.1 entry with zai-family thinking compat", () => { - const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1"); - expect(model).toBeDefined(); - expect(model.id).toBe("GLM-5.1"); - expect(model.provider).toBe("wafer-pass"); - expect(model.api).toBe("openai-completions"); - expect(model.baseUrl).toBe("https://pass.wafer.ai/v1"); - expect(model.reasoning).toBe(true); - expect(model.input).toEqual(["text"]); - expect(model.compatConfig?.thinkingFormat).toBe("zai"); - expect(model.compatConfig?.reasoningContentField).toBe("reasoning_content"); - expect(model.compatConfig?.supportsDeveloperRole).toBe(false); - }); - - it("ships a bundled Qwen3.5-397B-A17B entry with vision input and no reasoning", () => { - const model = getBundledModel<"openai-completions">("wafer-pass", "Qwen3.5-397B-A17B"); - expect(model).toBeDefined(); - expect(model.id).toBe("Qwen3.5-397B-A17B"); - expect(model.provider).toBe("wafer-pass"); - expect(model.reasoning).toBe(false); - expect(model.input).toEqual(["text", "image"]); - }); - - it("preserves the catalog id verbatim on the wire (no rewrite, case-sensitive)", async () => { - const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1"); - const captured: { url: string | null; body: string | null } = { url: null, body: null }; - const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => { - captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; - captured.body = typeof init?.body === "string" ? init.body : null; - return sseResponse(["[DONE]"]); - }; - - const context: Context = { - systemPrompt: ["t"], - messages: [{ role: "user", content: "hi", timestamp: Date.now() }], - }; - const stream = streamOpenAICompletions(model as Model<"openai-completions">, context, { - apiKey: "wfr_test", - fetch: fetchMock, - }); - for await (const _event of stream) { - /* drain */ - } - - expect(captured.url).toBe("https://pass.wafer.ai/v1/chat/completions"); - expect(captured.body).not.toBeNull(); - const parsed = JSON.parse(captured.body ?? "{}") as { model?: unknown }; - // Wafer's docs note model names are case-insensitive on input, but the - // canonical id has mixed case; we must round-trip it unchanged so users - // who pin `GLM-5.1` don't end up with usage rows under `glm-5.1` or - // hitting the upstream 404 path. - expect(parsed.model).toBe("GLM-5.1"); - }); -}); - describe("Wafer Serverless provider", () => { it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Qwen3.6, Qwen3.7-Max, Kimi-K2.6, DeepSeek V4 Flash/Pro)", () => { const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1"); @@ -143,15 +70,8 @@ describe("Wafer Serverless provider", () => { expect(dsPro.reasoning).toBe(true); expect(dsPro.compatConfig?.thinkingFormat).toBeUndefined(); }); - - it("does not expose Serverless-only ids on the Wafer Pass catalog", () => { - expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined(); - expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined(); - expect(getBundledModel("wafer-pass", "qwen3.7-max")).toBeUndefined(); - expect(getBundledModel("wafer-pass", "deepseek-v4-flash")).toBeUndefined(); - expect(getBundledModel("wafer-pass", "deepseek-v4-pro")).toBeUndefined(); - }); }); + describe("Wafer dynamic discovery mapper", () => { // Synthetic /v1/models response that exercises every upstream provider Wafer // announces, including a deliberately-unknown one. The mapper must: @@ -224,11 +144,9 @@ describe("Wafer dynamic discovery mapper", () => { } }); - it("zeros cost for the Pass SKU and applies retail × 0.0125 for Serverless", async () => { - // Same upstream record served via both SKUs — `wafer.pricing` in cents/M: - // 120/360/12. Pass is a flat-rate subscription (no per-token charge), so - // `mapWaferModel` zeros the cost regardless of envelope values. Serverless - // is pay-as-you-go and applies the empirical × 0.0125 conversion to match + it("applies retail × 0.0125 for Serverless pricing", async () => { + // `wafer.pricing` is in Wafer's internal cents/M unit. Serverless is + // pay-as-you-go and applies the empirical × 0.0125 conversion to match // wafer.ai's published retail rates (120 cents → $1.50/M). const sharedEntry = { id: "Shared-fake", @@ -249,13 +167,6 @@ describe("Wafer dynamic discovery mapper", () => { }, }; - const fetchMock = mockWaferModelsResponse([sharedEntry]); - const passManager = createModelManager(waferPassModelManagerOptions({ apiKey: "wfr_test", fetch: fetchMock })); - const passResult = await passManager.refresh("online"); - const passModel = passResult.models.find(m => m.id === "Shared-fake"); - expect(passModel).toBeDefined(); - expect(passModel?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }); - const srvFetchMock = mockWaferModelsResponse([sharedEntry]); const srvManager = createModelManager( waferServerlessModelManagerOptions({ apiKey: "wfr_test", fetch: srvFetchMock }), diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 009d8abd6..505aad553 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -9,6 +9,14 @@ - Fixed `omp --approval-mode=yolo acp` and other global option flags placed before a subcommand being rewritten to `launch` with the subcommand swallowed as prompt text; the CLI resolver now skips leading global flags (using the launch parser's value-consumption contract) and dispatches the real subcommand with the flags applied, so ACP mode honors the configured approval policy. ([#2970](https://github.com/can1357/oh-my-pi/issues/2970)) - Fixed `/mcp enable` and `/mcp disable` reconnecting unrelated MCP servers by scoping toggle reconnect/disconnect work to the named server. ([#3157](https://github.com/can1357/oh-my-pi/issues/3157)) +### Fixed + +- Stopped the local llama.cpp no-auth login placeholder from being sent as a discovery bearer token. + +### Removed + +- Removed Wafer Pass from CLI credential help; Wafer Serverless remains available. + ## [16.1.8] - 2026-06-20 ### Added diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 3e000658a..488826561 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -298,7 +298,6 @@ export function getExtraHelpText(): string { OPENCODE_API_KEY - OpenCode Zen/OpenCode Go models CURSOR_ACCESS_TOKEN - Cursor AI models AI_GATEWAY_API_KEY - Vercel AI Gateway - WAFER_PASS_API_KEY - Wafer Pass (flat-rate subscription; GLM-5.1, Qwen3.5) WAFER_SERVERLESS_API_KEY - Wafer Serverless (pay-as-you-go) ${chalk.dim("# Cloud Providers")} diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 265b9c601..da80b07a6 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -25,8 +25,9 @@ import { } from "@oh-my-pi/pi-catalog/variant-collapse"; // Sentinels for local-only OAuth tokens — declared inline to avoid loading -// provider modules at startup. Must match packages/ai/src/registry/lm-studio.ts -// and packages/ai/src/registry/vllm.ts. +// provider modules at startup. Must match packages/ai/src/registry/llama-cpp.ts, +// packages/ai/src/registry/lm-studio.ts, and packages/ai/src/registry/vllm.ts. +const DEFAULT_LLAMA_CPP_LOCAL_TOKEN = "llama-cpp-local"; const DEFAULT_LOCAL_TOKEN = "lm-studio-local"; const DEFAULT_VLLM_LOCAL_TOKEN = "vllm-local"; @@ -85,7 +86,12 @@ export function isAuthenticated(apiKey: string | undefined | null): apiKey is st } function isDiscoveryBearerApiKey(apiKey: string | undefined | null): apiKey is string { - return isAuthenticated(apiKey) && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== DEFAULT_VLLM_LOCAL_TOKEN; + return ( + isAuthenticated(apiKey) && + apiKey !== DEFAULT_LLAMA_CPP_LOCAL_TOKEN && + apiKey !== DEFAULT_LOCAL_TOKEN && + apiKey !== DEFAULT_VLLM_LOCAL_TOKEN + ); } /** Provider override config (baseUrl, headers, apiKey, compat, transport) without custom models */ diff --git a/packages/coding-agent/test/issue-970-custom-provider-discovery.test.ts b/packages/coding-agent/test/issue-970-custom-provider-discovery.test.ts index 7468f5064..89f1cfa7c 100644 --- a/packages/coding-agent/test/issue-970-custom-provider-discovery.test.ts +++ b/packages/coding-agent/test/issue-970-custom-provider-discovery.test.ts @@ -470,4 +470,50 @@ describe("issue #970 custom provider discovery", () => { expect(registry.getProviderDiscoveryState("vllm")?.status).toBe("ok"); }); + + test("does not send llama.cpp-local placeholder as discovery bearer", async () => { + fs.writeFileSync( + modelsPath, + [ + "providers:", + " llama.cpp:", + " baseUrl: http://127.0.0.1:8080", + " apiKey: llama-cpp-local", + " api: openai-responses", + " discovery:", + " type: llama.cpp", + ].join("\n"), + ); + + const fetchMock: (input: string | URL | Request, init?: RequestInit) => Promise = async ( + input, + init, + ) => { + const url = String(input); + if (url === "http://127.0.0.1:8080/props") { + const headers = init?.headers as Headers | Record | undefined; + const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization; + expect(authHeader).toBeUndefined(); + return new Response(JSON.stringify({ default_generation_settings: { n_ctx: 8192 } }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + } + if (url !== "http://127.0.0.1:8080/models") { + throw new Error(`Unexpected URL: ${url}`); + } + const headers = init?.headers as Headers | Record | undefined; + const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization; + expect(authHeader).toBeUndefined(); + return new Response(JSON.stringify({ data: [{ id: "local-llama" }] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock }); + await registry.refreshProvider("llama.cpp"); + + expect(registry.getProviderDiscoveryState("llama.cpp")?.status).toBe("ok"); + }); });