Remove Wafer Pass provider

This commit is contained in:
oldschoola
2026-06-20 17:21:01 -07:00
committed by can1357
parent 4702818870
commit 13bfc7b9c0
19 changed files with 118 additions and 278 deletions
+1 -1
View File
@@ -292,7 +292,7 @@ Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google A
Subscription-routed. `/login` attaches the session.
Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · Wafer Pass `plan` · OpenCode Go · OpenCode Zen
Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen
### Run it yourself
-1
View File
@@ -82,7 +82,6 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
| `DEEPSEEK_API_KEY` | DeepSeek auth | Using DeepSeek models | |
| `KILO_API_KEY` | Kilo auth | Using Kilo models | |
| `OLLAMA_CLOUD_API_KEY` | Ollama Cloud auth | Using `ollama-cloud` provider | |
| `WAFER_PASS_API_KEY` | Wafer Pass auth | Using `wafer-pass` provider | Flat-rate Wafer subscription; validated against `https://pass.wafer.ai/v1/models` |
| `WAFER_SERVERLESS_API_KEY` | Wafer Serverless auth | Using `wafer-serverless` provider | Pay-as-you-go Wafer SKU; validated against `https://pass.wafer.ai/v1/models` |
| `GITLAB_TOKEN` | GitLab Duo auth | Using `gitlab-duo` provider | |
+1 -2
View File
@@ -123,7 +123,6 @@ Each provider has one or more environment variables that supply a key when no st
| `gitlab-duo` | `GITLAB_TOKEN` |
| `opencode-zen`, `opencode-go` | `OPENCODE_API_KEY` |
| `firepass` | `FIREPASS_API_KEY` |
| `wafer-pass` | `WAFER_PASS_API_KEY` |
| `wafer-serverless` | `WAFER_SERVERLESS_API_KEY` |
| `xiaomi` | `XIAOMI_API_KEY` |
| `ollama-cloud` | `OLLAMA_CLOUD_API_KEY` |
@@ -131,7 +130,7 @@ Each provider has one or more environment variables that supply a key when no st
| `lm-studio` | `LM_STUDIO_API_KEY` (optional; keyless by default) |
| `llama.cpp` | `LLAMA_CPP_API_KEY` (only when the server requires auth) |
OAuth-backed providers such as `anthropic`, `github-copilot`, `cursor`, `ollama-cloud`, `qwen-portal`, `kimi-code`, `xai-oauth`, `wafer-pass`, `wafer-serverless`, `google-gemini-cli`, and `google-antigravity` are normally reached through `/login` rather than an environment variable. See [Environment variables](./environment-variables.md) for search-tool and configuration variables not listed here.
OAuth-backed providers such as `anthropic`, `github-copilot`, `cursor`, `ollama-cloud`, `qwen-portal`, `kimi-code`, `xai-oauth`, `wafer-serverless`, `google-gemini-cli`, and `google-antigravity` are normally reached through `/login` rather than an environment variable. See [Environment variables](./environment-variables.md) for search-tool and configuration variables not listed here.
### `.env` discovery and precedence
+4
View File
@@ -10,6 +10,10 @@
- Added `llama.cpp` to the interactive `/login` provider list, accepting an optional API key while defaulting to local no-auth mode.
### Removed
- Removed Wafer Pass (`wafer-pass`) login support; Wafer Serverless remains available as `wafer-serverless`.
## [16.1.8] - 2026-06-20
### Changed
-1
View File
@@ -62,7 +62,6 @@ Unified LLM API with automatic model discovery, provider configuration, token an
- **Hugging Face Inference**
- **xAI**
- **Venice** (requires `VENICE_API_KEY`)
- **Wafer Pass** (requires `WAFER_PASS_API_KEY`; flat-rate subscription, includes GLM-5.1 and Qwen3.5-397B-A17B)
- **Wafer Serverless** (requires `WAFER_SERVERLESS_API_KEY`; pay-as-you-go)
- **OpenRouter**
- **Kilo Gateway** (supports OAuth `/login kilo` or `KILO_API_KEY`)
+4 -30
View File
@@ -1,41 +1,15 @@
/**
* Wafer login flows.
* Wafer Serverless login flow.
*
* Wafer (https://wafer.ai) exposes a single OpenAI-compatible base URL
* (`https://pass.wafer.ai/v1`) for two SKUs:
*
* - **Wafer Pass** — flat-rate subscription. The key authorizes models whose
* catalog entries carry `wafer.tier = "pass_included"`.
* - **Wafer Serverless** — pay-as-you-go. Superset of Pass; the same `/v1/models`
* endpoint returns the full per-account model list.
*
* Both SKUs issue `wfr_…` keys. The key prefix alone does not distinguish
* tiers — the entitlement is per-account on the server side — so we expose
* two parallel logins / env vars (`WAFER_PASS_API_KEY`, `WAFER_SERVERLESS_API_KEY`)
* mirroring the firepass/fireworks split, letting users with both
* subscriptions switch between them without re-pasting.
*
* Validation uses the shared `/v1/models` endpoint, which works for both
* tiers and is cheap (no token spend).
* Wafer (https://wafer.ai) exposes a pay-as-you-go OpenAI-compatible SKU at
* `https://pass.wafer.ai/v1`. Keys use the `wfr_…` prefix and are validated
* against `/v1/models`, which is cheap (no token spend).
*/
import { createApiKeyLogin } from "../api-key-login";
const WAFER_AUTH_URL = "https://wafer.ai/dashboard";
const WAFER_MODELS_URL = "https://pass.wafer.ai/v1/models";
export const loginWaferPass = createApiKeyLogin({
providerLabel: "Wafer Pass",
authUrl: WAFER_AUTH_URL,
instructions: "Create or copy your Wafer Pass API key from the Wafer dashboard",
promptMessage: "Paste your Wafer Pass API key",
placeholder: "wfr_...",
validation: {
kind: "models-endpoint",
provider: "Wafer Pass",
modelsUrl: WAFER_MODELS_URL,
},
});
export const loginWaferServerless = createApiKeyLogin({
providerLabel: "Wafer Serverless",
authUrl: WAFER_AUTH_URL,
-2
View File
@@ -51,7 +51,6 @@ import { umansProvider } from "./umans";
import { veniceProvider } from "./venice";
import { vercelAiGatewayProvider } from "./vercel-ai-gateway";
import { vllmProvider } from "./vllm";
import { waferPassProvider } from "./wafer-pass";
import { waferServerlessProvider } from "./wafer-serverless";
import { xaiProvider } from "./xai";
import { xaiOauthProvider } from "./xai-oauth";
@@ -96,7 +95,6 @@ const ALL = [
xiaomiTokenPlanAmsProvider,
xiaomiTokenPlanCnProvider,
firepassProvider,
waferPassProvider,
deepseekProvider,
moonshotProvider,
cerebrasProvider,
-12
View File
@@ -1,12 +0,0 @@
import type { OAuthLoginCallbacks } from "./oauth/types";
import type { ProviderDefinition } from "./types";
export const waferPassProvider = {
id: "wafer-pass",
name: "Wafer Pass (flat-rate subscription)",
login: async (cb: OAuthLoginCallbacks) => {
// Lazy import: keep heavy OAuth flow modules out of the eager registry graph.
const { loginWaferPass } = await import("./oauth/wafer");
return loginWaferPass(cb);
},
} as const satisfies ProviderDefinition;
+10 -11
View File
@@ -1,25 +1,24 @@
/**
* Live Wafer Pass smoke. NOT part of the bun test suite — run manually:
* WAFER_PASS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts
* Live Wafer Serverless smoke. NOT part of the bun test suite — run manually:
* WAFER_SERVERLESS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts
*
* Validates that the bundled `wafer-pass/GLM-5.1` entry round-trips a real
* streaming chat completion against `https://pass.wafer.ai/v1`, with the wire
* `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a non-empty
* assistant text returned.
* Validates that the bundled `wafer-serverless/GLM-5.1` entry round-trips a
* real streaming chat completion against `https://pass.wafer.ai/v1`, with the
* wire `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a
* non-empty assistant text returned.
*/
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const apiKey = process.env.WAFER_PASS_API_KEY ?? process.env.WAFER_SERVERLESS_API_KEY;
const apiKey = process.env.WAFER_SERVERLESS_API_KEY;
if (!apiKey) {
console.error("WAFER_PASS_API_KEY (or WAFER_SERVERLESS_API_KEY) env var is required");
console.error("WAFER_SERVERLESS_API_KEY env var is required");
process.exit(2);
}
const providerId = process.env.WAFER_PASS_API_KEY ? "wafer-pass" : "wafer-serverless";
const model = getBundledModel<"openai-completions">(providerId, "GLM-5.1");
const model = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1");
console.log(`Model: ${model.provider}/${model.id} -> ${model.baseUrl}`);
console.log(`compat.thinkingFormat: ${model.compat?.thinkingFormat ?? "(none)"}`);
@@ -89,6 +88,6 @@ if (text.trim().length === 0) {
}
console.log(
`\nLIVE OK — Wafer ${providerId} round-trip: GLM-5.1 endpoint preserved, ` +
`\nLIVE OK — Wafer Serverless round-trip: GLM-5.1 endpoint preserved, ` +
`${inputTokens}→${outputTokens} tokens, stopReason=${stopReason}.`,
);
+4
View File
@@ -6,6 +6,10 @@
- Fixed the `moonshot` provider with no path to the Kimi China API: model discovery now honors a `MOONSHOT_BASE_URL` override (redirecting to `api.moonshot.cn`), and `KIMI_API_KEY` resolves as a fallback for `MOONSHOT_API_KEY`. ([#2883](https://github.com/can1357/oh-my-pi/issues/2883))
### Removed
- Removed bundled Wafer Pass (`wafer-pass`) catalog entries and generation support; Wafer Serverless remains available as `wafer-serverless`.
## [16.1.8] - 2026-06-20
### Fixed
+3 -1
View File
@@ -61,6 +61,7 @@ const packageRoot = path.join(import.meta.dir, "..");
* and never written to models.json.
*/
const DISCOVERY_ONLY_PROVIDERS = new Set(["ollama", "vllm", "lm-studio", "litellm"]);
const RETIRED_PROVIDERS = new Set(["wafer-pass"]);
async function resolveProviderApiKey(providerId: string, catalog: CatalogDiscoveryConfig): Promise<string | undefined> {
for (const envVar of catalog.envVars ?? []) {
@@ -532,6 +533,7 @@ async function generateModels() {
if (
!fetchedKeys.has(`${model.provider}/${model.id}`) &&
!DISCOVERY_ONLY_PROVIDERS.has(model.provider) &&
!RETIRED_PROVIDERS.has(model.provider) &&
!authoritativeCatalogProviders.has(model.provider) &&
!modelsDevSnapshotExcludedProviders.has(model.provider)
) {
@@ -569,7 +571,7 @@ async function generateModels() {
// Group by provider and sort each provider's models
const providers: Record<string, Record<string, ModelSpec>> = {};
for (const model of allModels) {
if (DISCOVERY_ONLY_PROVIDERS.has(model.provider)) continue;
if (DISCOVERY_ONLY_PROVIDERS.has(model.provider) || RETIRED_PROVIDERS.has(model.provider)) continue;
if (!providers[model.provider]) {
providers[model.provider] = {};
}
-58
View File
@@ -77590,64 +77590,6 @@
}
}
},
"wafer-pass": {
"GLM-5.1": {
"id": "GLM-5.1",
"name": "GLM-5.1",
"api": "openai-completions",
"provider": "wafer-pass",
"baseUrl": "https://pass.wafer.ai/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 202752,
"maxTokens": 65536,
"compat": {
"supportsDeveloperRole": false,
"thinkingFormat": "zai",
"reasoningContentField": "reasoning_content"
},
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen3.5-397B-A17B": {
"id": "Qwen3.5-397B-A17B",
"name": "Qwen3.5-397B-A17B",
"api": "openai-completions",
"provider": "wafer-pass",
"baseUrl": "https://pass.wafer.ai/v1",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536,
"compat": {
"supportsDeveloperRole": false
}
}
},
"wafer-serverless": {
"deepseek-v4-flash": {
"id": "deepseek-v4-flash",
@@ -41,7 +41,6 @@ import {
veniceModelManagerOptions,
vercelAiGatewayModelManagerOptions,
vllmModelManagerOptions,
waferPassModelManagerOptions,
waferServerlessModelManagerOptions,
xaiModelManagerOptions,
xaiOAuthModelManagerOptions,
@@ -349,13 +348,6 @@ export const CATALOG_PROVIDERS = [
createModelManagerOptions: (config: ModelManagerConfig) => vllmModelManagerOptions(config),
catalogDiscovery: { label: "vLLM", allowUnauthenticated: true },
},
{
id: "wafer-pass",
defaultModel: "GLM-5.1",
envVars: ["WAFER_PASS_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => waferPassModelManagerOptions(config),
catalogDiscovery: { label: "Wafer Pass", oauthProvider: "wafer-pass" },
},
{
id: "wafer-serverless",
defaultModel: "GLM-5.1",
@@ -1568,7 +1568,7 @@ export function firepassModelManagerOptions(
}
// ---------------------------------------------------------------------------
// 7.7 Wafer (Pass + Serverless)
// 7.7 Wafer Serverless
// ---------------------------------------------------------------------------
export interface WaferModelManagerConfig {
@@ -1581,13 +1581,14 @@ const WAFER_DEFAULT_BASE_URL = "https://pass.wafer.ai/v1";
const WAFER_MAX_TOKENS_CAP = 65536;
/**
* Shared mapper for Wafer's `/v1/models` records.
* Mapper for Wafer Serverless `/v1/models` records.
*
* Wafer wraps each entry with a `wafer` envelope describing tier, capabilities,
* and cents-per-million pricing. The mapper folds that metadata into the
* canonical `ModelSpec<"openai-completions">` shape and applies zai-family thinking
* compat when the entry advertises reasoning support (GLM-family on the Pass
* SKU). Cents-per-million → dollars-per-million via /100.
* Wafer wraps each entry with a `wafer` envelope describing capabilities and
* pricing. The mapper folds that metadata into the canonical
* `ModelSpec<"openai-completions">` shape and applies upstream-specific thinking
* compat when the entry advertises reasoning support. Wafer pricing is exposed
* through internal wholesale units; the public Serverless rate equals
* `cents × 125 / 10000`.
*/
interface WaferRecord {
context_length?: unknown;
@@ -1608,7 +1609,7 @@ function readWaferRecord(entry: OpenAICompatibleModelRecord): WaferRecord | unde
}
function mapWaferModel(
providerId: "wafer-pass" | "wafer-serverless",
providerId: "wafer-serverless",
baseUrl: string,
entry: OpenAICompatibleModelRecord,
defaults: ModelSpec<"openai-completions">,
@@ -1624,25 +1625,12 @@ function mapWaferModel(
);
const maxTokens = contextWindow !== null ? Math.min(contextWindow, WAFER_MAX_TOKENS_CAP) : null;
const pricing = wafer?.pricing ?? {};
// Wafer's `/v1/models` exposes pricing through `*_cents_per_million` fields,
// but the values are an internal wholesale unit, not literal cents — across
// every published Serverless model on wafer.ai the user-facing rate equals
// `cents × 125 / 10000` (i.e. wholesale × 1.25 / 100; GLM-5.1's `120` →
// $1.50/M, Kimi-K2.6's `88` → $1.10/M, etc.). The multiply-first form keeps
// the result a finite dyadic for every observed value.
// For the Pass SKU the per-token rate is bundled in the flat-rate
// subscription, so we follow the convention shared with
// `kimi-code`/`firepass`/`alibaba-coding-plan` and seed every Pass model with
// `cost: 0` regardless of what the upstream envelope says.
const isPassSku = providerId === "wafer-pass";
const cost = isPassSku
? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }
: {
input: (toPositiveNumber(pricing.input_cents_per_million, 0) * 125) / 10000,
output: (toPositiveNumber(pricing.output_cents_per_million, 0) * 125) / 10000,
cacheRead: (toPositiveNumber(pricing.cache_read_cents_per_million, 0) * 125) / 10000,
cacheWrite: 0,
};
const cost = {
input: (toPositiveNumber(pricing.input_cents_per_million, 0) * 125) / 10000,
output: (toPositiveNumber(pricing.output_cents_per_million, 0) * 125) / 10000,
cacheRead: (toPositiveNumber(pricing.cache_read_cents_per_million, 0) * 125) / 10000,
cacheWrite: 0,
};
const name = toModelName(wafer?.display_name, defaults.name);
const base: ModelSpec<"openai-completions"> = {
...defaults,
@@ -1688,13 +1676,12 @@ function mapWaferModel(
};
}
function createWaferOptions(
providerId: "wafer-pass" | "wafer-serverless",
config: WaferModelManagerConfig | undefined,
export function waferServerlessModelManagerOptions(
config?: WaferModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
const apiKey = config?.apiKey;
const baseUrl = config?.baseUrl ?? WAFER_DEFAULT_BASE_URL;
const passOnly = providerId === "wafer-pass";
const providerId = "wafer-serverless" as const;
return {
providerId,
...(apiKey && {
@@ -1704,11 +1691,6 @@ function createWaferOptions(
provider: providerId,
baseUrl,
apiKey,
filterModel: entry => {
if (!passOnly) return true;
const wafer = readWaferRecord(entry);
return wafer?.tier === "pass_included";
},
mapModel: (entry, defaults) => mapWaferModel(providerId, baseUrl, entry, defaults),
fetch: config?.fetch,
}),
@@ -1716,18 +1698,6 @@ function createWaferOptions(
};
}
export function waferPassModelManagerOptions(
config?: WaferModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
return createWaferOptions("wafer-pass", config);
}
export function waferServerlessModelManagerOptions(
config?: WaferModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
return createWaferOptions("wafer-serverless", config);
}
// ---------------------------------------------------------------------------
// 7. Mistral
// ---------------------------------------------------------------------------
+10 -99
View File
@@ -1,90 +1,17 @@
/**
* Wafer Pass + Wafer Serverless provider wiring.
* Wafer Serverless provider wiring.
*
* Wafer exposes a single OpenAI-compatible base URL (`https://pass.wafer.ai/v1`)
* for two SKUs whose entitlement differs server-side:
* - `wafer-pass` (flat-rate)
* - `wafer-serverless` (pay-as-you-go)
*
* Both providers route through `openai-completions` and the catalog id matches
* the wire id (no rewrite). These tests defend the bundled catalog contract and
* the case-sensitive id pass-through against the wire.
* Wafer Serverless exposes an OpenAI-compatible base URL
* (`https://pass.wafer.ai/v1`) and routes through `openai-completions`. The
* catalog id matches the wire id (no rewrite). These tests defend the bundled
* catalog contract and the case-sensitive id pass-through against the wire.
*/
import { describe, expect, it } from "bun:test";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context } from "@oh-my-pi/pi-ai/types";
import { createModelManager } from "@oh-my-pi/pi-catalog/model-manager";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import {
waferPassModelManagerOptions,
waferServerlessModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import { waferServerlessModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types";
function sseResponse(events: unknown[]): Response {
const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`;
return new Response(payload, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
describe("Wafer Pass provider", () => {
it("ships a bundled GLM-5.1 entry with zai-family thinking compat", () => {
const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1");
expect(model).toBeDefined();
expect(model.id).toBe("GLM-5.1");
expect(model.provider).toBe("wafer-pass");
expect(model.api).toBe("openai-completions");
expect(model.baseUrl).toBe("https://pass.wafer.ai/v1");
expect(model.reasoning).toBe(true);
expect(model.input).toEqual(["text"]);
expect(model.compatConfig?.thinkingFormat).toBe("zai");
expect(model.compatConfig?.reasoningContentField).toBe("reasoning_content");
expect(model.compatConfig?.supportsDeveloperRole).toBe(false);
});
it("ships a bundled Qwen3.5-397B-A17B entry with vision input and no reasoning", () => {
const model = getBundledModel<"openai-completions">("wafer-pass", "Qwen3.5-397B-A17B");
expect(model).toBeDefined();
expect(model.id).toBe("Qwen3.5-397B-A17B");
expect(model.provider).toBe("wafer-pass");
expect(model.reasoning).toBe(false);
expect(model.input).toEqual(["text", "image"]);
});
it("preserves the catalog id verbatim on the wire (no rewrite, case-sensitive)", async () => {
const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1");
const captured: { url: string | null; body: string | null } = { url: null, body: null };
const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => {
captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
captured.body = typeof init?.body === "string" ? init.body : null;
return sseResponse(["[DONE]"]);
};
const context: Context = {
systemPrompt: ["t"],
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
};
const stream = streamOpenAICompletions(model as Model<"openai-completions">, context, {
apiKey: "wfr_test",
fetch: fetchMock,
});
for await (const _event of stream) {
/* drain */
}
expect(captured.url).toBe("https://pass.wafer.ai/v1/chat/completions");
expect(captured.body).not.toBeNull();
const parsed = JSON.parse(captured.body ?? "{}") as { model?: unknown };
// Wafer's docs note model names are case-insensitive on input, but the
// canonical id has mixed case; we must round-trip it unchanged so users
// who pin `GLM-5.1` don't end up with usage rows under `glm-5.1` or
// hitting the upstream 404 path.
expect(parsed.model).toBe("GLM-5.1");
});
});
describe("Wafer Serverless provider", () => {
it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Qwen3.6, Qwen3.7-Max, Kimi-K2.6, DeepSeek V4 Flash/Pro)", () => {
const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1");
@@ -143,15 +70,8 @@ describe("Wafer Serverless provider", () => {
expect(dsPro.reasoning).toBe(true);
expect(dsPro.compatConfig?.thinkingFormat).toBeUndefined();
});
it("does not expose Serverless-only ids on the Wafer Pass catalog", () => {
expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined();
expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined();
expect(getBundledModel("wafer-pass", "qwen3.7-max")).toBeUndefined();
expect(getBundledModel("wafer-pass", "deepseek-v4-flash")).toBeUndefined();
expect(getBundledModel("wafer-pass", "deepseek-v4-pro")).toBeUndefined();
});
});
describe("Wafer dynamic discovery mapper", () => {
// Synthetic /v1/models response that exercises every upstream provider Wafer
// announces, including a deliberately-unknown one. The mapper must:
@@ -224,11 +144,9 @@ describe("Wafer dynamic discovery mapper", () => {
}
});
it("zeros cost for the Pass SKU and applies retail × 0.0125 for Serverless", async () => {
// Same upstream record served via both SKUs — `wafer.pricing` in cents/M:
// 120/360/12. Pass is a flat-rate subscription (no per-token charge), so
// `mapWaferModel` zeros the cost regardless of envelope values. Serverless
// is pay-as-you-go and applies the empirical × 0.0125 conversion to match
it("applies retail × 0.0125 for Serverless pricing", async () => {
// `wafer.pricing` is in Wafer's internal cents/M unit. Serverless is
// pay-as-you-go and applies the empirical × 0.0125 conversion to match
// wafer.ai's published retail rates (120 cents → $1.50/M).
const sharedEntry = {
id: "Shared-fake",
@@ -249,13 +167,6 @@ describe("Wafer dynamic discovery mapper", () => {
},
};
const fetchMock = mockWaferModelsResponse([sharedEntry]);
const passManager = createModelManager(waferPassModelManagerOptions({ apiKey: "wfr_test", fetch: fetchMock }));
const passResult = await passManager.refresh("online");
const passModel = passResult.models.find(m => m.id === "Shared-fake");
expect(passModel).toBeDefined();
expect(passModel?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
const srvFetchMock = mockWaferModelsResponse([sharedEntry]);
const srvManager = createModelManager(
waferServerlessModelManagerOptions({ apiKey: "wfr_test", fetch: srvFetchMock }),
+8
View File
@@ -9,6 +9,14 @@
- Fixed `omp --approval-mode=yolo acp` and other global option flags placed before a subcommand being rewritten to `launch` with the subcommand swallowed as prompt text; the CLI resolver now skips leading global flags (using the launch parser's value-consumption contract) and dispatches the real subcommand with the flags applied, so ACP mode honors the configured approval policy. ([#2970](https://github.com/can1357/oh-my-pi/issues/2970))
- Fixed `/mcp enable` and `/mcp disable` reconnecting unrelated MCP servers by scoping toggle reconnect/disconnect work to the named server. ([#3157](https://github.com/can1357/oh-my-pi/issues/3157))
### Fixed
- Stopped the local llama.cpp no-auth login placeholder from being sent as a discovery bearer token.
### Removed
- Removed Wafer Pass from CLI credential help; Wafer Serverless remains available.
## [16.1.8] - 2026-06-20
### Added
-1
View File
@@ -298,7 +298,6 @@ export function getExtraHelpText(): string {
OPENCODE_API_KEY - OpenCode Zen/OpenCode Go models
CURSOR_ACCESS_TOKEN - Cursor AI models
AI_GATEWAY_API_KEY - Vercel AI Gateway
WAFER_PASS_API_KEY - Wafer Pass (flat-rate subscription; GLM-5.1, Qwen3.5)
WAFER_SERVERLESS_API_KEY - Wafer Serverless (pay-as-you-go)
${chalk.dim("# Cloud Providers")}
@@ -25,8 +25,9 @@ import {
} from "@oh-my-pi/pi-catalog/variant-collapse";
// Sentinels for local-only OAuth tokens — declared inline to avoid loading
// provider modules at startup. Must match packages/ai/src/registry/lm-studio.ts
// and packages/ai/src/registry/vllm.ts.
// provider modules at startup. Must match packages/ai/src/registry/llama-cpp.ts,
// packages/ai/src/registry/lm-studio.ts, and packages/ai/src/registry/vllm.ts.
const DEFAULT_LLAMA_CPP_LOCAL_TOKEN = "llama-cpp-local";
const DEFAULT_LOCAL_TOKEN = "lm-studio-local";
const DEFAULT_VLLM_LOCAL_TOKEN = "vllm-local";
@@ -85,7 +86,12 @@ export function isAuthenticated(apiKey: string | undefined | null): apiKey is st
}
function isDiscoveryBearerApiKey(apiKey: string | undefined | null): apiKey is string {
return isAuthenticated(apiKey) && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== DEFAULT_VLLM_LOCAL_TOKEN;
return (
isAuthenticated(apiKey) &&
apiKey !== DEFAULT_LLAMA_CPP_LOCAL_TOKEN &&
apiKey !== DEFAULT_LOCAL_TOKEN &&
apiKey !== DEFAULT_VLLM_LOCAL_TOKEN
);
}
/** Provider override config (baseUrl, headers, apiKey, compat, transport) without custom models */
@@ -470,4 +470,50 @@ describe("issue #970 custom provider discovery", () => {
expect(registry.getProviderDiscoveryState("vllm")?.status).toBe("ok");
});
test("does not send llama.cpp-local placeholder as discovery bearer", async () => {
fs.writeFileSync(
modelsPath,
[
"providers:",
" llama.cpp:",
" baseUrl: http://127.0.0.1:8080",
" apiKey: llama-cpp-local",
" api: openai-responses",
" discovery:",
" type: llama.cpp",
].join("\n"),
);
const fetchMock: (input: string | URL | Request, init?: RequestInit) => Promise<Response> = async (
input,
init,
) => {
const url = String(input);
if (url === "http://127.0.0.1:8080/props") {
const headers = init?.headers as Headers | Record<string, string> | undefined;
const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization;
expect(authHeader).toBeUndefined();
return new Response(JSON.stringify({ default_generation_settings: { n_ctx: 8192 } }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
if (url !== "http://127.0.0.1:8080/models") {
throw new Error(`Unexpected URL: ${url}`);
}
const headers = init?.headers as Headers | Record<string, string> | undefined;
const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization;
expect(authHeader).toBeUndefined();
return new Response(JSON.stringify({ data: [{ id: "local-llama" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("llama.cpp");
expect(registry.getProviderDiscoveryState("llama.cpp")?.status).toBe("ok");
});
});