Remove Wafer Pass provider
This commit is contained in:
@@ -292,7 +292,7 @@ Anthropic `oauth` · OpenAI · OpenAI Codex `oauth` · Google Gemini · Google A
|
||||
|
||||
Subscription-routed. `/login` attaches the session.
|
||||
|
||||
Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · Wafer Pass `plan` · OpenCode Go · OpenCode Zen
|
||||
Cursor `oauth` · GitHub Copilot `oauth` · GitLab Duo · Kimi Code `plan` · Moonshot · MiniMax Coding Plan `plan` · MiniMax Coding Plan CN `plan` · Alibaba Coding Plan `plan` · Qwen Portal · Z.AI / GLM Coding Plan `plan` · Xiaomi MiMo · Qianfan · NanoGPT · Venice · Kilo · ZenMux · OpenCode Go · OpenCode Zen
|
||||
|
||||
### Run it yourself
|
||||
|
||||
|
||||
@@ -82,7 +82,6 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
|
||||
| `DEEPSEEK_API_KEY` | DeepSeek auth | Using DeepSeek models | |
|
||||
| `KILO_API_KEY` | Kilo auth | Using Kilo models | |
|
||||
| `OLLAMA_CLOUD_API_KEY` | Ollama Cloud auth | Using `ollama-cloud` provider | |
|
||||
| `WAFER_PASS_API_KEY` | Wafer Pass auth | Using `wafer-pass` provider | Flat-rate Wafer subscription; validated against `https://pass.wafer.ai/v1/models` |
|
||||
| `WAFER_SERVERLESS_API_KEY` | Wafer Serverless auth | Using `wafer-serverless` provider | Pay-as-you-go Wafer SKU; validated against `https://pass.wafer.ai/v1/models` |
|
||||
| `GITLAB_TOKEN` | GitLab Duo auth | Using `gitlab-duo` provider | |
|
||||
|
||||
|
||||
+1
-2
@@ -123,7 +123,6 @@ Each provider has one or more environment variables that supply a key when no st
|
||||
| `gitlab-duo` | `GITLAB_TOKEN` |
|
||||
| `opencode-zen`, `opencode-go` | `OPENCODE_API_KEY` |
|
||||
| `firepass` | `FIREPASS_API_KEY` |
|
||||
| `wafer-pass` | `WAFER_PASS_API_KEY` |
|
||||
| `wafer-serverless` | `WAFER_SERVERLESS_API_KEY` |
|
||||
| `xiaomi` | `XIAOMI_API_KEY` |
|
||||
| `ollama-cloud` | `OLLAMA_CLOUD_API_KEY` |
|
||||
@@ -131,7 +130,7 @@ Each provider has one or more environment variables that supply a key when no st
|
||||
| `lm-studio` | `LM_STUDIO_API_KEY` (optional; keyless by default) |
|
||||
| `llama.cpp` | `LLAMA_CPP_API_KEY` (only when the server requires auth) |
|
||||
|
||||
OAuth-backed providers such as `anthropic`, `github-copilot`, `cursor`, `ollama-cloud`, `qwen-portal`, `kimi-code`, `xai-oauth`, `wafer-pass`, `wafer-serverless`, `google-gemini-cli`, and `google-antigravity` are normally reached through `/login` rather than an environment variable. See [Environment variables](./environment-variables.md) for search-tool and configuration variables not listed here.
|
||||
OAuth-backed providers such as `anthropic`, `github-copilot`, `cursor`, `ollama-cloud`, `qwen-portal`, `kimi-code`, `xai-oauth`, `wafer-serverless`, `google-gemini-cli`, and `google-antigravity` are normally reached through `/login` rather than an environment variable. See [Environment variables](./environment-variables.md) for search-tool and configuration variables not listed here.
|
||||
|
||||
### `.env` discovery and precedence
|
||||
|
||||
|
||||
@@ -10,6 +10,10 @@
|
||||
|
||||
- Added `llama.cpp` to the interactive `/login` provider list, accepting an optional API key while defaulting to local no-auth mode.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed Wafer Pass (`wafer-pass`) login support; Wafer Serverless remains available as `wafer-serverless`.
|
||||
|
||||
## [16.1.8] - 2026-06-20
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -62,7 +62,6 @@ Unified LLM API with automatic model discovery, provider configuration, token an
|
||||
- **Hugging Face Inference**
|
||||
- **xAI**
|
||||
- **Venice** (requires `VENICE_API_KEY`)
|
||||
- **Wafer Pass** (requires `WAFER_PASS_API_KEY`; flat-rate subscription, includes GLM-5.1 and Qwen3.5-397B-A17B)
|
||||
- **Wafer Serverless** (requires `WAFER_SERVERLESS_API_KEY`; pay-as-you-go)
|
||||
- **OpenRouter**
|
||||
- **Kilo Gateway** (supports OAuth `/login kilo` or `KILO_API_KEY`)
|
||||
|
||||
@@ -1,41 +1,15 @@
|
||||
/**
|
||||
* Wafer login flows.
|
||||
* Wafer Serverless login flow.
|
||||
*
|
||||
* Wafer (https://wafer.ai) exposes a single OpenAI-compatible base URL
|
||||
* (`https://pass.wafer.ai/v1`) for two SKUs:
|
||||
*
|
||||
* - **Wafer Pass** — flat-rate subscription. The key authorizes models whose
|
||||
* catalog entries carry `wafer.tier = "pass_included"`.
|
||||
* - **Wafer Serverless** — pay-as-you-go. Superset of Pass; the same `/v1/models`
|
||||
* endpoint returns the full per-account model list.
|
||||
*
|
||||
* Both SKUs issue `wfr_…` keys. The key prefix alone does not distinguish
|
||||
* tiers — the entitlement is per-account on the server side — so we expose
|
||||
* two parallel logins / env vars (`WAFER_PASS_API_KEY`, `WAFER_SERVERLESS_API_KEY`)
|
||||
* mirroring the firepass/fireworks split, letting users with both
|
||||
* subscriptions switch between them without re-pasting.
|
||||
*
|
||||
* Validation uses the shared `/v1/models` endpoint, which works for both
|
||||
* tiers and is cheap (no token spend).
|
||||
* Wafer (https://wafer.ai) exposes a pay-as-you-go OpenAI-compatible SKU at
|
||||
* `https://pass.wafer.ai/v1`. Keys use the `wfr_…` prefix and are validated
|
||||
* against `/v1/models`, which is cheap (no token spend).
|
||||
*/
|
||||
import { createApiKeyLogin } from "../api-key-login";
|
||||
|
||||
const WAFER_AUTH_URL = "https://wafer.ai/dashboard";
|
||||
const WAFER_MODELS_URL = "https://pass.wafer.ai/v1/models";
|
||||
|
||||
export const loginWaferPass = createApiKeyLogin({
|
||||
providerLabel: "Wafer Pass",
|
||||
authUrl: WAFER_AUTH_URL,
|
||||
instructions: "Create or copy your Wafer Pass API key from the Wafer dashboard",
|
||||
promptMessage: "Paste your Wafer Pass API key",
|
||||
placeholder: "wfr_...",
|
||||
validation: {
|
||||
kind: "models-endpoint",
|
||||
provider: "Wafer Pass",
|
||||
modelsUrl: WAFER_MODELS_URL,
|
||||
},
|
||||
});
|
||||
|
||||
export const loginWaferServerless = createApiKeyLogin({
|
||||
providerLabel: "Wafer Serverless",
|
||||
authUrl: WAFER_AUTH_URL,
|
||||
|
||||
@@ -51,7 +51,6 @@ import { umansProvider } from "./umans";
|
||||
import { veniceProvider } from "./venice";
|
||||
import { vercelAiGatewayProvider } from "./vercel-ai-gateway";
|
||||
import { vllmProvider } from "./vllm";
|
||||
import { waferPassProvider } from "./wafer-pass";
|
||||
import { waferServerlessProvider } from "./wafer-serverless";
|
||||
import { xaiProvider } from "./xai";
|
||||
import { xaiOauthProvider } from "./xai-oauth";
|
||||
@@ -96,7 +95,6 @@ const ALL = [
|
||||
xiaomiTokenPlanAmsProvider,
|
||||
xiaomiTokenPlanCnProvider,
|
||||
firepassProvider,
|
||||
waferPassProvider,
|
||||
deepseekProvider,
|
||||
moonshotProvider,
|
||||
cerebrasProvider,
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
import type { OAuthLoginCallbacks } from "./oauth/types";
|
||||
import type { ProviderDefinition } from "./types";
|
||||
|
||||
export const waferPassProvider = {
|
||||
id: "wafer-pass",
|
||||
name: "Wafer Pass (flat-rate subscription)",
|
||||
login: async (cb: OAuthLoginCallbacks) => {
|
||||
// Lazy import: keep heavy OAuth flow modules out of the eager registry graph.
|
||||
const { loginWaferPass } = await import("./oauth/wafer");
|
||||
return loginWaferPass(cb);
|
||||
},
|
||||
} as const satisfies ProviderDefinition;
|
||||
@@ -1,25 +1,24 @@
|
||||
/**
|
||||
* Live Wafer Pass smoke. NOT part of the bun test suite — run manually:
|
||||
* WAFER_PASS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts
|
||||
* Live Wafer Serverless smoke. NOT part of the bun test suite — run manually:
|
||||
* WAFER_SERVERLESS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts
|
||||
*
|
||||
* Validates that the bundled `wafer-pass/GLM-5.1` entry round-trips a real
|
||||
* streaming chat completion against `https://pass.wafer.ai/v1`, with the wire
|
||||
* `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a non-empty
|
||||
* assistant text returned.
|
||||
* Validates that the bundled `wafer-serverless/GLM-5.1` entry round-trips a
|
||||
* real streaming chat completion against `https://pass.wafer.ai/v1`, with the
|
||||
* wire `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a
|
||||
* non-empty assistant text returned.
|
||||
*/
|
||||
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
const apiKey = process.env.WAFER_PASS_API_KEY ?? process.env.WAFER_SERVERLESS_API_KEY;
|
||||
const apiKey = process.env.WAFER_SERVERLESS_API_KEY;
|
||||
if (!apiKey) {
|
||||
console.error("WAFER_PASS_API_KEY (or WAFER_SERVERLESS_API_KEY) env var is required");
|
||||
console.error("WAFER_SERVERLESS_API_KEY env var is required");
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const providerId = process.env.WAFER_PASS_API_KEY ? "wafer-pass" : "wafer-serverless";
|
||||
const model = getBundledModel<"openai-completions">(providerId, "GLM-5.1");
|
||||
const model = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1");
|
||||
console.log(`Model: ${model.provider}/${model.id} -> ${model.baseUrl}`);
|
||||
console.log(`compat.thinkingFormat: ${model.compat?.thinkingFormat ?? "(none)"}`);
|
||||
|
||||
@@ -89,6 +88,6 @@ if (text.trim().length === 0) {
|
||||
}
|
||||
|
||||
console.log(
|
||||
`\nLIVE OK — Wafer ${providerId} round-trip: GLM-5.1 endpoint preserved, ` +
|
||||
`\nLIVE OK — Wafer Serverless round-trip: GLM-5.1 endpoint preserved, ` +
|
||||
`${inputTokens}→${outputTokens} tokens, stopReason=${stopReason}.`,
|
||||
);
|
||||
|
||||
@@ -6,6 +6,10 @@
|
||||
|
||||
- Fixed the `moonshot` provider with no path to the Kimi China API: model discovery now honors a `MOONSHOT_BASE_URL` override (redirecting to `api.moonshot.cn`), and `KIMI_API_KEY` resolves as a fallback for `MOONSHOT_API_KEY`. ([#2883](https://github.com/can1357/oh-my-pi/issues/2883))
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed bundled Wafer Pass (`wafer-pass`) catalog entries and generation support; Wafer Serverless remains available as `wafer-serverless`.
|
||||
|
||||
## [16.1.8] - 2026-06-20
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -61,6 +61,7 @@ const packageRoot = path.join(import.meta.dir, "..");
|
||||
* and never written to models.json.
|
||||
*/
|
||||
const DISCOVERY_ONLY_PROVIDERS = new Set(["ollama", "vllm", "lm-studio", "litellm"]);
|
||||
const RETIRED_PROVIDERS = new Set(["wafer-pass"]);
|
||||
|
||||
async function resolveProviderApiKey(providerId: string, catalog: CatalogDiscoveryConfig): Promise<string | undefined> {
|
||||
for (const envVar of catalog.envVars ?? []) {
|
||||
@@ -532,6 +533,7 @@ async function generateModels() {
|
||||
if (
|
||||
!fetchedKeys.has(`${model.provider}/${model.id}`) &&
|
||||
!DISCOVERY_ONLY_PROVIDERS.has(model.provider) &&
|
||||
!RETIRED_PROVIDERS.has(model.provider) &&
|
||||
!authoritativeCatalogProviders.has(model.provider) &&
|
||||
!modelsDevSnapshotExcludedProviders.has(model.provider)
|
||||
) {
|
||||
@@ -569,7 +571,7 @@ async function generateModels() {
|
||||
// Group by provider and sort each provider's models
|
||||
const providers: Record<string, Record<string, ModelSpec>> = {};
|
||||
for (const model of allModels) {
|
||||
if (DISCOVERY_ONLY_PROVIDERS.has(model.provider)) continue;
|
||||
if (DISCOVERY_ONLY_PROVIDERS.has(model.provider) || RETIRED_PROVIDERS.has(model.provider)) continue;
|
||||
if (!providers[model.provider]) {
|
||||
providers[model.provider] = {};
|
||||
}
|
||||
|
||||
@@ -77590,64 +77590,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"wafer-pass": {
|
||||
"GLM-5.1": {
|
||||
"id": "GLM-5.1",
|
||||
"name": "GLM-5.1",
|
||||
"api": "openai-completions",
|
||||
"provider": "wafer-pass",
|
||||
"baseUrl": "https://pass.wafer.ai/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 202752,
|
||||
"maxTokens": 65536,
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false,
|
||||
"thinkingFormat": "zai",
|
||||
"reasoningContentField": "reasoning_content"
|
||||
},
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
]
|
||||
}
|
||||
},
|
||||
"Qwen3.5-397B-A17B": {
|
||||
"id": "Qwen3.5-397B-A17B",
|
||||
"name": "Qwen3.5-397B-A17B",
|
||||
"api": "openai-completions",
|
||||
"provider": "wafer-pass",
|
||||
"baseUrl": "https://pass.wafer.ai/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 65536,
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false
|
||||
}
|
||||
}
|
||||
},
|
||||
"wafer-serverless": {
|
||||
"deepseek-v4-flash": {
|
||||
"id": "deepseek-v4-flash",
|
||||
|
||||
@@ -41,7 +41,6 @@ import {
|
||||
veniceModelManagerOptions,
|
||||
vercelAiGatewayModelManagerOptions,
|
||||
vllmModelManagerOptions,
|
||||
waferPassModelManagerOptions,
|
||||
waferServerlessModelManagerOptions,
|
||||
xaiModelManagerOptions,
|
||||
xaiOAuthModelManagerOptions,
|
||||
@@ -349,13 +348,6 @@ export const CATALOG_PROVIDERS = [
|
||||
createModelManagerOptions: (config: ModelManagerConfig) => vllmModelManagerOptions(config),
|
||||
catalogDiscovery: { label: "vLLM", allowUnauthenticated: true },
|
||||
},
|
||||
{
|
||||
id: "wafer-pass",
|
||||
defaultModel: "GLM-5.1",
|
||||
envVars: ["WAFER_PASS_API_KEY"],
|
||||
createModelManagerOptions: (config: ModelManagerConfig) => waferPassModelManagerOptions(config),
|
||||
catalogDiscovery: { label: "Wafer Pass", oauthProvider: "wafer-pass" },
|
||||
},
|
||||
{
|
||||
id: "wafer-serverless",
|
||||
defaultModel: "GLM-5.1",
|
||||
|
||||
@@ -1568,7 +1568,7 @@ export function firepassModelManagerOptions(
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7.7 Wafer (Pass + Serverless)
|
||||
// 7.7 Wafer Serverless
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export interface WaferModelManagerConfig {
|
||||
@@ -1581,13 +1581,14 @@ const WAFER_DEFAULT_BASE_URL = "https://pass.wafer.ai/v1";
|
||||
const WAFER_MAX_TOKENS_CAP = 65536;
|
||||
|
||||
/**
|
||||
* Shared mapper for Wafer's `/v1/models` records.
|
||||
* Mapper for Wafer Serverless `/v1/models` records.
|
||||
*
|
||||
* Wafer wraps each entry with a `wafer` envelope describing tier, capabilities,
|
||||
* and cents-per-million pricing. The mapper folds that metadata into the
|
||||
* canonical `ModelSpec<"openai-completions">` shape and applies zai-family thinking
|
||||
* compat when the entry advertises reasoning support (GLM-family on the Pass
|
||||
* SKU). Cents-per-million → dollars-per-million via /100.
|
||||
* Wafer wraps each entry with a `wafer` envelope describing capabilities and
|
||||
* pricing. The mapper folds that metadata into the canonical
|
||||
* `ModelSpec<"openai-completions">` shape and applies upstream-specific thinking
|
||||
* compat when the entry advertises reasoning support. Wafer pricing is exposed
|
||||
* through internal wholesale units; the public Serverless rate equals
|
||||
* `cents × 125 / 10000`.
|
||||
*/
|
||||
interface WaferRecord {
|
||||
context_length?: unknown;
|
||||
@@ -1608,7 +1609,7 @@ function readWaferRecord(entry: OpenAICompatibleModelRecord): WaferRecord | unde
|
||||
}
|
||||
|
||||
function mapWaferModel(
|
||||
providerId: "wafer-pass" | "wafer-serverless",
|
||||
providerId: "wafer-serverless",
|
||||
baseUrl: string,
|
||||
entry: OpenAICompatibleModelRecord,
|
||||
defaults: ModelSpec<"openai-completions">,
|
||||
@@ -1624,25 +1625,12 @@ function mapWaferModel(
|
||||
);
|
||||
const maxTokens = contextWindow !== null ? Math.min(contextWindow, WAFER_MAX_TOKENS_CAP) : null;
|
||||
const pricing = wafer?.pricing ?? {};
|
||||
// Wafer's `/v1/models` exposes pricing through `*_cents_per_million` fields,
|
||||
// but the values are an internal wholesale unit, not literal cents — across
|
||||
// every published Serverless model on wafer.ai the user-facing rate equals
|
||||
// `cents × 125 / 10000` (i.e. wholesale × 1.25 / 100; GLM-5.1's `120` →
|
||||
// $1.50/M, Kimi-K2.6's `88` → $1.10/M, etc.). The multiply-first form keeps
|
||||
// the result a finite dyadic for every observed value.
|
||||
// For the Pass SKU the per-token rate is bundled in the flat-rate
|
||||
// subscription, so we follow the convention shared with
|
||||
// `kimi-code`/`firepass`/`alibaba-coding-plan` and seed every Pass model with
|
||||
// `cost: 0` regardless of what the upstream envelope says.
|
||||
const isPassSku = providerId === "wafer-pass";
|
||||
const cost = isPassSku
|
||||
? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }
|
||||
: {
|
||||
input: (toPositiveNumber(pricing.input_cents_per_million, 0) * 125) / 10000,
|
||||
output: (toPositiveNumber(pricing.output_cents_per_million, 0) * 125) / 10000,
|
||||
cacheRead: (toPositiveNumber(pricing.cache_read_cents_per_million, 0) * 125) / 10000,
|
||||
cacheWrite: 0,
|
||||
};
|
||||
const cost = {
|
||||
input: (toPositiveNumber(pricing.input_cents_per_million, 0) * 125) / 10000,
|
||||
output: (toPositiveNumber(pricing.output_cents_per_million, 0) * 125) / 10000,
|
||||
cacheRead: (toPositiveNumber(pricing.cache_read_cents_per_million, 0) * 125) / 10000,
|
||||
cacheWrite: 0,
|
||||
};
|
||||
const name = toModelName(wafer?.display_name, defaults.name);
|
||||
const base: ModelSpec<"openai-completions"> = {
|
||||
...defaults,
|
||||
@@ -1688,13 +1676,12 @@ function mapWaferModel(
|
||||
};
|
||||
}
|
||||
|
||||
function createWaferOptions(
|
||||
providerId: "wafer-pass" | "wafer-serverless",
|
||||
config: WaferModelManagerConfig | undefined,
|
||||
export function waferServerlessModelManagerOptions(
|
||||
config?: WaferModelManagerConfig,
|
||||
): ModelManagerOptions<"openai-completions"> {
|
||||
const apiKey = config?.apiKey;
|
||||
const baseUrl = config?.baseUrl ?? WAFER_DEFAULT_BASE_URL;
|
||||
const passOnly = providerId === "wafer-pass";
|
||||
const providerId = "wafer-serverless" as const;
|
||||
return {
|
||||
providerId,
|
||||
...(apiKey && {
|
||||
@@ -1704,11 +1691,6 @@ function createWaferOptions(
|
||||
provider: providerId,
|
||||
baseUrl,
|
||||
apiKey,
|
||||
filterModel: entry => {
|
||||
if (!passOnly) return true;
|
||||
const wafer = readWaferRecord(entry);
|
||||
return wafer?.tier === "pass_included";
|
||||
},
|
||||
mapModel: (entry, defaults) => mapWaferModel(providerId, baseUrl, entry, defaults),
|
||||
fetch: config?.fetch,
|
||||
}),
|
||||
@@ -1716,18 +1698,6 @@ function createWaferOptions(
|
||||
};
|
||||
}
|
||||
|
||||
export function waferPassModelManagerOptions(
|
||||
config?: WaferModelManagerConfig,
|
||||
): ModelManagerOptions<"openai-completions"> {
|
||||
return createWaferOptions("wafer-pass", config);
|
||||
}
|
||||
|
||||
export function waferServerlessModelManagerOptions(
|
||||
config?: WaferModelManagerConfig,
|
||||
): ModelManagerOptions<"openai-completions"> {
|
||||
return createWaferOptions("wafer-serverless", config);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Mistral
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -1,90 +1,17 @@
|
||||
/**
|
||||
* Wafer Pass + Wafer Serverless provider wiring.
|
||||
* Wafer Serverless provider wiring.
|
||||
*
|
||||
* Wafer exposes a single OpenAI-compatible base URL (`https://pass.wafer.ai/v1`)
|
||||
* for two SKUs whose entitlement differs server-side:
|
||||
* - `wafer-pass` (flat-rate)
|
||||
* - `wafer-serverless` (pay-as-you-go)
|
||||
*
|
||||
* Both providers route through `openai-completions` and the catalog id matches
|
||||
* the wire id (no rewrite). These tests defend the bundled catalog contract and
|
||||
* the case-sensitive id pass-through against the wire.
|
||||
* Wafer Serverless exposes an OpenAI-compatible base URL
|
||||
* (`https://pass.wafer.ai/v1`) and routes through `openai-completions`. The
|
||||
* catalog id matches the wire id (no rewrite). These tests defend the bundled
|
||||
* catalog contract and the case-sensitive id pass-through against the wire.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { Context } from "@oh-my-pi/pi-ai/types";
|
||||
import { createModelManager } from "@oh-my-pi/pi-catalog/model-manager";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import {
|
||||
waferPassModelManagerOptions,
|
||||
waferServerlessModelManagerOptions,
|
||||
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import { waferServerlessModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function sseResponse(events: unknown[]): Response {
|
||||
const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`;
|
||||
return new Response(payload, {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
}
|
||||
|
||||
describe("Wafer Pass provider", () => {
|
||||
it("ships a bundled GLM-5.1 entry with zai-family thinking compat", () => {
|
||||
const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1");
|
||||
expect(model).toBeDefined();
|
||||
expect(model.id).toBe("GLM-5.1");
|
||||
expect(model.provider).toBe("wafer-pass");
|
||||
expect(model.api).toBe("openai-completions");
|
||||
expect(model.baseUrl).toBe("https://pass.wafer.ai/v1");
|
||||
expect(model.reasoning).toBe(true);
|
||||
expect(model.input).toEqual(["text"]);
|
||||
expect(model.compatConfig?.thinkingFormat).toBe("zai");
|
||||
expect(model.compatConfig?.reasoningContentField).toBe("reasoning_content");
|
||||
expect(model.compatConfig?.supportsDeveloperRole).toBe(false);
|
||||
});
|
||||
|
||||
it("ships a bundled Qwen3.5-397B-A17B entry with vision input and no reasoning", () => {
|
||||
const model = getBundledModel<"openai-completions">("wafer-pass", "Qwen3.5-397B-A17B");
|
||||
expect(model).toBeDefined();
|
||||
expect(model.id).toBe("Qwen3.5-397B-A17B");
|
||||
expect(model.provider).toBe("wafer-pass");
|
||||
expect(model.reasoning).toBe(false);
|
||||
expect(model.input).toEqual(["text", "image"]);
|
||||
});
|
||||
|
||||
it("preserves the catalog id verbatim on the wire (no rewrite, case-sensitive)", async () => {
|
||||
const model = getBundledModel<"openai-completions">("wafer-pass", "GLM-5.1");
|
||||
const captured: { url: string | null; body: string | null } = { url: null, body: null };
|
||||
const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => {
|
||||
captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
||||
captured.body = typeof init?.body === "string" ? init.body : null;
|
||||
return sseResponse(["[DONE]"]);
|
||||
};
|
||||
|
||||
const context: Context = {
|
||||
systemPrompt: ["t"],
|
||||
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
|
||||
};
|
||||
const stream = streamOpenAICompletions(model as Model<"openai-completions">, context, {
|
||||
apiKey: "wfr_test",
|
||||
fetch: fetchMock,
|
||||
});
|
||||
for await (const _event of stream) {
|
||||
/* drain */
|
||||
}
|
||||
|
||||
expect(captured.url).toBe("https://pass.wafer.ai/v1/chat/completions");
|
||||
expect(captured.body).not.toBeNull();
|
||||
const parsed = JSON.parse(captured.body ?? "{}") as { model?: unknown };
|
||||
// Wafer's docs note model names are case-insensitive on input, but the
|
||||
// canonical id has mixed case; we must round-trip it unchanged so users
|
||||
// who pin `GLM-5.1` don't end up with usage rows under `glm-5.1` or
|
||||
// hitting the upstream 404 path.
|
||||
expect(parsed.model).toBe("GLM-5.1");
|
||||
});
|
||||
});
|
||||
|
||||
describe("Wafer Serverless provider", () => {
|
||||
it("ships the documented Serverless catalog (GLM-5.1, Qwen3.5, Qwen3.6, Qwen3.7-Max, Kimi-K2.6, DeepSeek V4 Flash/Pro)", () => {
|
||||
const glm = getBundledModel<"openai-completions">("wafer-serverless", "GLM-5.1");
|
||||
@@ -143,15 +70,8 @@ describe("Wafer Serverless provider", () => {
|
||||
expect(dsPro.reasoning).toBe(true);
|
||||
expect(dsPro.compatConfig?.thinkingFormat).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not expose Serverless-only ids on the Wafer Pass catalog", () => {
|
||||
expect(getBundledModel("wafer-pass", "Kimi-K2.6")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "Qwen3.6-35B-A3B")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "qwen3.7-max")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "deepseek-v4-flash")).toBeUndefined();
|
||||
expect(getBundledModel("wafer-pass", "deepseek-v4-pro")).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("Wafer dynamic discovery mapper", () => {
|
||||
// Synthetic /v1/models response that exercises every upstream provider Wafer
|
||||
// announces, including a deliberately-unknown one. The mapper must:
|
||||
@@ -224,11 +144,9 @@ describe("Wafer dynamic discovery mapper", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("zeros cost for the Pass SKU and applies retail × 0.0125 for Serverless", async () => {
|
||||
// Same upstream record served via both SKUs — `wafer.pricing` in cents/M:
|
||||
// 120/360/12. Pass is a flat-rate subscription (no per-token charge), so
|
||||
// `mapWaferModel` zeros the cost regardless of envelope values. Serverless
|
||||
// is pay-as-you-go and applies the empirical × 0.0125 conversion to match
|
||||
it("applies retail × 0.0125 for Serverless pricing", async () => {
|
||||
// `wafer.pricing` is in Wafer's internal cents/M unit. Serverless is
|
||||
// pay-as-you-go and applies the empirical × 0.0125 conversion to match
|
||||
// wafer.ai's published retail rates (120 cents → $1.50/M).
|
||||
const sharedEntry = {
|
||||
id: "Shared-fake",
|
||||
@@ -249,13 +167,6 @@ describe("Wafer dynamic discovery mapper", () => {
|
||||
},
|
||||
};
|
||||
|
||||
const fetchMock = mockWaferModelsResponse([sharedEntry]);
|
||||
const passManager = createModelManager(waferPassModelManagerOptions({ apiKey: "wfr_test", fetch: fetchMock }));
|
||||
const passResult = await passManager.refresh("online");
|
||||
const passModel = passResult.models.find(m => m.id === "Shared-fake");
|
||||
expect(passModel).toBeDefined();
|
||||
expect(passModel?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
|
||||
|
||||
const srvFetchMock = mockWaferModelsResponse([sharedEntry]);
|
||||
const srvManager = createModelManager(
|
||||
waferServerlessModelManagerOptions({ apiKey: "wfr_test", fetch: srvFetchMock }),
|
||||
|
||||
@@ -9,6 +9,14 @@
|
||||
- Fixed `omp --approval-mode=yolo acp` and other global option flags placed before a subcommand being rewritten to `launch` with the subcommand swallowed as prompt text; the CLI resolver now skips leading global flags (using the launch parser's value-consumption contract) and dispatches the real subcommand with the flags applied, so ACP mode honors the configured approval policy. ([#2970](https://github.com/can1357/oh-my-pi/issues/2970))
|
||||
- Fixed `/mcp enable` and `/mcp disable` reconnecting unrelated MCP servers by scoping toggle reconnect/disconnect work to the named server. ([#3157](https://github.com/can1357/oh-my-pi/issues/3157))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Stopped the local llama.cpp no-auth login placeholder from being sent as a discovery bearer token.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed Wafer Pass from CLI credential help; Wafer Serverless remains available.
|
||||
|
||||
## [16.1.8] - 2026-06-20
|
||||
|
||||
### Added
|
||||
|
||||
@@ -298,7 +298,6 @@ export function getExtraHelpText(): string {
|
||||
OPENCODE_API_KEY - OpenCode Zen/OpenCode Go models
|
||||
CURSOR_ACCESS_TOKEN - Cursor AI models
|
||||
AI_GATEWAY_API_KEY - Vercel AI Gateway
|
||||
WAFER_PASS_API_KEY - Wafer Pass (flat-rate subscription; GLM-5.1, Qwen3.5)
|
||||
WAFER_SERVERLESS_API_KEY - Wafer Serverless (pay-as-you-go)
|
||||
|
||||
${chalk.dim("# Cloud Providers")}
|
||||
|
||||
@@ -25,8 +25,9 @@ import {
|
||||
} from "@oh-my-pi/pi-catalog/variant-collapse";
|
||||
|
||||
// Sentinels for local-only OAuth tokens — declared inline to avoid loading
|
||||
// provider modules at startup. Must match packages/ai/src/registry/lm-studio.ts
|
||||
// and packages/ai/src/registry/vllm.ts.
|
||||
// provider modules at startup. Must match packages/ai/src/registry/llama-cpp.ts,
|
||||
// packages/ai/src/registry/lm-studio.ts, and packages/ai/src/registry/vllm.ts.
|
||||
const DEFAULT_LLAMA_CPP_LOCAL_TOKEN = "llama-cpp-local";
|
||||
const DEFAULT_LOCAL_TOKEN = "lm-studio-local";
|
||||
const DEFAULT_VLLM_LOCAL_TOKEN = "vllm-local";
|
||||
|
||||
@@ -85,7 +86,12 @@ export function isAuthenticated(apiKey: string | undefined | null): apiKey is st
|
||||
}
|
||||
|
||||
function isDiscoveryBearerApiKey(apiKey: string | undefined | null): apiKey is string {
|
||||
return isAuthenticated(apiKey) && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== DEFAULT_VLLM_LOCAL_TOKEN;
|
||||
return (
|
||||
isAuthenticated(apiKey) &&
|
||||
apiKey !== DEFAULT_LLAMA_CPP_LOCAL_TOKEN &&
|
||||
apiKey !== DEFAULT_LOCAL_TOKEN &&
|
||||
apiKey !== DEFAULT_VLLM_LOCAL_TOKEN
|
||||
);
|
||||
}
|
||||
|
||||
/** Provider override config (baseUrl, headers, apiKey, compat, transport) without custom models */
|
||||
|
||||
@@ -470,4 +470,50 @@ describe("issue #970 custom provider discovery", () => {
|
||||
|
||||
expect(registry.getProviderDiscoveryState("vllm")?.status).toBe("ok");
|
||||
});
|
||||
|
||||
test("does not send llama.cpp-local placeholder as discovery bearer", async () => {
|
||||
fs.writeFileSync(
|
||||
modelsPath,
|
||||
[
|
||||
"providers:",
|
||||
" llama.cpp:",
|
||||
" baseUrl: http://127.0.0.1:8080",
|
||||
" apiKey: llama-cpp-local",
|
||||
" api: openai-responses",
|
||||
" discovery:",
|
||||
" type: llama.cpp",
|
||||
].join("\n"),
|
||||
);
|
||||
|
||||
const fetchMock: (input: string | URL | Request, init?: RequestInit) => Promise<Response> = async (
|
||||
input,
|
||||
init,
|
||||
) => {
|
||||
const url = String(input);
|
||||
if (url === "http://127.0.0.1:8080/props") {
|
||||
const headers = init?.headers as Headers | Record<string, string> | undefined;
|
||||
const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization;
|
||||
expect(authHeader).toBeUndefined();
|
||||
return new Response(JSON.stringify({ default_generation_settings: { n_ctx: 8192 } }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
if (url !== "http://127.0.0.1:8080/models") {
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
}
|
||||
const headers = init?.headers as Headers | Record<string, string> | undefined;
|
||||
const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization;
|
||||
expect(authHeader).toBeUndefined();
|
||||
return new Response(JSON.stringify({ data: [{ id: "local-llama" }] }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
};
|
||||
|
||||
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
|
||||
await registry.refreshProvider("llama.cpp");
|
||||
|
||||
expect(registry.getProviderDiscoveryState("llama.cpp")?.status).toBe("ok");
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user