f6ca76728b
Wafer (https://wafer.ai) exposes a single OpenAI-compatible endpoint (`https://pass.wafer.ai/v1`) for two SKUs whose entitlement differs server-side, so we model them as two parallel providers — mirroring the firepass/fireworks split so a user with both subscriptions can switch without re-pasting: - `wafer-pass` — flat-rate. `/v1/models` is filtered to entries whose `wafer.tier === "pass_included"`. - `wafer-serverless` — pay-as-you-go superset of Pass. Both issue `wfr_…` keys. `/login wafer-pass` and `/login wafer-serverless` paste-and-validate via `/v1/models`. `WAFER_PASS_API_KEY` and `WAFER_SERVERLESS_API_KEY` are wired through `getEnvApiKey`. Bundled catalog: - `wafer-pass`: GLM-5.1, Qwen3.5-397B-A17B. - `wafer-serverless`: GLM-5.1, Qwen3.5-397B-A17B, Kimi-K2.6, Qwen3.6-35B-A3B. Dynamic discovery via `/v1/models` overlays additional models at runtime and folds the `wafer` envelope (tier, capabilities, cents/M pricing) into the canonical `Model<"openai-completions">` shape. GLM-family entries carry the zai-style thinking compat (`thinkingFormat: "zai"`, `reasoningContentField: "reasoning_content"`) so reasoning tokens land in the right field. Cents-per-million → dollars-per-million via /100. Tests (`packages/ai/test/wafer.test.ts`, 5 cases): bundled catalog contract for both providers and wire-id pass-through (case-sensitive, no rewrite — `GLM-5.1` must round-trip verbatim or upstream 404s). Optional `packages/ai/test/wafer.live.ts` exercises a real round-trip against `pass.wafer.ai` when `WAFER_PASS_API_KEY` is set.
94 lines
3.6 KiB
TypeScript
94 lines
3.6 KiB
TypeScript
/**
|
|
* Live Wafer Pass smoke. NOT part of the bun test suite — run manually:
|
|
* WAFER_PASS_API_KEY=wfr_... bun packages/ai/test/wafer.live.ts
|
|
*
|
|
* Validates that the bundled `wafer-pass/GLM-5.1` entry round-trips a real
|
|
* streaming chat completion against `https://pass.wafer.ai/v1`, with the wire
|
|
* `model` field preserved verbatim (`GLM-5.1`, not lowercased) and a non-empty
|
|
* assistant text returned.
|
|
*/
|
|
import { getBundledModel } from "../src/models";
|
|
import { streamOpenAICompletions } from "../src/providers/openai-completions";
|
|
import type { Context, Model } from "../src/types";
|
|
|
|
const apiKey = process.env.WAFER_PASS_API_KEY ?? process.env.WAFER_SERVERLESS_API_KEY;
|
|
if (!apiKey) {
|
|
console.error("WAFER_PASS_API_KEY (or WAFER_SERVERLESS_API_KEY) env var is required");
|
|
process.exit(2);
|
|
}
|
|
|
|
const providerId = process.env.WAFER_PASS_API_KEY ? "wafer-pass" : "wafer-serverless";
|
|
const model = getBundledModel<"openai-completions">(providerId, "GLM-5.1");
|
|
console.log(`Model: ${model.provider}/${model.id} -> ${model.baseUrl}`);
|
|
console.log(`compat.thinkingFormat: ${model.compat?.thinkingFormat ?? "(none)"}`);
|
|
|
|
interface CapturedRequest {
|
|
url: string;
|
|
body: string | null;
|
|
}
|
|
|
|
const originalFetch = global.fetch;
|
|
const captured: { value: CapturedRequest | null } = { value: null };
|
|
type FetchInput = Parameters<typeof fetch>[0];
|
|
global.fetch = (async (input: FetchInput, init?: RequestInit) => {
|
|
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
|
captured.value = { url, body: typeof init?.body === "string" ? init.body : null };
|
|
return originalFetch(input as Parameters<typeof fetch>[0], init);
|
|
}) as typeof global.fetch;
|
|
|
|
const context: Context = {
|
|
systemPrompt: ["Reply with exactly two words."],
|
|
messages: [{ role: "user", content: "Say hi.", timestamp: Date.now() }],
|
|
};
|
|
|
|
const stream = streamOpenAICompletions(model as Model<"openai-completions">, context, { apiKey });
|
|
let text = "";
|
|
let stopReason: string | undefined;
|
|
let cost = 0;
|
|
let firstError: unknown;
|
|
let inputTokens = 0;
|
|
let outputTokens = 0;
|
|
for await (const ev of stream) {
|
|
if (ev.type === "text_delta") text += ev.delta;
|
|
else if (ev.type === "done") {
|
|
stopReason = ev.reason;
|
|
const usage = ev.message.usage;
|
|
cost = usage?.cost?.total ?? 0;
|
|
inputTokens = usage?.input ?? 0;
|
|
outputTokens = usage?.output ?? 0;
|
|
} else if (ev.type === "error") {
|
|
firstError = ev.error.errorMessage ?? ev.error;
|
|
stopReason = ev.reason;
|
|
}
|
|
}
|
|
|
|
const snapshot = (captured as { value: CapturedRequest | null }).value;
|
|
const parsedBody = snapshot?.body ? (JSON.parse(snapshot.body) as { model?: unknown }) : null;
|
|
console.log("wire url:", snapshot?.url);
|
|
console.log("wire model:", parsedBody?.model);
|
|
console.log("text:", JSON.stringify(text.slice(0, 200)));
|
|
console.log("stopReason:", stopReason);
|
|
console.log("usage:", { input: inputTokens, output: outputTokens, costUSD: cost });
|
|
|
|
if (firstError) {
|
|
console.error("\nLIVE FAIL — Wafer rejected the request:", firstError);
|
|
process.exit(1);
|
|
}
|
|
if (snapshot?.url !== "https://pass.wafer.ai/v1/chat/completions") {
|
|
console.error("\nLIVE FAIL — wire url was not the documented endpoint");
|
|
process.exit(1);
|
|
}
|
|
if (parsedBody?.model !== "GLM-5.1") {
|
|
console.error("\nLIVE FAIL — wire model id was not preserved verbatim:", parsedBody?.model);
|
|
process.exit(1);
|
|
}
|
|
if (text.trim().length === 0) {
|
|
console.error("\nLIVE FAIL — assistant returned empty text");
|
|
process.exit(1);
|
|
}
|
|
|
|
console.log(
|
|
`\nLIVE OK — Wafer ${providerId} round-trip: GLM-5.1 endpoint preserved, ` +
|
|
`${inputTokens}→${outputTokens} tokens, stopReason=${stopReason}.`,
|
|
);
|