diff --git a/AGENTS.md b/AGENTS.md index c67bf4193..482ef2b9e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -8,23 +8,24 @@ This repo contains multiple packages, but **`packages/coding-agent/`** is the pr ### Package Structure -| Package | Description | -| ----------------------- | ---------------------------------------------------- | -| `packages/ai` | Multi-provider LLM client with streaming support | +| Package | Description | +| ----------------------- | --------------------------------------------------------------------------------------- | +| `packages/ai` | Multi-provider LLM client with streaming support | | `packages/catalog` | Model catalog: bundled models.json, provider descriptors, model identity/classification | -| `packages/agent` | Agent runtime with tool calling and state management | -| `packages/coding-agent` | Main CLI application (primary focus) | -| `packages/tui` | Terminal UI library with differential rendering | -| `packages/natives` | Bindings for native text/image/grep operations | -| `packages/stats` | Local observability dashboard (`omp stats`) | -| `packages/utils` | Shared utilities (logger, streams, temp files) | -| `crates/pi-natives` | Rust crate for performance-critical text/grep ops | +| `packages/agent` | Agent runtime with tool calling and state management | +| `packages/coding-agent` | Main CLI application (primary focus) | +| `packages/tui` | Terminal UI library with differential rendering | +| `packages/natives` | Bindings for native text/image/grep operations | +| `packages/stats` | Local observability dashboard (`omp stats`) | +| `packages/utils` | Shared utilities (logger, streams, temp files) | +| `crates/pi-natives` | Rust crate for performance-critical text/grep ops | -**Catalog import convention**: code in this repo imports catalog *values* (bundled models, model-thinking helpers, identity, descriptors, model manager/cache) from `@oh-my-pi/pi-catalog/` — never via `@oh-my-pi/pi-ai`. The pi-ai barrel re-exports only the model/effort *types* its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, …); type-only imports of those from `@oh-my-pi/pi-ai` are fine. +**Catalog import convention**: code in this repo imports catalog _values_ (bundled models, model-thinking helpers, identity, descriptors, model manager/cache) from `@oh-my-pi/pi-catalog/` — never via `@oh-my-pi/pi-ai`. The pi-ai barrel re-exports only the model/effort _types_ its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, …); type-only imports of those from `@oh-my-pi/pi-ai` are fine. ## GitHub Unless user tells you exactly what to write: + - **Never comment on GitHub** (issues, PRs, discussions). - **Never create issues on GitHub**. @@ -64,20 +65,20 @@ Use Bun APIs where they provide a cleaner alternative; fall back to `node:*` onl ### Quick reference -| Operation | Use | Not | -| --------------- | ----------------------------------------- | ------------------------------- | -| File read/write | `Bun.file()`, `Bun.write()` | `readFileSync`, `writeFileSync` | -| Spawn process | `` $`cmd` ``, `Bun.spawn()` | `child_process` | -| Sleep | `Bun.sleep(ms)` | `setTimeout` promise | -| Binary lookup | `$which("git")` from `@oh-my-pi/pi-utils` | `spawnSync(["which", "git"])` | -| HTTP server | `Bun.serve()` | `http.createServer()` | -| SQLite | `bun:sqlite` | `better-sqlite3` | -| Hashing | `Bun.hash()`, `Bun.password.*`, WebCrypto | `node:crypto` | -| Path resolution | `import.meta.dir`, `import.meta.path` | `fileURLToPath` dance | -| JSON5 | `Bun.JSON5.parse()` / `.stringify()` | `json5` package | +| Operation | Use | Not | +| --------------- | ----------------------------------------- | ---------------------------------- | +| File read/write | `Bun.file()`, `Bun.write()` | `readFileSync`, `writeFileSync` | +| Spawn process | `` $`cmd` ``, `Bun.spawn()` | `child_process` | +| Sleep | `Bun.sleep(ms)` | `setTimeout` promise | +| Binary lookup | `$which("git")` from `@oh-my-pi/pi-utils` | `spawnSync(["which", "git"])` | +| HTTP server | `Bun.serve()` | `http.createServer()` | +| SQLite | `bun:sqlite` | `better-sqlite3` | +| Hashing | `Bun.hash()`, `Bun.password.*`, WebCrypto | `node:crypto` | +| Path resolution | `import.meta.dir`, `import.meta.path` | `fileURLToPath` dance | +| JSON5 | `Bun.JSON5.parse()` / `.stringify()` | `json5` package | | JSONL | `Bun.JSONL.parse()` / `.parseChunk()` | `text.split("\n").map(JSON.parse)` | -| String width | `Bun.stringWidth()` | `get-east-asian-width`, custom | -| Text wrapping | `Bun.wrapAnsi()` | custom ANSI-aware wrappers | +| String width | `Bun.stringWidth()` | `get-east-asian-width`, custom | +| Text wrapping | `Bun.wrapAnsi()` | custom ANSI-aware wrappers | ### Process execution @@ -99,6 +100,7 @@ Methods: `.quiet()`, `.nothrow()`, `.text()`, `.cwd(path)`. Use `Bun.spawn`/`Bun.spawnSync` only for: long-running processes (LSP, kernels), streaming stdin/stdout/stderr (SSE, JSON-RPC), or process control (signals, kill, complex lifecycle). When using `pipe` mode, cast the stream: + ```typescript const child = Bun.spawn(["cmd"], { stdout: "pipe", stderr: "pipe" }); const reader = (child.stdout as ReadableStream).getReader(); @@ -120,6 +122,7 @@ import * as os from "node:os"; ### File I/O Prefer Bun: + ```typescript const text = await Bun.file(path).text(); const data = await Bun.file(path).json(); @@ -129,6 +132,7 @@ await Bun.write(path, data); // auto-creates parent dirs Use `node:fs/promises` for directory ops (`fs.mkdir`, `fs.rm`, `fs.readdir`) — Bun has no native directory APIs. Avoid sync APIs in async flows; use sync only when forced by a synchronous interface. **Anti-patterns:** + - `existsSync`/`readFileSync`/`writeFileSync` in async code → `Bun.file()` APIs. - `mkdir(dirname(path), …)` before `Bun.write(path, …)` → redundant; `Bun.write` handles it. - `if (await file.exists()) { await file.json() }` → two syscalls plus race. Use try-catch with `isEnoent`: @@ -148,11 +152,15 @@ Use `node:fs/promises` for directory ops (`fs.mkdir`, `fs.rm`, `fs.readdir`) — ### Streams Prefer centralized helpers: + ```typescript import { readStream, readLines } from "./utils/stream"; const text = await readStream(child.stdout); -for await (const line of readLines(stream)) { /* ... */ } +for await (const line of readLines(stream)) { + /* ... */ +} ``` + Manual reader loops only when the protocol requires it (SSE, streaming JSON-RPC). ### Misc @@ -164,9 +172,10 @@ Manual reader loops only when the protocol requires it (SSE, streaming JSON-RPC) ## Generated Files -**NEVER edit `packages/catalog/src/models.json` directly.** It is generated from upstream sources (models.dev, provider catalog discovery, OpenCode docs) by `packages/catalog/scripts/generate-models.ts` and the descriptors/resolvers in `packages/catalog/src/provider-models/`. Hand-edits get overwritten on the next regen. +**NEVER edit `packages/catalog/src/models.json` directly.** It is generated from upstream sources (stencil.so, provider catalog discovery, OpenCode docs) by `packages/catalog/scripts/generate-models.ts` and the descriptors/resolvers in `packages/catalog/src/provider-models/`. Hand-edits get overwritten on the next regen. To change an entry, fix the source: + - **Resolution rules / per-id overrides** → relevant resolver in `packages/catalog/src/provider-models/openai-compat.ts` (e.g. `createOpenCodeApiResolution`'s id-override map). - **Provider catalog entries** (default model, discovery factory/flags) → the `CATALOG_PROVIDERS` table in `packages/catalog/src/provider-models/descriptors.ts`. - **Generator-level fixups** (premium multipliers, codex pricing fallback, fallback models, post-processing) → `packages/catalog/scripts/generate-models.ts`. @@ -193,12 +202,14 @@ Logs go to `~/.omp/logs/omp.YYYY-MM-DD.log` with automatic rotation. Standalone All text displayed in tool renderers must be sanitized. Raw content (file contents, error messages, tool output) breaks terminal rendering: tabs → visual holes, long lines → overflow, paths → leak home directory. **Rules:** + - **Tabs → spaces** via `replaceTabs()` (from `@oh-my-pi/pi-tui` or `../tools/render-utils`). - **Truncate** lines with `truncateToWidth()` / `ui.truncate()`. Use `TRUNCATE_LENGTHS` constants. - **Shorten paths** with `shortenPath()` (replaces home with `~`). - **Preview limits** from `PREVIEW_LIMITS`. No ad-hoc numbers. **Apply to every render path**, not just the happy one: + - Success output (file previews, command output, search results). - **Error messages** — these often embed file content (e.g., patch failure messages include unmatched lines). If a message contains file content, it needs `replaceTabs()`. - Diff content (added and removed). @@ -209,6 +220,7 @@ All text displayed in tool renderers must be sanitized. Raw content (file conten Tool-call previews can have **multiple render paths**. If you add preview-only fields or depend on partially streamed args, update every path — not only the final renderer. Streamed argument buffers decode into display args via `decodeStreamedToolArgs` / `ToolArgsRevealController` (`modes/controllers/tool-args-reveal.ts`); both the live event path and transcript rebuilds must go through them — never spread provider-parsed `arguments` next to a raw `__partialJson` (parsed args lag the stream by a throttled parse window). For the bash tool specifically: + - The pending preview may need raw `partialJson`, not just parsed `arguments`. Parsed args lag until a JSON object closes, which makes inline env assignments appear only at the end. - Preserve preview-only fields (e.g. `__partialJson`) through `event-controller.ts`, transcript rebuilds in `ui-helpers.ts`, and merged call/result rendering in `tool-execution.ts`. Missing one path causes inconsistent previews. - `ToolExecutionComponent.#buildRenderContext()` for bash must work even before a result exists — the renderer uses call args plus render context to show the command preview while streaming. @@ -235,7 +247,7 @@ Test the contract the system exposes — not the easiest internal detail to asse - Smoke tests are acceptable only when they catch a failure mode narrower tests would miss. "Package boots" or "command starts" alone is not enough. - Assert exact strings, ordering, and formatting only when downstream code parses or depends on the exact bytes. Otherwise assert semantic content. - Compile-time guarantees → type checks/type tests, not runtime placeholders. -- **Never source-grep.** A test that reads an implementation file (`.ts`/`.rs`/build script) and asserts on its *text* — `expect(src).toContain("someCall()")`, `.toMatch(/import .../)`, `.not.toContain("oldName")`, or "comment must say X" — is banned. It tests how code *looks*, not what it *does*: it breaks on harmless refactors (comment reflow, rename, import reorder) and passes while the behavior is broken. Assert the observable contract instead (run the code, check output/state/error), use the runtime smoke probe for wiring you cannot exercise in-process, and enforce structural invariants (no value-import of X, no self-import) with a type test or a lint/biome rule — never a string scan of the source. (Reading a file your code *wrote* — apply-patch result, generated bundle, temp fixture — and asserting on that output is fine; that is behavior, not a source grep.) +- **Never source-grep.** A test that reads an implementation file (`.ts`/`.rs`/build script) and asserts on its _text_ — `expect(src).toContain("someCall()")`, `.toMatch(/import .../)`, `.not.toContain("oldName")`, or "comment must say X" — is banned. It tests how code _looks_, not what it _does_: it breaks on harmless refactors (comment reflow, rename, import reorder) and passes while the behavior is broken. Assert the observable contract instead (run the code, check output/state/error), use the runtime smoke probe for wiring you cannot exercise in-process, and enforce structural invariants (no value-import of X, no self-import) with a type test or a lint/biome rule — never a string scan of the source. (Reading a file your code _wrote_ — apply-patch result, generated bundle, temp fixture — and asserting on that output is fine; that is behavior, not a source grep.) - Don't add tests for tiny low-risk changes unless they protect a real contract or fix a regression-prone edge case. - Prefer focused package-local verification for the changed area. @@ -244,6 +256,7 @@ Test the contract the system exposes — not the easiest internal detail to asse Location: `packages/*/CHANGELOG.md` (per package). **Format** — sections under `## [Unreleased]`: + - `### Breaking Changes` (first if present) - `### Added` - `### Changed` @@ -251,11 +264,13 @@ Location: `packages/*/CHANGELOG.md` (per package). - `### Removed` **Rules:** + - New entries always go under `## [Unreleased]`. - Never modify already-released sections (e.g., `## [0.12.2]`) — they are immutable. - Don't flag changelog section order or formatting in reviews or PRs — `bun run release` runs `fix-changelogs` which normalizes everything automatically. **Attribution:** + - Internal (from issues): `Fixed foo bar ([#123](https://github.com/can1357/oh-my-pi/issues/123))`. - External contributions: `Added feature X ([#456](https://github.com/can1357/oh-my-pi/pull/456) by [@username](https://github.com/username))`. diff --git a/packages/catalog/README.md b/packages/catalog/README.md index 93477f3e9..3196a1ca0 100644 --- a/packages/catalog/README.md +++ b/packages/catalog/README.md @@ -4,24 +4,24 @@ Model catalog for [oh-my-pi](https://github.com/can1357/oh-my-pi): bundled model ## What's inside -| Module | Purpose | -| --- | --- | -| `models.json` + `models` | Bundled model database (pricing, context windows, modalities, thinking support) | -| `provider-models` | Provider catalog descriptors (`CATALOG_PROVIDERS`), per-provider model resolution rules | -| `discovery` | Runtime model discovery for OpenAI-compatible endpoints, Gemini, Codex, Cursor, Antigravity, Ollama | -| `identity` | Model id parsing and classification (family/version), reference resolution, equivalence, selection priority | -| `model-thinking` | Thinking/reasoning metadata and generated per-model policies | -| `model-manager` / `model-cache` | Runtime model registry with discovery refresh and on-disk caching | -| `variant-collapse` | Collapsing provider-specific variants of the same underlying model | -| `compat` | Request/response compatibility fixups for OpenAI- and Anthropic-shaped APIs | -| `wire` | Wire-level helpers: Codex, Gemini headers, GitHub Copilot | -| `effort` | Reasoning-effort level definitions | +| Module | Purpose | +| ------------------------------- | ----------------------------------------------------------------------------------------------------------- | +| `models.json` + `models` | Bundled model database (pricing, context windows, modalities, thinking support) | +| `provider-models` | Provider catalog descriptors (`CATALOG_PROVIDERS`), per-provider model resolution rules | +| `discovery` | Runtime model discovery for OpenAI-compatible endpoints, Gemini, Codex, Cursor, Antigravity, Ollama | +| `identity` | Model id parsing and classification (family/version), reference resolution, equivalence, selection priority | +| `model-thinking` | Thinking/reasoning metadata and generated per-model policies | +| `model-manager` / `model-cache` | Runtime model registry with discovery refresh and on-disk caching | +| `variant-collapse` | Collapsing provider-specific variants of the same underlying model | +| `compat` | Request/response compatibility fixups for OpenAI- and Anthropic-shaped APIs | +| `wire` | Wire-level helpers: Codex, Gemini headers, GitHub Copilot | +| `effort` | Reasoning-effort level definitions | Import from subpaths (`@oh-my-pi/pi-catalog/`) or the root barrel. ## models.json is generated -Never edit `src/models.json` by hand — it is produced from upstream sources (models.dev, provider catalog discovery, OpenCode docs) by `scripts/generate-models.ts` and the resolvers in `src/provider-models/`. Regenerate with: +Never edit `src/models.json` by hand — it is produced from upstream sources (stencil.so, provider catalog discovery, OpenCode docs) by `scripts/generate-models.ts` and the resolvers in `src/provider-models/`. Regenerate with: ```sh bun run gen:models diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 55d7378de..a60423732 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -35,6 +35,7 @@ import { buildXaiOAuthStaticSeed, clampFireworksKimiMaxTokens, clampKimiK27CodeMaxTokens, + fetchWellKnownModels, GMI_CLOUD_STATIC_MODELS, isFireworksKimiK2ModelId, isKimiK27CodeModelId, @@ -154,15 +155,14 @@ async function fetchProviderModelsFromCatalog( async function loadModelsDevData(): Promise { try { - console.log("Fetching models from models.dev API..."); - const response = await fetch("https://models.dev/api.json"); - const data = await response.json(); + console.log("Fetching stencil.so catalog from catalog.stencil.so..."); + const data = await fetchWellKnownModels(); const models = mapModelsDevToModels(data as Record, MODELS_DEV_PROVIDER_DESCRIPTORS); models.sort((a, b) => a.id.localeCompare(b.id)); - console.log(`Loaded ${models.length} tool-capable models from models.dev`); + console.log(`Loaded ${models.length} tool-capable models from stencil.so`); return models; } catch (error) { - console.error("Failed to load models.dev data:", error); + console.error("Failed to load stencil.so data:", error); return []; } } @@ -212,7 +212,7 @@ function applyGlobalModelsDevFallback( name: reference.name, reasoning: reference.reasoning, input: reference.input, - // Fill unknown endpoint limits from same-id models.dev references, but keep + // Fill unknown endpoint limits from same-id stencil.so references, but keep // provider-specific values when discovery returned them explicitly. contextWindow: model.contextWindow ?? reference.contextWindow, maxTokens: model.maxTokens ?? reference.maxTokens, @@ -248,7 +248,7 @@ function applyUmansPricingFallback(models: readonly ModelSpec[], modelsDevModels } // The public endpoint exposes this technical alias for Umans Flash, but - // models.dev publishes pricing only for the recommended `umans-flash` id. + // stencil.so publishes pricing only for the recommended `umans-flash` id. const flashCost = paygCosts.get("umans-flash"); if (flashCost) { paygCosts.set("umans-qwen3.6-35b-a3b", flashCost); @@ -502,7 +502,7 @@ async function generateModels() { })), ); // A provider is authoritative once its endpoint snapshot can replace the - // models.dev / previous-snapshot rows. Requiring fetched models keeps a + // stencil.so / previous-snapshot rows. Requiring fetched models keeps a // flaky empty-but-200 discovery from silently wiping another provider's // bundled catalog; only alibaba-token-plan treats an empty success as // authoritative, because its `/models` allowlist reflects the subscribed @@ -520,8 +520,8 @@ async function generateModels() { const bundledModelsDevModels = modelsDevModels.filter(model => !authoritativeCatalogProviders.has(model.provider)); // getGitLabDuoModels returns built models; project back to spec stage for the bundle. const gitLabDuoModels = getGitLabDuoModels().map(model => toModelSpec(model)); - // Combine models. models.dev has priority unless a provider's successful endpoint - // discovery is authoritative; those endpoint snapshots replace models.dev rows. + // Combine models. stencil.so has priority unless a provider's successful endpoint + // discovery is authoritative; those endpoint snapshots replace stencil.so rows. let allModels = applyGlobalModelsDevFallback( [...bundledModelsDevModels, ...catalogProviderModels, ...gitLabDuoModels], modelsDevModels, @@ -531,7 +531,7 @@ async function generateModels() { allModels.push(CLOUDFLARE_FALLBACK_MODEL as ModelSpec<"anthropic-messages">); } - // xai-oauth is not in models.dev; its descriptor's catalogDiscovery fetch + // xai-oauth is not in stencil.so; its descriptor's catalogDiscovery fetch // only succeeds with live SuperGrok OAuth credentials (and on success the // dynamic entries — already overlaid by applyXAIOAuthCuration — win dedup // below). Always push the curated seed so a regen without credentials, or @@ -547,7 +547,7 @@ async function generateModels() { allModels.push(...ALIBABA_TOKEN_PLAN_STATIC_MODELS); } // Seed Anthropic models that are live on the first-party API or in limited - // release but that models.dev has not catalogued yet (e.g. Claude Fable 5 / + // release but that stencil.so has not catalogued yet (e.g. Claude Fable 5 / // Mythos 5). Deduped behind upstream entries; metadata is pinned in // applyAnthropicCatalogPolicy. allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS); @@ -617,7 +617,7 @@ async function generateModels() { } // Merge previous models.json entries as fallback for provider/model pairs not // fetched dynamically. Providers covered by authoritative endpoint discovery - // or authoritative models.dev sources keep that upstream list exactly, so + // or authoritative stencil.so sources keep that upstream list exactly, so // retired entries from the previous snapshot do not reappear during regeneration. // Discovery-only providers (local inference servers) — never bundle static models. const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`)); diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index 77a0bab37..4e3d0c0fb 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -52,7 +52,7 @@ export const CLOUDFLARE_FALLBACK_MODEL: ModelSpec<"anthropic-messages"> = { }; /** - * `models.dev` currently lists `jp.anthropic.claude-opus-5`, but AWS's own + * `stencil.so` currently lists `jp.anthropic.claude-opus-5`, but AWS's own * Bedrock model card documents only `anthropic.claude-opus-5` plus the `us.`, * `eu.`, `au.`, and `global.` Geo/Global inference-profile IDs under * Programmatic Access; Japan regions are marked unsupported for Geo inference @@ -321,7 +321,7 @@ function applyGeneratedModelPolicy(model: ModelSpec): void { } function applyAnthropicCatalogPolicy(model: ModelSpec, parsedModel: AnthropicModel): void { - // Claude Opus 4.5: models.dev reports 3x the correct cache pricing. + // Claude Opus 4.5: stencil.so reports 3x the correct cache pricing. if (model.provider === "anthropic" && parsedModel.kind === "opus" && semverEqual(parsedModel.version, "4.5")) { model.cost.cacheRead = 0.5; model.cost.cacheWrite = 6.25; @@ -336,7 +336,7 @@ function applyAnthropicCatalogPolicy(model: ModelSpec, parsedModel: Anthrop } // Claude Fable/Mythos 5: Anthropic's /v1/models omits token limits and - // pricing, and models.dev lags new releases. Pin authoritative values from + // pricing, and stencil.so lags new releases. Pin authoritative values from // the model card (1M context / 128k output) and pricing docs ($10 in / $50 // out per MTok). if (model.provider === "anthropic" && isFableOrMythos(parsedModel.kind)) { diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index a11fdc0ae..185477c68 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -14,10 +14,10 @@ const NON_AUTHORITATIVE_RETRY_MS = 5 * 60 * 1000; export type ModelRefreshStrategy = "online" | "offline" | "online-if-uncached"; /** - * Hook for loading and mapping models.dev fallback data into canonical model objects. + * Hook for loading and mapping stencil.so fallback data into canonical model objects. */ export interface ModelsDevFallback { - /** Fetches raw fallback payload (for example from models.dev). */ + /** Fetches raw fallback payload (for example from stencil.so). */ fetch(): Promise; /** Maps payload into provider models. */ map(payload: TPayload, providerId: Provider): readonly ModelSpec[]; @@ -51,7 +51,7 @@ export interface ModelManagerOptions; /** Optional dynamic endpoint fetcher. */ fetchDynamicModels?: () => Promise[] | null>; - /** Optional models.dev fallback hook. */ + /** Optional stencil.so fallback hook. */ modelsDev?: ModelsDevFallback; /** Clock override for deterministic tests. */ now?: () => number; @@ -171,7 +171,7 @@ function restoreCachedModelHeaders( /** * Resolves provider models with source precedence: - * static -> models.dev -> cache -> dynamic. + * static -> stencil.so -> cache -> dynamic. * * Later sources override earlier ones by model id. */ diff --git a/packages/catalog/src/models.ts b/packages/catalog/src/models.ts index 7204e6bc2..0c43aa1d6 100644 --- a/packages/catalog/src/models.ts +++ b/packages/catalog/src/models.ts @@ -6,7 +6,7 @@ import type { Api, KnownProvider, Model, ModelSpec, Usage } from "./types"; * Static bundled model registry loaded from `models.json`. * * This module intentionally exposes compile-time defaults only. - * It does not include runtime discovery, models.dev overlays, or on-disk cache state. + * It does not include runtime discovery, stencil.so overlays, or on-disk cache state. * * For runtime-aware resolution, use `createModelManager()` / `resolveProviderModels()`. */ diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index bf386a71f..04c791c02 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1,3 +1,4 @@ +import { VERSION } from "@oh-my-pi/pi-utils"; import * as logger from "@oh-my-pi/pi-utils/logger"; import { fetchOpenAICompatibleModels, @@ -31,7 +32,10 @@ import { import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references"; import { getDefaultModelDiscoveryBaseUrl, resolveModelCacheProviderId } from "./cache-provider-id"; -const MODELS_DEV_URL = "https://models.dev/api.json"; +const MODELS_DEV_URL = "https://catalog.stencil.so/models.json.zstd"; + +/** Little-endian magic number opening every zstd frame (RFC 8878). */ +const ZSTD_MAGIC = 0xfd2fb528; /** * Uses a cancellable timer rather than the native abort-timeout helper so @@ -93,16 +97,74 @@ function toInputCapabilities(value: unknown): ("text" | "image")[] { return supportsImage ? ["text", "image"] : ["text"]; } -async function fetchModelsDevPayload(fetchImpl: FetchImpl = discoveryFetch(), signal?: AbortSignal): Promise { - const response = await fetchImpl(MODELS_DEV_URL, { - method: "GET", - headers: { Accept: "application/json" }, - signal, - }); - if (!response.ok) { - throw new Error(`models.dev fetch failed: ${response.status}`); +/** + * Process-wide catalog session: the first call downloads the payload (the one + * request the server logs); later calls revalidate with `If-None-Match` and + * reuse the decoded payload on `304`. Failure after a successful load falls + * back to the session copy. + */ +const catalogSession: { + inflight: Promise | null; + payload: unknown; + etag: string | null; + hasPayload: boolean; +} = { inflight: null, payload: undefined, etag: null, hasPayload: false }; + +const CATALOG_USER_AGENT = `omp/${VERSION} (+https://omp.sh)`; + +/** + * Fetches the models.dev catalog via catalog.stencil.so, which serves a + * field-pruned copy precompressed as a zstd blob (~93 KB vs ~3.3 MB raw). + * The frame magic is sniffed rather than trusting content-type so plain-JSON + * responses (test stubs, fallback mirrors) parse identically. + * + * Fetched fully once per process: concurrent callers share the in-flight + * request, repeat callers send a conditional GET that the server answers + * (and deliberately does not log) with `304`. + */ +export function fetchWellKnownModels(fetchImpl?: FetchImpl, signal?: AbortSignal): Promise { + if (!catalogSession.inflight) { + catalogSession.inflight = fetchCatalogPayload(fetchImpl ?? discoveryFetch(), signal).finally(() => { + catalogSession.inflight = null; + }); } - return response.json(); + return catalogSession.inflight; +} + +async function fetchCatalogPayload(fetchImpl: FetchImpl, signal?: AbortSignal): Promise { + const headers: Record = { + Accept: "application/zstd, application/json", + "User-Agent": CATALOG_USER_AGENT, + }; + if (catalogSession.hasPayload && catalogSession.etag) { + headers["If-None-Match"] = catalogSession.etag; + } + let response: Response; + try { + response = await fetchImpl(MODELS_DEV_URL, { method: "GET", headers, signal }); + } catch (error) { + if (catalogSession.hasPayload) { + return catalogSession.payload; + } + throw error; + } + if (response.status === 304 && catalogSession.hasPayload) { + return catalogSession.payload; + } + if (!response.ok) { + if (catalogSession.hasPayload) { + return catalogSession.payload; + } + throw new Error(`models catalog fetch failed: ${response.status}`); + } + const bytes = new Uint8Array(await response.arrayBuffer()); + const isZstd = bytes.length >= 4 && new DataView(bytes.buffer, bytes.byteOffset).getUint32(0, true) === ZSTD_MAGIC; + const text = new TextDecoder().decode(isZstd ? await Bun.zstdDecompress(bytes) : bytes); + const payload: unknown = JSON.parse(text); + catalogSession.payload = payload; + catalogSession.etag = response.headers.get("etag"); + catalogSession.hasPayload = true; + return payload; } function mapAnthropicModelsDev(payload: unknown, baseUrl: string): ModelSpec<"anthropic-messages">[] { @@ -1566,7 +1628,7 @@ async function loadSiliconFlowModelsDevReferences( // Bounded: this enrichment is optional, so a stalled models.dev must not // hold back the authoritative endpoint request that runs after it. const payload = await withCatalogDiscoveryTimeout(SILICONFLOW_MODELS_DEV_REFERENCE_TIMEOUT_MS, signal => - fetchModelsDevPayload(fetchImpl, signal), + fetchWellKnownModels(fetchImpl, signal), ); return createModelsDevReferenceMap<"openai-completions">( mapModelsDevToModels(payload as Record, [descriptor]), @@ -2024,7 +2086,7 @@ function createModelsDevReferenceMap( async function loadModelsDevReferences(fetchImpl?: FetchImpl): Promise>> { try { - const payload = await fetchModelsDevPayload(fetchImpl); + const payload = await fetchWellKnownModels(fetchImpl); return createModelsDevReferenceMap( mapModelsDevToModels(payload as Record, MODELS_DEV_PROVIDER_DESCRIPTORS), ); @@ -4842,12 +4904,12 @@ export function anthropicModelManagerOptions( return { providerId: "anthropic", modelsDev: { - fetch: () => fetchModelsDevPayload(config?.fetch), + fetch: () => fetchWellKnownModels(config?.fetch), map: payload => mapAnthropicModelsDev(payload, baseUrl), }, ...(apiKey && { fetchDynamicModels: async () => { - const modelsDevModels = await fetchModelsDevPayload(config?.fetch) + const modelsDevModels = await fetchWellKnownModels(config?.fetch) .then(payload => mapAnthropicModelsDev(payload, baseUrl)) .catch(() => []); const references = buildAnthropicReferenceMap(modelsDevModels); diff --git a/packages/catalog/test/amazon-bedrock-opus-5.test.ts b/packages/catalog/test/amazon-bedrock-opus-5.test.ts index d91ceddd8..d1e54e6ca 100644 --- a/packages/catalog/test/amazon-bedrock-opus-5.test.ts +++ b/packages/catalog/test/amazon-bedrock-opus-5.test.ts @@ -18,8 +18,8 @@ const AWS_DOCUMENTED_OPUS_5_IDS = [ "global.anthropic.claude-opus-5", ]; -// A representative `models.dev` "amazon-bedrock" payload for Claude Opus 5. -// models.dev lists each inference-profile prefix as its own row (the `eu.` +// A representative `stencil.so` "amazon-bedrock" payload for Claude Opus 5. +// stencil.so lists each inference-profile prefix as its own row (the `eu.` // row even carries distinct EU pricing), including the `jp.` profile that AWS // does not actually expose for this model. We reproduce that shape so the test // exercises the real source → catalog path — `mapModelsDevToModels` plus the @@ -65,7 +65,7 @@ const OPUS_5_MODELS_DEV_FIXTURE = { describe("Amazon Bedrock Claude Opus 5", () => { test("source mapping plus generation policy yields exactly the AWS-documented inference-profile IDs", () => { - // Guard the source (models.dev descriptor + exclusion policy), not the + // Guard the source (stencil.so descriptor + exclusion policy), not the // bundled snapshot: the assertion must break if the mapping or policy // stops reproducing the documented IDs, and must not falsely fail when // upstream metadata legitimately shifts. @@ -80,10 +80,10 @@ describe("Amazon Bedrock Claude Opus 5", () => { // Set semantics: the descriptor also derives an `eu.` variant from the // bare `anthropic.` row, so `eu.` legitimately arrives from both that - // derivation and the standalone models.dev row (deduped downstream by + // derivation and the standalone stencil.so row (deduped downstream by // the generator). We assert the documented ID coverage, not row count. expect(new Set(opus5Ids)).toEqual(new Set(AWS_DOCUMENTED_OPUS_5_IDS)); - // `models.dev` lists `jp.anthropic.claude-opus-5`, but Bedrock has no such + // `stencil.so` lists `jp.anthropic.claude-opus-5`, but Bedrock has no such // inference profile for this model and would reject it, so the generation // policy must drop it before it reaches the catalog. expect(opus5Ids).not.toContain("jp.anthropic.claude-opus-5"); diff --git a/packages/catalog/test/azure-provider.test.ts b/packages/catalog/test/azure-provider.test.ts index c213bcb98..01d2325d2 100644 --- a/packages/catalog/test/azure-provider.test.ts +++ b/packages/catalog/test/azure-provider.test.ts @@ -10,7 +10,7 @@ import { } from "@oh-my-pi/pi-catalog/provider-models"; import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; -// A models.dev "azure" payload: two OpenAI-family models (one reasoning), a +// A stencil.so "azure" payload: two OpenAI-family models (one reasoning), a // non-tool-capable instruct model, and a Foundry-hosted third party served via // a per-model `provider` override (claude over .services.ai.azure.com). const AZURE_MODELS_DEV_FIXTURE = { @@ -37,7 +37,7 @@ describe("azure catalog provider", () => { expect(DEFAULT_MODEL_PER_PROVIDER.azure).toBe("gpt-5.5"); }); - test("models.dev descriptor keeps only OpenAI-family Responses models, baseUrl resolved at runtime", () => { + test("stencil.so descriptor keeps only OpenAI-family Responses models, baseUrl resolved at runtime", () => { const azure = mapModelsDevToModels(AZURE_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( model => model.provider === "azure", ); diff --git a/packages/catalog/test/coreweave-provider.test.ts b/packages/catalog/test/coreweave-provider.test.ts index d695eaeea..5dd5ab952 100644 --- a/packages/catalog/test/coreweave-provider.test.ts +++ b/packages/catalog/test/coreweave-provider.test.ts @@ -89,7 +89,7 @@ describe("CoreWeave Serverless Inference provider support", () => { }); }); - test("maps models.dev wandb metadata into OpenAI chat completions models", () => { + test("maps stencil.so wandb metadata into OpenAI chat completions models", () => { const mapped = mapModelsDevToModels( { wandb: { diff --git a/packages/catalog/test/fireworks-serverless-discovery.test.ts b/packages/catalog/test/fireworks-serverless-discovery.test.ts index 834d70833..31367fa5b 100644 --- a/packages/catalog/test/fireworks-serverless-discovery.test.ts +++ b/packages/catalog/test/fireworks-serverless-discovery.test.ts @@ -79,9 +79,9 @@ function createMockFetch(): { fetch: FetchImpl; controlPlaneUrls: string[] } { const controlPlaneUrls: string[] = []; const fetch = (async (input: string | URL | Request): Promise => { const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; - // models.dev reference fetch — return an empty catalog so the mapper relies + // stencil.so reference fetch — return an empty catalog so the mapper relies // purely on control-plane + bundled references. - if (url.startsWith("https://models.dev")) { + if (url.startsWith("https://stencil.so")) { return jsonResponse({}); } if (url.includes("/v1/accounts/fireworks/models")) { @@ -167,7 +167,7 @@ describe("Fireworks control-plane serverless discovery", () => { it("returns null on a control-plane transport failure so the manager keeps its cache", async () => { const fetch = (async (input: string | URL | Request): Promise => { const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; - if (url.startsWith("https://models.dev")) return jsonResponse({}); + if (url.startsWith("https://stencil.so")) return jsonResponse({}); return new Response("server error", { status: 500 }); }) as unknown as FetchImpl; const options = fireworksModelManagerOptions({ apiKey: "fw_test_key", fetch }); diff --git a/packages/catalog/test/google-vertex-discovery.test.ts b/packages/catalog/test/google-vertex-discovery.test.ts index 7b7cf814e..06b5c3e4b 100644 --- a/packages/catalog/test/google-vertex-discovery.test.ts +++ b/packages/catalog/test/google-vertex-discovery.test.ts @@ -44,7 +44,7 @@ const googleVertexModelsDevPayload = { } satisfies Record; describe("google-vertex model catalog", () => { - it("maps the models.dev Vertex catalog instead of the project discovery endpoint", () => { + it("maps the stencil.so Vertex catalog instead of the project discovery endpoint", () => { const models = mapModelsDevToModels(googleVertexModelsDevPayload, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( model => model.provider === "google-vertex", ); diff --git a/packages/catalog/test/issue-1617-repro.test.ts b/packages/catalog/test/issue-1617-repro.test.ts index e80ae88da..f07089ed7 100644 --- a/packages/catalog/test/issue-1617-repro.test.ts +++ b/packages/catalog/test/issue-1617-repro.test.ts @@ -6,7 +6,7 @@ * `<|minimax|>`) leaking into the UI because OMP POSTs anthropic-shaped * requests to /v1/messages and the gateway returns non-Anthropic responses. * - * models.dev declares these ids with `provider.npm = "@ai-sdk/anthropic"`, + * stencil.so declares these ids with `provider.npm = "@ai-sdk/anthropic"`, * which by default would resolve to anthropic-messages on opencode-zen/-go. * The descriptor must override these specific ids to openai-completions so * that regenerated models.json keeps the correct routing, AND so the @@ -29,9 +29,9 @@ describe("opencode-zen/-go resolver routes MiniMax M3 to openai-completions (iss const zenDescriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-zen"); const goDescriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-go"); - // Per upstream models.dev (verified 2026-06-02 against - // https://models.dev/api.json["opencode"].models and - // https://models.dev/api.json["opencode-go"].models), the affected ids + // Per upstream stencil.so (verified 2026-06-02 against + // https://stencil.so/api.json["opencode"].models and + // https://stencil.so/api.json["opencode-go"].models), the affected ids // carry `provider.npm = "@ai-sdk/anthropic"`. The naive @ai-sdk/anthropic // rule would route them to /v1/messages on opencode.ai/zen[/go] which // 404s. Per-id overrides must win. diff --git a/packages/catalog/test/issue-2883-moonshot-china.test.ts b/packages/catalog/test/issue-2883-moonshot-china.test.ts index cb0404e39..bb49472cd 100644 --- a/packages/catalog/test/issue-2883-moonshot-china.test.ts +++ b/packages/catalog/test/issue-2883-moonshot-china.test.ts @@ -4,7 +4,7 @@ import { moonshotModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-model import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; import { $pickenv } from "@oh-my-pi/pi-utils"; -const MODELS_DEV_URL = "https://models.dev/api.json"; +const MODELS_DEV_URL = "https://catalog.stencil.so/models.json.zstd"; const ORIGINAL_ENV: Record = { MOONSHOT_BASE_URL: Bun.env.MOONSHOT_BASE_URL, diff --git a/packages/catalog/test/issue-5598-repro.test.ts b/packages/catalog/test/issue-5598-repro.test.ts index 08390eb0a..51cc769b0 100644 --- a/packages/catalog/test/issue-5598-repro.test.ts +++ b/packages/catalog/test/issue-5598-repro.test.ts @@ -5,12 +5,12 @@ import { } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; // Z.AI GLM coding-plan token costs all showed as "Free" (issue #5598): the `zai` -// provider descriptor sourced the models.dev `zai-coding-plan` key, which reports +// provider descriptor sourced the stencil.so `zai-coding-plan` key, which reports // all-$0 subscription rates. The `zai` (pay-as-you-go) key carries the real // per-token rates for the identical GLM ids, matching how other subscription // providers surface comparison pricing in `/models`. -describe("zai GLM pricing sources the PAYG models.dev key (issue #5598)", () => { - test("descriptor maps the `zai` models.dev key, not `zai-coding-plan`", () => { +describe("zai GLM pricing sources the PAYG stencil.so key (issue #5598)", () => { + test("descriptor maps the `zai` stencil.so key, not `zai-coding-plan`", () => { const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "zai"); expect(descriptor).toBeDefined(); expect(descriptor?.modelsDevKey).toBe("zai"); diff --git a/packages/catalog/test/issue-5756-repro.test.ts b/packages/catalog/test/issue-5756-repro.test.ts index 50452f157..f1aab7ef0 100644 --- a/packages/catalog/test/issue-5756-repro.test.ts +++ b/packages/catalog/test/issue-5756-repro.test.ts @@ -2,7 +2,7 @@ * Issue #5756 — `moonshot/kimi-k3 is incorrectly shown as free` * * The native Moonshot `kimi-k3` entry is dynamically discovered but has no - * bundled/models.dev reference, so `mapWithBundledReference` produced the + * bundled/stencil.so reference, so `mapWithBundledReference` produced the * generic dynamic defaults: zero token cost, null limits, text-only input, * and `reasoning: false`. `/models` then labeled the paid model "Free". * diff --git a/packages/catalog/test/issue-6563-repro.test.ts b/packages/catalog/test/issue-6563-repro.test.ts index 318389f11..b04be3cdd 100644 --- a/packages/catalog/test/issue-6563-repro.test.ts +++ b/packages/catalog/test/issue-6563-repro.test.ts @@ -7,7 +7,7 @@ * override verbatim, so generic discovery requested * `https://api.anthropic.com/models` (404) instead of `/v1/models`. The failed * refresh then retained a stale text-only cache row, which `mergeDynamicModel` - * treated as authoritative over fresh models.dev vision metadata — leaving + * treated as authoritative over fresh stencil.so vision metadata — leaving * `claude-opus-5` marked text-only and snapcompact refusing to run. * * The fix normalizes the discovery URL to always end in `/v1` while model rows @@ -22,7 +22,7 @@ function modelsDevResponse(): Response { const body = { anthropic: { models: { - // Not in the bundled catalog: capability must come from models.dev + // Not in the bundled catalog: capability must come from stencil.so // through the discovery path under repair. "claude-test-vision-1": { name: "Claude Test Vision", @@ -54,7 +54,7 @@ describe("issue #6563 — anthropic discovery base URL missing /v1", () => { const fetchMock = (async (input: string | URL | Request): Promise => { const url = String(input instanceof Request ? input.url : input); requestedUrls.push(url); - if (url === "https://models.dev/api.json") return modelsDevResponse(); + if (url === "https://catalog.stencil.so/models.json.zstd") return modelsDevResponse(); if (url === `${PROVIDER_BASE_URL}/v1/models`) return anthropicModelsResponse(); return new Response("not found", { status: 404 }); }) as typeof fetch; @@ -78,7 +78,7 @@ describe("issue #6563 — anthropic discovery base URL missing /v1", () => { // change that would make stale capabilities authoritative. expect(opus5?.baseUrl).toBe(PROVIDER_BASE_URL); - // A model absent from the bundled catalog picks up vision from models.dev. + // A model absent from the bundled catalog picks up vision from stencil.so. const unbundled = models?.find(m => m.id === "claude-test-vision-1"); expect(unbundled?.input).toContain("image"); expect(unbundled?.baseUrl).toBe(PROVIDER_BASE_URL); diff --git a/packages/catalog/test/issue-830-repro.test.ts b/packages/catalog/test/issue-830-repro.test.ts index 011e7148e..b65b980a0 100644 --- a/packages/catalog/test/issue-830-repro.test.ts +++ b/packages/catalog/test/issue-830-repro.test.ts @@ -34,7 +34,7 @@ describe("deepseek built-in provider (issue #830)", () => { } }); - test("models.dev mapping descriptor uses api.deepseek.com and forces reasoning_content + no tool_choice", () => { + test("stencil.so mapping descriptor uses api.deepseek.com and forces reasoning_content + no tool_choice", () => { const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "deepseek"); expect(descriptor).toBeDefined(); expect(descriptor?.modelsDevKey).toBe("deepseek"); diff --git a/packages/catalog/test/issue-887-repro.test.ts b/packages/catalog/test/issue-887-repro.test.ts index e91d728a4..2fac1dbee 100644 --- a/packages/catalog/test/issue-887-repro.test.ts +++ b/packages/catalog/test/issue-887-repro.test.ts @@ -3,7 +3,7 @@ * because the resolver routes them to anthropic-messages /v1/messages while * the OpenCode Go gateway only serves them at /v1/chat/completions. * - * models.dev declares these ids with `provider.npm = "@ai-sdk/anthropic"`, + * stencil.so declares these ids with `provider.npm = "@ai-sdk/anthropic"`, * which by default would resolve to anthropic-messages on opencode-go. The * descriptor must override these specific ids to openai-completions so that * regenerated models.json keeps the correct routing. @@ -20,8 +20,8 @@ const OPENCODE_GO_BASE = "https://opencode.ai/zen/go/v1"; describe("opencode-go resolver routes 404-ing ids to openai-completions (issue #887)", () => { const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-go"); - // Per upstream models.dev (verified 2026-05-02 against - // https://models.dev/api.json["opencode-go"].models), these three ids carry + // Per upstream stencil.so (verified 2026-05-02 against + // https://stencil.so/api.json["opencode-go"].models), these three ids carry // `provider.npm = "@ai-sdk/anthropic"`. The naive @ai-sdk/anthropic rule // would route them to /v1/messages on opencode.ai/zen/go which 404s. const npmAnthropic: ModelsDevModel = { provider: { npm: "@ai-sdk/anthropic" }, tool_call: true }; @@ -35,7 +35,7 @@ describe("opencode-go resolver routes 404-ing ids to openai-completions (issue # ); test("minimax-m2.5 (control: works empirically) also resolves to openai-completions", () => { - // models.dev currently lists minimax-m2.5 without an explicit provider.npm, + // stencil.so currently lists minimax-m2.5 without an explicit provider.npm, // so it falls through to the default openai-completions resolution. const m25: ModelsDevModel = { tool_call: true }; const resolved = descriptor?.resolveApi?.("minimax-m2.5", m25); diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index 24183fd24..619a1b6a0 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -4,7 +4,7 @@ import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; import * as logger from "@oh-my-pi/pi-utils/logger"; const ORIGINAL_LITELLM_BASE_URL = Bun.env.LITELLM_BASE_URL; -const MODELS_DEV_URL = "https://models.dev/api.json"; +const MODELS_DEV_URL = "https://catalog.stencil.so/models.json.zstd"; function makeLiteLLMSentinelPlaceholder(modelGroup: string) { return { model_group: modelGroup, @@ -156,7 +156,7 @@ describe("LiteLLM provider discovery", () => { expect(models?.[0]?.baseUrl).toBe("http://litellm-config.example:4200/v1"); }); - test("keeps LiteLLM transport when models.dev has a colliding provider model id", async () => { + test("keeps LiteLLM transport when stencil.so has a colliding provider model id", async () => { const fetchMock = makeCollisionFetchMock(); const options = litellmModelManagerOptions({ @@ -306,7 +306,7 @@ describe("LiteLLM provider discovery", () => { expect(warnSpy).not.toHaveBeenCalled(); }); - test("maps LiteLLM per-token cost onto cost.input/output for models missing from models.dev", async () => { + test("maps LiteLLM per-token cost onto cost.input/output for models missing from stencil.so", async () => { const fetchMock = vi.fn(async (input: string | URL | Request) => { const url = inputUrl(input); if (url === MODELS_DEV_URL) { @@ -349,7 +349,7 @@ describe("LiteLLM provider discovery", () => { }); }); - test("enriches LiteLLM rich models missing from models.dev with bundled reasoning metadata", async () => { + test("enriches LiteLLM rich models missing from stencil.so with bundled reasoning metadata", async () => { const fetchMock = vi.fn(async (input: string | URL | Request) => { const url = inputUrl(input); if (url === MODELS_DEV_URL) { @@ -778,7 +778,7 @@ describe("LiteLLM provider discovery", () => { }); }); - test("enriches LiteLLM /v1/models fallback entries missing from models.dev with bundled reasoning metadata", async () => { + test("enriches LiteLLM /v1/models fallback entries missing from stencil.so with bundled reasoning metadata", async () => { const calls: string[] = []; const fetchMock = vi.fn(async (input: string | URL | Request) => { const url = inputUrl(input); diff --git a/packages/catalog/test/siliconflow-provider.test.ts b/packages/catalog/test/siliconflow-provider.test.ts index 5aba51586..57c192fc0 100644 --- a/packages/catalog/test/siliconflow-provider.test.ts +++ b/packages/catalog/test/siliconflow-provider.test.ts @@ -82,7 +82,7 @@ describe("siliconflow built-in providers", () => { expect(entry?.dynamicModelsAuthoritative).toBe(true); expect(entry?.catalogDiscovery).toBeUndefined(); } - // Runtime: no models.dev mapping may feed the generator either. + // Runtime: no stencil.so mapping may feed the generator either. expect(MODELS_DEV_PROVIDER_DESCRIPTORS.some(d => d.providerId === "siliconflow")).toBe(false); expect(MODELS_DEV_PROVIDER_DESCRIPTORS.some(d => d.providerId === "siliconflow-cn")).toBe(false); }); @@ -106,12 +106,12 @@ describe("siliconflow built-in providers", () => { }); }); - test("dynamic discovery filters non-chat ids and hydrates metadata from models.dev and bundled references", async () => { + test("dynamic discovery filters non-chat ids and hydrates metadata from stencil.so and bundled references", async () => { const seen: { urls: string[]; authorization?: string } = { urls: [] }; const stubFetch: FetchImpl = async (input, init) => { const url = String(input); seen.urls.push(url); - if (url.startsWith("https://models.dev/")) { + if (url.startsWith("https://catalog.stencil.so/")) { return new Response(JSON.stringify(MODELS_DEV_STUB_PAYLOAD), { status: 200, headers: { "content-type": "application/json" }, @@ -142,7 +142,7 @@ describe("siliconflow built-in providers", () => { expect(models).not.toBeNull(); expect((models ?? []).map(model => model.id)).toEqual(["deepseek-ai/DeepSeek-V4-Pro", "zai-org/GLM-5.1"]); - // Tier 1: models.dev carries the id — pricing, limits, and reasoning hydrate. + // Tier 1: stencil.so carries the id — pricing, limits, and reasoning hydrate. const glm = models?.find(model => model.id === "zai-org/GLM-5.1"); expect(glm?.reasoning).toBe(true); expect(glm?.contextWindow).toBe(205000); @@ -152,7 +152,7 @@ describe("siliconflow built-in providers", () => { expect(glm?.api).toBe("openai-completions"); expect(glm?.baseUrl).toBe("https://api.siliconflow.com/v1"); - // Tier 2: absent from models.dev — reasoning and canonical limits recover + // Tier 2: absent from stencil.so — reasoning and canonical limits recover // from the bundled upstream/reseller reference, but provider-specific // pricing stays unknown instead of inheriting another host's values. const canonical = resolveModelReference("deepseek-ai/DeepSeek-V4-Pro", getBundledModelReferenceIndex()); @@ -168,16 +168,16 @@ describe("siliconflow built-in providers", () => { } expect(seen.urls).toContain("https://api.siliconflow.com/v1/models"); - expect(seen.urls.some(url => url.startsWith("https://models.dev/"))).toBe(true); + expect(seen.urls.some(url => url.startsWith("https://catalog.stencil.so/"))).toBe(true); expect(seen.authorization).toBe("Bearer sk-test"); }); - test("cn variant discovers against the China endpoint with cn models.dev pricing", async () => { + test("cn variant discovers against the China endpoint with cn stencil.so pricing", async () => { const seen: { urls: string[] } = { urls: [] }; const stubFetch: FetchImpl = async input => { const url = String(input); seen.urls.push(url); - if (url.startsWith("https://models.dev/")) { + if (url.startsWith("https://catalog.stencil.so/")) { return new Response(JSON.stringify(MODELS_DEV_STUB_PAYLOAD), { status: 200, headers: { "content-type": "application/json" }, @@ -204,11 +204,11 @@ describe("siliconflow built-in providers", () => { expect(seen.urls).toContain("https://api.siliconflow.cn/v1/models"); }); - test("models.dev lookup failure still yields endpoint-discovered models", async () => { + test("stencil.so lookup failure still yields endpoint-discovered models", async () => { const stubFetch: FetchImpl = async input => { const url = String(input); - if (url.startsWith("https://models.dev/")) { - throw new Error("models.dev stalled"); + if (url.startsWith("https://catalog.stencil.so/")) { + throw new Error("stencil.so stalled"); } const payload = { object: "list", @@ -223,7 +223,7 @@ describe("siliconflow built-in providers", () => { const options = siliconflowCnModelManagerOptions({ apiKey: "sk-test", fetch: stubFetch }); const models = await options.fetchDynamicModels?.(); expect(models?.map(model => model.id)).toEqual(["deepseek-ai/DeepSeek-V4-Pro"]); - // Canonical fallback still hydrates reasoning when models.dev is unreachable. + // Canonical fallback still hydrates reasoning when stencil.so is unreachable. expect(models?.[0]?.reasoning).toBe(true); }); }); diff --git a/packages/catalog/test/umans-provider.test.ts b/packages/catalog/test/umans-provider.test.ts index 197ac6431..8ad3cafbc 100644 --- a/packages/catalog/test/umans-provider.test.ts +++ b/packages/catalog/test/umans-provider.test.ts @@ -228,7 +228,7 @@ describe("umans provider catalog", () => { } }); - it("maps the models.dev Umans PAYG pricing to the Anthropic endpoint", () => { + it("maps the stencil.so Umans PAYG pricing to the Anthropic endpoint", () => { const models = mapModelsDevToModels( { "umans-ai": { diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index c408dac34..d3d66fca1 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -154,7 +154,7 @@ describe("ModelRegistry runtime discovery", () => { const endpointPrefix = "https://api.anthropic.com/"; return async (input, init) => { const url = String(input); - if (url === "https://models.dev/api.json") { + if (url === "https://catalog.stencil.so/models.json.zstd") { return Response.json({}); } if (url.startsWith(endpointPrefix) && url.endsWith("/models")) { @@ -1870,6 +1870,9 @@ describe("ModelRegistry runtime discovery", () => { test("llama.cpp selected model refresh does not resolve command api keys", async () => { const commandLogPath = path.join(tempDir, "llama-cpp-key-command.log"); + // Pre-create so the before/after comparison works whether or not + // registry construction happens to invoke the key command itself. + fs.writeFileSync(commandLogPath, ""); writeRawModelsJson({ "llama.cpp": { baseUrl: "http://127.0.0.1:8080", diff --git a/packages/coding-agent/test/model-registry-runtime-provider.test.ts b/packages/coding-agent/test/model-registry-runtime-provider.test.ts index dc3f5f25c..f7280e48c 100644 --- a/packages/coding-agent/test/model-registry-runtime-provider.test.ts +++ b/packages/coding-agent/test/model-registry-runtime-provider.test.ts @@ -27,7 +27,7 @@ describe("ModelRegistry runtime provider registration", () => { // Stub transport: reject every request so refresh("online") drives the full // online discovery path with deterministic, instant failures instead of real - // network. Provider fetches (dynamic + models.dev) are caught and swallowed, + // network. Provider fetches (dynamic + stencil.so) are caught and swallowed, // leaving the registry with its bundled catalog plus runtime overlays. const offlineFetch: FetchImpl = () => Promise.reject(new Error("network disabled in model-registry runtime test")); diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 5ee260cdf..6e3694f9b 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -702,7 +702,7 @@ describe("ModelRegistry", () => { }); const fetchMock: FetchImpl = async input => { const url = String(input); - if (url === "https://models.dev/api.json") return Response.json({}); + if (url === "https://catalog.stencil.so/models.json.zstd") return Response.json({}); if (url === "https://proxy.example/v1/models") { return Response.json({ data: [{ id: "claude-sonnet-5", display_name: "Claude Sonnet 5" }],