fix(catalog): self-heal a corrupt models.db model cache

The shared SQLite model cache wrapped every read/write in a blanket catch that swallowed unrecoverable SQLITE_CORRUPT*/SQLITE_NOTADB failures as best-effort misses, and getSharedDb cached the broken handle. A physically corrupt models.db therefore permanently disabled cached catalogs across processes: a successful live discovery could never overwrite the corrupt cache, so a runtime extension with no bundled catalog was stuck with only its bootstrap model.

On those unrecoverable codes the cache now self-heals: close the handle, quarantine models.db(+-wal/-shm) to models.db.corrupt-<ts>, recreate a fresh database, and retry the operation once. SQLITE_BUSY, permission, and unrelated errors keep their existing best-effort paths. The SQLITE_BUSY/corruption classifiers moved to @oh-my-pi/pi-utils so the credential store and model cache share one implementation.

Fixes #8867
This commit is contained in:
roboomp
2026-08-19 09:00:07 +00:00
parent d94bdfa1bb
commit 289cc19325
6 changed files with 227 additions and 30 deletions
+11 -24
View File
@@ -8,7 +8,13 @@ import { Database, type Statement } from "bun:sqlite";
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import { getAgentDbPath, getDbBusyTimeoutMs, logger } from "@oh-my-pi/pi-utils";
import {
getAgentDbPath,
getDbBusyTimeoutMs,
isSqliteBusyError,
isSqliteCorruptionError,
logger,
} from "@oh-my-pi/pi-utils";
import type {
AuthCredential,
AuthCredentialStore,
@@ -87,29 +93,10 @@ const LEGACY_CODEX_BLOCK_PROVIDER_KEY = "openai-codex:oauth";
const LEGACY_CODEX_BLOCK_SCOPE = "shared";
const CODEX_METER_BLOCK_SCOPES = ["chat", "spark"] as const;
/**
* SQLite's busy result code family — base `SQLITE_BUSY` plus the extended
* variants `SQLITE_BUSY_RECOVERY` (concurrent WAL recovery), `SQLITE_BUSY_SNAPSHOT`,
* and `SQLITE_BUSY_TIMEOUT`. All warrant the same backoff-and-retry treatment.
*/
export function isSqliteBusyError(err: unknown): boolean {
if (err === null || typeof err !== "object") return false;
const code = (err as { code?: unknown }).code;
return typeof code === "string" && code.startsWith("SQLITE_BUSY");
}
/**
* SQLite's unrecoverable-corruption result codes — the `SQLITE_CORRUPT` family
* (base plus extended variants like `SQLITE_CORRUPT_VTAB` / `SQLITE_CORRUPT_INDEX`)
* and `SQLITE_NOTADB` (the file header is not a database). Unlike
* {@link isSqliteBusyError}, these never clear by retrying: the store must be
* repaired or replaced, so callers latch and stop touching it.
*/
export function isSqliteCorruptionError(err: unknown): boolean {
if (err === null || typeof err !== "object" || !("code" in err)) return false;
const code = err.code;
return typeof code === "string" && (code.startsWith("SQLITE_CORRUPT") || code === "SQLITE_NOTADB");
}
// SQLite error classifiers live in pi-utils so the credential store and the
// model cache share one implementation; re-exported here to preserve the
// pre-existing `@oh-my-pi/pi-ai/auth-storage` surface.
export { isSqliteBusyError, isSqliteCorruptionError };
function normalizeStoredAccountId(accountId: string | null | undefined): string | null {
const normalized = accountId?.trim();
+2
View File
@@ -8,6 +8,8 @@
### Fixed
- Fixed a physically corrupt `models.db` (`SQLITE_CORRUPT*` / `SQLITE_NOTADB`, "database disk image is malformed") permanently disabling the model cache. The shared read/write paths swallowed unrecoverable SQLite corruption as a best-effort miss and cached the broken handle, so a successful live catalog could never overwrite the corrupt cache and every later process repeated the miss — a runtime provider extension with no bundled catalog was left with only its bootstrap model. Corruption now self-heals: the cache closes the handle, quarantines `models.db`(+`-wal`/`-shm`) aside, recreates a fresh database, and retries the operation once; `SQLITE_BUSY`, permission, and unrelated errors keep their existing best-effort paths ([#8867](https://github.com/can1357/oh-my-pi/issues/8867)).
- Fixed local Qwen 3.8+ models (llama.cpp, vLLM, loopback custom providers) exposing the generic `minimal..high` thinking ladder instead of the chat template's real `low`/`medium`/`xhigh` `reasoning_effort` tiers. The derived metadata now marks thinking as mandatory (the official 3.8 template raises on `enable_thinking: false`), vLLM-served Qwen routes through the `chat_template_kwargs` dialect (top-level `enable_thinking` is ignored by vLLM), and vLLM discovery lights up the reasoning dial for Qwen 3.8+ ids its `/v1/models` endpoint reports as non-reasoning.
- Fixed `deepseek-v4-pro-0813` surfacing from Alibaba Token Plan discovery with `contextWindow`/`maxTokens` of `null`. The dated DeepSeek V4 Pro snapshot was missing from `ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS`, so unlike its `deepseek-v4-flash-0731` sibling it fell through to unknown limits ([#8847](https://github.com/can1357/oh-my-pi/issues/8847)).
- Cloud Code Assist Gemini 3.6/3.7 Flash no longer maps user `minimal` to wire `thinkingLevel: MINIMAL` when that effort is aliased onto the `-low` SKU. The request now sends `LOW`, which those SKUs accept.
+67 -6
View File
@@ -3,7 +3,8 @@
* Replaces per-provider JSON files with a single cache.db.
*/
import { Database } from "bun:sqlite";
import { getModelDbPath } from "@oh-my-pi/pi-utils";
import { renameSync } from "node:fs";
import { getModelDbPath, isEnoent, isSqliteCorruptionError, logger } from "@oh-my-pi/pi-utils";
import type { Api, Model, ModelSpec } from "./types";
// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
@@ -91,13 +92,14 @@ function openDb(resolvedPath: string): Database {
return db;
}
function getSharedDb(): Database {
const resolvedPath = getModelDbPath();
function getSharedDb(resolvedPath: string): Database {
if (sharedDb && sharedDbPath === resolvedPath) {
return sharedDb;
}
if (sharedDb) {
sharedDb.close();
sharedDb = null;
sharedDbPath = null;
}
const db = openDb(resolvedPath);
sharedDb = db;
@@ -105,9 +107,9 @@ function getSharedDb(): Database {
return db;
}
function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) => T): T {
if (!dbPath) return useDb(getSharedDb());
const db = openDb(dbPath);
function runModelCacheDb<T>(resolvedPath: string, shared: boolean, useDb: (db: Database) => T): T {
if (shared) return useDb(getSharedDb(resolvedPath));
const db = openDb(resolvedPath);
try {
return useDb(db);
} finally {
@@ -115,6 +117,65 @@ function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) =
}
}
// Paths already reported corrupt this process: the first unrecoverable failure
// is logged at `error`, later heals at `debug`, so a dying disk cannot spam.
const reportedCorruptPaths = new Set<string>();
/**
* Move a physically corrupt `models.db` (plus its `-wal`/`-shm` sidecars) aside
* so {@link openDb} can recreate a fresh cache at the original path. Renames are
* best-effort: a vanished sidecar (already healed by a peer process) is fine,
* and any other rename failure is left for {@link openDb} to surface.
*/
function quarantineCorruptModelCache(resolvedPath: string): void {
const stamp = Date.now();
for (const suffix of ["", "-wal", "-shm"]) {
try {
renameSync(`${resolvedPath}${suffix}`, `${resolvedPath}.corrupt-${stamp}${suffix}`);
} catch (err) {
if (!isEnoent(err)) {
logger.debug("model cache: could not quarantine corrupt file", { path: `${resolvedPath}${suffix}` });
}
}
}
}
/**
* Recover from unrecoverable `models.db` corruption: drop the cached handle,
* quarantine the broken files, and let the next open recreate the cache. A
* corrupt cache would otherwise be re-queried on every read/write forever,
* permanently masking a successful live catalog (issue #8867). Only
* {@link isSqliteCorruptionError} codes reach here; BUSY/permission errors keep
* their existing best-effort paths.
*/
function healCorruptModelCache(resolvedPath: string, shared: boolean, err: unknown): void {
if (shared && sharedDb) {
sharedDb.close();
sharedDb = null;
sharedDbPath = null;
}
quarantineCorruptModelCache(resolvedPath);
const code = err && typeof err === "object" && "code" in err ? err.code : undefined;
if (reportedCorruptPaths.has(resolvedPath)) {
logger.debug("model cache: re-healed corrupt database", { path: resolvedPath, code });
} else {
reportedCorruptPaths.add(resolvedPath);
logger.error("model cache corrupt; quarantined and recreated a fresh cache", { path: resolvedPath, code });
}
}
function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) => T): T {
const resolvedPath = dbPath ?? getModelDbPath();
const shared = dbPath === undefined;
try {
return runModelCacheDb(resolvedPath, shared, useDb);
} catch (err) {
if (!isSqliteCorruptionError(err)) throw err;
healCorruptModelCache(resolvedPath, shared, err);
return runModelCacheDb(resolvedPath, shared, useDb);
}
}
function migrateCacheSchema(db: Database): void {
const stmt = db.prepare("PRAGMA table_info(model_cache)");
try {
@@ -0,0 +1,113 @@
// Contract (#8867): a physically corrupt models.db must not permanently
// disable the model cache. On an unrecoverable SQLITE_CORRUPT/NOTADB failure
// the shared cache quarantines the broken file, recreates a fresh database,
// and retries the operation once — so a successful live catalog can be
// persisted and read back by later processes. Non-corruption reads of a
// healthy cache never quarantine anything.
import { Database } from "bun:sqlite";
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { readModelCache, writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import type { Model } from "@oh-my-pi/pi-catalog/types";
import { removeWithRetries } from "../../utils/src/temp";
const TTL_MS = 24 * 60 * 60 * 1000;
function createModel(id: string): Model<"openai-completions"> {
return buildModel({
id,
name: id,
api: "openai-completions",
provider: "runtime-ext",
baseUrl: "https://ext.example/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 4096,
maxTokens: 1024,
});
}
async function foldWalIntoMainDb(dbPath: string): Promise<void> {
const db = new Database(dbPath);
db.run("PRAGMA wal_checkpoint(TRUNCATE)");
db.close();
await removeWithRetries(`${dbPath}-wal`);
await removeWithRetries(`${dbPath}-shm`);
}
/** Clobber every byte after the 100-byte header — a valid header over garbage pages yields SQLITE_CORRUPT. */
async function corruptDbPages(dbPath: string): Promise<void> {
await foldWalIntoMainDb(dbPath);
const buf = await fs.readFile(dbPath);
buf.fill(0xff, 100);
await fs.writeFile(dbPath, buf);
}
async function quarantinedFiles(dir: string): Promise<string[]> {
const entries = await fs.readdir(dir);
return entries.filter(name => name.startsWith("models.db.corrupt-"));
}
describe("model cache corruption self-heal (#8867)", () => {
let tempDir = "";
let dbPath = "";
beforeEach(async () => {
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-corrupt-cache-"));
dbPath = path.join(tempDir, "models.db");
});
afterEach(async () => {
if (tempDir) {
await removeWithRetries(tempDir);
tempDir = "";
dbPath = "";
}
});
it("quarantines a SQLITE_CORRUPT cache and persists the authoritative catalog", async () => {
writeModelCache("runtime-ext", Date.now(), [createModel("bootstrap")], true, "fp1", dbPath);
await corruptDbPages(dbPath);
// A read of the corrupt file must not throw and must self-heal to an empty cache.
expect(readModelCache<"openai-completions">("runtime-ext", TTL_MS, Date.now, dbPath)).toBeNull();
expect((await quarantinedFiles(tempDir)).length).toBeGreaterThan(0);
// The successful live catalog now persists into the recreated database...
writeModelCache(
"runtime-ext",
Date.now(),
[createModel("discovered-a"), createModel("discovered-b")],
true,
"fp2",
dbPath,
);
// ...and a later process (fresh read) sees it instead of a permanent miss.
const healed = readModelCache<"openai-completions">("runtime-ext", TTL_MS, Date.now, dbPath);
expect(healed?.models.map(model => model.id)).toEqual(["discovered-a", "discovered-b"]);
});
it("recreates a SQLITE_NOTADB cache on write so discovery can persist", async () => {
writeModelCache("runtime-ext", Date.now(), [createModel("bootstrap")], true, "fp1", dbPath);
// Overwrite with bytes that are not a SQLite database at all.
await fs.writeFile(dbPath, Buffer.from("not a database".repeat(64)));
writeModelCache("runtime-ext", Date.now(), [createModel("discovered")], true, "fp2", dbPath);
expect((await quarantinedFiles(tempDir)).length).toBeGreaterThan(0);
const healed = readModelCache<"openai-completions">("runtime-ext", TTL_MS, Date.now, dbPath);
expect(healed?.models.map(model => model.id)).toEqual(["discovered"]);
});
it("never quarantines a healthy cache", async () => {
writeModelCache("runtime-ext", Date.now(), [createModel("bootstrap")], true, "fp1", dbPath);
const cached = readModelCache<"openai-completions">("runtime-ext", TTL_MS, Date.now, dbPath);
expect(cached?.models.map(model => model.id)).toEqual(["bootstrap"]);
expect(await quarantinedFiles(tempDir)).toEqual([]);
});
});
+1
View File
@@ -28,6 +28,7 @@ export { AbortError, ChildProcess, Exception, NonZeroExitError } from "./ptree";
export * from "./runtime-install";
export * from "./sanitize-text";
export * from "./snowflake";
export * from "./sqlite";
export * from "./stderr-guard";
export * from "./stream";
export * from "./tab-spacing";
+33
View File
@@ -0,0 +1,33 @@
/**
* Shared classifiers for `bun:sqlite` error result codes.
*
* Every omp SQLite store (`agent.db` credential/usage store, `models.db` model
* cache, `history.db`) needs the same two distinctions: a transient BUSY that
* clears by retrying, and an unrecoverable corruption that never does. Keeping
* one implementation here prevents the classifiers from drifting between the
* credential store and the model cache.
*/
/**
* SQLite's busy result-code family — base `SQLITE_BUSY` plus the extended
* variants `SQLITE_BUSY_RECOVERY` (concurrent WAL recovery), `SQLITE_BUSY_SNAPSHOT`,
* and `SQLITE_BUSY_TIMEOUT`. All warrant the same backoff-and-retry treatment.
*/
export function isSqliteBusyError(err: unknown): boolean {
if (err === null || typeof err !== "object" || !("code" in err)) return false;
const code = err.code;
return typeof code === "string" && code.startsWith("SQLITE_BUSY");
}
/**
* SQLite's unrecoverable-corruption result codes — the `SQLITE_CORRUPT` family
* (base plus extended variants like `SQLITE_CORRUPT_VTAB` / `SQLITE_CORRUPT_INDEX`)
* and `SQLITE_NOTADB` (the file header is not a database). Unlike
* {@link isSqliteBusyError}, these never clear by retrying: the store must be
* repaired or replaced, so callers latch, quarantine, or recreate the file.
*/
export function isSqliteCorruptionError(err: unknown): boolean {
if (err === null || typeof err !== "object" || !("code" in err)) return false;
const code = err.code;
return typeof code === "string" && (code.startsWith("SQLITE_CORRUPT") || code === "SQLITE_NOTADB");
}