From dd2e7364768e68c88f0f389738ca82180178b960 Mon Sep 17 00:00:00 2001 From: metaphorics <152830360+metaphorics@users.noreply.github.com> Date: Sun, 14 Jun 2026 05:55:21 +0900 Subject: [PATCH] feat(mnemopi): add embedding variant setting (English / multilingual) (#2476) --- packages/coding-agent/CHANGELOG.md | 4 + .../src/config/settings-schema.ts | 28 +++- packages/coding-agent/src/mnemopi/config.ts | 10 +- .../test/mnemopi-embedding-variant.test.ts | 36 +++++ packages/mnemopi/CHANGELOG.md | 4 + packages/mnemopi/src/core/beam/store.ts | 75 +++++++++- packages/mnemopi/src/core/embeddings.ts | 2 +- packages/mnemopi/src/core/memory.ts | 5 + .../test/embedding-model-reconcile.test.ts | 140 ++++++++++++++++++ 9 files changed, 300 insertions(+), 4 deletions(-) create mode 100644 packages/coding-agent/test/mnemopi-embedding-variant.test.ts create mode 100644 packages/mnemopi/test/embedding-model-reconcile.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7fc96cb5e..4306590fd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added the `mnemopi.embeddingVariant` setting (`en` | `multilingual`) selecting a stronger SOTA local embedding model — `en` → `BAAI/bge-base-en-v1.5` (768d), `multilingual` → `intfloat/multilingual-e5-large` (1024d). `mnemopi.embeddingModel` still works as an advanced explicit override (it wins over the variant). Changing the active model wipes and rebuilds stored embeddings on the next start ([#2476](https://github.com/can1357/oh-my-pi/issues/2476)) + ## [15.12.5] - 2026-06-13 ### Changed diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 10bc8f316..edcd075fd 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -1894,6 +1894,31 @@ export const SETTINGS_SCHEMA = { condition: "mnemopiActive", }, }, + "mnemopi.embeddingVariant": { + type: "enum", + values: ["en", "multilingual"] as const, + default: "en", + ui: { + tab: "memory", + group: "Mnemopi", + label: "Embedding variant", + description: + "Local embedding model family. en = stronger English model; multilingual = cross-language model. Changing this rebuilds existing memory embeddings on next start.", + options: [ + { + value: "en", + label: "English (bge-base-en-v1.5)", + description: "BAAI/bge-base-en-v1.5 (768d), English-only", + }, + { + value: "multilingual", + label: "Multilingual (multilingual-e5-large)", + description: "intfloat/multilingual-e5-large (1024d), cross-language recall", + }, + ], + condition: "mnemopiActive", + }, + }, "mnemopi.autoRecall": { type: "boolean", default: true, @@ -1956,7 +1981,8 @@ export const SETTINGS_SCHEMA = { tab: "memory", group: "Mnemopi", label: "Mnemopi Embedding Model", - description: "Optional embedding model override passed to Mnemopi", + description: + "Advanced: explicit embedding model id that overrides the variant. Leave empty to use mnemopi.embeddingVariant.", condition: "mnemopiActive", }, }, diff --git a/packages/coding-agent/src/mnemopi/config.ts b/packages/coding-agent/src/mnemopi/config.ts index 7c6b51647..f0def843b 100644 --- a/packages/coding-agent/src/mnemopi/config.ts +++ b/packages/coding-agent/src/mnemopi/config.ts @@ -48,6 +48,14 @@ export function loadMnemopiConfig(settings: Settings, agentDir: string): Mnemopi const recallBanks = scoping === "global" ? scope.recallBanks : extendRecallWithLegacyBanks(scope.recallBanks, dbPath, cwd); const llmMode = settings.get("mnemopi.llmMode"); + const embeddingOverride = settings.get("mnemopi.embeddingModel"); + const embeddingVariant = settings.get("mnemopi.embeddingVariant"); + // Map the variant explicitly rather than indexing an object with the raw config + // value (which could resolve an inherited property like `__proto__`); any value + // other than the multilingual variant falls back to the English default. + const variantModel = + embeddingVariant === "multilingual" ? "intfloat/multilingual-e5-large" : "BAAI/bge-base-en-v1.5"; + const embeddingModel = embeddingOverride?.trim() || variantModel; return { dbPath, baseBank: scope.baseBank, @@ -69,7 +77,7 @@ export function loadMnemopiConfig(settings: Settings, agentDir: string): Mnemopi providerOptions: { noEmbeddings: settings.get("mnemopi.noEmbeddings"), debug: settings.get("mnemopi.debug"), - embeddingModel: settings.get("mnemopi.embeddingModel"), + embeddingModel, embeddingApiUrl: settings.get("mnemopi.embeddingApiUrl"), embeddingApiKey: settings.get("mnemopi.embeddingApiKey"), llm: diff --git a/packages/coding-agent/test/mnemopi-embedding-variant.test.ts b/packages/coding-agent/test/mnemopi-embedding-variant.test.ts new file mode 100644 index 000000000..8cbaa17db --- /dev/null +++ b/packages/coding-agent/test/mnemopi-embedding-variant.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from "bun:test"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { loadMnemopiConfig } from "@oh-my-pi/pi-coding-agent/mnemopi/config"; + +// `mnemopi.embeddingVariant` selects the concrete local embedding model, while an +// explicit `mnemopi.embeddingModel` is an advanced override that wins. Scoping is +// pinned to "global" so the resolver stays pure (no legacy-bank disk probing). +function embeddingModelFor(overrides: Record): string | undefined { + const settings = Settings.isolated({ "mnemopi.scoping": "global", ...overrides }); + return loadMnemopiConfig(settings, "/tmp/mnemopi-embedding-variant-test").providerOptions.embeddingModel; +} + +describe("loadMnemopiConfig embedding variant resolution", () => { + it("maps the en variant to BAAI/bge-base-en-v1.5", () => { + expect(embeddingModelFor({ "mnemopi.embeddingVariant": "en" })).toBe("BAAI/bge-base-en-v1.5"); + }); + + it("maps the multilingual variant to intfloat/multilingual-e5-large", () => { + expect(embeddingModelFor({ "mnemopi.embeddingVariant": "multilingual" })).toBe("intfloat/multilingual-e5-large"); + }); + + it("lets an explicit embeddingModel override win over the variant", () => { + expect( + embeddingModelFor({ + "mnemopi.embeddingVariant": "multilingual", + "mnemopi.embeddingModel": "openai/text-embedding-3-small", + }), + ).toBe("openai/text-embedding-3-small"); + }); + + it("ignores a blank override and falls back to the variant", () => { + expect(embeddingModelFor({ "mnemopi.embeddingVariant": "en", "mnemopi.embeddingModel": " " })).toBe( + "BAAI/bge-base-en-v1.5", + ); + }); +}); diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index 01bc96a28..a4c8ad3cd 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added a wipe-and-rebuild reconcile (`reconcileEmbeddingModel`) that runs when the configured embedding model changes. At store open, if the model stamped on stored `memory_embeddings` rows differs from the active `currentEmbeddingModel()`, the stale embeddings and their binary vectors are dropped and every existing memory is enqueued for background re-embedding (in bounded batches) at the new model/dimension. The destructive wipe is skipped whenever it could not be rebuilt — embeddings disabled via the runtime option or the `MNEMOPI_NO_EMBEDDINGS` env, or an unresolved (empty) active model — so a stale-but-valid corpus is never destroyed without a replacement. Recall degrades gracefully (FTS-only) for memories whose vectors are not yet rebuilt ([#2476](https://github.com/can1357/oh-my-pi/issues/2476)) + ## [15.12.4] - 2026-06-13 ### Fixed diff --git a/packages/mnemopi/src/core/beam/store.ts b/packages/mnemopi/src/core/beam/store.ts index 0a89b9da1..5a81ff144 100644 --- a/packages/mnemopi/src/core/beam/store.ts +++ b/packages/mnemopi/src/core/beam/store.ts @@ -1,12 +1,14 @@ import type { Database, SQLQueryBindings } from "bun:sqlite"; +import { logger } from "@oh-my-pi/pi-utils"; import { transaction } from "../../db"; import { toUtcIso } from "../../util/datetime"; import { generateId } from "../../util/ids"; +import { currentEmbeddingModel, embeddingsDisabled } from "../embeddings"; import { EpisodicGraph } from "../episodic-graph"; import { extractFactsSafe } from "../extraction"; import { getMnemopiRuntimeOptions, withMnemopiRuntimeOptions } from "../runtime-options"; import { storeFactStrings } from "./consolidate"; -import { scheduleEmbedding, vecAvailable, vecInsert } from "./helpers"; +import { type EmbedItem, scheduleEmbedding, vecAvailable, vecInsert } from "./helpers"; import type { BeamEvent, BeamMemoryState, @@ -248,6 +250,77 @@ function rowToDict(row: Row): Row { return { ...row }; } +/** Re-embedding batch size for a model-change rebuild — bounds each background + * embedding request instead of embedding the whole corpus in one call. */ +const EMBED_REBUILD_BATCH = 128; + +/** + * Reconcile stored embeddings against the active embedding model at store open. + * + * Every `memory_embeddings` row is stamped with the model that produced it (see + * `runEmbedding` in `helpers.ts`). When the configured embedding model changes, + * its vector dimension changes too, so the previously-stored vectors are no + * longer comparable. On a mismatch we wipe every stored vector — the + * `memory_embeddings` table, the `episodic_memory.binary_vector` column, and the + * sqlite-vec `vec_episodes` index — then enqueue all live memories for + * background re-embedding under the new model via `scheduleEmbedding`. + * + * Runs once per store open; a fresh store (no embeddings) or an already-current + * store is a no-op. The destructive wipe is skipped whenever it could not be + * rebuilt — embeddings disabled via the runtime option OR the + * `MNEMOPI_NO_EMBEDDINGS` env, or an unresolved (empty) active model — so a + * stale-but-valid corpus is never destroyed without a replacement. MUST run + * inside the active runtime-options scope so `currentEmbeddingModel()` / + * `embeddingsDisabled()` reflect the per-instance configuration. + */ +export function reconcileEmbeddingModel(beam: BeamMemoryState): void { + if (embeddingsDisabled()) return; + const active = currentEmbeddingModel().trim(); + if (active === "") return; + + // Stop at the first row whose stamped model differs from the active one + // (NULL/unstamped counts as a mismatch via `IS NOT`); an empty store or an + // all-current store short-circuits here without gathering the DISTINCT set. + const mismatch = beam.db.query("SELECT 1 FROM memory_embeddings WHERE model IS NOT ? LIMIT 1").get(active); + if (!mismatch) return; + + const staleModels = beam.db + .query("SELECT DISTINCT model FROM memory_embeddings WHERE model IS NOT ?") + .all(active) as { model: string | null }[]; + + const live = beam.db + .query(` + SELECT id AS memoryId, content FROM working_memory WHERE superseded_by IS NULL + UNION ALL + SELECT id AS memoryId, content FROM episodic_memory WHERE superseded_by IS NULL + `) + .all() as EmbedItem[]; + + transaction(beam.db, () => { + beam.db.prepare("DELETE FROM memory_embeddings").run(); + beam.db.prepare("UPDATE episodic_memory SET binary_vector = NULL").run(); + if (vecAvailable(beam.db)) { + try { + beam.db.prepare("DELETE FROM vec_episodes").run(); + } catch { + // sqlite-vec cleanup is best-effort; rebuild correctness takes precedence. + } + } + }); + + logger.info("mnemopi: embedding model changed, rebuilding", { + from: staleModels.map(row => row.model ?? "(unstamped)"), + to: active, + count: live.length, + }); + + // Re-embed in bounded batches so a corpus-wide rebuild never issues one giant + // embedding request; each batch is its own tracked background task. + for (let offset = 0; offset < live.length; offset += EMBED_REBUILD_BATCH) { + scheduleEmbedding(beam, live.slice(offset, offset + EMBED_REBUILD_BATCH)); + } +} + export function remember(beam: BeamMemoryState, content: string, options: StoreRememberOptions = {}): string { const source = options.source ?? "conversation"; const importance = options.importance ?? 0.5; diff --git a/packages/mnemopi/src/core/embeddings.ts b/packages/mnemopi/src/core/embeddings.ts index b098ddf4c..2119360bc 100644 --- a/packages/mnemopi/src/core/embeddings.ts +++ b/packages/mnemopi/src/core/embeddings.ts @@ -99,7 +99,7 @@ function inTestRuntime(): boolean { return $env.NODE_ENV === "test" || $env.BUN_ENV === "test"; } -function embeddingsDisabled(): boolean { +export function embeddingsDisabled(): boolean { const active = activeEmbeddingOptions(); if (active?.disabled !== undefined) { return active.disabled; diff --git a/packages/mnemopi/src/core/memory.ts b/packages/mnemopi/src/core/memory.ts index b7e7e62cd..90d911710 100644 --- a/packages/mnemopi/src/core/memory.ts +++ b/packages/mnemopi/src/core/memory.ts @@ -7,6 +7,7 @@ import type { MemoryInput, Metadata } from "../types"; import { AnnotationStore } from "./annotations"; import { BankManager } from "./banks"; import { BeamMemory, initBeam } from "./beam/index"; +import { reconcileEmbeddingModel } from "./beam/store"; import type { RecallEnhancedOptions, RecallOptions, RecallResult, SleepResult } from "./beam/types"; import { EpisodicGraph } from "./episodic-graph"; import { @@ -388,6 +389,10 @@ export class Mnemopi { } this.conn = this.beam.db; this.db = this.beam.db; + // Wipe-and-rebuild stale embeddings when the configured model changed since + // the vectors were written. Runs inside the runtime scope so + // `currentEmbeddingModel()` reflects this instance's configured model. + this.#withRuntimeOptions(() => reconcileEmbeddingModel(this.beam)); } close(): void { diff --git a/packages/mnemopi/test/embedding-model-reconcile.test.ts b/packages/mnemopi/test/embedding-model-reconcile.test.ts new file mode 100644 index 000000000..9612fde59 --- /dev/null +++ b/packages/mnemopi/test/embedding-model-reconcile.test.ts @@ -0,0 +1,140 @@ +/** + * Wipe-and-rebuild reconcile: each `memory_embeddings` row is stamped with the + * model that produced it. When the configured embedding model changes the vector + * dimension changes too, so on store open `reconcileEmbeddingModel` (wired into + * the `Mnemopi` constructor) wipes every stale vector and enqueues all live + * memories for re-embedding under the new model. A matching model is a no-op. + */ + +import { Database } from "bun:sqlite"; +import { describe, expect, it } from "bun:test"; +import "./setup"; +import { initBeam } from "@oh-my-pi/pi-mnemopi/core/beam"; +import { Mnemopi } from "@oh-my-pi/pi-mnemopi/core/memory"; + +const OLD_MODEL = "BAAI/bge-small-en-v1.5"; +const NEW_MODEL = "intfloat/multilingual-e5-large"; + +// Deterministic fastembed-shaped provider so the background rebuild actually +// writes rows (and stamps them with the active model) under the test runtime. +function fakeEmbed() { + return async function* embed(texts: readonly string[]) { + yield texts.map(() => [0.1, 0.2, 0.3, 0.4]); + }; +} + +function seedDb(model: string): { db: Database; ids: string[] } { + const db = new Database(":memory:"); + initBeam(db); + const ts = new Date().toISOString(); + db.prepare( + "INSERT INTO working_memory (id, content, source, timestamp, session_id) VALUES (?, ?, 'test', ?, 'default')", + ).run("wm-1", "alpha working memory", ts); + db.prepare( + "INSERT INTO episodic_memory (id, content, source, timestamp, session_id, binary_vector) VALUES (?, ?, 'test', ?, 'default', ?)", + ).run("ep-1", "beta episodic memory", ts, new Uint8Array([1, 2, 3, 4])); + for (const id of ["wm-1", "ep-1"]) { + db.prepare("INSERT INTO memory_embeddings (memory_id, embedding_json, model) VALUES (?, ?, ?)").run( + id, + JSON.stringify([1, 0, 0, 0]), + model, + ); + } + return { db, ids: ["wm-1", "ep-1"] }; +} + +function countEmbeddings(memory: Mnemopi): number { + return (memory.conn.query("SELECT COUNT(*) AS n FROM memory_embeddings").get() as { n: number }).n; +} + +describe("reconcileEmbeddingModel on store open", () => { + it("wipes stale embeddings + binary vectors and re-embeds when the model changed", async () => { + const { db, ids } = seedDb(OLD_MODEL); + const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } }); + try { + // Reconcile fired in the constructor: stale vector rows are gone and the + // episodic binary vector was cleared. The async rebuild is enqueued but + // has not yet run. + expect(countEmbeddings(memory)).toBe(0); + const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as { + v: Uint8Array | null; + }; + expect(ep.v).toBeNull(); + expect(memory.beam.pendingExtractions.size).toBeGreaterThanOrEqual(1); + + // The background rebuild repopulates every live memory, stamped with the + // new model. + await memory.flushExtractions(); + const rows = memory.conn.query("SELECT memory_id, model FROM memory_embeddings ORDER BY memory_id").all() as { + memory_id: string; + model: string; + }[]; + expect(rows.map(row => row.memory_id).sort()).toEqual([...ids].sort()); + expect(rows.every(row => row.model === NEW_MODEL)).toBe(true); + } finally { + memory.close(); + db.close(); + } + }); + + it("leaves embeddings untouched when the stored model already matches", () => { + const { db } = seedDb(NEW_MODEL); + const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } }); + try { + // No mismatch -> no wipe, no rebuild enqueued. + expect(countEmbeddings(memory)).toBe(2); + expect(memory.beam.pendingExtractions.size).toBe(0); + // The no-op path must preserve the episodic binary vector too (regression + // guard against an unconditional clear that would silently drop vectors). + const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as { + v: Uint8Array | null; + }; + expect(ep.v).not.toBeNull(); + expect(Array.from(ep.v as Uint8Array)).toEqual([1, 2, 3, 4]); + } finally { + memory.close(); + db.close(); + } + }); + + it("does not wipe when embeddings are disabled via the MNEMOPI_NO_EMBEDDINGS env", () => { + const { db } = seedDb(OLD_MODEL); + const previous = process.env.MNEMOPI_NO_EMBEDDINGS; + process.env.MNEMOPI_NO_EMBEDDINGS = "1"; + let memory: Mnemopi | undefined; + try { + // The model differs, but with embeddings disabled the rebuild would + // produce nothing — so the stale-but-present vectors must survive. + memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } }); + expect(countEmbeddings(memory)).toBe(2); + expect(memory.beam.pendingExtractions.size).toBe(0); + const ep = memory.conn.query("SELECT binary_vector AS v FROM episodic_memory WHERE id = 'ep-1'").get() as { + v: Uint8Array | null; + }; + expect(ep.v).not.toBeNull(); + } finally { + if (previous === undefined) { + delete process.env.MNEMOPI_NO_EMBEDDINGS; + } else { + process.env.MNEMOPI_NO_EMBEDDINGS = previous; + } + memory?.close(); + db.close(); + } + }); + + it("does not wipe when the active embedding model is empty", () => { + const { db } = seedDb(OLD_MODEL); + let memory: Mnemopi | undefined; + try { + // An explicit empty model resolves to no embedder; wiping would be + // unrecoverable, so the reconcile must skip. + memory = new Mnemopi({ db, embeddings: { model: "", provider: fakeEmbed() } }); + expect(countEmbeddings(memory)).toBe(2); + expect(memory.beam.pendingExtractions.size).toBe(0); + } finally { + memory?.close(); + db.close(); + } + }); +});